brandonmusic commited on
Commit
43a635b
·
verified ·
1 Parent(s): f8336d6

Publish selected TP4 reference, full speed evidence and matched FP8/NVFP4 cache KLD

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. README.md +181 -58
  2. compose.yaml +8 -3
  3. image-record.json +34 -126
  4. results/image-record-historical-20260908.json +217 -0
  5. results/kld-reference-20260909/README.md +33 -0
  6. results/kld-reference-20260909/audit.json +30 -0
  7. results/kld-reference-20260909/comparison.json +1053 -0
  8. results/speed-20260909/README.md +40 -0
  9. results/speed-20260909/benchmark-index.json +0 -0
  10. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-command.json +27 -0
  11. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-receipt.json +5 -0
  12. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json +1394 -0
  13. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.log +123 -0
  14. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-command.json +31 -0
  15. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-receipt.json +5 -0
  16. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json +396 -0
  17. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.log +56 -0
  18. results/speed-20260909/evidence/batch16384-speed-window-01/results-01/failure.json +3 -0
  19. results/speed-20260909/evidence/candidate-graph-analysis-02/REPORT.md +20 -0
  20. results/speed-20260909/evidence/candidate-graph-analysis-02/SUMMARY.json +242 -0
  21. results/speed-20260909/evidence/candidate-graph-profile-01/results-01/failure.json +3 -0
  22. results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c1-8k/result.json +345 -0
  23. results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c4-8k/result.json +381 -0
  24. results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-32k/result.json +257 -0
  25. results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-64k/result.json +257 -0
  26. results/speed-20260909/evidence/candidate-graph-profile-02/results-01/result.json +7 -0
  27. results/speed-20260909/evidence/candidate-kernel-window-02/results-01/result.json +23 -0
  28. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-command.json +27 -0
  29. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-receipt.json +5 -0
  30. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json +2366 -0
  31. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.log +126 -0
  32. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-command.json +27 -0
  33. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-receipt.json +5 -0
  34. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json +1394 -0
  35. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.log +123 -0
  36. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-command.json +31 -0
  37. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-receipt.json +5 -0
  38. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json +396 -0
  39. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.log +56 -0
  40. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-command.json +27 -0
  41. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-receipt.json +5 -0
  42. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json +2342 -0
  43. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.log +126 -0
  44. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-command.json +27 -0
  45. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-receipt.json +5 -0
  46. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json +1394 -0
  47. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.log +123 -0
  48. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-command.json +31 -0
  49. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-receipt.json +5 -0
  50. results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json +396 -0
README.md CHANGED
@@ -18,13 +18,175 @@ tags:
18
 
19
  # GLM-5.3-Flash TrellisMX-MXFP8
20
 
21
- **Weight upload complete; remote inventory and all 168 sidecar SHA-256 identities verified.**
22
- The HF-layout GPU serving test is still pending. Consult `release-status.json`.
 
 
23
 
24
- This is the measured **17-K5 / 25-K4 coupled checkpoint**, covering all
25
- 42 routed layers (3–44). Historical DCP1 mean true-decode KLD is **0.0341811459** on 32
26
- already-opened conditional-fit development windows. The routed tensors store
27
- **4.6587417643 bits/weight**, including metadata; they are not uniform 4.25 bpw.
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
28
 
29
  ## What TrellisMX does
30
 
@@ -111,61 +273,20 @@ Existing prior art makes that broad wording inappropriate. The distinctive
111
  combination above describes this implementation without asserting an
112
  unverified first-in-history claim or general superiority over its sources.
113
 
114
- ## Required runtime and model layout
115
 
116
- This repository contains **168 TP4 routed-expert sidecars (177,269,057,440
117
- bytes)** plus serving metadata, scripts, license and evaluation receipts.
118
- They replace the carrier's routed experts at load time. They are not an
119
- additional expert ensemble, and this is **not a standalone Transformers
120
- checkpoint**. Do not point stock Transformers or unmodified vLLM at it.
121
 
122
- The working runtime additionally requires the stock carrier:
123
- [`local-inference-lab/GLM-5.3-Flash-NVFP4`](https://huggingface.co/local-inference-lab/GLM-5.3-Flash-NVFP4/tree/520de24eabf507659eaef7c70f14fd584527facc),
124
- revision `520de24eabf507659eaef7c70f14fd584527facc`. Its attention, shared
125
- experts, embeddings, output head and MTP components remain part of the served
126
- model. Keeping this exact dependency preserves the existing loader contract;
127
- the reported logical model payload is not a claim that the two download
128
- directories together occupy only that many bytes.
129
 
130
- The Docker image is already published on
131
- [Docker Hub](https://hub.docker.com/r/verdictai/trellismx):
 
 
 
132
 
133
- ```text
134
- verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f
135
- ```
136
-
137
- The image stays in the container registry; this card links its immutable
138
- digest and includes `compose.yaml` and `serve.sh`. No encoder, calibration
139
- corpus or teacher logits are included in this HF release.
140
-
141
- ## Serving
142
-
143
- Requires four RTX PRO 6000 Blackwell 96GB SM120 GPUs and the NVIDIA Container
144
- Toolkit. The current r27 recipe uses TP4/DCP4, probabilistic MTP3 with standard rejection, B12X_MLA_SPARSE attention,
145
- NVFP4 MLA KV (`nvfp4_ds_mla`), native P8 MoE with E4M3 activations, CUDA graphs,
146
- B12X PCIe collectives, prefix caching, a 4096-token scheduler batch and at most 16 sequences. GPU memory utilization is 0.97.
147
- It does not alter GPU power limits, memory clocks or host services.
148
-
149
- After `release-status.json` reports upload completion:
150
-
151
- ```bash
152
- hf download brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8 --local-dir ./trellismx-p8
153
- hf download local-inference-lab/GLM-5.3-Flash-NVFP4 \
154
- --revision 520de24eabf507659eaef7c70f14fd584527facc --local-dir ./glm53-carrier
155
- cd trellismx-p8
156
- export MODEL_ROOT=/absolute/path/to/glm53-carrier
157
- docker compose -f compose.yaml config --quiet
158
- docker compose -f compose.yaml up -d
159
- ```
160
-
161
- Default port: 8000. Default maximum model length: **1,000,000 tokens**,
162
- within the model's declared 1,048,576 positions. That new launch limit is
163
- **not a tested 1M-context accuracy claim**. The HF-layout launch has CPU
164
- configuration checks only; a clean GPU download-to-serving test is pending.
165
- Use a firewall or authenticated gateway before exposing the endpoint beyond
166
- a trusted network. API credentials are not baked into the image.
167
-
168
- ## Current r27 DCP4 results
169
 
170
  | Measurement | Result | Conditions |
171
  | --- | --- | --- |
@@ -190,7 +311,7 @@ client snapshot and redaction hashes](results/r27-20260908/README.md). The
190
  benchmark client's larger generic capacity estimate is not valid for this
191
  split-cache layout.
192
 
193
- ### Current r27 KV-cache KLD comparison
194
 
195
  | Runtime / measurement image | KV dtype | Attention backend | Mean true-decode KLD | Window BCa 95% interval |
196
  | --- | --- | --- | ---: | --- |
@@ -301,6 +422,8 @@ not reported by this release; no values are inferred for missing metrics.
301
  See the [full GitHub results and receipts](https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/blob/03527b1f092cfbf274ae3ca78cabb23077c37057/results/RC5_RELEASE_RESULTS_20260907.md),
302
  and the included `results/` and `evidence/` files.
303
 
 
 
304
  ## Codec and limitations
305
 
306
  Procedural MCG trellis streams decode in registers to E4M3 for native
 
18
 
19
  # GLM-5.3-Flash TrellisMX-MXFP8
20
 
21
+ A compressed GLM-5.3-Flash checkpoint whose routed experts reconstruct directly
22
+ into **FP8 Tensor Core operands**. The September 9 reference stack is the selected
23
+ runtime: **222.100 tokens/s C1 decode at 8K** and **8,457 tokens/s prefill at 32K**
24
+ in separate `llm_decode_bench` runs on four RTX PRO 6000 Blackwell GPUs at 300W each.
25
 
26
+ ## Recommended reference stack September 9, 2026
27
+
28
+ [Docker Hub](https://hub.docker.com/r/verdictai/trellismx) image:
29
+
30
+ ```text
31
+ verdictai/trellismx:glm53-flash-p8-r27-reference-20260909
32
+ ```
33
+
34
+ Immutable image reference:
35
+
36
+ ```text
37
+ verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf
38
+ ```
39
+
40
+ Use the supplied [Docker Compose configuration](compose.yaml) and
41
+ [serving script](serve.sh). The [reference runtime recipe](runtime-reference-20260909/Dockerfile)
42
+ and [source manifest](runtime-reference-20260909/source-manifest.json) record the
43
+ inference overlays. The checkpoint has not been re-encoded for this update.
44
+
45
+ | Setting | Selected reference |
46
+ | --- | --- |
47
+ | GPUs / measured power | 4 × RTX PRO 6000 Blackwell 96GB (SM120), 300W per GPU |
48
+ | Parallelism | TP4 / DCP4 |
49
+ | Routed-expert math | E4M3 FP8, native UE8M0 scales per 32 elements |
50
+ | MLA KV cache | NVFP4 (`nvfp4_ds_mla`) |
51
+ | Speculative decoding | Probabilistic MTP3, standard rejection |
52
+ | Execution | CUDA graphs, B12X MLA attention and PCIe collectives |
53
+ | Scheduler | 24 sequence slots, 4,096-token batch |
54
+ | Memory / maximum request length | 0.97 GPU memory utilization / 1,000,000 tokens |
55
+
56
+ **FP8 math and NVFP4 KV cache describe different parts of the model.** Compressed
57
+ K4/K5 expert weights decode to FP8 operands for multiplication. NVFP4 stores the
58
+ MLA attention cache. Choosing FP8 KV instead changes the cache representation;
59
+ it does not turn this into an FP4-math checkpoint. The P4 design is separate.
60
+
61
+ Weight upload is complete: the remote inventory and all **168 sidecar SHA-256
62
+ identities** were verified. Local serving and measurements use the same checkpoint.
63
+ A separate clean download-to-GPU test of the HF directory layout has not been run;
64
+ see [release status](release-status.json).
65
+
66
+ ## Measured speed and KV capacity
67
+
68
+ Rates below are **tokens/s from `llm_decode_bench`**. C1 is one request;
69
+ C4 and C8 are aggregate throughput across concurrent requests, not per user.
70
+ Context labels are nominal benchmark cells; raw logs retain actual prompt sizes.
71
+
72
+ | Context | Prefill | C1 decode | C4 decode, aggregate | C8 decode, aggregate |
73
+ | --- | ---: | ---: | ---: | ---: |
74
+ | 0K | — | 204.611 | 327.670 | — |
75
+ | 8K | — | 222.100 | 318.560 | 594.639 |
76
+ | 16K | — | 216.636 | 319.157 | 602.965 |
77
+ | 32K | 8,457 | 204.930 | 329.343 | — |
78
+ | 64K | 8,443 | 199.446 | 330.096 | — |
79
+ | 128K | 8,323 | 204.445 | 334.086 | — |
80
+
81
+ This table combines the short-context screen with the separate expanded-context
82
+ pass. The earlier comparison screen measured **8,407 tokens/s prefill at both
83
+ 32K and 64K**. All runs are retained rather than selecting one figure as a repeated-run average.
84
+ The reference was selected for balanced use: it had the highest observed C1
85
+ throughput at 0K and 8K among the three finalists. Task-count performed better
86
+ at 16K C4 (**353.532 tokens/s**), but neither alternative was measured at C1
87
+ 32K–128K. Reference is not established as the fastest candidate for every workload.
88
+
89
+ The engine reported **23,562,091 aggregate effective KV tokens** in the expanded
90
+ reference run; separate short-context sessions reported **23,568,627**. This is
91
+ allocated capacity across requests, not a tested single-request context length
92
+ or a full-capacity stress result. The configured request limit is 1,000,000 tokens.
93
+ Use the engine's split-cache metric, not the benchmark client's generic
94
+ block-count-times-DCP estimate.
95
+
96
+ These are exploratory measurements, with one server preparation per configuration.
97
+ Later screens used a minimum 90-second idle period and required all GPUs at or below
98
+ 55°C for 30 continuous seconds before each cell; earlier runs retain their original
99
+ protocols. MTP acceptance, generated output and clocks can vary. No independently
100
+ repeated speed qualification or long-context accuracy claim is implied.
101
+
102
+ [All September 9 results, candidate comparison and protocol](results/speed-20260909/README.md)
103
+ include the full benchmark index, JSON, logs, negative results and diagnostic runs.
104
+ Profiled diagnostics are identified separately from serving speed measurements.
105
+
106
+ ## Reference-stack KLD: matched FP8 and NVFP4 MLA cache
107
+
108
+ Lower KLD means the student's next-token distribution is closer to the BF16
109
+ teacher on the measured inputs. It is not a direct score for general answer quality.
110
+
111
+ | Cache on the September 9 reference | Mean true-decode KLD | Status |
112
+ | --- | ---: | --- |
113
+ | FP8 KV | **0.0319451732** | 32/32; receipt audit passed |
114
+ | NVFP4 MLA KV | **0.0354562238** | 32/32; receipt audit passed |
115
+
116
+ FP8 has lower observed KLD in **22/32 windows**. The paired mean difference
117
+ (FP8 minus NVFP4) is **−0.0035110506**, with a window BCa 95% interval
118
+ [−0.0081718772, −0.0013675941]. [Full results and audit](results/kld-reference-20260909/README.md).
119
+
120
+ Both arms use the **same 32 previously opened conditional-fit development windows**
121
+ as the prior measurement, with the same token arrays and BF16 teacher. Each window
122
+ contains 2,048 input tokens and 2,047 prediction rows; row zero is excluded, leaving
123
+ **2,046 true-decode rows per window**. Scores use CPU FP64 KL(teacher || student)
124
+ and the equal mean of the 32 window means.
125
+
126
+ The matched setup is TP4/DCP4, **MTP off**, one sequence, a 4,096-token batch and
127
+ zero prefix-cache hits. FP8 runs first, then NVFP4. The logits-capture image is
128
+ `sha256:0405a1c0dc128b51069798a5d00b346257bbd006c0deb7e6530a3d005d75de71`;
129
+ it adds the capture/warmup seam to the selected serving image and is not the public
130
+ serving tag. This quality measurement therefore does not measure the MTP3 speed
131
+ configuration's complete output behavior.
132
+
133
+ The paired result and confidence interval passed the receipt/score audit. These
134
+ already-opened windows provide a development comparison, **not untouched final qualification**.
135
+ No matched claim against EXL3, TR3 4bpw or another weight format is established by
136
+ this cache comparison.
137
+
138
+ The archived historical `.0341811459` cache label is under provenance review:
139
+ the prior card labels it NVFP4, while the owner's historical account identifies
140
+ FP8. The prior text and linked receipts are retained below without silently
141
+ changing either number. Do not use that disputed historical label as a matched
142
+ cache comparison. New reference results are recorded separately above.
143
+
144
+ ## Run the model
145
+
146
+ Requires the custom runtime, four supported GPUs and NVIDIA Container Toolkit.
147
+ This is a sidecar checkpoint used with a separate carrier model; stock Transformers
148
+ or unmodified vLLM cannot load it as a standalone model.
149
+
150
+ ```bash
151
+ hf download brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8 --local-dir ./trellismx-p8
152
+ hf download local-inference-lab/GLM-5.3-Flash-NVFP4 \
153
+ --revision 520de24eabf507659eaef7c70f14fd584527facc --local-dir ./glm53-carrier
154
+ cd trellismx-p8
155
+ export MODEL_ROOT=/absolute/path/to/glm53-carrier
156
+ docker compose -f compose.yaml config --quiet
157
+ docker compose -f compose.yaml up -d
158
+ ```
159
+
160
+ The default endpoint is port 8000. The scripts do not set GPU power limits or
161
+ memory clocks. The measured 300W per GPU setting is part of the benchmark setup,
162
+ not a guarantee that a new host will reproduce its speed. The 1,000,000-token
163
+ launch limit is within the model's declared 1,048,576 positions; it is not a
164
+ measured 1M-context accuracy result. API credentials are not baked into the image.
165
+
166
+ ## Required runtime and model layout
167
+
168
+ This repository contains **168 TP4 routed-expert sidecars (177,269,057,440
169
+ bytes)** plus serving metadata, scripts, license and evaluation receipts.
170
+ They replace the carrier's routed experts at load time. They are not an
171
+ additional expert ensemble, and this is **not a standalone Transformers
172
+ checkpoint**. Do not point stock Transformers or unmodified vLLM at it.
173
+
174
+ The working runtime additionally requires the stock carrier:
175
+ [`local-inference-lab/GLM-5.3-Flash-NVFP4`](https://huggingface.co/local-inference-lab/GLM-5.3-Flash-NVFP4/tree/520de24eabf507659eaef7c70f14fd584527facc),
176
+ revision `520de24eabf507659eaef7c70f14fd584527facc`. Its attention, shared
177
+ experts, embeddings, output head and MTP components remain part of the served
178
+ model. Keeping this exact dependency preserves the existing loader contract;
179
+ the reported logical model payload is not a claim that the two download
180
+ directories together occupy only that many bytes.
181
+
182
+ ## Checkpoint and implementation details
183
+
184
+ This is the **17-K5 / 25-K4 coupled checkpoint**, covering all 42 routed layers
185
+ (3–44). Routed tensors occupy **4.6587417643 bits/weight including metadata**;
186
+ this is not a uniform 4.25 bpw model.
187
+
188
+ <details>
189
+ <summary>How the compressed weights, FP8 math and prior work fit together</summary>
190
 
191
  ## What TrellisMX does
192
 
 
273
  combination above describes this implementation without asserting an
274
  unverified first-in-history claim or general superiority over its sources.
275
 
276
+ </details>
277
 
278
+ ## Earlier measurements and provenance
 
 
 
 
279
 
280
+ <details>
281
+ <summary>September 8, RC5 and DCP1 results — archived runtime settings and receipts</summary>
 
 
 
 
 
282
 
283
+ The text below preserves the prior release's results and links. References to
284
+ “current” or “production” inside this archive describe the earlier release,
285
+ not the September 9 reference above. In particular, the old 16-sequence recipe
286
+ and old image digest are superseded. The historical `.0341811459` cache label
287
+ is disputed as described above; retaining the original text is not a resolution.
288
 
289
+ ## September 8 r27 DCP4 results (superseded runtime)
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
290
 
291
  | Measurement | Result | Conditions |
292
  | --- | --- | --- |
 
311
  benchmark client's larger generic capacity estimate is not valid for this
312
  split-cache layout.
313
 
314
+ ### September 8 r27 KV-cache KLD comparison
315
 
316
  | Runtime / measurement image | KV dtype | Attention backend | Mean true-decode KLD | Window BCa 95% interval |
317
  | --- | --- | --- | ---: | --- |
 
422
  See the [full GitHub results and receipts](https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/blob/03527b1f092cfbf274ae3ca78cabb23077c37057/results/RC5_RELEASE_RESULTS_20260907.md),
423
  and the included `results/` and `evidence/` files.
424
 
425
+ </details>
426
+
427
  ## Codec and limitations
428
 
429
  Procedural MCG trellis streams decode in registers to E4M3 for native
compose.yaml CHANGED
@@ -1,12 +1,12 @@
1
  services:
2
  trellismx:
3
- image: verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f
4
  restart: "no"
5
  network_mode: host
6
  ipc: host
7
  gpus: all
8
  working_dir: /
9
- entrypoint: ["/bin/bash", "/opt/trellismx/serve.sh"]
10
  environment:
11
  PORT: "8000"
12
  SERVED_MODEL_NAME: "glm53-flash-trellismx-p8-k45"
@@ -19,11 +19,16 @@ services:
19
  NUM_SPECULATIVE_TOKENS: "3"
20
  KV_CACHE_DTYPE: "nvfp4_ds_mla"
21
  MAX_MODEL_LEN: "1000000"
22
- MAX_NUM_SEQS: "16"
23
  MAX_NUM_BATCHED_TOKENS: "4096"
24
  GPU_MEMORY_UTILIZATION: "0.97"
25
  CP_KV_CACHE_INTERLEAVE_SIZE: "4"
26
  DCP_CKV_GATHER: "auto"
 
 
 
 
 
27
  VLLM_NO_USAGE_STATS: "1"
28
  volumes:
29
  - ${MODEL_ROOT:?Set MODEL_ROOT to the pinned carrier directory}:/model:ro
 
1
  services:
2
  trellismx:
3
+ image: verdictai/trellismx:glm53-flash-p8-r27-reference-20260909@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf
4
  restart: "no"
5
  network_mode: host
6
  ipc: host
7
  gpus: all
8
  working_dir: /
9
+ entrypoint: ["/bin/bash", "/release/serve-r27-production.sh"]
10
  environment:
11
  PORT: "8000"
12
  SERVED_MODEL_NAME: "glm53-flash-trellismx-p8-k45"
 
19
  NUM_SPECULATIVE_TOKENS: "3"
20
  KV_CACHE_DTYPE: "nvfp4_ds_mla"
21
  MAX_MODEL_LEN: "1000000"
22
+ MAX_NUM_SEQS: "24"
23
  MAX_NUM_BATCHED_TOKENS: "4096"
24
  GPU_MEMORY_UTILIZATION: "0.97"
25
  CP_KV_CACHE_INTERLEAVE_SIZE: "4"
26
  DCP_CKV_GATHER: "auto"
27
+ NCCL_MIN_NCHANNELS: "8"
28
+ NCCL_MAX_NCHANNELS: "8"
29
+ VLLM_PCIE_ONESHOT_ALLREDUCE_MAX_SIZE: "131072"
30
+ VLLM_PCIE_ONESHOT_FUSED_ADD_RMS_NORM_MAX_SIZE: "86016"
31
+ VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD: "4096"
32
  VLLM_NO_USAGE_STATS: "1"
33
  volumes:
34
  - ${MODEL_ROOT:?Set MODEL_ROOT to the pinned carrier directory}:/model:ro
image-record.json CHANGED
@@ -1,8 +1,8 @@
1
  {
2
  "schema_version": 4,
3
- "record_id": "trellismx-r27-dcp4-20260908",
4
- "title": "TrellisMX P8 GLM-5.3-Flash r27 DCP4",
5
- "summary": "Reproducible r27 inference overlay with source parity, baked production launcher, measured speed receipts and audited paired NVFP4/FP8 KV development KLD.",
6
  "model_family": "GLM-5.3-Flash",
7
  "release_class": "experimental",
8
  "distribution_role": "custom",
@@ -10,9 +10,9 @@
10
  "maintenance_status": "ephemeral",
11
  "image": {
12
  "repository": "verdictai/trellismx",
13
- "tag": "glm53-flash-p8-r27-dcp4-20260908",
14
- "digest": "sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f",
15
- "reference": "verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f"
16
  },
17
  "base_image": {
18
  "repository": "voipmonitor/vllm",
@@ -33,69 +33,29 @@
33
  "relationship": "Community source contract; does not qualify this custom P8 runtime."
34
  },
35
  "build": {
36
- "recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/53d93dfbc9002df7b73178327dfe99773efb680e/runtime",
37
- "recipe_commit": "53d93dfbc9002df7b73178327dfe99773efb680e",
38
- "build_command": "docker build --pull=false -t trellismx-r27-rebuild runtime",
39
- "components": [
40
- {
41
- "name": "vllm",
42
- "repository": "https://github.com/local-inference-lab/vllm",
43
- "commit": "c88fb8847dc8bc18bca56640759f137198750085",
44
- "release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
45
- "pull_requests": [],
46
- "patches": [
47
- {
48
- "path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
49
- "sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
50
- "purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
51
- "authors": [
52
- "Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
53
- ]
54
- }
55
- ],
56
- "overlays": []
57
- },
58
- {
59
- "name": "b12x",
60
- "repository": "https://github.com/local-inference-lab/b12x",
61
- "commit": "7093ad77849cf181bcf0c30b897c54fd32dac40e",
62
- "release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
63
- "pull_requests": [],
64
- "patches": [
65
- {
66
- "path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
67
- "sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
68
- "purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
69
- "authors": [
70
- "Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
71
- ]
72
- }
73
- ],
74
- "overlays": []
75
- }
76
- ],
77
- "package_changes": [
78
- "No dependency installs; r27 native extension binaries and packages inherited."
79
- ],
80
- "build_arguments": [
81
- "JOVIAN_IMAGE=voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512"
82
- ],
83
  "environment_defaults": [
84
- "TP=4 DCP=4 KV_CACHE_DTYPE=nvfp4_ds_mla NUM_SPECULATIVE_TOKENS=3 GPU_MEMORY_UTILIZATION=0.97 MAX_NUM_SEQS=16 MAX_NUM_BATCHED_TOKENS=4096 MAX_MODEL_LEN=1000000"
 
 
85
  ],
86
  "entrypoint_changes": [
87
- "Bakes deployed launcher at /opt/trellismx/serve.sh; executes inherited r27 serve-glm53-flash launcher. PORT=8000."
88
- ],
89
- "result_tree": "N/A",
90
- "integration_patch_sha256": "N/A"
91
  },
92
  "changes": {
93
  "inherited": [
94
  "r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
95
  ],
96
  "introduced": [
97
- "Native TrellisMX P8 checkpoint loader and B12X expert path; DCP4 persistent workspace integration; complete inference source in recipe.",
98
- "Production launcher defaults baked into image; licenses included."
99
  ],
100
  "compatibility_impact": [
101
  "TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
@@ -103,7 +63,7 @@
103
  },
104
  "tested_configurations": [
105
  {
106
- "name": "Measured source-equivalent r27 DCP4 local deployment; public image CPU checked",
107
  "hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
108
  "topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
109
  "power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
@@ -111,86 +71,34 @@
111
  "cuda_runtime": "13.3",
112
  "pytorch": "2.13.0",
113
  "nccl": "2.31.2",
114
- "engine_source": "c88fb8847dc8bc18bca56640759f137198750085 overlay on r27 binaries",
115
  "model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
116
  "quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
117
  "parallelism": "TP4/DCP4, EP off",
118
- "kv_cache": "nvfp4_ds_mla; engine reports 23424836 effective aggregate tokens; hybrid split layout",
119
  "speculative_mode": "MTP3 probabilistic draft plus standard rejection",
120
  "graph_mode": "FULL_AND_PIECEWISE",
121
- "scheduler_limits": "16 sequences; 4096 batched tokens; 1000000 model length; prefill share 0.4",
122
  "cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
123
  "launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
124
  }
125
  ],
126
  "validation": {
127
- "commands": [
128
- "docker build --pull=false -t trellismx-r27-rebuild runtime",
129
- "MODEL_ROOT=/model docker compose -f compose.yaml config --quiet",
130
- "docker run --rm --entrypoint python -v /absolute/path/to/checkpoint:/checkpoint:ro verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f -c 'from vllm.utils.trellismx import load_overlay; print(len(load_overlay(\"/checkpoint\").records))'",
131
- "Pinned llm_decode_bench.py commands and effective arguments in results/r27-20260908/README.md and JSON."
132
- ],
133
- "results": [
134
- {
135
- "name": "Deployed inference source parity",
136
- "status": "passed",
137
- "conditions": "Current local image57a9967b75e6 vs public build context",
138
- "measurement": "SHA256 every included deployed inference file",
139
- "result": "413 files matched",
140
- "conclusion": "Source parity only; rebuilt image is separately identified and is not a new GPU benchmark.",
141
- "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/runtime-parity.json",
142
- "evidence_sha256": "sha256:033767e89483ccf6290f427decbbef96911511d704358da970cee40b512207fa",
143
- "category": "correctness"
144
- },
145
- {
146
- "name": "Public image CPU overlay import and manifest inventory",
147
- "status": "passed",
148
- "conditions": "Public image, no GPUs passed, local HF-layout checkpoint read-only",
149
- "measurement": "Import native overlay validator; verify168 file inventory, sizes, and design identities and launcher SHA",
150
- "result": "168 records loaded",
151
- "conclusion": "CPU manifest/inventory gate only, no complete sidecar rehash or fresh GPU launch claim.",
152
- "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/public-image-cpu-check.txt",
153
- "evidence_sha256": "sha256:ea5d2725e24abafef7edfd7266efbde829f61153a62426e93df67274eb454d62",
154
- "category": "smoke"
155
- },
156
- {
157
- "name": "Recorded current-server speed observations",
158
- "status": "passed",
159
- "conditions": "Four SM120 GPUs at300W; TP4/DCP4 MTP3 NVFP4 MLA KV; one run each, no matched baseline",
160
- "measurement": "C1/C2 20s streaming cells; separate10s cold-prefill targets; full JSON/logs retained",
161
- "result": "C1 zero202.45t/s; prefill7692/7940/8006t/s at8200/16227/32316tokens",
162
- "conclusion": "Exploratory absolute observations; no advantage or independent qualification claim; workload and MTP confounders detailed.",
163
- "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/README.md",
164
- "evidence_sha256": "sha256:16e1e1fedf7b49dcdc1d79c156438ee016d47817861c8abdfe32964d9fe2277b",
165
- "category": "performance"
166
- },
167
- {
168
- "name": "Paired current r27 NVFP4 versus FP8 MLA KV development KLD",
169
- "status": "passed",
170
- "conditions": "Same32 opened conditional-fit windows; TP4/DCP4, MTP off, maxseq1; fixed NVFP4 then FP8 order, one server start per arm.",
171
- "measurement": "CPU FP64 teacher-to-student KL over2046 true-decode rows/window;20,000-resample paired window BCa95.",
172
- "result": "NVFP4=0.0350078183; FP8=0.0310574767; FP8-minus-NVFP4=-0.0039503416, paired BCa95[-0.0092510457,-0.0018107907].",
173
- "conclusion": "Audited development comparison only; no final holdout, independent replication, MTP or long-context qualification.",
174
- "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-kv-cf32-20260908/comparison.json",
175
- "evidence_sha256": "sha256:9376acdce0d66738f5af7bc54188ad98d298dbb47dfd3cd3faece459becb77d6",
176
- "category": "correctness"
177
- }
178
- ],
179
- "performance_claims": [
180
- {
181
- "name": "N/A"
182
- }
183
- ]
184
  },
185
  "limitations": {
186
  "known": [
187
- "Existing absolute speed observations belong to source-equivalent local server image57a9967b75e6, not a newly benchmarked registry rebuild.",
188
- "Historical KLD0.0341811459 uses DCP1; current paired DCP4 results use the same opened windows, MTP off and2046 true-decode rows per window. Different historical runtime/topology prevents KV-only attribution.",
189
- "Weight upload is still in progress; release-status.json governs remote completeness.",
190
- "Engine aggregate cache capacity does not establish1M-context accuracy or full-capacity stress."
191
  ],
192
  "untested": [
193
- "Clean download-to-GPU serving of public image",
194
  "Independent repeated speed qualification"
195
  ],
196
  "unsupported": [
 
1
  {
2
  "schema_version": 4,
3
+ "record_id": "trellismx-reference-20260909",
4
+ "title": "TrellisMX GLM-5.3 Flash selected TP4 reference",
5
+ "summary": "Selected September9 FP8-math reference image with measured speed, exact source overlays and audited matched FP8/NVFP4 MLA KLD.",
6
  "model_family": "GLM-5.3-Flash",
7
  "release_class": "experimental",
8
  "distribution_role": "custom",
 
10
  "maintenance_status": "ephemeral",
11
  "image": {
12
  "repository": "verdictai/trellismx",
13
+ "tag": "glm53-flash-p8-r27-reference-20260909",
14
+ "digest": "sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf",
15
+ "reference": "verdictai/trellismx:glm53-flash-p8-r27-reference-20260909@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf"
16
  },
17
  "base_image": {
18
  "repository": "voipmonitor/vllm",
 
33
  "relationship": "Community source contract; does not qualify this custom P8 runtime."
34
  },
35
  "build": {
36
+ "recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/main/runtime-reference-20260909",
37
+ "recipe_commit": "See containing HF commit",
38
+ "build_command": "docker build --pull=false -t trellismx-reference-rebuild runtime-reference-20260909",
39
+ "source_manifest": "runtime-reference-20260909/source-manifest.json",
40
+ "source_manifest_sha256": "2629fdb6094b54ce1cc1cc59b7dc0fe2617e0fe2ed39ff49a4e92ed72a28b252",
41
+ "base": "verdictai/trellismx@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f",
42
+ "qualification": "Published image is the retained measured image, retagged without a rebuild. Recipe reproduces Python overlay; fresh rebuild not GPU tested.",
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
43
  "environment_defaults": [
44
+ "TP4 DCP4 MTP3 maxseq24 batch4096 maxlen1000000 GMU0.97 KVnvfp4_ds_mla",
45
+ "NCCL_MIN_NCHANNELS=8 NCCL_MAX_NCHANNELS=8",
46
+ "VLLM_PCIE_ONESHOT_ALLREDUCE_MAX_SIZE=131072 VLLM_PCIE_ONESHOT_FUSED_ADD_RMS_NORM_MAX_SIZE=86016 VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD=4096"
47
  ],
48
  "entrypoint_changes": [
49
+ "compose explicitly launches /bin/bash /release/serve-r27-production.sh; baked launcher defaults16, public compose/wrapper override24"
50
+ ]
 
 
51
  },
52
  "changes": {
53
  "inherited": [
54
  "r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
55
  ],
56
  "introduced": [
57
+ "Route-hoisted FP8 native expert dispatch, tile policy and grouped FC2 paths; exact changed Python files in source manifest.",
58
+ "Selected collective settings and24slot serving wrapper."
59
  ],
60
  "compatibility_impact": [
61
  "TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
 
63
  },
64
  "tested_configurations": [
65
  {
66
+ "name": "Retained September9 selected reference image",
67
  "hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
68
  "topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
69
  "power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
 
71
  "cuda_runtime": "13.3",
72
  "pytorch": "2.13.0",
73
  "nccl": "2.31.2",
74
+ "engine_source": "Exact selected image Python source and manifest in runtime-reference-20260909",
75
  "model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
76
  "quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
77
  "parallelism": "TP4/DCP4, EP off",
78
+ "kv_cache": "nvfp4_ds_mla; expanded benchmark engine capacity23562091 effective aggregate tokens",
79
  "speculative_mode": "MTP3 probabilistic draft plus standard rejection",
80
  "graph_mode": "FULL_AND_PIECEWISE",
81
+ "scheduler_limits": "24 sequences;4096 batched tokens;1000000 model length",
82
  "cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
83
  "launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
84
  }
85
  ],
86
  "validation": {
87
+ "speed_evidence": "results/speed-20260909/README.md",
88
+ "all_speed_index": "results/speed-20260909/benchmark-index.json",
89
+ "kld_evidence": "results/kld-reference-20260909/comparison.json",
90
+ "kld_audit": "results/kld-reference-20260909/audit.json",
91
+ "kld_level": "Already-opened conditional-fit development comparison; single server per arm; MTPoff; not independent replication or final qualification."
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
92
  },
93
  "limitations": {
94
  "known": [
95
+ "Recipe Python-source reconstruction is separate from the measured immutable Docker image.",
96
+ "Historical comparisons differ in image/topology; new matched KLD holds reference checkpoint/TP4/DCP4 fixed and changes KV specialization.",
97
+ "TR3/EXL3 matched KLD requested, not measured yet.",
98
+ "KV aggregate capacity is not1M-context accuracy or full-capacity stress."
99
  ],
100
  "untested": [
101
+ "Fresh downloaded deployment GPU test",
102
  "Independent repeated speed qualification"
103
  ],
104
  "unsupported": [
results/image-record-historical-20260908.json ADDED
@@ -0,0 +1,217 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 4,
3
+ "record_id": "trellismx-r27-dcp4-20260908",
4
+ "title": "TrellisMX P8 GLM-5.3-Flash r27 DCP4",
5
+ "summary": "Reproducible r27 inference overlay with source parity, baked production launcher, measured speed receipts and audited paired NVFP4/FP8 KV development KLD.",
6
+ "model_family": "GLM-5.3-Flash",
7
+ "release_class": "experimental",
8
+ "distribution_role": "custom",
9
+ "qualification_status": "implemented",
10
+ "maintenance_status": "ephemeral",
11
+ "image": {
12
+ "repository": "verdictai/trellismx",
13
+ "tag": "glm53-flash-p8-r27-dcp4-20260908",
14
+ "digest": "sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f",
15
+ "reference": "verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f"
16
+ },
17
+ "base_image": {
18
+ "repository": "voipmonitor/vllm",
19
+ "tag": "jovian-judgement-community-20260906-r27",
20
+ "digest": "sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
21
+ "reference": "voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
22
+ "credit": "Local Inference Lab Jovian Judgement GLM r27; vLLM, B12X and component license obligations remain applicable."
23
+ },
24
+ "recommended_image": {
25
+ "reference": "voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
26
+ "relationship": "Pinned community runtime inherited by this custom overlay; not a matched numerical baseline."
27
+ },
28
+ "community_wiki": {
29
+ "repository": "https://github.com/local-inference-lab/rtx6kpro",
30
+ "commit": "94b71ac2a5f9c75f6b60dd1b6e6dffda492a4942",
31
+ "runbook_path": "models/glm-5.3-flash.md",
32
+ "runbook_url": "https://github.com/local-inference-lab/rtx6kpro/blob/94b71ac2a5f9c75f6b60dd1b6e6dffda492a4942/models/glm-5.3-flash.md",
33
+ "relationship": "Community source contract; does not qualify this custom P8 runtime."
34
+ },
35
+ "build": {
36
+ "recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/53d93dfbc9002df7b73178327dfe99773efb680e/runtime",
37
+ "recipe_commit": "53d93dfbc9002df7b73178327dfe99773efb680e",
38
+ "build_command": "docker build --pull=false -t trellismx-r27-rebuild runtime",
39
+ "components": [
40
+ {
41
+ "name": "vllm",
42
+ "repository": "https://github.com/local-inference-lab/vllm",
43
+ "commit": "c88fb8847dc8bc18bca56640759f137198750085",
44
+ "release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
45
+ "pull_requests": [],
46
+ "patches": [
47
+ {
48
+ "path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
49
+ "sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
50
+ "purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
51
+ "authors": [
52
+ "Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
53
+ ]
54
+ }
55
+ ],
56
+ "overlays": []
57
+ },
58
+ {
59
+ "name": "b12x",
60
+ "repository": "https://github.com/local-inference-lab/b12x",
61
+ "commit": "7093ad77849cf181bcf0c30b897c54fd32dac40e",
62
+ "release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
63
+ "pull_requests": [],
64
+ "patches": [
65
+ {
66
+ "path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
67
+ "sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
68
+ "purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
69
+ "authors": [
70
+ "Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
71
+ ]
72
+ }
73
+ ],
74
+ "overlays": []
75
+ }
76
+ ],
77
+ "package_changes": [
78
+ "No dependency installs; r27 native extension binaries and packages inherited."
79
+ ],
80
+ "build_arguments": [
81
+ "JOVIAN_IMAGE=voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512"
82
+ ],
83
+ "environment_defaults": [
84
+ "TP=4 DCP=4 KV_CACHE_DTYPE=nvfp4_ds_mla NUM_SPECULATIVE_TOKENS=3 GPU_MEMORY_UTILIZATION=0.97 MAX_NUM_SEQS=16 MAX_NUM_BATCHED_TOKENS=4096 MAX_MODEL_LEN=1000000"
85
+ ],
86
+ "entrypoint_changes": [
87
+ "Bakes deployed launcher at /opt/trellismx/serve.sh; executes inherited r27 serve-glm53-flash launcher. PORT=8000."
88
+ ],
89
+ "result_tree": "N/A",
90
+ "integration_patch_sha256": "N/A"
91
+ },
92
+ "changes": {
93
+ "inherited": [
94
+ "r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
95
+ ],
96
+ "introduced": [
97
+ "Native TrellisMX P8 checkpoint loader and B12X expert path; DCP4 persistent workspace integration; complete inference source in recipe.",
98
+ "Production launcher defaults baked into image; licenses included."
99
+ ],
100
+ "compatibility_impact": [
101
+ "TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
102
+ ]
103
+ },
104
+ "tested_configurations": [
105
+ {
106
+ "name": "Measured source-equivalent r27 DCP4 local deployment; public image CPU checked",
107
+ "hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
108
+ "topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
109
+ "power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
110
+ "driver": "610.57.04",
111
+ "cuda_runtime": "13.3",
112
+ "pytorch": "2.13.0",
113
+ "nccl": "2.31.2",
114
+ "engine_source": "c88fb8847dc8bc18bca56640759f137198750085 overlay on r27 binaries",
115
+ "model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
116
+ "quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
117
+ "parallelism": "TP4/DCP4, EP off",
118
+ "kv_cache": "nvfp4_ds_mla; engine reports 23424836 effective aggregate tokens; hybrid split layout",
119
+ "speculative_mode": "MTP3 probabilistic draft plus standard rejection",
120
+ "graph_mode": "FULL_AND_PIECEWISE",
121
+ "scheduler_limits": "16 sequences; 4096 batched tokens; 1000000 model length; prefill share 0.4",
122
+ "cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
123
+ "launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
124
+ }
125
+ ],
126
+ "validation": {
127
+ "commands": [
128
+ "docker build --pull=false -t trellismx-r27-rebuild runtime",
129
+ "MODEL_ROOT=/model docker compose -f compose.yaml config --quiet",
130
+ "docker run --rm --entrypoint python -v /absolute/path/to/checkpoint:/checkpoint:ro verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f -c 'from vllm.utils.trellismx import load_overlay; print(len(load_overlay(\"/checkpoint\").records))'",
131
+ "Pinned llm_decode_bench.py commands and effective arguments in results/r27-20260908/README.md and JSON."
132
+ ],
133
+ "results": [
134
+ {
135
+ "name": "Deployed inference source parity",
136
+ "status": "passed",
137
+ "conditions": "Current local image57a9967b75e6 vs public build context",
138
+ "measurement": "SHA256 every included deployed inference file",
139
+ "result": "413 files matched",
140
+ "conclusion": "Source parity only; rebuilt image is separately identified and is not a new GPU benchmark.",
141
+ "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/runtime-parity.json",
142
+ "evidence_sha256": "sha256:033767e89483ccf6290f427decbbef96911511d704358da970cee40b512207fa",
143
+ "category": "correctness"
144
+ },
145
+ {
146
+ "name": "Public image CPU overlay import and manifest inventory",
147
+ "status": "passed",
148
+ "conditions": "Public image, no GPUs passed, local HF-layout checkpoint read-only",
149
+ "measurement": "Import native overlay validator; verify168 file inventory, sizes, and design identities and launcher SHA",
150
+ "result": "168 records loaded",
151
+ "conclusion": "CPU manifest/inventory gate only, no complete sidecar rehash or fresh GPU launch claim.",
152
+ "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/public-image-cpu-check.txt",
153
+ "evidence_sha256": "sha256:ea5d2725e24abafef7edfd7266efbde829f61153a62426e93df67274eb454d62",
154
+ "category": "smoke"
155
+ },
156
+ {
157
+ "name": "Recorded current-server speed observations",
158
+ "status": "passed",
159
+ "conditions": "Four SM120 GPUs at300W; TP4/DCP4 MTP3 NVFP4 MLA KV; one run each, no matched baseline",
160
+ "measurement": "C1/C2 20s streaming cells; separate10s cold-prefill targets; full JSON/logs retained",
161
+ "result": "C1 zero202.45t/s; prefill7692/7940/8006t/s at8200/16227/32316tokens",
162
+ "conclusion": "Exploratory absolute observations; no advantage or independent qualification claim; workload and MTP confounders detailed.",
163
+ "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/README.md",
164
+ "evidence_sha256": "sha256:16e1e1fedf7b49dcdc1d79c156438ee016d47817861c8abdfe32964d9fe2277b",
165
+ "category": "performance"
166
+ },
167
+ {
168
+ "name": "Paired current r27 NVFP4 versus FP8 MLA KV development KLD",
169
+ "status": "passed",
170
+ "conditions": "Same32 opened conditional-fit windows; TP4/DCP4, MTP off, maxseq1; fixed NVFP4 then FP8 order, one server start per arm.",
171
+ "measurement": "CPU FP64 teacher-to-student KL over2046 true-decode rows/window;20,000-resample paired window BCa95.",
172
+ "result": "NVFP4=0.0350078183; FP8=0.0310574767; FP8-minus-NVFP4=-0.0039503416, paired BCa95[-0.0092510457,-0.0018107907].",
173
+ "conclusion": "Audited development comparison only; no final holdout, independent replication, MTP or long-context qualification.",
174
+ "evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-kv-cf32-20260908/comparison.json",
175
+ "evidence_sha256": "sha256:9376acdce0d66738f5af7bc54188ad98d298dbb47dfd3cd3faece459becb77d6",
176
+ "category": "correctness"
177
+ }
178
+ ],
179
+ "performance_claims": [
180
+ {
181
+ "name": "N/A"
182
+ }
183
+ ]
184
+ },
185
+ "limitations": {
186
+ "known": [
187
+ "Existing absolute speed observations belong to source-equivalent local server image57a9967b75e6, not a newly benchmarked registry rebuild.",
188
+ "Historical KLD0.0341811459 uses DCP1; current paired DCP4 results use the same opened windows, MTP off and2046 true-decode rows per window. Different historical runtime/topology prevents KV-only attribution.",
189
+ "Weight upload is still in progress; release-status.json governs remote completeness.",
190
+ "Engine aggregate cache capacity does not establish1M-context accuracy or full-capacity stress."
191
+ ],
192
+ "untested": [
193
+ "Clean download-to-GPU serving of public image",
194
+ "Independent repeated speed qualification"
195
+ ],
196
+ "unsupported": [
197
+ "Other GPU architectures, TP sizes, arbitrary model encoders; inference-only release."
198
+ ]
199
+ },
200
+ "support": {
201
+ "owner": "Brandon Music",
202
+ "contact": "https://github.com/brandonmmusic-max",
203
+ "issue_tracker": "https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/issues",
204
+ "support_thread": "https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/issues/5",
205
+ "thread_status": "active",
206
+ "support_commitment": "ephemeral",
207
+ "triage_policy": "Keep reports in this thread. Escalate upstream only after reproduction on the recommended image or a minimal reproducer identifies the responsible source change.",
208
+ "superseded_by": "N/A"
209
+ },
210
+ "publication": {
211
+ "record_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/image-record.json",
212
+ "main_channel_link": "N/A",
213
+ "main_channel_link_count": 0,
214
+ "bot_listing": "not-applicable",
215
+ "maintainer_approval_url": "N/A"
216
+ }
217
+ }
results/kld-reference-20260909/README.md ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Reference-stack KV-cache KLD — September 9, 2026
2
+
3
+ | Cache | Mean KL(teacher || student) | Window BCa 95% interval |
4
+ | --- | ---: | --- |
5
+ | Current r27 DCP4 / NVFP4 MLA KV | 0.0354562238 | [0.0295558848, 0.0434620368] |
6
+ | Current r27 DCP4 / FP8 MLA KV | 0.0319451732 | [0.0268267482, 0.0387948613] |
7
+
8
+ FP8 minus NVFP4: -0.0035110506; paired-window BCa 95% interval
9
+ [-0.0081718772, -0.0013675941].
10
+ FP8 has lower observed KLD in 22/32 windows.
11
+
12
+ Exact same 32 previously opened conditional-fit windows and BF16 teacher as the
13
+ September 8 measurement: 2048 input tokens, 2047 captured predictions, exclude
14
+ row zero, leaving 2046 true-decode rows/window. CPU FP64 KL over vocabulary154880;
15
+ equal mean of window means; BCa20000 resamples with seed20260902. Teacher stored F32.
16
+ TP4/DCP4, MTP off, maxseq1, batch4096, GMU0.97, maxlen1M; prefix caching enabled,
17
+ all prefix-hit counters zero. FP8 first then NVFP4, one server preparation each.
18
+ Window uncertainty does not estimate server-run variability. These are development
19
+ measurements, not untouched-final qualification or independent reproduction.
20
+
21
+ Selected image: verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf.
22
+ Capture-only derivative image, seal, source identities and all64 window scores are
23
+ in comparison.json. Capture seam preserves pre-mask logits and forces exact histories.
24
+ NCCL8, plain one-shot cutoff131072, fused cutoff86016, shared expert threshold4096.
25
+ Checkpoint and FP8 weight/activation math held fixed; KV dtype and native cache
26
+ specialization vary. Capture timing is not a throughput measurement.
27
+
28
+ Raw logits retired only after hashes, durable numerical scores and receipts.
29
+ Audit recomputes aggregates from retained FP64 scores and verifies metadata,
30
+ zero prefix hits, row masks and receipt hashes; it is not independent recapture.
31
+ No measured windows excluded or rerun. Prior image/cache values remain historical.
32
+ The separate production recipe uses MTP3 and24slots; its KV capacity must not be
33
+ confused with capacity from this MTP-disabled correctness profile.
results/kld-reference-20260909/audit.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "passed",
3
+ "arms": [
4
+ {
5
+ "arm": "fp8",
6
+ "windows": 32,
7
+ "prediction_rows": 65504,
8
+ "true_decode_rows": 65472,
9
+ "all_receipt_hashes_and_masks_pass": true,
10
+ "all_prefix_hit_counters_zero": true,
11
+ "raw_logits_retired_after_scoring": true,
12
+ "source_runtime_audit_sha256": "6a3393053b83b33b41a7d49de2fab1803b1efc4843ec5862c1acd8467b993fa4",
13
+ "source_startup_sha256": "00ce74cfce281b627289f3322829d357467f5bc1f026677ef27cda93ed43fce3"
14
+ },
15
+ {
16
+ "arm": "nvfp4",
17
+ "windows": 32,
18
+ "prediction_rows": 65504,
19
+ "true_decode_rows": 65472,
20
+ "all_receipt_hashes_and_masks_pass": true,
21
+ "all_prefix_hit_counters_zero": true,
22
+ "raw_logits_retired_after_scoring": true,
23
+ "source_runtime_audit_sha256": "74d7d1914a55a73b5e9de196f557c658603a9501acd2fafb1e540d496b1c7185",
24
+ "source_startup_sha256": "cae3cb749c61cc18fd23286f41f79266e32bfdc551e0575b3515d21eb993036a"
25
+ }
26
+ ],
27
+ "producing_analysis_sha256": "0b9a6cb0c0f1d1b4b7784190d38e6f8fdf3dbefe5cb619be6f0dc065a742d8a3",
28
+ "verifier_sha256": "124494f28c018e0945349fb4479d3be9f9164cdf7ce298989de25367f4106ade",
29
+ "claim": "Receipt and score audit by the same operator; not independent replication."
30
+ }
results/kld-reference-20260909/comparison.json ADDED
@@ -0,0 +1,1053 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema": "trellismx.r27-kv-cf32.v1",
3
+ "evidence": "controlled comparison on already-opened conditional-fit; one prepared run per KV mode, fixed order; not final qualification",
4
+ "arms": {
5
+ "fp8": {
6
+ "mean_true_decode_kld": 0.03194517316654619,
7
+ "window_bca95": [
8
+ 0.026826748240418287,
9
+ 0.03879486133125964
10
+ ],
11
+ "per_window": [
12
+ {
13
+ "window_id": "conditional-fit-0056",
14
+ "domain": "axis1_general",
15
+ "prediction_rows": 2047,
16
+ "true_decode_rows": 2046,
17
+ "true_decode_mean_kld": 0.02440979598998064,
18
+ "including_prefill_mean_kld": 0.02501134181859523,
19
+ "prefill_row_kld": 1.2557741071640407,
20
+ "raw_sha256": "72bc5007903d7885f04ff28c03128901bfd15d7a94385772af929125edb650ff",
21
+ "score_sha256": "3d00fe1e7f7505cb4186e5d67c6e7c5df53b904b6c041230b9e227ad34afcd93",
22
+ "teacher_sha256": "55a70a330f39c5f23294c23d270a1ddc53031def99a42206e2706d4d0bb4466b",
23
+ "token_sha256": "219d8cd8873c6cf41f52782297cc5cfa9c4ed9666f2bc30d84cee9caa68ea744",
24
+ "raw_retirement": "after durable score and verified raw hash",
25
+ "raw_bytes": 1268157440
26
+ },
27
+ {
28
+ "window_id": "conditional-fit-0008",
29
+ "domain": "axis1_general",
30
+ "prediction_rows": 2047,
31
+ "true_decode_rows": 2046,
32
+ "true_decode_mean_kld": 0.06994665102705569,
33
+ "including_prefill_mean_kld": 0.07040344776705193,
34
+ "prefill_row_kld": 1.0050095777993453,
35
+ "raw_sha256": "f9c863fed0667b09b18edeba791c9ab4018c68e010ff488c0b73c1b0a0a67721",
36
+ "score_sha256": "a1bfbe0a93478e2939a51800dea0ad26d56c6aca01ad3633f18a3709e1601bad",
37
+ "teacher_sha256": "040641247b2e060035d89c2e7345e029d7c142983585e7c5d63dc818486f650d",
38
+ "token_sha256": "581d72e3e78b15c02d9e73f6bc742260cb57cdcd74337f4419c8f90cbc7c9c11",
39
+ "raw_retirement": "after durable score and verified raw hash",
40
+ "raw_bytes": 1268157440
41
+ },
42
+ {
43
+ "window_id": "conditional-fit-0100",
44
+ "domain": "axis1_general",
45
+ "prediction_rows": 2047,
46
+ "true_decode_rows": 2046,
47
+ "true_decode_mean_kld": 0.022149540339761863,
48
+ "including_prefill_mean_kld": 0.023738251318587308,
49
+ "prefill_row_kld": 3.2742409139954454,
50
+ "raw_sha256": "7b082ebc4250bb8b032c82bdb02a15019b947dfe54131e776a0af539fab8b2d8",
51
+ "score_sha256": "cef09f41fd3bc3e395b4721344d0b546796c981917ac64c997c78cbcedebe6b9",
52
+ "teacher_sha256": "f62991383b3ef5e8db0ed71442b17c428972e58da1c868c6aeb0f3fa143c865b",
53
+ "token_sha256": "7f008ec99068f373c6e15dd059ee977fb70e6dc832f46e009f0f299d48d19892",
54
+ "raw_retirement": "after durable score and verified raw hash",
55
+ "raw_bytes": 1268157440
56
+ },
57
+ {
58
+ "window_id": "conditional-fit-0094",
59
+ "domain": "axis1_general",
60
+ "prediction_rows": 2047,
61
+ "true_decode_rows": 2046,
62
+ "true_decode_mean_kld": 0.016832794344000746,
63
+ "including_prefill_mean_kld": 0.016934019957995376,
64
+ "prefill_row_kld": 0.22404162619101314,
65
+ "raw_sha256": "29d51a000848b3f9a158070703dc4b25be593e4890deb83f572291379b54668d",
66
+ "score_sha256": "7ade9eefb18e3bdcffc63d64abcc88dd7260bf9b97b4be788da8e8c492710e35",
67
+ "teacher_sha256": "656911b8fb34d7f42e984dd21311caf9ef56db0b7618b414cad4089562d93cba",
68
+ "token_sha256": "6b6a9b21c49a9984cc8c5ea90e0bda6db5761d039c7cdd335dc65e30444e017b",
69
+ "raw_retirement": "after durable score and verified raw hash",
70
+ "raw_bytes": 1268157440
71
+ },
72
+ {
73
+ "window_id": "conditional-fit-0083",
74
+ "domain": "axis2_legal",
75
+ "prediction_rows": 2047,
76
+ "true_decode_rows": 2046,
77
+ "true_decode_mean_kld": 0.03015300663995215,
78
+ "including_prefill_mean_kld": 0.03048617506449642,
79
+ "prefill_row_kld": 0.7121487716820719,
80
+ "raw_sha256": "d71da9253ed6946f0fb5632c845d2ffb2f7c03b7b6f7cc25b959b0e14bfe8484",
81
+ "score_sha256": "0f1cc2f89d5738573eea4d2b3dc1a6fa7fc02180cd2e7cd8a2a5ba68892007a6",
82
+ "teacher_sha256": "66a7870a1f13c8b5d647c1681fca005bb0f55a6e502e94f282da673a6bed3f1b",
83
+ "token_sha256": "3df9d2872815e061ff8d32a9a19da0d11d80ac67070cccd28579b24ec3079552",
84
+ "raw_retirement": "after durable score and verified raw hash",
85
+ "raw_bytes": 1268157440
86
+ },
87
+ {
88
+ "window_id": "conditional-fit-0021",
89
+ "domain": "axis2_legal",
90
+ "prediction_rows": 2047,
91
+ "true_decode_rows": 2046,
92
+ "true_decode_mean_kld": 0.0390320697408021,
93
+ "including_prefill_mean_kld": 0.03997988073234378,
94
+ "prefill_row_kld": 1.9792011694266036,
95
+ "raw_sha256": "63ef61c53db88b1eaf3bd30a833593d48ea2bd25313a7c500c8d754e8142f04c",
96
+ "score_sha256": "6e41161d9b1edb5b19299d74682f682d5fd8643794befb9c111ff1557bcadc25",
97
+ "teacher_sha256": "ebf76910c4db8f6955ad03f2ec7352e2f8aaf15e0c939d28f9a7d78f6bd7451b",
98
+ "token_sha256": "dbd9f2539cb3e97127ae760a04ca5b3456b0b13d6ce8698f88b215cc68b8f1a9",
99
+ "raw_retirement": "after durable score and verified raw hash",
100
+ "raw_bytes": 1268157440
101
+ },
102
+ {
103
+ "window_id": "conditional-fit-0098",
104
+ "domain": "axis2_legal",
105
+ "prediction_rows": 2047,
106
+ "true_decode_rows": 2046,
107
+ "true_decode_mean_kld": 0.04434523531376396,
108
+ "including_prefill_mean_kld": 0.04552245091649955,
109
+ "prefill_row_kld": 2.4541055741135085,
110
+ "raw_sha256": "f2b99e96d0776ba966989383185fc18a776722a9de83a1d7d7d22f833cb0fe15",
111
+ "score_sha256": "33363d670f8d1e088ebadd3edbfc00e8aac3ebe3a47aac8c64edc9698f5bc44e",
112
+ "teacher_sha256": "16e3d9524100836c020535635f7722a4c93ce0fdddc2ad0781f25bda77e8bcc9",
113
+ "token_sha256": "749bda3291a0323d286b0018b7dc3e0f469de8afd5d194370cde357b7b315c7d",
114
+ "raw_retirement": "after durable score and verified raw hash",
115
+ "raw_bytes": 1268157440
116
+ },
117
+ {
118
+ "window_id": "conditional-fit-0080",
119
+ "domain": "axis2_legal",
120
+ "prediction_rows": 2047,
121
+ "true_decode_rows": 2046,
122
+ "true_decode_mean_kld": 0.04201227846375734,
123
+ "including_prefill_mean_kld": 0.04314755010220913,
124
+ "prefill_row_kld": 2.365913322374568,
125
+ "raw_sha256": "1646b7c5c5387e320517272052d9b5f01c168e44dbc8497ab66f3b55fd473db1",
126
+ "score_sha256": "8a1f8466c13bb37d2e1c8f04463d90905c9acddf054d1271678a2cea58b35bfe",
127
+ "teacher_sha256": "e9c1a77490a5db63d410297bd8febe06adf4df2a6f4e16552db187738c9289d2",
128
+ "token_sha256": "a31d43f38710eb305c684501a2b02909a4e82c43101c07e76e4a3ca4edb4d0dd",
129
+ "raw_retirement": "after durable score and verified raw hash",
130
+ "raw_bytes": 1268157440
131
+ },
132
+ {
133
+ "window_id": "conditional-fit-0081",
134
+ "domain": "axis3_code_agentic",
135
+ "prediction_rows": 2047,
136
+ "true_decode_rows": 2046,
137
+ "true_decode_mean_kld": 0.0255516960804676,
138
+ "including_prefill_mean_kld": 0.025679992777695365,
139
+ "prefill_row_kld": 0.28817503530570604,
140
+ "raw_sha256": "dfbb9d556272241c4e45c7a48302895559e7567d5cf85ddfdc788a6dd616a117",
141
+ "score_sha256": "601136d9bf9208532067fb0fe807c5200ae263e4dd85d01c21b1df89fbed8095",
142
+ "teacher_sha256": "5ae07760d91030d1eba925b76ea463309baa49c2e7d5f5e14409a23ced7fd20d",
143
+ "token_sha256": "350628488b94d59f3e4f3e99a418913544ea1fe7eacee14a642b996d9c614d16",
144
+ "raw_retirement": "after durable score and verified raw hash",
145
+ "raw_bytes": 1268157440
146
+ },
147
+ {
148
+ "window_id": "conditional-fit-0123",
149
+ "domain": "axis3_code_agentic",
150
+ "prediction_rows": 2047,
151
+ "true_decode_rows": 2046,
152
+ "true_decode_mean_kld": 0.033191169179572766,
153
+ "including_prefill_mean_kld": 0.03317634067093455,
154
+ "prefill_row_kld": 0.0028372119971554416,
155
+ "raw_sha256": "0cfa11435295489e24da83d30b42710f530f65a99e473c64755317a2391f8940",
156
+ "score_sha256": "0c5504a8e70139d48221451c5569e222ac3f55d13a17de9e65adfe9ae645bfa3",
157
+ "teacher_sha256": "4a24b5b750ba98c06d9bc69e19334aff89e7bd4f9429592598af5d62de7ecb42",
158
+ "token_sha256": "aab84ee426f91621d123aa68bd53f75e220fbaf8cb270db1feab1529a62ddbf0",
159
+ "raw_retirement": "after durable score and verified raw hash",
160
+ "raw_bytes": 1268157440
161
+ },
162
+ {
163
+ "window_id": "conditional-fit-0090",
164
+ "domain": "axis3_code_agentic",
165
+ "prediction_rows": 2047,
166
+ "true_decode_rows": 2046,
167
+ "true_decode_mean_kld": 0.03953329012985432,
168
+ "including_prefill_mean_kld": 0.04086400060183276,
169
+ "prefill_row_kld": 2.7634976262697193,
170
+ "raw_sha256": "8067287b0733598d6f9d01ad790e350dab5bc4bc32eba7430a05d38adbfdf455",
171
+ "score_sha256": "4ac0add16333368e2109970786f1ab2b8a6c1f0532218bdb3dbe94fb3045551e",
172
+ "teacher_sha256": "db2a5fe33c9e457a22ebeb31a7073c1dac666b877503e9b4e90a8d6abf7b8a32",
173
+ "token_sha256": "a06e76993db717afee847fcc4d86f10b52dd23cdce5dae916f32d238d08d886c",
174
+ "raw_retirement": "after durable score and verified raw hash",
175
+ "raw_bytes": 1268157440
176
+ },
177
+ {
178
+ "window_id": "conditional-fit-0062",
179
+ "domain": "axis3_code_agentic",
180
+ "prediction_rows": 2047,
181
+ "true_decode_rows": 2046,
182
+ "true_decode_mean_kld": 0.02626102483081332,
183
+ "including_prefill_mean_kld": 0.02663675046149981,
184
+ "prefill_row_kld": 0.7953713908460541,
185
+ "raw_sha256": "9f2f72b2e7ea4ffb425c094bd8d3df7452743a16ec6bcc22065dd25f613670f3",
186
+ "score_sha256": "9ecc5d9141126ec6f81b86eb28e9c785088f25d994ee8558b5bc7218ca325940",
187
+ "teacher_sha256": "d6f7883a533625c7c75119fe7403b48bf81885b1de0b311a1a0571949f7f231c",
188
+ "token_sha256": "7c6d2ef2f4b7c7091a517f6ccd778009f9ee144165c9aecf2f8f28aea7b3753f",
189
+ "raw_retirement": "after durable score and verified raw hash",
190
+ "raw_bytes": 1268157440
191
+ },
192
+ {
193
+ "window_id": "conditional-fit-0047",
194
+ "domain": "axis4_reasoning_termination",
195
+ "prediction_rows": 2047,
196
+ "true_decode_rows": 2046,
197
+ "true_decode_mean_kld": 0.02201209190929644,
198
+ "including_prefill_mean_kld": 0.026359335852221596,
199
+ "prefill_row_kld": 8.920820443077092,
200
+ "raw_sha256": "6594c75d2eb3fef5ac2049c9924e2be8119f2e0e9948b5c114f1e9b5ea9233b3",
201
+ "score_sha256": "92f15873c7c6837cebdb8a1d16767da58ec6a2a13091394ad3028b2e2c0fe716",
202
+ "teacher_sha256": "8513cdcb9940786c7c42bf14d2f5f2cd452791e4ec6036c06a1f973eb969018d",
203
+ "token_sha256": "89730ee6826d4302c352156d5df766c51622a8b6681b8052d4ca6e7d481177fe",
204
+ "raw_retirement": "after durable score and verified raw hash",
205
+ "raw_bytes": 1268157440
206
+ },
207
+ {
208
+ "window_id": "conditional-fit-0051",
209
+ "domain": "axis4_reasoning_termination",
210
+ "prediction_rows": 2047,
211
+ "true_decode_rows": 2046,
212
+ "true_decode_mean_kld": 0.013267071463714407,
213
+ "including_prefill_mean_kld": 0.013608489001681363,
214
+ "prefill_row_kld": 0.7121487716820719,
215
+ "raw_sha256": "43626b428d36c588448d0d2756b020f63e23237c04d852b958f79ca7f601298c",
216
+ "score_sha256": "30293b5eabd64e4bdfce7177844412ab5834b7f6c199aba692592cc6206665fa",
217
+ "teacher_sha256": "177477c19b54dc0b0230e19202d48ab71d52f2b1a37fe383511bb64f0aae2fcf",
218
+ "token_sha256": "c79269b16560e7301549593443c9394007e7a957ad149c46224e7ad271dc6a18",
219
+ "raw_retirement": "after durable score and verified raw hash",
220
+ "raw_bytes": 1268157440
221
+ },
222
+ {
223
+ "window_id": "conditional-fit-0055",
224
+ "domain": "axis4_reasoning_termination",
225
+ "prediction_rows": 2047,
226
+ "true_decode_rows": 2046,
227
+ "true_decode_mean_kld": 0.020627011525195565,
228
+ "including_prefill_mean_kld": 0.021473521460828236,
229
+ "prefill_row_kld": 1.7534328497652714,
230
+ "raw_sha256": "e470e9ded871cbe5d724b0416e973bb98b60bbe4cf76f6348feb871494464fbd",
231
+ "score_sha256": "2c690f7a591f3e932abfd73138a1185cdb253a37ce48cb85292b712ebd412ff8",
232
+ "teacher_sha256": "7679c828eae4bf08f17598d044904693fc90c05b9d3f189dddeeb846100e43a7",
233
+ "token_sha256": "feec9ac508172dd275ec6452a03493acab5ffdff9bb57161317c77f24daf90f2",
234
+ "raw_retirement": "after durable score and verified raw hash",
235
+ "raw_bytes": 1268157440
236
+ },
237
+ {
238
+ "window_id": "conditional-fit-0031",
239
+ "domain": "axis4_reasoning_termination",
240
+ "prediction_rows": 2047,
241
+ "true_decode_rows": 2046,
242
+ "true_decode_mean_kld": 0.01655011698258689,
243
+ "including_prefill_mean_kld": 0.020900029208329205,
244
+ "prefill_row_kld": 8.920820443077092,
245
+ "raw_sha256": "e94b496a3cec9e30b6bcbd429f1d4757f38bcd031ce303872442bf994404bec8",
246
+ "score_sha256": "4486c4b80fe284775cb7e57a99199749eb5b8892e397e56d1904c3d27459e1a7",
247
+ "teacher_sha256": "e04a6c04a951efb5a178927e91457535cc67745ce5c81e785cb6d963aac6ce83",
248
+ "token_sha256": "2cff3482b4212537d0c35b88b8a0f479502c8905952430c973d25b1ab0bb5818",
249
+ "raw_retirement": "after durable score and verified raw hash",
250
+ "raw_bytes": 1268157440
251
+ },
252
+ {
253
+ "window_id": "conditional-fit-0003",
254
+ "domain": "axis4_reasoning_termination",
255
+ "prediction_rows": 2047,
256
+ "true_decode_rows": 2046,
257
+ "true_decode_mean_kld": 0.019912232579636858,
258
+ "including_prefill_mean_kld": 0.022605333926454176,
259
+ "prefill_row_kld": 5.5326906895146735,
260
+ "raw_sha256": "2f0ef47bdc989be608cabfb611ac42a5710712f17f219bec543b0594ab7dadd6",
261
+ "score_sha256": "b72e81ff056ef12b3edc2b59ed5425223aef7120e8f8f144409329512b0cdb6c",
262
+ "teacher_sha256": "7677c89cc3dee8e2a64e3c7947add39475f828921a3a8fe2e92b379b3ed8320b",
263
+ "token_sha256": "d416537c6e8273747bf025e97287a0bb6e3ff047bfee65cf54e4dfe287569b93",
264
+ "raw_retirement": "after durable score and verified raw hash",
265
+ "raw_bytes": 1268157440
266
+ },
267
+ {
268
+ "window_id": "conditional-fit-0009",
269
+ "domain": "axis2_legal",
270
+ "prediction_rows": 2047,
271
+ "true_decode_rows": 2046,
272
+ "true_decode_mean_kld": 0.04340137042823817,
273
+ "including_prefill_mean_kld": 0.04350757595049225,
274
+ "prefill_row_kld": 0.2608040744823556,
275
+ "raw_sha256": "6e8cc5a72d7df8f166b66eb2e84d1fad1947b65002b6112bddf3e824f4adeb39",
276
+ "score_sha256": "5274f9fb536c8add3acf2f38dcb460002193f9eb13e9425607330a29a9389876",
277
+ "teacher_sha256": "00701ad8bf4a4a5eedcedd4eafd956830c9ff1c790b15074ca721a3462abf930",
278
+ "token_sha256": "fc1166707f5938a914faf364dab717b0de3ea0e8451ed1a3fb399469e659b105",
279
+ "raw_retirement": "after durable score and verified raw hash",
280
+ "raw_bytes": 1268157440
281
+ },
282
+ {
283
+ "window_id": "conditional-fit-0010",
284
+ "domain": "axis3_code_agentic",
285
+ "prediction_rows": 2047,
286
+ "true_decode_rows": 2046,
287
+ "true_decode_mean_kld": 0.016893295516952813,
288
+ "including_prefill_mean_kld": 0.016892189177210668,
289
+ "prefill_row_kld": 0.014628618064772496,
290
+ "raw_sha256": "30eddaed5438035e494759f7a8e5def1346ff8cc3d1144ff135459781ee8c36d",
291
+ "score_sha256": "a982b303a18638a696fd13660a23cfedfea7373a61663b08e0438d7834345726",
292
+ "teacher_sha256": "f9bf4519890b70d972dfdf9760181418eb7c55b6237c0108f01d44df89943278",
293
+ "token_sha256": "ddd9d769a76eba74bc8e680ff25736a3b9146371801d1d2b29e4dc48f6d256a2",
294
+ "raw_retirement": "after durable score and verified raw hash",
295
+ "raw_bytes": 1268157440
296
+ },
297
+ {
298
+ "window_id": "conditional-fit-0011",
299
+ "domain": "axis4_reasoning_termination",
300
+ "prediction_rows": 2047,
301
+ "true_decode_rows": 2046,
302
+ "true_decode_mean_kld": 0.016308449522884096,
303
+ "including_prefill_mean_kld": 0.02065847980796188,
304
+ "prefill_row_kld": 8.920820443077092,
305
+ "raw_sha256": "b6098a0ba73fd75f000bbfd23b395eac80f5c3529f5cc55b592cd44c49d3f272",
306
+ "score_sha256": "4c9d123aaaaf9eb62e494447f1cd08cb61b905d70df3e4b461ce06c48fd4b15b",
307
+ "teacher_sha256": "9f37c4ead38bb80a5d85f825b156e5b493b6d3f1ca9dabfad06b0416f37db286",
308
+ "token_sha256": "c4b96acfa47944be3485b20eab76eb16d59913c9fedb5001a537335a3ee13614",
309
+ "raw_retirement": "after durable score and verified raw hash",
310
+ "raw_bytes": 1268157440
311
+ },
312
+ {
313
+ "window_id": "conditional-fit-0030",
314
+ "domain": "axis3_code_agentic",
315
+ "prediction_rows": 2047,
316
+ "true_decode_rows": 2046,
317
+ "true_decode_mean_kld": 0.02726396346416777,
318
+ "including_prefill_mean_kld": 0.027262084282667205,
319
+ "prefill_row_kld": 0.023417278932510124,
320
+ "raw_sha256": "bc8f5f049ba9fea541f21abd4109e523067d52c9b5ef8cd3df6be759c2d1d954",
321
+ "score_sha256": "5f7dc8e0eafcb5371f8f2a34cd406787b5c35d15cede4c2b167f685629c2f84f",
322
+ "teacher_sha256": "90c08d120cfac926855e15c106dba1e242099bcd0ae6da7655bfaa54d5ce2420",
323
+ "token_sha256": "d644a79da2ba01cefc5f2020eab9e8662d45ba3025377844e4bf4d3f1d36802d",
324
+ "raw_retirement": "after durable score and verified raw hash",
325
+ "raw_bytes": 1268157440
326
+ },
327
+ {
328
+ "window_id": "conditional-fit-0032",
329
+ "domain": "axis1_general",
330
+ "prediction_rows": 2047,
331
+ "true_decode_rows": 2046,
332
+ "true_decode_mean_kld": 0.033507635709301424,
333
+ "including_prefill_mean_kld": 0.03461520895529943,
334
+ "prefill_row_kld": 2.3007100702672156,
335
+ "raw_sha256": "06846c021ae34fc753d82f7163b2c4fecb1ea4d093d1a3c4e494afbff734a9a0",
336
+ "score_sha256": "e561f17f20aa32b915f1df05ee4af640cb8d053ad9c8df19c6fffdf27d9f01b0",
337
+ "teacher_sha256": "1aa70ccf8e224c3a58834e2232a3a2477768db29ffeca39f5b27c17ad8c40c2b",
338
+ "token_sha256": "88745c333ead140a6fb05929031613429cbd42e25881d92add968a94a5d2d930",
339
+ "raw_retirement": "after durable score and verified raw hash",
340
+ "raw_bytes": 1268157440
341
+ },
342
+ {
343
+ "window_id": "conditional-fit-0035",
344
+ "domain": "axis4_reasoning_termination",
345
+ "prediction_rows": 2047,
346
+ "true_decode_rows": 2046,
347
+ "true_decode_mean_kld": 0.015753355570421665,
348
+ "including_prefill_mean_kld": 0.015747717272730583,
349
+ "prefill_row_kld": 0.004211760196777308,
350
+ "raw_sha256": "c6735072659f7f4c6f1c9dc5aa12bd496cd81be2c55f5198023d60e5b2bebcc8",
351
+ "score_sha256": "0d8da276e156afdb165e71e5982d7980741407a093267641016b7880cd2c0a9d",
352
+ "teacher_sha256": "4e958dce74e093eb62ba5ac210fe0a4e50f12f1875de678769700cde206b429d",
353
+ "token_sha256": "49f702136da39db9e762b6815a1952c0eba0a1fb04af22fb28a27bb63e2a41dc",
354
+ "raw_retirement": "after durable score and verified raw hash",
355
+ "raw_bytes": 1268157440
356
+ },
357
+ {
358
+ "window_id": "conditional-fit-0040",
359
+ "domain": "axis1_general",
360
+ "prediction_rows": 2047,
361
+ "true_decode_rows": 2046,
362
+ "true_decode_mean_kld": 0.04597109826675457,
363
+ "including_prefill_mean_kld": 0.04706420401414631,
364
+ "prefill_row_kld": 2.2835585631776425,
365
+ "raw_sha256": "615796ddf57e73c358a672505908dee823ae3da813ad0e43d69d2e5b15c07d80",
366
+ "score_sha256": "53206f5c9c253f5406ea7b054cfa816db5112665f5c2be6561e2016d1fa402ed",
367
+ "teacher_sha256": "f4c96798fc9eb5564d9aec52779cc0ce88bc55312dcd7663240938d711c8a90d",
368
+ "token_sha256": "50f3bc08ab37c551a05364eb48f0d726742f19331e79114dd5a9fb3c42ce6d4f",
369
+ "raw_retirement": "after durable score and verified raw hash",
370
+ "raw_bytes": 1268157440
371
+ },
372
+ {
373
+ "window_id": "conditional-fit-0041",
374
+ "domain": "axis2_legal",
375
+ "prediction_rows": 2047,
376
+ "true_decode_rows": 2046,
377
+ "true_decode_mean_kld": 0.07671995932113433,
378
+ "including_prefill_mean_kld": 0.077568246897156,
379
+ "prefill_row_kld": 1.8131646274374966,
380
+ "raw_sha256": "f3486e51a876933be29f0ff6656920a2b778351cbc722f24c0829d6e5735901e",
381
+ "score_sha256": "f94f09523579b471bd22d57bee5657add70718f81bf13271592ae266f0c401a3",
382
+ "teacher_sha256": "c2bbe4ea31532baceff966423c6ce6e042aebb5a2eb1de45e382a7d3df6f6827",
383
+ "token_sha256": "8acf2614f0f787c185af112f21bcdf247c9b6ae76d292ada28f476e2a7b0241d",
384
+ "raw_retirement": "after durable score and verified raw hash",
385
+ "raw_bytes": 1268157440
386
+ },
387
+ {
388
+ "window_id": "conditional-fit-0046",
389
+ "domain": "axis3_code_agentic",
390
+ "prediction_rows": 2047,
391
+ "true_decode_rows": 2046,
392
+ "true_decode_mean_kld": 0.07234383010849382,
393
+ "including_prefill_mean_kld": 0.07241689437092738,
394
+ "prefill_row_kld": 0.22190637530999638,
395
+ "raw_sha256": "04c26659f459b862e0584309731b22e5d21d01a9e4449b726053a16b3796d642",
396
+ "score_sha256": "fcd1f44a5795aa1c23ff4c21f90db35b93a5419ece26a0553a4518b3e566970b",
397
+ "teacher_sha256": "3caa5c58fb0b0e80129d551b39867fd52fefd93830060bb5655862fdd0e3b8ce",
398
+ "token_sha256": "3f2df1c090abf4473163a078c168fec33ab98ed6415985b92449caa861c04444",
399
+ "raw_retirement": "after durable score and verified raw hash",
400
+ "raw_bytes": 1268157440
401
+ },
402
+ {
403
+ "window_id": "conditional-fit-0060",
404
+ "domain": "axis1_general",
405
+ "prediction_rows": 2047,
406
+ "true_decode_rows": 2046,
407
+ "true_decode_mean_kld": 0.019628373069938383,
408
+ "including_prefill_mean_kld": 0.020660688213730436,
409
+ "prefill_row_kld": 2.1327774724122643,
410
+ "raw_sha256": "688a90f82f45f3dad38f0a72a3b663e0a71ffc5e9ca80528e53e43879bba4610",
411
+ "score_sha256": "d96ce7e3e1f13a8b49afb666fc3051cb23760b27f7ede74d4c5405d18de22f44",
412
+ "teacher_sha256": "d9e6344664b92527ee58d3bca038702676b357e7052ade03ac4bc9c70908cc34",
413
+ "token_sha256": "04ffe38aafcd9407b2846d59e8d797bee04b97f00683f8bc74d475a2d6b8592b",
414
+ "raw_retirement": "after durable score and verified raw hash",
415
+ "raw_bytes": 1268157440
416
+ },
417
+ {
418
+ "window_id": "conditional-fit-0063",
419
+ "domain": "axis4_reasoning_termination",
420
+ "prediction_rows": 2047,
421
+ "true_decode_rows": 2046,
422
+ "true_decode_mean_kld": 0.026129365452294863,
423
+ "including_prefill_mean_kld": 0.026535091377105725,
424
+ "prefill_row_kld": 0.8566503335401285,
425
+ "raw_sha256": "01445f5b0df02c5d68900360da39e4d7c52ffc099dc8f89d58ceecce3d06b305",
426
+ "score_sha256": "b34a6f8bb4a69bfc147bc76df6b8ca1fae7425c53a02ba4cd570b92df5cefc77",
427
+ "teacher_sha256": "a9bb59659437c1df5b3f3ef0b54eda61402842a2597b76b0124f58b59ecf3ab4",
428
+ "token_sha256": "38003e2d26db850dd8ff39f2e95fa3d1be9a34838c70d9f7e87056b7ef50b67d",
429
+ "raw_retirement": "after durable score and verified raw hash",
430
+ "raw_bytes": 1268157440
431
+ },
432
+ {
433
+ "window_id": "conditional-fit-0074",
434
+ "domain": "axis2_legal",
435
+ "prediction_rows": 2047,
436
+ "true_decode_rows": 2046,
437
+ "true_decode_mean_kld": 0.04218677374740498,
438
+ "including_prefill_mean_kld": 0.04276227996673919,
439
+ "prefill_row_kld": 1.2202480047245254,
440
+ "raw_sha256": "6b1b3fadfdc741d4dae747cc9bba50570a49e4e33a23221c4693775904ec7b7b",
441
+ "score_sha256": "67e46c4c9f47e3dcdace4de241760311be0a3d8e50d34e2efa8f801ad0c51ee1",
442
+ "teacher_sha256": "440341730d063ae27bbc15bcbfc879d9d16b3b25691d97b8cfd60be8f32f2313",
443
+ "token_sha256": "6f6618094ffce3f44b50f94cc533d338a3bfbbda7432890cd5ad2c13efd50129",
444
+ "raw_retirement": "after durable score and verified raw hash",
445
+ "raw_bytes": 1268157440
446
+ },
447
+ {
448
+ "window_id": "conditional-fit-0099",
449
+ "domain": "axis3_code_agentic",
450
+ "prediction_rows": 2047,
451
+ "true_decode_rows": 2046,
452
+ "true_decode_mean_kld": 0.014557495693748427,
453
+ "including_prefill_mean_kld": 0.014584803788056863,
454
+ "prefill_row_kld": 0.07045716474311164,
455
+ "raw_sha256": "d17316f3263c27d2276a4eaf0e0c9a0db0ff6442d7414be32326aab7fbeb7152",
456
+ "score_sha256": "9af87d56dc24e6b616ffd759b55c7c64c3698effc4a5432e1a93ff9ca68725a1",
457
+ "teacher_sha256": "da8dea840dd57564f0b3675872eaf297631b38a6d8dabcde7f8b54db06623e9a",
458
+ "token_sha256": "07e37956eb606026bdd7479ac5516275e92c0ee834c3c0439a3f300f9099608d",
459
+ "raw_retirement": "after durable score and verified raw hash",
460
+ "raw_bytes": 1268157440
461
+ },
462
+ {
463
+ "window_id": "conditional-fit-0113",
464
+ "domain": "axis2_legal",
465
+ "prediction_rows": 2047,
466
+ "true_decode_rows": 2046,
467
+ "true_decode_mean_kld": 0.053326012340420864,
468
+ "including_prefill_mean_kld": 0.05418572832239306,
469
+ "prefill_row_kld": 1.8131646274374966,
470
+ "raw_sha256": "c4e382d7ab1b99a2bc226cabd882f2c7c4e0763163aaff6a1e1374372378ee32",
471
+ "score_sha256": "ca2567079ef41e93fe9ba19a536fa45619090dcf218a0dd1c1b8b40924fee362",
472
+ "teacher_sha256": "6dc5f38f3436d6833b0ab68ce34e0f7ee5cde80359c26602526d8de512fd3207",
473
+ "token_sha256": "742ab50951388e712d20dab49b12ad7eb31490cfe5091a2fbc76f31ed8fa1b09",
474
+ "raw_retirement": "after durable score and verified raw hash",
475
+ "raw_bytes": 1268157440
476
+ },
477
+ {
478
+ "window_id": "conditional-fit-0118",
479
+ "domain": "axis1_general",
480
+ "prediction_rows": 2047,
481
+ "true_decode_rows": 2046,
482
+ "true_decode_mean_kld": 0.012467486577109195,
483
+ "including_prefill_mean_kld": 0.012881541710171803,
484
+ "prefill_row_kld": 0.8600383439562671,
485
+ "raw_sha256": "6051661893d329024bb0b34ae027b8ec6075f01634cea83d4814916c1440f807",
486
+ "score_sha256": "a3e717d8dbdf162e89757147d2143be0ca13d6aee8396556d949646361133efc",
487
+ "teacher_sha256": "e296287593aa0dc89b695a6a082a224a85904757c4a3cefa69729394d1469f89",
488
+ "token_sha256": "2e03fdfcf4d54fc3174ee4620d5efd9f2e5c10e2e6e0b1217577f08392a09749",
489
+ "raw_retirement": "after durable score and verified raw hash",
490
+ "raw_bytes": 1268157440
491
+ }
492
+ ],
493
+ "domains": {
494
+ "axis1_general": 0.030614171915487813,
495
+ "axis2_legal": 0.04639708824943424,
496
+ "axis3_code_agentic": 0.031949470625508854,
497
+ "axis4_reasoning_termination": 0.018819961875753848
498
+ },
499
+ "correctness_profile_engine_capacity_tokens": [
500
+ 19355555
501
+ ]
502
+ },
503
+ "nvfp4": {
504
+ "mean_true_decode_kld": 0.03545622377215776,
505
+ "window_bca95": [
506
+ 0.029555884771305184,
507
+ 0.04346203678097485
508
+ ],
509
+ "per_window": [
510
+ {
511
+ "window_id": "conditional-fit-0056",
512
+ "domain": "axis1_general",
513
+ "prediction_rows": 2047,
514
+ "true_decode_rows": 2046,
515
+ "true_decode_mean_kld": 0.023563083317839253,
516
+ "including_prefill_mean_kld": 0.025769605740888088,
517
+ "prefill_row_kld": 4.540314483298789,
518
+ "raw_sha256": "435e6c97b6f953bed64d5bc5fe1a66a22a35b5c8a0a5e181b0054577b69c89cd",
519
+ "score_sha256": "76b63f8ba02de8d5c913bcc6870fcf73890f72f8eeecae981741c148a0886c71",
520
+ "teacher_sha256": "55a70a330f39c5f23294c23d270a1ddc53031def99a42206e2706d4d0bb4466b",
521
+ "token_sha256": "219d8cd8873c6cf41f52782297cc5cfa9c4ed9666f2bc30d84cee9caa68ea744",
522
+ "raw_retirement": "after durable score and verified raw hash",
523
+ "raw_bytes": 1268157440
524
+ },
525
+ {
526
+ "window_id": "conditional-fit-0008",
527
+ "domain": "axis1_general",
528
+ "prediction_rows": 2047,
529
+ "true_decode_rows": 2046,
530
+ "true_decode_mean_kld": 0.06856101759452098,
531
+ "including_prefill_mean_kld": 0.06958029764551615,
532
+ "prefill_row_kld": 2.1550272819816234,
533
+ "raw_sha256": "0bcd5339efda6b717b15343f5de53e83d0757e1fd96aa9b8e08840b11fd32be2",
534
+ "score_sha256": "592c290d69ded521dfbceb1bfda56d0a23550bf6bc90c1b33692cecfe34aa2ee",
535
+ "teacher_sha256": "040641247b2e060035d89c2e7345e029d7c142983585e7c5d63dc818486f650d",
536
+ "token_sha256": "581d72e3e78b15c02d9e73f6bc742260cb57cdcd74337f4419c8f90cbc7c9c11",
537
+ "raw_retirement": "after durable score and verified raw hash",
538
+ "raw_bytes": 1268157440
539
+ },
540
+ {
541
+ "window_id": "conditional-fit-0100",
542
+ "domain": "axis1_general",
543
+ "prediction_rows": 2047,
544
+ "true_decode_rows": 2046,
545
+ "true_decode_mean_kld": 0.02667201133477343,
546
+ "including_prefill_mean_kld": 0.027072764083237236,
547
+ "prefill_row_kld": 0.8470128874401833,
548
+ "raw_sha256": "b75fba8ee764641d60da5e9aa16fc0d8331c1f6fc90bf9da54c8fd9d3583dd6b",
549
+ "score_sha256": "83558033705994a6c4563da73feca28252fde4e18e6fdb0e1a1597fbaf785ebe",
550
+ "teacher_sha256": "f62991383b3ef5e8db0ed71442b17c428972e58da1c868c6aeb0f3fa143c865b",
551
+ "token_sha256": "7f008ec99068f373c6e15dd059ee977fb70e6dc832f46e009f0f299d48d19892",
552
+ "raw_retirement": "after durable score and verified raw hash",
553
+ "raw_bytes": 1268157440
554
+ },
555
+ {
556
+ "window_id": "conditional-fit-0094",
557
+ "domain": "axis1_general",
558
+ "prediction_rows": 2047,
559
+ "true_decode_rows": 2046,
560
+ "true_decode_mean_kld": 0.01980333069056471,
561
+ "including_prefill_mean_kld": 0.019910660629571305,
562
+ "prefill_row_kld": 0.2395077158370706,
563
+ "raw_sha256": "387038afed9bd884c7430caed10c37159f96b475f782f74c42f39d161289c416",
564
+ "score_sha256": "4a29ea890ee1c5fb5ef4b38af4faf3a2d0ccbf4b9e039cd4e2fae795ab9adad0",
565
+ "teacher_sha256": "656911b8fb34d7f42e984dd21311caf9ef56db0b7618b414cad4089562d93cba",
566
+ "token_sha256": "6b6a9b21c49a9984cc8c5ea90e0bda6db5761d039c7cdd335dc65e30444e017b",
567
+ "raw_retirement": "after durable score and verified raw hash",
568
+ "raw_bytes": 1268157440
569
+ },
570
+ {
571
+ "window_id": "conditional-fit-0083",
572
+ "domain": "axis2_legal",
573
+ "prediction_rows": 2047,
574
+ "true_decode_rows": 2046,
575
+ "true_decode_mean_kld": 0.029895885238883084,
576
+ "including_prefill_mean_kld": 0.03010492019686818,
577
+ "prefill_row_kld": 0.4577904442343803,
578
+ "raw_sha256": "caeca2ce6c7e3883a51eb1fe7af4f3f945fac4122c763b7ff6770fa3e8fc3532",
579
+ "score_sha256": "cf6e4f767052c8393f8f7f422237981e13589db63db732ef56478b955b0cfec2",
580
+ "teacher_sha256": "66a7870a1f13c8b5d647c1681fca005bb0f55a6e502e94f282da673a6bed3f1b",
581
+ "token_sha256": "3df9d2872815e061ff8d32a9a19da0d11d80ac67070cccd28579b24ec3079552",
582
+ "raw_retirement": "after durable score and verified raw hash",
583
+ "raw_bytes": 1268157440
584
+ },
585
+ {
586
+ "window_id": "conditional-fit-0021",
587
+ "domain": "axis2_legal",
588
+ "prediction_rows": 2047,
589
+ "true_decode_rows": 2046,
590
+ "true_decode_mean_kld": 0.04397075315325536,
591
+ "including_prefill_mean_kld": 0.044495762997015596,
592
+ "prefill_row_kld": 1.1186659033304427,
593
+ "raw_sha256": "ff68e13dedc4ed15d16f1e3b7651f3b030cd406b4e0840480b33dc2fa2dbdfdb",
594
+ "score_sha256": "677465d5f820174bf8485fb26bc023cc12882f11a493e62ca2ce0b114ac7bcef",
595
+ "teacher_sha256": "ebf76910c4db8f6955ad03f2ec7352e2f8aaf15e0c939d28f9a7d78f6bd7451b",
596
+ "token_sha256": "dbd9f2539cb3e97127ae760a04ca5b3456b0b13d6ce8698f88b215cc68b8f1a9",
597
+ "raw_retirement": "after durable score and verified raw hash",
598
+ "raw_bytes": 1268157440
599
+ },
600
+ {
601
+ "window_id": "conditional-fit-0098",
602
+ "domain": "axis2_legal",
603
+ "prediction_rows": 2047,
604
+ "true_decode_rows": 2046,
605
+ "true_decode_mean_kld": 0.05041230633987613,
606
+ "including_prefill_mean_kld": 0.051236972335366136,
607
+ "prefill_row_kld": 1.738503599107944,
608
+ "raw_sha256": "4b290f9027a497d7af5bc5c06174dae87e2723fb95b93f3d2e02ffcf355d5234",
609
+ "score_sha256": "1e3fcd31518d465b473c11fbed05d857b572785311b32c88e7fa6ba9a7e75a18",
610
+ "teacher_sha256": "16e3d9524100836c020535635f7722a4c93ce0fdddc2ad0781f25bda77e8bcc9",
611
+ "token_sha256": "749bda3291a0323d286b0018b7dc3e0f469de8afd5d194370cde357b7b315c7d",
612
+ "raw_retirement": "after durable score and verified raw hash",
613
+ "raw_bytes": 1268157440
614
+ },
615
+ {
616
+ "window_id": "conditional-fit-0080",
617
+ "domain": "axis2_legal",
618
+ "prediction_rows": 2047,
619
+ "true_decode_rows": 2046,
620
+ "true_decode_mean_kld": 0.047528968125352955,
621
+ "including_prefill_mean_kld": 0.04910846403209107,
622
+ "prefill_row_kld": 3.2807570892182745,
623
+ "raw_sha256": "68f5c356fbf444f5d0022938812e368f1ba97c6cc2f7541072c1609923160030",
624
+ "score_sha256": "87dda561523fd7657fe08d596e9f14bc21a47d1cba5f5f425e392b4ea4d436dc",
625
+ "teacher_sha256": "e9c1a77490a5db63d410297bd8febe06adf4df2a6f4e16552db187738c9289d2",
626
+ "token_sha256": "a31d43f38710eb305c684501a2b02909a4e82c43101c07e76e4a3ca4edb4d0dd",
627
+ "raw_retirement": "after durable score and verified raw hash",
628
+ "raw_bytes": 1268157440
629
+ },
630
+ {
631
+ "window_id": "conditional-fit-0081",
632
+ "domain": "axis3_code_agentic",
633
+ "prediction_rows": 2047,
634
+ "true_decode_rows": 2046,
635
+ "true_decode_mean_kld": 0.025434826960204975,
636
+ "including_prefill_mean_kld": 0.026087226927511527,
637
+ "prefill_row_kld": 1.360897560036721,
638
+ "raw_sha256": "55a8ed57d62691e0e1252f3f244d14fe16a7382025f276fd851612b4b5d71ffc",
639
+ "score_sha256": "b3a51927b33af5a34e3a67c19f25e8cb70e335a25e09b270ea8916213a430f55",
640
+ "teacher_sha256": "5ae07760d91030d1eba925b76ea463309baa49c2e7d5f5e14409a23ced7fd20d",
641
+ "token_sha256": "350628488b94d59f3e4f3e99a418913544ea1fe7eacee14a642b996d9c614d16",
642
+ "raw_retirement": "after durable score and verified raw hash",
643
+ "raw_bytes": 1268157440
644
+ },
645
+ {
646
+ "window_id": "conditional-fit-0123",
647
+ "domain": "axis3_code_agentic",
648
+ "prediction_rows": 2047,
649
+ "true_decode_rows": 2046,
650
+ "true_decode_mean_kld": 0.03625410701112817,
651
+ "including_prefill_mean_kld": 0.03623887373163693,
652
+ "prefill_row_kld": 0.0050715838925684655,
653
+ "raw_sha256": "9b464ec3a977d1ee3cebc601ffb2a2169d0cf3a9ce70673ce1ea96d807947b12",
654
+ "score_sha256": "47f8a43f8c66147cabaedafeea53145f78935ed971f3cc45d5ec7f8870970954",
655
+ "teacher_sha256": "4a24b5b750ba98c06d9bc69e19334aff89e7bd4f9429592598af5d62de7ecb42",
656
+ "token_sha256": "aab84ee426f91621d123aa68bd53f75e220fbaf8cb270db1feab1529a62ddbf0",
657
+ "raw_retirement": "after durable score and verified raw hash",
658
+ "raw_bytes": 1268157440
659
+ },
660
+ {
661
+ "window_id": "conditional-fit-0090",
662
+ "domain": "axis3_code_agentic",
663
+ "prediction_rows": 2047,
664
+ "true_decode_rows": 2046,
665
+ "true_decode_mean_kld": 0.04104338344204917,
666
+ "including_prefill_mean_kld": 0.04195193367340904,
667
+ "prefill_row_kld": 1.9008457070357228,
668
+ "raw_sha256": "890278e84ba9ed1f86276a9702da857639ec6542257c80a8c84ccfb5c9485f63",
669
+ "score_sha256": "5e6e6d5b823b9f005abb50e738519564f2fc407837530abdf906b7cad8348e39",
670
+ "teacher_sha256": "db2a5fe33c9e457a22ebeb31a7073c1dac666b877503e9b4e90a8d6abf7b8a32",
671
+ "token_sha256": "a06e76993db717afee847fcc4d86f10b52dd23cdce5dae916f32d238d08d886c",
672
+ "raw_retirement": "after durable score and verified raw hash",
673
+ "raw_bytes": 1268157440
674
+ },
675
+ {
676
+ "window_id": "conditional-fit-0062",
677
+ "domain": "axis3_code_agentic",
678
+ "prediction_rows": 2047,
679
+ "true_decode_rows": 2046,
680
+ "true_decode_mean_kld": 0.02918700579461782,
681
+ "including_prefill_mean_kld": 0.030030454221016484,
682
+ "prefill_row_kld": 1.7557259346326815,
683
+ "raw_sha256": "4ee3b4a4a812c85274c3c836dcd86d0b123e38e4e33bbab437385e1955636035",
684
+ "score_sha256": "6a54c1c1ebae3e5b81823df2342e1d41c4b77ee7c9bbf26d205163b937fcca75",
685
+ "teacher_sha256": "d6f7883a533625c7c75119fe7403b48bf81885b1de0b311a1a0571949f7f231c",
686
+ "token_sha256": "7c6d2ef2f4b7c7091a517f6ccd778009f9ee144165c9aecf2f8f28aea7b3753f",
687
+ "raw_retirement": "after durable score and verified raw hash",
688
+ "raw_bytes": 1268157440
689
+ },
690
+ {
691
+ "window_id": "conditional-fit-0047",
692
+ "domain": "axis4_reasoning_termination",
693
+ "prediction_rows": 2047,
694
+ "true_decode_rows": 2046,
695
+ "true_decode_mean_kld": 0.025365434304478303,
696
+ "including_prefill_mean_kld": 0.03031257573187498,
697
+ "prefill_row_kld": 10.152163936185477,
698
+ "raw_sha256": "dd5c4505db35935c58c9f74bb6e74a6edd3c19374f835664d81a6e8de5b33eb9",
699
+ "score_sha256": "d6e3feb817873e71e0dd76cdfa4a8f638ca2b3b8ee42264f726be54b4d8f0845",
700
+ "teacher_sha256": "8513cdcb9940786c7c42bf14d2f5f2cd452791e4ec6036c06a1f973eb969018d",
701
+ "token_sha256": "89730ee6826d4302c352156d5df766c51622a8b6681b8052d4ca6e7d481177fe",
702
+ "raw_retirement": "after durable score and verified raw hash",
703
+ "raw_bytes": 1268157440
704
+ },
705
+ {
706
+ "window_id": "conditional-fit-0051",
707
+ "domain": "axis4_reasoning_termination",
708
+ "prediction_rows": 2047,
709
+ "true_decode_rows": 2046,
710
+ "true_decode_mean_kld": 0.012924581711263027,
711
+ "including_prefill_mean_kld": 0.013141907486799479,
712
+ "prefill_row_kld": 0.4577904442343803,
713
+ "raw_sha256": "d3d0a1b6cbfdcecf0842b87193c339608d8d51fb39007b7db0ec03f073db51a7",
714
+ "score_sha256": "c590acb2bb149f97ecf9be58ea5884de08396d26e18935a548f2d11c56fd8ce8",
715
+ "teacher_sha256": "177477c19b54dc0b0230e19202d48ab71d52f2b1a37fe383511bb64f0aae2fcf",
716
+ "token_sha256": "c79269b16560e7301549593443c9394007e7a957ad149c46224e7ad271dc6a18",
717
+ "raw_retirement": "after durable score and verified raw hash",
718
+ "raw_bytes": 1268157440
719
+ },
720
+ {
721
+ "window_id": "conditional-fit-0055",
722
+ "domain": "axis4_reasoning_termination",
723
+ "prediction_rows": 2047,
724
+ "true_decode_rows": 2046,
725
+ "true_decode_mean_kld": 0.023650142194727084,
726
+ "including_prefill_mean_kld": 0.02412364237713099,
727
+ "prefill_row_kld": 0.9929050155755266,
728
+ "raw_sha256": "0ab63005f5783e2d377aa953c64abe217cc6d1957ba9d164a3daf62b4a4f3693",
729
+ "score_sha256": "c7ca6c112700a4f07e16ad2806d6f2b18483a929964476311de92a1594a43746",
730
+ "teacher_sha256": "7679c828eae4bf08f17598d044904693fc90c05b9d3f189dddeeb846100e43a7",
731
+ "token_sha256": "feec9ac508172dd275ec6452a03493acab5ffdff9bb57161317c77f24daf90f2",
732
+ "raw_retirement": "after durable score and verified raw hash",
733
+ "raw_bytes": 1268157440
734
+ },
735
+ {
736
+ "window_id": "conditional-fit-0031",
737
+ "domain": "axis4_reasoning_termination",
738
+ "prediction_rows": 2047,
739
+ "true_decode_rows": 2046,
740
+ "true_decode_mean_kld": 0.02037624180208121,
741
+ "including_prefill_mean_kld": 0.02532582054872674,
742
+ "prefill_row_kld": 10.152163936185477,
743
+ "raw_sha256": "0e31a3893a84a0ec7e1fdb6fb71c2a14d4fb2175229b0019bc57e59728d0380a",
744
+ "score_sha256": "74ac2939aafb8dbe1f1654aacbb4abd0b3b295da5ce29ee0788188acb38dc105",
745
+ "teacher_sha256": "e04a6c04a951efb5a178927e91457535cc67745ce5c81e785cb6d963aac6ce83",
746
+ "token_sha256": "2cff3482b4212537d0c35b88b8a0f479502c8905952430c973d25b1ab0bb5818",
747
+ "raw_retirement": "after durable score and verified raw hash",
748
+ "raw_bytes": 1268157440
749
+ },
750
+ {
751
+ "window_id": "conditional-fit-0003",
752
+ "domain": "axis4_reasoning_termination",
753
+ "prediction_rows": 2047,
754
+ "true_decode_rows": 2046,
755
+ "true_decode_mean_kld": 0.02690073030649202,
756
+ "including_prefill_mean_kld": 0.029946398068044746,
757
+ "prefill_row_kld": 6.2613826382049185,
758
+ "raw_sha256": "bc400c83ee2139f177dabf51511a8683e1a73e03f57fc40554c77f563c7da8e0",
759
+ "score_sha256": "dbfa39c66bc06fa999aaac571e8377131c13782d16e47b119caf725f1c7d28f3",
760
+ "teacher_sha256": "7677c89cc3dee8e2a64e3c7947add39475f828921a3a8fe2e92b379b3ed8320b",
761
+ "token_sha256": "d416537c6e8273747bf025e97287a0bb6e3ff047bfee65cf54e4dfe287569b93",
762
+ "raw_retirement": "after durable score and verified raw hash",
763
+ "raw_bytes": 1268157440
764
+ },
765
+ {
766
+ "window_id": "conditional-fit-0009",
767
+ "domain": "axis2_legal",
768
+ "prediction_rows": 2047,
769
+ "true_decode_rows": 2046,
770
+ "true_decode_mean_kld": 0.04604780270271153,
771
+ "including_prefill_mean_kld": 0.046153134975366834,
772
+ "prefill_row_kld": 0.26166296482811324,
773
+ "raw_sha256": "b425869ce07538ae045c4d809021ae9564be0171c2776a9e7ab6465ffe11c5d7",
774
+ "score_sha256": "7ce0191e8d42f8aa8b9a45d584ac950a90b92023e131bb2dc60ffc35ee081583",
775
+ "teacher_sha256": "00701ad8bf4a4a5eedcedd4eafd956830c9ff1c790b15074ca721a3462abf930",
776
+ "token_sha256": "fc1166707f5938a914faf364dab717b0de3ea0e8451ed1a3fb399469e659b105",
777
+ "raw_retirement": "after durable score and verified raw hash",
778
+ "raw_bytes": 1268157440
779
+ },
780
+ {
781
+ "window_id": "conditional-fit-0010",
782
+ "domain": "axis3_code_agentic",
783
+ "prediction_rows": 2047,
784
+ "true_decode_rows": 2046,
785
+ "true_decode_mean_kld": 0.01563325095084946,
786
+ "including_prefill_mean_kld": 0.015634848716246288,
787
+ "prefill_row_kld": 0.018903876718152742,
788
+ "raw_sha256": "0020ca7d06c49315301acd377253c6fa721a5e8db71c88efaa33eb4ce96ee31b",
789
+ "score_sha256": "53fa079c5a28e23de66c8f16e72d56125b2721d5943a1ea387b4d2ac9ffaa5cf",
790
+ "teacher_sha256": "f9bf4519890b70d972dfdf9760181418eb7c55b6237c0108f01d44df89943278",
791
+ "token_sha256": "ddd9d769a76eba74bc8e680ff25736a3b9146371801d1d2b29e4dc48f6d256a2",
792
+ "raw_retirement": "after durable score and verified raw hash",
793
+ "raw_bytes": 1268157440
794
+ },
795
+ {
796
+ "window_id": "conditional-fit-0011",
797
+ "domain": "axis4_reasoning_termination",
798
+ "prediction_rows": 2047,
799
+ "true_decode_rows": 2046,
800
+ "true_decode_mean_kld": 0.018165048102336066,
801
+ "including_prefill_mean_kld": 0.023115707060852503,
802
+ "prefill_row_kld": 10.152163936185477,
803
+ "raw_sha256": "e5460f80bb681f0001beea1baead1660b76ba8565ec4aa06cbbb13e7f59b256f",
804
+ "score_sha256": "1870f3a7be94e65bf76df49d9fdf3690cd4c8563146acbc0186efc9e7c6c7cce",
805
+ "teacher_sha256": "9f37c4ead38bb80a5d85f825b156e5b493b6d3f1ca9dabfad06b0416f37db286",
806
+ "token_sha256": "c4b96acfa47944be3485b20eab76eb16d59913c9fedb5001a537335a3ee13614",
807
+ "raw_retirement": "after durable score and verified raw hash",
808
+ "raw_bytes": 1268157440
809
+ },
810
+ {
811
+ "window_id": "conditional-fit-0030",
812
+ "domain": "axis3_code_agentic",
813
+ "prediction_rows": 2047,
814
+ "true_decode_rows": 2046,
815
+ "true_decode_mean_kld": 0.024984801688836707,
816
+ "including_prefill_mean_kld": 0.024978325612706537,
817
+ "prefill_row_kld": 0.011728273850378172,
818
+ "raw_sha256": "9b7a816b646bd6c974b15105d1f7f92d162d594ca262d7c79fa1bcf1afdb980a",
819
+ "score_sha256": "fd35a10287015d380956b9985ce62b7df231d250fc9dbf2f923d3e974a7d29f6",
820
+ "teacher_sha256": "90c08d120cfac926855e15c106dba1e242099bcd0ae6da7655bfaa54d5ce2420",
821
+ "token_sha256": "d644a79da2ba01cefc5f2020eab9e8662d45ba3025377844e4bf4d3f1d36802d",
822
+ "raw_retirement": "after durable score and verified raw hash",
823
+ "raw_bytes": 1268157440
824
+ },
825
+ {
826
+ "window_id": "conditional-fit-0032",
827
+ "domain": "axis1_general",
828
+ "prediction_rows": 2047,
829
+ "true_decode_rows": 2046,
830
+ "true_decode_mean_kld": 0.021718153455878218,
831
+ "including_prefill_mean_kld": 0.022319411629929006,
832
+ "prefill_row_kld": 1.2524936357378378,
833
+ "raw_sha256": "b364f38f47ad85d8efe4e8426747acb301a9e63117501450139127fc75039721",
834
+ "score_sha256": "fb18bbe03c546a46151b364d26cf895f7991b8f30fe53489b9447d7cc7e0fa7c",
835
+ "teacher_sha256": "1aa70ccf8e224c3a58834e2232a3a2477768db29ffeca39f5b27c17ad8c40c2b",
836
+ "token_sha256": "88745c333ead140a6fb05929031613429cbd42e25881d92add968a94a5d2d930",
837
+ "raw_retirement": "after durable score and verified raw hash",
838
+ "raw_bytes": 1268157440
839
+ },
840
+ {
841
+ "window_id": "conditional-fit-0035",
842
+ "domain": "axis4_reasoning_termination",
843
+ "prediction_rows": 2047,
844
+ "true_decode_rows": 2046,
845
+ "true_decode_mean_kld": 0.032422367247282986,
846
+ "including_prefill_mean_kld": 0.03463729855715423,
847
+ "prefill_row_kld": 4.566386758553708,
848
+ "raw_sha256": "a4c51cbc1ac997dd9c3bb9215dadeca429e88747a33aa61d37445974baac480c",
849
+ "score_sha256": "37eefbf56891e3d8af693c6f8685bbd172337a6f8bf081849e805e78b3148866",
850
+ "teacher_sha256": "4e958dce74e093eb62ba5ac210fe0a4e50f12f1875de678769700cde206b429d",
851
+ "token_sha256": "49f702136da39db9e762b6815a1952c0eba0a1fb04af22fb28a27bb63e2a41dc",
852
+ "raw_retirement": "after durable score and verified raw hash",
853
+ "raw_bytes": 1268157440
854
+ },
855
+ {
856
+ "window_id": "conditional-fit-0040",
857
+ "domain": "axis1_general",
858
+ "prediction_rows": 2047,
859
+ "true_decode_rows": 2046,
860
+ "true_decode_mean_kld": 0.05400247271460698,
861
+ "including_prefill_mean_kld": 0.0549407898731226,
862
+ "prefill_row_kld": 1.9747376961960739,
863
+ "raw_sha256": "fd04d2487f3904f117be851809799cc8c320f950d589b3c328db83f9cda9dfcd",
864
+ "score_sha256": "c79be837a60e6ba89adb4b80483b1b10d49aeac719841479ca4eb8980618e9b7",
865
+ "teacher_sha256": "f4c96798fc9eb5564d9aec52779cc0ce88bc55312dcd7663240938d711c8a90d",
866
+ "token_sha256": "50f3bc08ab37c551a05364eb48f0d726742f19331e79114dd5a9fb3c42ce6d4f",
867
+ "raw_retirement": "after durable score and verified raw hash",
868
+ "raw_bytes": 1268157440
869
+ },
870
+ {
871
+ "window_id": "conditional-fit-0041",
872
+ "domain": "axis2_legal",
873
+ "prediction_rows": 2047,
874
+ "true_decode_rows": 2046,
875
+ "true_decode_mean_kld": 0.08657045244168773,
876
+ "including_prefill_mean_kld": 0.08720775297518737,
877
+ "prefill_row_kld": 1.3911246445154333,
878
+ "raw_sha256": "8bbe8f464764b145ae367b5ffdbafd78c30bacf85f8bc90f22bbe2db6e7da597",
879
+ "score_sha256": "b571a88deab449256a1c881ef8550aad88b9ea274c98f6182267c11da48e79bf",
880
+ "teacher_sha256": "c2bbe4ea31532baceff966423c6ce6e042aebb5a2eb1de45e382a7d3df6f6827",
881
+ "token_sha256": "8acf2614f0f787c185af112f21bcdf247c9b6ae76d292ada28f476e2a7b0241d",
882
+ "raw_retirement": "after durable score and verified raw hash",
883
+ "raw_bytes": 1268157440
884
+ },
885
+ {
886
+ "window_id": "conditional-fit-0046",
887
+ "domain": "axis3_code_agentic",
888
+ "prediction_rows": 2047,
889
+ "true_decode_rows": 2046,
890
+ "true_decode_mean_kld": 0.06463482692444868,
891
+ "including_prefill_mean_kld": 0.0648417166461593,
892
+ "prefill_row_kld": 0.48813808726609126,
893
+ "raw_sha256": "daa52d49b4550a1fd0bdf38c70a737b613369f485bd5051fafae07cb8ce6a5eb",
894
+ "score_sha256": "b25aa33f209bc0e72888e76efe9feef250e2c106ca34142f4c0dbd353a355532",
895
+ "teacher_sha256": "3caa5c58fb0b0e80129d551b39867fd52fefd93830060bb5655862fdd0e3b8ce",
896
+ "token_sha256": "3f2df1c090abf4473163a078c168fec33ab98ed6415985b92449caa861c04444",
897
+ "raw_retirement": "after durable score and verified raw hash",
898
+ "raw_bytes": 1268157440
899
+ },
900
+ {
901
+ "window_id": "conditional-fit-0060",
902
+ "domain": "axis1_general",
903
+ "prediction_rows": 2047,
904
+ "true_decode_rows": 2046,
905
+ "true_decode_mean_kld": 0.015704907348830933,
906
+ "including_prefill_mean_kld": 0.01629535873627198,
907
+ "prefill_row_kld": 1.2243588974406492,
908
+ "raw_sha256": "e4067368b626a98b028811ad6772aeaeadd85f04dcdeb9f91dab6658078b3a74",
909
+ "score_sha256": "bab86bd14f05d787511ea71f9bdf76096edf975451528afd499e35b3701a5454",
910
+ "teacher_sha256": "d9e6344664b92527ee58d3bca038702676b357e7052ade03ac4bc9c70908cc34",
911
+ "token_sha256": "04ffe38aafcd9407b2846d59e8d797bee04b97f00683f8bc74d475a2d6b8592b",
912
+ "raw_retirement": "after durable score and verified raw hash",
913
+ "raw_bytes": 1268157440
914
+ },
915
+ {
916
+ "window_id": "conditional-fit-0063",
917
+ "domain": "axis4_reasoning_termination",
918
+ "prediction_rows": 2047,
919
+ "true_decode_rows": 2046,
920
+ "true_decode_mean_kld": 0.03132559147375623,
921
+ "including_prefill_mean_kld": 0.031814505567999064,
922
+ "prefill_row_kld": 1.0321327423888638,
923
+ "raw_sha256": "614a33a2276349c6d3ef14734996f4579e50cc22ef755b409471361e7b24699d",
924
+ "score_sha256": "17cfb59297e5f31cfa88eae36930a85afbfd41d4edfdbd51e9c612bfe75af43f",
925
+ "teacher_sha256": "a9bb59659437c1df5b3f3ef0b54eda61402842a2597b76b0124f58b59ecf3ab4",
926
+ "token_sha256": "38003e2d26db850dd8ff39f2e95fa3d1be9a34838c70d9f7e87056b7ef50b67d",
927
+ "raw_retirement": "after durable score and verified raw hash",
928
+ "raw_bytes": 1268157440
929
+ },
930
+ {
931
+ "window_id": "conditional-fit-0074",
932
+ "domain": "axis2_legal",
933
+ "prediction_rows": 2047,
934
+ "true_decode_rows": 2046,
935
+ "true_decode_mean_kld": 0.08556837423676537,
936
+ "including_prefill_mean_kld": 0.0867002508399446,
937
+ "prefill_row_kld": 2.4025197809446555,
938
+ "raw_sha256": "9abd2d8047eef0aa95de284928fde34187426078a2759b214f339b67fafea057",
939
+ "score_sha256": "02b7030207fb7736d57c060fbc8208a57254ef4f66b81b6954e3b8b1cb2ca972",
940
+ "teacher_sha256": "440341730d063ae27bbc15bcbfc879d9d16b3b25691d97b8cfd60be8f32f2313",
941
+ "token_sha256": "6f6618094ffce3f44b50f94cc533d338a3bfbbda7432890cd5ad2c13efd50129",
942
+ "raw_retirement": "after durable score and verified raw hash",
943
+ "raw_bytes": 1268157440
944
+ },
945
+ {
946
+ "window_id": "conditional-fit-0099",
947
+ "domain": "axis3_code_agentic",
948
+ "prediction_rows": 2047,
949
+ "true_decode_rows": 2046,
950
+ "true_decode_mean_kld": 0.01659116406314401,
951
+ "including_prefill_mean_kld": 0.016611742343204117,
952
+ "prefill_row_kld": 0.058714903346186786,
953
+ "raw_sha256": "5cb40de2fe89feb4655bd5ab724266f6376e2d8559eb6f14838fff92e0f57eda",
954
+ "score_sha256": "c52bf6a766f6db58d17972e8cdb67aaec0d255c27eee59affc117e2bf110f423",
955
+ "teacher_sha256": "da8dea840dd57564f0b3675872eaf297631b38a6d8dabcde7f8b54db06623e9a",
956
+ "token_sha256": "07e37956eb606026bdd7479ac5516275e92c0ee834c3c0439a3f300f9099608d",
957
+ "raw_retirement": "after durable score and verified raw hash",
958
+ "raw_bytes": 1268157440
959
+ },
960
+ {
961
+ "window_id": "conditional-fit-0113",
962
+ "domain": "axis2_legal",
963
+ "prediction_rows": 2047,
964
+ "true_decode_rows": 2046,
965
+ "true_decode_mean_kld": 0.0561128193667057,
966
+ "including_prefill_mean_kld": 0.056764999056568295,
967
+ "prefill_row_kld": 1.3911246445154333,
968
+ "raw_sha256": "c232c29068dada31c94b8aba188f91fd007fd4bc2170cec51b51ada2659f0593",
969
+ "score_sha256": "19dcdfe7a722a3f784e204d6b7126be29a6fd82803522814a9f13efba2d99c8c",
970
+ "teacher_sha256": "6dc5f38f3436d6833b0ab68ce34e0f7ee5cde80359c26602526d8de512fd3207",
971
+ "token_sha256": "742ab50951388e712d20dab49b12ad7eb31490cfe5091a2fbc76f31ed8fa1b09",
972
+ "raw_retirement": "after durable score and verified raw hash",
973
+ "raw_bytes": 1268157440
974
+ },
975
+ {
976
+ "window_id": "conditional-fit-0118",
977
+ "domain": "axis1_general",
978
+ "prediction_rows": 2047,
979
+ "true_decode_rows": 2046,
980
+ "true_decode_mean_kld": 0.013573318669100111,
981
+ "including_prefill_mean_kld": 0.01402729782126711,
982
+ "prefill_row_kld": 0.9428686431549456,
983
+ "raw_sha256": "1dd59d9376e947ed299c24653b27953232374ae9f1b124692f8dfb9534afdb85",
984
+ "score_sha256": "4f2f5d0ac0658964a5bc7efa3ddb34089a34adc96e31a3e17a428179686992d0",
985
+ "teacher_sha256": "e296287593aa0dc89b695a6a082a224a85904757c4a3cefa69729394d1469f89",
986
+ "token_sha256": "2e03fdfcf4d54fc3174ee4620d5efd9f2e5c10e2e6e0b1217577f08392a09749",
987
+ "raw_retirement": "after durable score and verified raw hash",
988
+ "raw_bytes": 1268157440
989
+ }
990
+ ],
991
+ "domains": {
992
+ "axis1_general": 0.030449786890764326,
993
+ "axis2_legal": 0.05576342020065474,
994
+ "axis3_code_agentic": 0.031720420854409875,
995
+ "axis4_reasoning_termination": 0.023891267142802115
996
+ },
997
+ "correctness_profile_engine_capacity_tokens": [
998
+ 31557971
999
+ ]
1000
+ }
1001
+ },
1002
+ "fp8_minus_nvfp4": -0.003511050605611574,
1003
+ "paired_window_bca95": [
1004
+ -0.008171877159734635,
1005
+ -0.001367594090752266
1006
+ ],
1007
+ "fp8_lower_windows": 22,
1008
+ "true_decode_rows_per_window": 2046,
1009
+ "windows": 32,
1010
+ "historical_dcp1_context_only": 0.03418114591027796,
1011
+ "limits": [
1012
+ "DCP4 new versus DCP1 historical; different runtime image.",
1013
+ "MTP off and maxseq1 forced decode; not MTP quality or throughput evidence.",
1014
+ "Fixed FP8 then NVFP4 order; one server preparation per arm, window intervals do not capture server-run variability."
1015
+ ],
1016
+ "receipt_audit": [
1017
+ {
1018
+ "arm": "fp8",
1019
+ "windows": 32,
1020
+ "prediction_rows": 65504,
1021
+ "true_decode_rows": 65472,
1022
+ "all_receipt_hashes_and_masks_pass": true,
1023
+ "all_prefix_hit_counters_zero": true,
1024
+ "raw_logits_retired_after_scoring": true,
1025
+ "source_runtime_audit_sha256": "6a3393053b83b33b41a7d49de2fab1803b1efc4843ec5862c1acd8467b993fa4",
1026
+ "source_startup_sha256": "00ce74cfce281b627289f3322829d357467f5bc1f026677ef27cda93ed43fce3"
1027
+ },
1028
+ {
1029
+ "arm": "nvfp4",
1030
+ "windows": 32,
1031
+ "prediction_rows": 65504,
1032
+ "true_decode_rows": 65472,
1033
+ "all_receipt_hashes_and_masks_pass": true,
1034
+ "all_prefix_hit_counters_zero": true,
1035
+ "raw_logits_retired_after_scoring": true,
1036
+ "source_runtime_audit_sha256": "74d7d1914a55a73b5e9de196f557c658603a9501acd2fafb1e540d496b1c7185",
1037
+ "source_startup_sha256": "cae3cb749c61cc18fd23286f41f79266e32bfdc551e0575b3515d21eb993036a"
1038
+ }
1039
+ ],
1040
+ "public_base_image": "verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf",
1041
+ "capture_image_local_id": "sha256:0405a1c0dc128b51069798a5d00b346257bbd006c0deb7e6530a3d005d75de71",
1042
+ "plan_seal": {
1043
+ "algorithm": "sha256",
1044
+ "created_at": "2026-09-09T20:32:50.484730+00:00",
1045
+ "plan_sha256": "8bd19d67a09ddc6ef6bea1afe67306f1a6859e927cb31969730b51e5d13e24fe"
1046
+ },
1047
+ "metric_sha256": "752af6740595791f300cf47f15ef81f8ca85245d5282fc600ef26ee57eae41e6",
1048
+ "role_manifest_sha256": "b5d7e4524eb98ddfbd230a5d9a44de0dc5dbeb796c03e859898b5838e4463d14",
1049
+ "teacher_revision": "7c378d5f17dba158c4c803eff27c346dd0615660",
1050
+ "teacher_model_revision": "a6c167b62691b2bac901344b65cb651a70f53e43",
1051
+ "capture_cpu_test_sha256": "d470ec0385cb9164d789a3a04c5aa638464b656f1c6a0ac12fdc9c2f8a5f37bc",
1052
+ "prior_attempts": []
1053
+ }
results/speed-20260909/README.md ADDED
@@ -0,0 +1,40 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # September 9 TrellisMX FP8 speed experiments
2
+
3
+ Selected runtime: **token-map-hoist reference**, local image identity `sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf`. TP4/DCP4, MTP3 probabilistic, CUDA graphs, NVFP4 MLA KV, 24 sequences, batch4096, four RTX PRO6000 GPUs at300W. Expert math remains E4M3 FP8 with UE8M0 scales. Checkpoint unchanged.
4
+
5
+ All rates are tokens/sec; C4 is aggregate. Single exploratory server runs; MTP acceptance/output trajectories and sustained clocks differ. Later cooled screens require90seconds minimum idle and all GPUs<=55C continuously30seconds before each cell. Earlier runs retain their original protocols; do not silently treat them as cooled replications. No independent replication or superiority claim. No measurement reached300t/s C1 or20000t/s prefill.
6
+
7
+ | Measurement | Selected reference | Task-count | Selective grid-floor |
8
+ |---|---:|---:|---:|
9
+ | 32K prefill | 8407.000 | 8439.000 | 8423.000 |
10
+ | 64K prefill | 8407.000 | 8451.000 | 8448.000 |
11
+ | 8K C1 decode | 222.100 | 212.098 | 214.588 |
12
+ | 8K C4 decode aggregate | 318.560 | 325.083 | 326.620 |
13
+ | 16K C1 decode | 216.636 | 208.708 | 217.293 |
14
+ | 16K C4 decode aggregate | 319.157 | 353.532 | 343.859 |
15
+ | 0K C1 decode | 204.611 | 195.372 | 185.422 |
16
+ | 0K C4 decode aggregate | 327.670 | 333.729 | 325.106 |
17
+
18
+ ## Selected reference extended context
19
+
20
+ | Measurement | Tokens/sec |
21
+ |---|---:|
22
+ | 32K prefill | 8457.000 |
23
+ | 64K prefill | 8443.000 |
24
+ | 128K prefill | 8323.000 |
25
+ | 32K C1 decode | 204.930 |
26
+ | 32K C4 decode aggregate | 329.343 |
27
+ | 64K C1 decode | 199.446 |
28
+ | 64K C4 decode aggregate | 330.096 |
29
+ | 128K C1 decode | 204.445 |
30
+ | 128K C4 decode aggregate | 334.086 |
31
+ | 8K C8 decode aggregate | 594.639 |
32
+ | 16K C8 decode aggregate | 602.965 |
33
+ | 0K C1 decode | 204.611 |
34
+ | 0K C4 decode aggregate | 327.670 |
35
+
36
+ Task-count and grid-floor were not measured at C1 contexts32K/64K/128K. Their shorter-context results cannot rank long-context C1. Reference KV capacity is23,562,091 tokens in the expanded run; separate zero-context reference sessions reported23,568,627. Use the explicit engine KV metric, not the benchmark's generic block-times-DCP estimate.
37
+
38
+ Reference is retained for balanced use: best observed C1 at0K/8K and effectively tied at16K. Task-count is a concurrent-throughput alternative, especially16K C4. Neither alternative has a meaningful prefill improvement established. Both alternatives passed172 exact representative K4/K5 component comparisons and FP8 compiled-opcode checks; these are not full-model quality evaluations.
39
+
40
+ `benchmark-index.json` indexes every discovered llm_decode_bench result in today's campaign, including earlier/negative/profile diagnostics. Results under profiling paths are diagnostic, not serving throughput qualification. `evidence/` includes raw benchmark JSON and associated logs/receipts plus candidate summaries/failures. Local host/path prefixes are redacted; `public-file-inventory.json` records both original and public hashes. Historical raw receipt hashes refer to the originals, not the redacted copies. Raw local archives are retained separately. KLD is published separately with cache dtype and capture-image provenance.
results/speed-20260909/benchmark-index.json ADDED
The diff for this file is too large to render. See raw diff
 
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-command.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json",
20
+ "--contexts",
21
+ "0,8k,32k",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--skip-prefill",
25
+ "--cell-warmup-timeout-seconds",
26
+ "180"
27
+ ]
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "cd13e14d3d3f0b8b357a7d0c2f80a963c62ba2aa169ff9bbde9ad08d9d3cd255"
5
+ }
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json ADDED
@@ -0,0 +1,1394 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T07:18:01.203111",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 32768,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 17031168,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0,
42
+ 8192,
43
+ 32768
44
+ ],
45
+ "startup_diagnostics_available": true,
46
+ "nvidia_p2p_override_effective": true,
47
+ "p2pmark_status": "not_run",
48
+ "amd_fabric_status": "not_run"
49
+ },
50
+ "startup_diagnostics": {
51
+ "version": "0.4.29",
52
+ "server_url": "http://127.0.0.1:8001",
53
+ "hostname": "<host>",
54
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
55
+ "env": {},
56
+ "args": {
57
+ "concurrency": "1,2,4",
58
+ "contexts": "0,8k,32k",
59
+ "max_tokens": 8192,
60
+ "duration": 20.0,
61
+ "request_count": 0,
62
+ "run_burst": false,
63
+ "standalone_prefill": false,
64
+ "prefill_only": false,
65
+ "skip_prefill": true,
66
+ "prefill_contexts": "8k,64k,128k",
67
+ "prefill_metric": "client",
68
+ "dcp_size": 0,
69
+ "kv_budget": 0
70
+ },
71
+ "nvidia_p2p_override": {
72
+ "effective": true,
73
+ "configured": true,
74
+ "params_path": "/proc/driver/nvidia/params",
75
+ "params_available": true,
76
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
77
+ "modprobe_available": true,
78
+ "runtime": {
79
+ "ForceP2P": "0x11",
80
+ "RMForceP2PType": "1",
81
+ "RMPcieP2PType": "2",
82
+ "GrdmaPciTopoCheckOverride": "1",
83
+ "EnableResizableBar": "1",
84
+ "DmaRemapPeerMmio": "1"
85
+ },
86
+ "expected": {
87
+ "ForceP2P": "0x11",
88
+ "RMForceP2PType": "1",
89
+ "RMPcieP2PType": "2",
90
+ "GrdmaPciTopoCheckOverride": "1",
91
+ "EnableResizableBar": "1"
92
+ },
93
+ "missing": [],
94
+ "mismatched": {},
95
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
96
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
97
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
98
+ },
99
+ "p2pmark": {
100
+ "status": "not_run"
101
+ },
102
+ "amd_fabric": {
103
+ "status": "not_run"
104
+ },
105
+ "nvidia_smi_query": {
106
+ "cmd": [
107
+ "nvidia-smi",
108
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
109
+ "--format=csv,noheader,nounits"
110
+ ],
111
+ "returncode": 0,
112
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
113
+ "stderr": ""
114
+ },
115
+ "nvidia_smi_topo": {
116
+ "cmd": [
117
+ "nvidia-smi",
118
+ "topo",
119
+ "-m"
120
+ ],
121
+ "returncode": 0,
122
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
123
+ "stderr": ""
124
+ }
125
+ },
126
+ "nvidia_p2p_override": {
127
+ "effective": true,
128
+ "configured": true,
129
+ "params_path": "/proc/driver/nvidia/params",
130
+ "params_available": true,
131
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
132
+ "modprobe_available": true,
133
+ "runtime": {
134
+ "ForceP2P": "0x11",
135
+ "RMForceP2PType": "1",
136
+ "RMPcieP2PType": "2",
137
+ "GrdmaPciTopoCheckOverride": "1",
138
+ "EnableResizableBar": "1",
139
+ "DmaRemapPeerMmio": "1"
140
+ },
141
+ "expected": {
142
+ "ForceP2P": "0x11",
143
+ "RMForceP2PType": "1",
144
+ "RMPcieP2PType": "2",
145
+ "GrdmaPciTopoCheckOverride": "1",
146
+ "EnableResizableBar": "1"
147
+ },
148
+ "missing": [],
149
+ "mismatched": {},
150
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
151
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
152
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
153
+ },
154
+ "p2pmark": {
155
+ "status": "not_run"
156
+ },
157
+ "amd_fabric": {
158
+ "status": "not_run"
159
+ },
160
+ "hardware_run_summary": {
161
+ "samples": 112,
162
+ "duration_seconds": 267.276,
163
+ "gpu_count": 4,
164
+ "cpu_util_avg_pct": 10.31,
165
+ "cpu_temp_max_c": 77.25,
166
+ "gpu_util_avg_pct": 91.56,
167
+ "gpu_util_max_pct": 100.0,
168
+ "mem_util_avg_pct": 39.56,
169
+ "mem_util_max_pct": 55.0,
170
+ "temp_avg_c": 67.18,
171
+ "temp_max_c": 84.0,
172
+ "power_total_avg_w": 1101.05,
173
+ "power_total_max_w": 1178.9,
174
+ "power_limit_total_w": 1200.0,
175
+ "vram_used_avg_mb": 380174.0,
176
+ "vram_used_max_mb": 380174.0,
177
+ "vram_total_mb": 391548.0,
178
+ "vram_used_avg_pct": 97.1,
179
+ "vram_used_max_pct": 97.1,
180
+ "pcie_rx_avg_mb_s": 11487.79,
181
+ "pcie_rx_max_mb_s": 71064.0,
182
+ "pcie_tx_avg_mb_s": 11408.48,
183
+ "pcie_tx_max_mb_s": 66265.0
184
+ },
185
+ "event_log": [
186
+ "07:13:31 benchmark start engine=vllm",
187
+ "07:13:31 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
188
+ "07:13:31 startup decode concurrency=1,2,4 contexts=0,8k,32k",
189
+ "07:13:31 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
190
+ "07:13:31 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
191
+ "07:13:31 startup KV cache budget from vLLM metrics: 17,031,168 tokens (2079 blocks x 2048; local 4,257,792 \u00d7 CP 4; CP source: local process)",
192
+ "07:13:31 startup model context length: 1,000,000 tokens",
193
+ "07:13:31 startup prefill tests: skipped",
194
+ "07:13:31 startup calibrating padding text run=wwbkdgkvesfu up_to=32k",
195
+ "07:13:31 startup context 8k: 50,544 chars (8,192 prompt tokens via /tokenize)",
196
+ "07:13:31 startup context 32k: 205,140 chars (32,768 prompt tokens via /tokenize)",
197
+ "07:13:31 startup token targeting: /tokenize exact",
198
+ "07:13:31 startup startup preparation done",
199
+ "07:13:31 hardware monitor interval=2s",
200
+ "07:13:31 decode warmup start",
201
+ "07:13:31 decode warmup start C=1 ctx=32k 3s",
202
+ "07:13:31 cell start C=1 ctx=32k",
203
+ "07:13:39 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
204
+ "07:13:42 cell done C=1 ctx=32k 175.7 tok/s",
205
+ "07:13:42 decode warmup done C=1 ctx=32k",
206
+ "07:13:44 cell start C=1 ctx=0",
207
+ "07:13:50 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
208
+ "07:14:10 cell done C=1 ctx=0 167.2 tok/s",
209
+ "07:14:12 cell start C=1 ctx=8k",
210
+ "07:14:18 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
211
+ "07:14:38 cell done C=1 ctx=8k 173.0 tok/s",
212
+ "07:14:40 cell start C=1 ctx=32k",
213
+ "07:14:49 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
214
+ "07:15:09 cell done C=1 ctx=32k 186.4 tok/s",
215
+ "07:15:11 cell start C=2 ctx=0",
216
+ "07:15:17 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
217
+ "07:15:37 cell done C=2 ctx=0 244.6 tok/s",
218
+ "07:15:39 cell start C=4 ctx=0",
219
+ "07:15:44 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
220
+ "07:16:04 cell done C=4 ctx=0 292.9 tok/s",
221
+ "07:16:06 cell start C=2 ctx=8k",
222
+ "07:16:12 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
223
+ "07:16:32 cell done C=2 ctx=8k 246.0 tok/s",
224
+ "07:16:34 cell start C=4 ctx=8k",
225
+ "07:16:41 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
226
+ "07:17:01 cell done C=4 ctx=8k 298.4 tok/s",
227
+ "07:17:04 cell start C=2 ctx=32k",
228
+ "07:17:09 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
229
+ "07:17:29 cell done C=2 ctx=32k 252.6 tok/s",
230
+ "07:17:31 cell start C=4 ctx=32k",
231
+ "07:17:39 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
232
+ "07:17:59 cell done C=4 ctx=32k 297.7 tok/s"
233
+ ],
234
+ "prefill": {},
235
+ "results": [
236
+ {
237
+ "concurrency": 1,
238
+ "context_tokens": 0,
239
+ "benchmark_mode": "duration",
240
+ "request_count_target": 0,
241
+ "warmup_request_count": 0,
242
+ "measurement_seconds": 19.988585,
243
+ "measurement_wall_seconds": 20.00075,
244
+ "client_output_tokens": 3342,
245
+ "server_output_tokens": 3342,
246
+ "aggregate_source": "openai_continuous_usage",
247
+ "aggregate_tps": 167.19542416324822,
248
+ "per_request_avg_tps": 167.19542416324822,
249
+ "ttft_avg": 0.07083372515626252,
250
+ "ttft_p50": 0.07083372515626252,
251
+ "ttft_p90": 0.07083372515626252,
252
+ "ttft_p99": 0.07083372515626252,
253
+ "time_to_second_token_avg": 0.012936464976519346,
254
+ "time_to_second_token_p50": 0.012936464976519346,
255
+ "time_to_second_token_p90": 0.012936464976519346,
256
+ "time_to_second_token_p99": 0.012936464976519346,
257
+ "request_latency_avg": 0.0,
258
+ "request_latency_p50": 0.0,
259
+ "request_latency_p90": 0.0,
260
+ "request_latency_p99": 0.0,
261
+ "inter_token_latency_avg": 0.005911081440074571,
262
+ "inter_token_latency_p50": 0.005911081440074571,
263
+ "inter_token_latency_p90": 0.005911081440074571,
264
+ "inter_token_latency_p99": 0.005911081440074571,
265
+ "output_tps_per_user_avg": 169.17378150475705,
266
+ "output_tps_per_user_p50": 169.17378150475705,
267
+ "output_tps_per_user_p90": 169.17378150475705,
268
+ "output_tps_per_user_p99": 169.17378150475705,
269
+ "e2e_output_tps_per_user_avg": 0.0,
270
+ "e2e_output_tps_per_user_p50": 0.0,
271
+ "e2e_output_tps_per_user_p90": 0.0,
272
+ "e2e_output_tps_per_user_p99": 0.0,
273
+ "chunk_inter_token_latency_avg": 0.01407805126159353,
274
+ "chunk_inter_token_latency_p50": 0.01407805126159353,
275
+ "chunk_inter_token_latency_p90": 0.01407805126159353,
276
+ "chunk_inter_token_latency_p99": 0.01407805126159353,
277
+ "input_seq_len_avg": 78.0,
278
+ "output_seq_len_avg": 4307.0,
279
+ "output_seq_len_p50": 4307.0,
280
+ "output_seq_len_p90": 4307.0,
281
+ "output_seq_len_p99": 4307.0,
282
+ "request_count": 1,
283
+ "completed_request_count": 0,
284
+ "request_samples": [
285
+ {
286
+ "ttft": 0.07083372515626252,
287
+ "time_to_second_token": 0.012936464976519346,
288
+ "latency": 0.0,
289
+ "inter_token_latency_avg": 0.005911081440074571,
290
+ "chunk_inter_token_latency_avg": 0.01407805126159353,
291
+ "input_tokens": 78,
292
+ "output_tokens": 4307,
293
+ "output_tps_per_user": 169.17378150475705,
294
+ "e2e_output_tps_per_user": 0.0,
295
+ "completed": false
296
+ }
297
+ ],
298
+ "total_tokens": 3342,
299
+ "wall_time": 25.542422840138897,
300
+ "num_completed": 1,
301
+ "num_errors": 0,
302
+ "server_gen_throughput": 167.0490340328882,
303
+ "server_utilization": 0.0101058710298364,
304
+ "server_spec_accept_rate": 0.49765258215962443,
305
+ "server_spec_accept_length": 0.0,
306
+ "avg_running_reqs": 1,
307
+ "max_running_reqs": 1,
308
+ "effective_concurrency": 1,
309
+ "avg_queue_reqs": 0,
310
+ "max_queue_reqs": 0,
311
+ "queue_fraction": 0.0,
312
+ "underfilled": false,
313
+ "warmup_timed_out": false,
314
+ "warmup_duration": 5.535,
315
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
316
+ "timeout_reason": "",
317
+ "capacity_limited": false,
318
+ "hardware_summary": {
319
+ "samples": 9,
320
+ "duration_seconds": 19.305,
321
+ "gpu_count": 4,
322
+ "cpu_util_avg_pct": 10.88,
323
+ "cpu_temp_max_c": 76.12,
324
+ "gpu_util_avg_pct": 99.0,
325
+ "gpu_util_max_pct": 99.0,
326
+ "mem_util_avg_pct": 48.47,
327
+ "mem_util_max_pct": 53.0,
328
+ "temp_avg_c": 65.39,
329
+ "temp_max_c": 80.0,
330
+ "power_total_avg_w": 1150.89,
331
+ "power_total_max_w": 1153.46,
332
+ "power_limit_total_w": 1200.0,
333
+ "vram_used_avg_mb": 380174.0,
334
+ "vram_used_max_mb": 380174.0,
335
+ "vram_total_mb": 391548.0,
336
+ "vram_used_avg_pct": 97.1,
337
+ "vram_used_max_pct": 97.1,
338
+ "pcie_rx_avg_mb_s": 8406.89,
339
+ "pcie_rx_max_mb_s": 8711.0,
340
+ "pcie_tx_avg_mb_s": 8352.33,
341
+ "pcie_tx_max_mb_s": 8558.0
342
+ }
343
+ },
344
+ {
345
+ "concurrency": 1,
346
+ "context_tokens": 8192,
347
+ "benchmark_mode": "duration",
348
+ "request_count_target": 0,
349
+ "warmup_request_count": 0,
350
+ "measurement_seconds": 19.99916,
351
+ "measurement_wall_seconds": 20.000275,
352
+ "client_output_tokens": 3459,
353
+ "server_output_tokens": 3459,
354
+ "aggregate_source": "openai_continuous_usage",
355
+ "aggregate_tps": 172.95726809626717,
356
+ "per_request_avg_tps": 172.95726809626717,
357
+ "ttft_avg": 0.5792566961608827,
358
+ "ttft_p50": 0.5792566961608827,
359
+ "ttft_p90": 0.5792566961608827,
360
+ "ttft_p99": 0.5792566961608827,
361
+ "time_to_second_token_avg": 0.0174833619967103,
362
+ "time_to_second_token_p50": 0.0174833619967103,
363
+ "time_to_second_token_p90": 0.0174833619967103,
364
+ "time_to_second_token_p99": 0.0174833619967103,
365
+ "request_latency_avg": 0.0,
366
+ "request_latency_p50": 0.0,
367
+ "request_latency_p90": 0.0,
368
+ "request_latency_p99": 0.0,
369
+ "inter_token_latency_avg": 0.005649759039614473,
370
+ "inter_token_latency_p50": 0.005649759039614473,
371
+ "inter_token_latency_p90": 0.005649759039614473,
372
+ "inter_token_latency_p99": 0.005649759039614473,
373
+ "output_tps_per_user_avg": 176.9986990574801,
374
+ "output_tps_per_user_p50": 176.9986990574801,
375
+ "output_tps_per_user_p90": 176.9986990574801,
376
+ "output_tps_per_user_p99": 176.9986990574801,
377
+ "e2e_output_tps_per_user_avg": 0.0,
378
+ "e2e_output_tps_per_user_p50": 0.0,
379
+ "e2e_output_tps_per_user_p90": 0.0,
380
+ "e2e_output_tps_per_user_p99": 0.0,
381
+ "chunk_inter_token_latency_avg": 0.014258915671407954,
382
+ "chunk_inter_token_latency_p50": 0.014258915671407954,
383
+ "chunk_inter_token_latency_p90": 0.014258915671407954,
384
+ "chunk_inter_token_latency_p99": 0.014258915671407954,
385
+ "input_seq_len_avg": 8192.0,
386
+ "output_seq_len_avg": 4241.0,
387
+ "output_seq_len_p50": 4241.0,
388
+ "output_seq_len_p90": 4241.0,
389
+ "output_seq_len_p99": 4241.0,
390
+ "request_count": 1,
391
+ "completed_request_count": 0,
392
+ "request_samples": [
393
+ {
394
+ "ttft": 0.5792566961608827,
395
+ "time_to_second_token": 0.0174833619967103,
396
+ "latency": 0.0,
397
+ "inter_token_latency_avg": 0.005649759039614473,
398
+ "chunk_inter_token_latency_avg": 0.014258915671407954,
399
+ "input_tokens": 8192,
400
+ "output_tokens": 4241,
401
+ "output_tps_per_user": 176.9986990574801,
402
+ "e2e_output_tps_per_user": 0.0,
403
+ "completed": false
404
+ }
405
+ ],
406
+ "total_tokens": 3459,
407
+ "wall_time": 26.074295089114457,
408
+ "num_completed": 1,
409
+ "num_errors": 0,
410
+ "server_gen_throughput": 172.904696376223,
411
+ "server_utilization": 0.010587102983638075,
412
+ "server_spec_accept_rate": 0.539906103286385,
413
+ "server_spec_accept_length": 0.0,
414
+ "avg_running_reqs": 1,
415
+ "max_running_reqs": 1,
416
+ "effective_concurrency": 1,
417
+ "avg_queue_reqs": 0,
418
+ "max_queue_reqs": 0,
419
+ "queue_fraction": 0.0,
420
+ "underfilled": false,
421
+ "warmup_timed_out": false,
422
+ "warmup_duration": 6.06,
423
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
424
+ "timeout_reason": "",
425
+ "capacity_limited": false,
426
+ "hardware_summary": {
427
+ "samples": 8,
428
+ "duration_seconds": 16.905,
429
+ "gpu_count": 4,
430
+ "cpu_util_avg_pct": 10.94,
431
+ "cpu_temp_max_c": 75.25,
432
+ "gpu_util_avg_pct": 99.0,
433
+ "gpu_util_max_pct": 99.0,
434
+ "mem_util_avg_pct": 49.0,
435
+ "mem_util_max_pct": 55.0,
436
+ "temp_avg_c": 66.22,
437
+ "temp_max_c": 81.0,
438
+ "power_total_avg_w": 1153.53,
439
+ "power_total_max_w": 1154.66,
440
+ "power_limit_total_w": 1200.0,
441
+ "vram_used_avg_mb": 380174.0,
442
+ "vram_used_max_mb": 380174.0,
443
+ "vram_total_mb": 391548.0,
444
+ "vram_used_avg_pct": 97.1,
445
+ "vram_used_max_pct": 97.1,
446
+ "pcie_rx_avg_mb_s": 8274.25,
447
+ "pcie_rx_max_mb_s": 8601.0,
448
+ "pcie_tx_avg_mb_s": 8292.38,
449
+ "pcie_tx_max_mb_s": 8476.0
450
+ }
451
+ },
452
+ {
453
+ "concurrency": 1,
454
+ "context_tokens": 32768,
455
+ "benchmark_mode": "duration",
456
+ "request_count_target": 0,
457
+ "warmup_request_count": 0,
458
+ "measurement_seconds": 19.999218,
459
+ "measurement_wall_seconds": 20.000322,
460
+ "client_output_tokens": 3727,
461
+ "server_output_tokens": 3727,
462
+ "aggregate_source": "openai_continuous_usage",
463
+ "aggregate_tps": 186.35728993398365,
464
+ "per_request_avg_tps": 186.35728993398365,
465
+ "ttft_avg": 0.5975865018554032,
466
+ "ttft_p50": 0.5975865018554032,
467
+ "ttft_p90": 0.5975865018554032,
468
+ "ttft_p99": 0.5975865018554032,
469
+ "time_to_second_token_avg": 0.011153812054544687,
470
+ "time_to_second_token_p50": 0.011153812054544687,
471
+ "time_to_second_token_p90": 0.011153812054544687,
472
+ "time_to_second_token_p99": 0.011153812054544687,
473
+ "request_latency_avg": 0.0,
474
+ "request_latency_p50": 0.0,
475
+ "request_latency_p90": 0.0,
476
+ "request_latency_p99": 0.0,
477
+ "inter_token_latency_avg": 0.005334254001541856,
478
+ "inter_token_latency_p50": 0.005334254001541856,
479
+ "inter_token_latency_p90": 0.005334254001541856,
480
+ "inter_token_latency_p99": 0.005334254001541856,
481
+ "output_tps_per_user_avg": 187.4676383447342,
482
+ "output_tps_per_user_p50": 187.4676383447342,
483
+ "output_tps_per_user_p90": 187.4676383447342,
484
+ "output_tps_per_user_p99": 187.4676383447342,
485
+ "e2e_output_tps_per_user_avg": 0.0,
486
+ "e2e_output_tps_per_user_p50": 0.0,
487
+ "e2e_output_tps_per_user_p90": 0.0,
488
+ "e2e_output_tps_per_user_p99": 0.0,
489
+ "chunk_inter_token_latency_avg": 0.014272389806152839,
490
+ "chunk_inter_token_latency_p50": 0.014272389806152839,
491
+ "chunk_inter_token_latency_p90": 0.014272389806152839,
492
+ "chunk_inter_token_latency_p99": 0.014272389806152839,
493
+ "input_seq_len_avg": 32768.0,
494
+ "output_seq_len_avg": 4488.0,
495
+ "output_seq_len_p50": 4488.0,
496
+ "output_seq_len_p90": 4488.0,
497
+ "output_seq_len_p99": 4488.0,
498
+ "request_count": 1,
499
+ "completed_request_count": 0,
500
+ "request_samples": [
501
+ {
502
+ "ttft": 0.5975865018554032,
503
+ "time_to_second_token": 0.011153812054544687,
504
+ "latency": 0.0,
505
+ "inter_token_latency_avg": 0.005334254001541856,
506
+ "chunk_inter_token_latency_avg": 0.014272389806152839,
507
+ "input_tokens": 32768,
508
+ "output_tokens": 4488,
509
+ "output_tps_per_user": 187.4676383447342,
510
+ "e2e_output_tps_per_user": 0.0,
511
+ "completed": false
512
+ }
513
+ ],
514
+ "total_tokens": 3727,
515
+ "wall_time": 29.12444451614283,
516
+ "num_completed": 1,
517
+ "num_errors": 0,
518
+ "server_gen_throughput": 186.30161122127362,
519
+ "server_utilization": 0.012030798845043322,
520
+ "server_spec_accept_rate": 0.4225352112676056,
521
+ "server_spec_accept_length": 0.0,
522
+ "avg_running_reqs": 1,
523
+ "max_running_reqs": 1,
524
+ "effective_concurrency": 1,
525
+ "avg_queue_reqs": 0,
526
+ "max_queue_reqs": 0,
527
+ "queue_fraction": 0.0,
528
+ "underfilled": false,
529
+ "warmup_timed_out": false,
530
+ "warmup_duration": 9.11,
531
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
532
+ "timeout_reason": "",
533
+ "capacity_limited": false,
534
+ "hardware_summary": {
535
+ "samples": 8,
536
+ "duration_seconds": 16.884,
537
+ "gpu_count": 4,
538
+ "cpu_util_avg_pct": 10.89,
539
+ "cpu_temp_max_c": 76.12,
540
+ "gpu_util_avg_pct": 99.03,
541
+ "gpu_util_max_pct": 100.0,
542
+ "mem_util_avg_pct": 48.38,
543
+ "mem_util_max_pct": 54.0,
544
+ "temp_avg_c": 67.03,
545
+ "temp_max_c": 82.0,
546
+ "power_total_avg_w": 1154.29,
547
+ "power_total_max_w": 1155.42,
548
+ "power_limit_total_w": 1200.0,
549
+ "vram_used_avg_mb": 380174.0,
550
+ "vram_used_max_mb": 380174.0,
551
+ "vram_total_mb": 391548.0,
552
+ "vram_used_avg_pct": 97.1,
553
+ "vram_used_max_pct": 97.1,
554
+ "pcie_rx_avg_mb_s": 8367.62,
555
+ "pcie_rx_max_mb_s": 8719.0,
556
+ "pcie_tx_avg_mb_s": 8179.25,
557
+ "pcie_tx_max_mb_s": 8438.0
558
+ }
559
+ },
560
+ {
561
+ "concurrency": 2,
562
+ "context_tokens": 0,
563
+ "benchmark_mode": "duration",
564
+ "request_count_target": 0,
565
+ "warmup_request_count": 0,
566
+ "measurement_seconds": 19.996612,
567
+ "measurement_wall_seconds": 20.000729,
568
+ "client_output_tokens": 4892,
569
+ "server_output_tokens": 4892,
570
+ "aggregate_source": "openai_continuous_usage",
571
+ "aggregate_tps": 244.64143701284553,
572
+ "per_request_avg_tps": 122.32071850642276,
573
+ "ttft_avg": 0.11364343203604221,
574
+ "ttft_p50": 0.11364343203604221,
575
+ "ttft_p90": 0.14825461693108083,
576
+ "ttft_p99": 0.1560421335324645,
577
+ "time_to_second_token_avg": 0.015344579587690532,
578
+ "time_to_second_token_p50": 0.015344579587690532,
579
+ "time_to_second_token_p90": 0.016974025568924845,
580
+ "time_to_second_token_p99": 0.017340650914702563,
581
+ "request_latency_avg": 0.0,
582
+ "request_latency_p50": 0.0,
583
+ "request_latency_p90": 0.0,
584
+ "request_latency_p99": 0.0,
585
+ "inter_token_latency_avg": 0.008021675154484334,
586
+ "inter_token_latency_p50": 0.008021675154484334,
587
+ "inter_token_latency_p90": 0.008142962393239039,
588
+ "inter_token_latency_p99": 0.008170252021958846,
589
+ "output_tps_per_user_avg": 124.70678698569867,
590
+ "output_tps_per_user_p50": 124.70678698569867,
591
+ "output_tps_per_user_p90": 126.59234599378327,
592
+ "output_tps_per_user_p99": 127.0165967706023,
593
+ "e2e_output_tps_per_user_avg": 0.0,
594
+ "e2e_output_tps_per_user_p50": 0.0,
595
+ "e2e_output_tps_per_user_p90": 0.0,
596
+ "e2e_output_tps_per_user_p99": 0.0,
597
+ "chunk_inter_token_latency_avg": 0.020790530252109002,
598
+ "chunk_inter_token_latency_p50": 0.020790530252109002,
599
+ "chunk_inter_token_latency_p90": 0.02081209152359912,
600
+ "chunk_inter_token_latency_p99": 0.020816942809684397,
601
+ "input_seq_len_avg": 78.0,
602
+ "output_seq_len_avg": 3170.5,
603
+ "output_seq_len_p50": 3170.5,
604
+ "output_seq_len_p90": 3214.1,
605
+ "output_seq_len_p99": 3223.91,
606
+ "request_count": 2,
607
+ "completed_request_count": 0,
608
+ "request_samples": [
609
+ {
610
+ "ttft": 0.07037945091724396,
611
+ "time_to_second_token": 0.013307772111147642,
612
+ "latency": 0.0,
613
+ "inter_token_latency_avg": 0.008173284202927714,
614
+ "chunk_inter_token_latency_avg": 0.02081748184147165,
615
+ "input_tokens": 78,
616
+ "output_tokens": 3116,
617
+ "output_tps_per_user": 122.34983822559292,
618
+ "e2e_output_tps_per_user": 0.0,
619
+ "completed": false
620
+ },
621
+ {
622
+ "ttft": 0.15690741315484047,
623
+ "time_to_second_token": 0.017381387064233422,
624
+ "latency": 0.0,
625
+ "inter_token_latency_avg": 0.007870066106040956,
626
+ "chunk_inter_token_latency_avg": 0.02076357866274635,
627
+ "input_tokens": 78,
628
+ "output_tokens": 3225,
629
+ "output_tps_per_user": 127.06373574580442,
630
+ "e2e_output_tps_per_user": 0.0,
631
+ "completed": false
632
+ }
633
+ ],
634
+ "total_tokens": 4892,
635
+ "wall_time": 25.551810663193464,
636
+ "num_completed": 2,
637
+ "num_errors": 0,
638
+ "server_gen_throughput": 244.53218903662054,
639
+ "server_utilization": 0.0202117420596728,
640
+ "server_spec_accept_rate": 0.5034013605442177,
641
+ "server_spec_accept_length": 0.0,
642
+ "avg_running_reqs": 2,
643
+ "max_running_reqs": 2,
644
+ "effective_concurrency": 2,
645
+ "avg_queue_reqs": 0,
646
+ "max_queue_reqs": 0,
647
+ "queue_fraction": 0.0,
648
+ "underfilled": false,
649
+ "warmup_timed_out": false,
650
+ "warmup_duration": 5.534,
651
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
652
+ "timeout_reason": "",
653
+ "capacity_limited": false,
654
+ "hardware_summary": {
655
+ "samples": 9,
656
+ "duration_seconds": 19.285,
657
+ "gpu_count": 4,
658
+ "cpu_util_avg_pct": 10.84,
659
+ "cpu_temp_max_c": 76.12,
660
+ "gpu_util_avg_pct": 100.0,
661
+ "gpu_util_max_pct": 100.0,
662
+ "mem_util_avg_pct": 45.14,
663
+ "mem_util_max_pct": 49.0,
664
+ "temp_avg_c": 67.78,
665
+ "temp_max_c": 83.0,
666
+ "power_total_avg_w": 1176.88,
667
+ "power_total_max_w": 1178.9,
668
+ "power_limit_total_w": 1200.0,
669
+ "vram_used_avg_mb": 380174.0,
670
+ "vram_used_max_mb": 380174.0,
671
+ "vram_total_mb": 391548.0,
672
+ "vram_used_avg_pct": 97.1,
673
+ "vram_used_max_pct": 97.1,
674
+ "pcie_rx_avg_mb_s": 11281.22,
675
+ "pcie_rx_max_mb_s": 11385.0,
676
+ "pcie_tx_avg_mb_s": 11144.0,
677
+ "pcie_tx_max_mb_s": 11241.0
678
+ }
679
+ },
680
+ {
681
+ "concurrency": 4,
682
+ "context_tokens": 0,
683
+ "benchmark_mode": "duration",
684
+ "request_count_target": 0,
685
+ "warmup_request_count": 0,
686
+ "measurement_seconds": 19.968882,
687
+ "measurement_wall_seconds": 20.001101,
688
+ "client_output_tokens": 5848,
689
+ "server_output_tokens": 5848,
690
+ "aggregate_source": "openai_continuous_usage",
691
+ "aggregate_tps": 292.85565306969767,
692
+ "per_request_avg_tps": 73.21391326742442,
693
+ "ttft_avg": 0.1554050333215855,
694
+ "ttft_p50": 0.18278046406339854,
695
+ "ttft_p90": 0.18279754216782748,
696
+ "ttft_p99": 0.18280153354629874,
697
+ "time_to_second_token_avg": 0.026175830571446568,
698
+ "time_to_second_token_p50": 0.030380802578292787,
699
+ "time_to_second_token_p90": 0.030411765072494747,
700
+ "time_to_second_token_p99": 0.030420526592060924,
701
+ "request_latency_avg": 0.0,
702
+ "request_latency_p50": 0.0,
703
+ "request_latency_p90": 0.0,
704
+ "request_latency_p99": 0.0,
705
+ "inter_token_latency_avg": 0.013267129396052025,
706
+ "inter_token_latency_p50": 0.013429554256049785,
707
+ "inter_token_latency_p90": 0.013534146595532138,
708
+ "inter_token_latency_p99": 0.013559169974355398,
709
+ "output_tps_per_user_avg": 75.43238689615268,
710
+ "output_tps_per_user_p50": 74.46328652304663,
711
+ "output_tps_per_user_p90": 77.75213899442375,
712
+ "output_tps_per_user_p99": 78.93575450758087,
713
+ "e2e_output_tps_per_user_avg": 0.0,
714
+ "e2e_output_tps_per_user_p50": 0.0,
715
+ "e2e_output_tps_per_user_p90": 0.0,
716
+ "e2e_output_tps_per_user_p99": 0.0,
717
+ "chunk_inter_token_latency_avg": 0.034184321112551014,
718
+ "chunk_inter_token_latency_p50": 0.03419339492657416,
719
+ "chunk_inter_token_latency_p90": 0.034255501869857555,
720
+ "chunk_inter_token_latency_p99": 0.03427054421702008,
721
+ "input_seq_len_avg": 78.0,
722
+ "output_seq_len_avg": 1913.0,
723
+ "output_seq_len_p50": 1890.5,
724
+ "output_seq_len_p90": 1969.7,
725
+ "output_seq_len_p99": 1999.67,
726
+ "request_count": 4,
727
+ "completed_request_count": 0,
728
+ "request_samples": [
729
+ {
730
+ "ttft": 0.0732572281267494,
731
+ "time_to_second_token": 0.013520217034965754,
732
+ "latency": 0.0,
733
+ "inter_token_latency_avg": 0.013469271168953313,
734
+ "chunk_inter_token_latency_avg": 0.03427221558892703,
735
+ "input_tokens": 78,
736
+ "output_tokens": 1889,
737
+ "output_tps_per_user": 74.24306686355838,
738
+ "e2e_output_tps_per_user": 0.0,
739
+ "completed": false
740
+ },
741
+ {
742
+ "ttft": 0.18280197703279555,
743
+ "time_to_second_token": 0.030421500094234943,
744
+ "latency": 0.0,
745
+ "inter_token_latency_avg": 0.012647458722328322,
746
+ "chunk_inter_token_latency_avg": 0.034216503192028784,
747
+ "input_tokens": 78,
748
+ "output_tokens": 2003,
749
+ "output_tps_per_user": 79.06726734237611,
750
+ "e2e_output_tps_per_user": 0.0,
751
+ "completed": false
752
+ },
753
+ {
754
+ "ttft": 0.18278719414956868,
755
+ "time_to_second_token": 0.030389050021767616,
756
+ "latency": 0.0,
757
+ "inter_token_latency_avg": 0.01338983734314626,
758
+ "chunk_inter_token_latency_avg": 0.034170286661119535,
759
+ "input_tokens": 78,
760
+ "output_tokens": 1892,
761
+ "output_tps_per_user": 74.68350618253487,
762
+ "e2e_output_tps_per_user": 0.0,
763
+ "completed": false
764
+ },
765
+ {
766
+ "ttft": 0.1827737339772284,
767
+ "time_to_second_token": 0.030372555134817958,
768
+ "latency": 0.0,
769
+ "inter_token_latency_avg": 0.013561950349780205,
770
+ "chunk_inter_token_latency_avg": 0.03407827900812872,
771
+ "input_tokens": 78,
772
+ "output_tokens": 1868,
773
+ "output_tps_per_user": 73.73570719614136,
774
+ "e2e_output_tps_per_user": 0.0,
775
+ "completed": false
776
+ }
777
+ ],
778
+ "total_tokens": 5848,
779
+ "wall_time": 25.541023290948942,
780
+ "num_completed": 4,
781
+ "num_errors": 0,
782
+ "server_gen_throughput": 292.3212847040937,
783
+ "server_utilization": 0.04042348411934549,
784
+ "server_spec_accept_rate": 0.49712643678160917,
785
+ "server_spec_accept_length": 0.0,
786
+ "avg_running_reqs": 4,
787
+ "max_running_reqs": 4,
788
+ "effective_concurrency": 4,
789
+ "avg_queue_reqs": 0,
790
+ "max_queue_reqs": 0,
791
+ "queue_fraction": 0.0,
792
+ "underfilled": false,
793
+ "warmup_timed_out": false,
794
+ "warmup_duration": 5.535,
795
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
796
+ "timeout_reason": "",
797
+ "capacity_limited": false,
798
+ "hardware_summary": {
799
+ "samples": 8,
800
+ "duration_seconds": 16.833,
801
+ "gpu_count": 4,
802
+ "cpu_util_avg_pct": 10.81,
803
+ "cpu_temp_max_c": 77.0,
804
+ "gpu_util_avg_pct": 100.0,
805
+ "gpu_util_max_pct": 100.0,
806
+ "mem_util_avg_pct": 39.75,
807
+ "mem_util_max_pct": 43.0,
808
+ "temp_avg_c": 68.5,
809
+ "temp_max_c": 84.0,
810
+ "power_total_avg_w": 1172.95,
811
+ "power_total_max_w": 1173.93,
812
+ "power_limit_total_w": 1200.0,
813
+ "vram_used_avg_mb": 380174.0,
814
+ "vram_used_max_mb": 380174.0,
815
+ "vram_total_mb": 391548.0,
816
+ "vram_used_avg_pct": 97.1,
817
+ "vram_used_max_pct": 97.1,
818
+ "pcie_rx_avg_mb_s": 7881.25,
819
+ "pcie_rx_max_mb_s": 8056.0,
820
+ "pcie_tx_avg_mb_s": 7872.12,
821
+ "pcie_tx_max_mb_s": 8191.0
822
+ }
823
+ },
824
+ {
825
+ "concurrency": 2,
826
+ "context_tokens": 8192,
827
+ "benchmark_mode": "duration",
828
+ "request_count_target": 0,
829
+ "warmup_request_count": 0,
830
+ "measurement_seconds": 19.997371,
831
+ "measurement_wall_seconds": 20.000506,
832
+ "client_output_tokens": 4919,
833
+ "server_output_tokens": 4919,
834
+ "aggregate_source": "openai_continuous_usage",
835
+ "aggregate_tps": 245.98233342981337,
836
+ "per_request_avg_tps": 122.99116671490668,
837
+ "ttft_avg": 0.9427531745750457,
838
+ "ttft_p50": 0.9427531745750457,
839
+ "ttft_p90": 1.225604160549119,
840
+ "ttft_p99": 1.2892456323932855,
841
+ "time_to_second_token_avg": 0.022013463429175317,
842
+ "time_to_second_token_p50": 0.022013463429175317,
843
+ "time_to_second_token_p90": 0.025811466271989048,
844
+ "time_to_second_token_p99": 0.026666016911622136,
845
+ "request_latency_avg": 0.0,
846
+ "request_latency_p50": 0.0,
847
+ "request_latency_p90": 0.0,
848
+ "request_latency_p99": 0.0,
849
+ "inter_token_latency_avg": 0.008114433301139572,
850
+ "inter_token_latency_p50": 0.008114433301139572,
851
+ "inter_token_latency_p90": 0.00823075733708024,
852
+ "inter_token_latency_p99": 0.008256930245166891,
853
+ "output_tps_per_user_avg": 123.27677949684329,
854
+ "output_tps_per_user_p50": 123.27677949684329,
855
+ "output_tps_per_user_p90": 125.04400734833438,
856
+ "output_tps_per_user_p99": 125.44163361491988,
857
+ "e2e_output_tps_per_user_avg": 0.0,
858
+ "e2e_output_tps_per_user_p50": 0.0,
859
+ "e2e_output_tps_per_user_p90": 0.0,
860
+ "e2e_output_tps_per_user_p99": 0.0,
861
+ "chunk_inter_token_latency_avg": 0.021151300976615273,
862
+ "chunk_inter_token_latency_p50": 0.021151300976615273,
863
+ "chunk_inter_token_latency_p90": 0.021420214613202804,
864
+ "chunk_inter_token_latency_p99": 0.021480720181434997,
865
+ "input_seq_len_avg": 8192.0,
866
+ "output_seq_len_avg": 2907.5,
867
+ "output_seq_len_p50": 2907.5,
868
+ "output_seq_len_p90": 2914.3,
869
+ "output_seq_len_p99": 2915.83,
870
+ "request_count": 2,
871
+ "completed_request_count": 0,
872
+ "request_samples": [
873
+ {
874
+ "ttft": 0.5891894421074539,
875
+ "time_to_second_token": 0.017265959875658154,
876
+ "latency": 0.0,
877
+ "inter_token_latency_avg": 0.008259838346065407,
878
+ "chunk_inter_token_latency_avg": 0.021487443022349687,
879
+ "input_tokens": 8192,
880
+ "output_tokens": 2899,
881
+ "output_tps_per_user": 121.06774468247944,
882
+ "e2e_output_tps_per_user": 0.0,
883
+ "completed": false
884
+ },
885
+ {
886
+ "ttft": 1.2963169070426375,
887
+ "time_to_second_token": 0.02676096698269248,
888
+ "latency": 0.0,
889
+ "inter_token_latency_avg": 0.007969028256213737,
890
+ "chunk_inter_token_latency_avg": 0.020815158930880862,
891
+ "input_tokens": 8192,
892
+ "output_tokens": 2916,
893
+ "output_tps_per_user": 125.48581431120715,
894
+ "e2e_output_tps_per_user": 0.0,
895
+ "completed": false
896
+ }
897
+ ],
898
+ "total_tokens": 4919,
899
+ "wall_time": 25.566069599008188,
900
+ "num_completed": 2,
901
+ "num_errors": 0,
902
+ "server_gen_throughput": 245.8762565579269,
903
+ "server_utilization": 0.02117420596727626,
904
+ "server_spec_accept_rate": 0.5510204081632653,
905
+ "server_spec_accept_length": 0.0,
906
+ "avg_running_reqs": 2,
907
+ "max_running_reqs": 2,
908
+ "effective_concurrency": 2,
909
+ "avg_queue_reqs": 0,
910
+ "max_queue_reqs": 0,
911
+ "queue_fraction": 0.0,
912
+ "underfilled": false,
913
+ "warmup_timed_out": false,
914
+ "warmup_duration": 5.547,
915
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
916
+ "timeout_reason": "",
917
+ "capacity_limited": false,
918
+ "hardware_summary": {
919
+ "samples": 8,
920
+ "duration_seconds": 16.882,
921
+ "gpu_count": 4,
922
+ "cpu_util_avg_pct": 10.89,
923
+ "cpu_temp_max_c": 75.62,
924
+ "gpu_util_avg_pct": 100.0,
925
+ "gpu_util_max_pct": 100.0,
926
+ "mem_util_avg_pct": 45.12,
927
+ "mem_util_max_pct": 49.0,
928
+ "temp_avg_c": 68.28,
929
+ "temp_max_c": 84.0,
930
+ "power_total_avg_w": 1178.0,
931
+ "power_total_max_w": 1178.89,
932
+ "power_limit_total_w": 1200.0,
933
+ "vram_used_avg_mb": 380174.0,
934
+ "vram_used_max_mb": 380174.0,
935
+ "vram_total_mb": 391548.0,
936
+ "vram_used_avg_pct": 97.1,
937
+ "vram_used_max_pct": 97.1,
938
+ "pcie_rx_avg_mb_s": 11233.75,
939
+ "pcie_rx_max_mb_s": 11359.0,
940
+ "pcie_tx_avg_mb_s": 11151.88,
941
+ "pcie_tx_max_mb_s": 11299.0
942
+ }
943
+ },
944
+ {
945
+ "concurrency": 4,
946
+ "context_tokens": 8192,
947
+ "benchmark_mode": "duration",
948
+ "request_count_target": 0,
949
+ "warmup_request_count": 0,
950
+ "measurement_seconds": 19.996605,
951
+ "measurement_wall_seconds": 20.000706,
952
+ "client_output_tokens": 5967,
953
+ "server_output_tokens": 5967,
954
+ "aggregate_source": "openai_continuous_usage",
955
+ "aggregate_tps": 298.40065698894307,
956
+ "per_request_avg_tps": 74.60016424723577,
957
+ "ttft_avg": 2.7296009025303647,
958
+ "ttft_p50": 3.442606173455715,
959
+ "ttft_p90": 3.4429665624164043,
960
+ "ttft_p99": 3.443046331773512,
961
+ "time_to_second_token_avg": 0.04401593271177262,
962
+ "time_to_second_token_p50": 0.05272976343985647,
963
+ "time_to_second_token_p90": 0.05277097581420094,
964
+ "time_to_second_token_p99": 0.052783893730957064,
965
+ "request_latency_avg": 0.0,
966
+ "request_latency_p50": 0.0,
967
+ "request_latency_p90": 0.0,
968
+ "request_latency_p99": 0.0,
969
+ "inter_token_latency_avg": 0.013244349253168272,
970
+ "inter_token_latency_p50": 0.01323177277428084,
971
+ "inter_token_latency_p90": 0.013676198067583552,
972
+ "inter_token_latency_p99": 0.013812555348605936,
973
+ "output_tps_per_user_avg": 75.57579651538197,
974
+ "output_tps_per_user_p50": 75.57923042394843,
975
+ "output_tps_per_user_p90": 78.00785040565367,
976
+ "output_tps_per_user_p99": 78.74432054765556,
977
+ "e2e_output_tps_per_user_avg": 0.0,
978
+ "e2e_output_tps_per_user_p50": 0.0,
979
+ "e2e_output_tps_per_user_p90": 0.0,
980
+ "e2e_output_tps_per_user_p99": 0.0,
981
+ "chunk_inter_token_latency_avg": 0.03463163790764121,
982
+ "chunk_inter_token_latency_p50": 0.03453151627338626,
983
+ "chunk_inter_token_latency_p90": 0.03481200221493731,
984
+ "chunk_inter_token_latency_p99": 0.03492015582317328,
985
+ "input_seq_len_avg": 8192.0,
986
+ "output_seq_len_avg": 1798.5,
987
+ "output_seq_len_p50": 1790.5,
988
+ "output_seq_len_p90": 1861.2,
989
+ "output_seq_len_p99": 1876.32,
990
+ "request_count": 4,
991
+ "completed_request_count": 0,
992
+ "request_samples": [
993
+ {
994
+ "ttft": 0.5901360681746155,
995
+ "time_to_second_token": 0.01781887491233647,
996
+ "latency": 0.0,
997
+ "inter_token_latency_avg": 0.013827706157608423,
998
+ "chunk_inter_token_latency_avg": 0.03493217289075506,
999
+ "input_tokens": 8192,
1000
+ "output_tokens": 1878,
1001
+ "output_tps_per_user": 72.31857465019748,
1002
+ "e2e_output_tps_per_user": 0.0,
1003
+ "completed": false
1004
+ },
1005
+ {
1006
+ "ttft": 3.443055195035413,
1007
+ "time_to_second_token": 0.052722041960805655,
1008
+ "latency": 0.0,
1009
+ "inter_token_latency_avg": 0.012686145306502984,
1010
+ "chunk_inter_token_latency_avg": 0.03453134619303727,
1011
+ "input_tokens": 8192,
1012
+ "output_tokens": 1822,
1013
+ "output_tps_per_user": 78.82615056343354,
1014
+ "e2e_output_tps_per_user": 0.0,
1015
+ "completed": false
1016
+ },
1017
+ {
1018
+ "ttft": 3.4427597529720515,
1019
+ "time_to_second_token": 0.052737484918907285,
1020
+ "latency": 0.0,
1021
+ "inter_token_latency_avg": 0.013322679190858855,
1022
+ "chunk_inter_token_latency_avg": 0.034531428575409945,
1023
+ "input_tokens": 8192,
1024
+ "output_tokens": 1735,
1025
+ "output_tps_per_user": 75.05997747706289,
1026
+ "e2e_output_tps_per_user": 0.0,
1027
+ "completed": false
1028
+ },
1029
+ {
1030
+ "ttft": 3.442452593939379,
1031
+ "time_to_second_token": 0.052785329055041075,
1032
+ "latency": 0.0,
1033
+ "inter_token_latency_avg": 0.013140866357702825,
1034
+ "chunk_inter_token_latency_avg": 0.03453160397136258,
1035
+ "input_tokens": 8192,
1036
+ "output_tokens": 1759,
1037
+ "output_tps_per_user": 76.09848337083397,
1038
+ "e2e_output_tps_per_user": 0.0,
1039
+ "completed": false
1040
+ }
1041
+ ],
1042
+ "total_tokens": 5967,
1043
+ "wall_time": 27.59907579794526,
1044
+ "num_completed": 4,
1045
+ "num_errors": 0,
1046
+ "server_gen_throughput": 298.2659723183749,
1047
+ "server_utilization": 0.04234841193455241,
1048
+ "server_spec_accept_rate": 0.4511494252873563,
1049
+ "server_spec_accept_length": 0.0,
1050
+ "avg_running_reqs": 4,
1051
+ "max_running_reqs": 4,
1052
+ "effective_concurrency": 4,
1053
+ "avg_queue_reqs": 0,
1054
+ "max_queue_reqs": 0,
1055
+ "queue_fraction": 0.0,
1056
+ "underfilled": false,
1057
+ "warmup_timed_out": false,
1058
+ "warmup_duration": 7.567,
1059
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1060
+ "timeout_reason": "",
1061
+ "capacity_limited": false,
1062
+ "hardware_summary": {
1063
+ "samples": 9,
1064
+ "duration_seconds": 19.211,
1065
+ "gpu_count": 4,
1066
+ "cpu_util_avg_pct": 10.83,
1067
+ "cpu_temp_max_c": 77.0,
1068
+ "gpu_util_avg_pct": 100.0,
1069
+ "gpu_util_max_pct": 100.0,
1070
+ "mem_util_avg_pct": 39.47,
1071
+ "mem_util_max_pct": 43.0,
1072
+ "temp_avg_c": 68.94,
1073
+ "temp_max_c": 84.0,
1074
+ "power_total_avg_w": 1173.61,
1075
+ "power_total_max_w": 1174.17,
1076
+ "power_limit_total_w": 1200.0,
1077
+ "vram_used_avg_mb": 380174.0,
1078
+ "vram_used_max_mb": 380174.0,
1079
+ "vram_total_mb": 391548.0,
1080
+ "vram_used_avg_pct": 97.1,
1081
+ "vram_used_max_pct": 97.1,
1082
+ "pcie_rx_avg_mb_s": 7879.33,
1083
+ "pcie_rx_max_mb_s": 8354.0,
1084
+ "pcie_tx_avg_mb_s": 7779.89,
1085
+ "pcie_tx_max_mb_s": 8190.0
1086
+ }
1087
+ },
1088
+ {
1089
+ "concurrency": 2,
1090
+ "context_tokens": 32768,
1091
+ "benchmark_mode": "duration",
1092
+ "request_count_target": 0,
1093
+ "warmup_request_count": 0,
1094
+ "measurement_seconds": 19.988436,
1095
+ "measurement_wall_seconds": 20.000537,
1096
+ "client_output_tokens": 5050,
1097
+ "server_output_tokens": 5050,
1098
+ "aggregate_source": "openai_continuous_usage",
1099
+ "aggregate_tps": 252.6460844518971,
1100
+ "per_request_avg_tps": 126.32304222594856,
1101
+ "ttft_avg": 0.951587092014961,
1102
+ "ttft_p50": 0.951587092014961,
1103
+ "ttft_p90": 1.232194907194935,
1104
+ "ttft_p99": 1.295331665610429,
1105
+ "time_to_second_token_avg": 0.0161319924518466,
1106
+ "time_to_second_token_p50": 0.0161319924518466,
1107
+ "time_to_second_token_p90": 0.019481185637414456,
1108
+ "time_to_second_token_p99": 0.020234754104167224,
1109
+ "request_latency_avg": 0.0,
1110
+ "request_latency_p50": 0.0,
1111
+ "request_latency_p90": 0.0,
1112
+ "request_latency_p99": 0.0,
1113
+ "inter_token_latency_avg": 0.007939165504351434,
1114
+ "inter_token_latency_p50": 0.007939165504351434,
1115
+ "inter_token_latency_p90": 0.008045622734254517,
1116
+ "inter_token_latency_p99": 0.008069575610982711,
1117
+ "output_tps_per_user_avg": 125.99321968668501,
1118
+ "output_tps_per_user_p50": 125.99321968668501,
1119
+ "output_tps_per_user_p90": 127.68267799903076,
1120
+ "output_tps_per_user_p99": 128.06280611930853,
1121
+ "e2e_output_tps_per_user_avg": 0.0,
1122
+ "e2e_output_tps_per_user_p50": 0.0,
1123
+ "e2e_output_tps_per_user_p90": 0.0,
1124
+ "e2e_output_tps_per_user_p99": 0.0,
1125
+ "chunk_inter_token_latency_avg": 0.02123036904289779,
1126
+ "chunk_inter_token_latency_p50": 0.02123036904289779,
1127
+ "chunk_inter_token_latency_p90": 0.021406878769873895,
1128
+ "chunk_inter_token_latency_p99": 0.02144659345844352,
1129
+ "input_seq_len_avg": 32768.0,
1130
+ "output_seq_len_avg": 2971.0,
1131
+ "output_seq_len_p50": 2971.0,
1132
+ "output_seq_len_p90": 3046.2,
1133
+ "output_seq_len_p99": 3063.12,
1134
+ "request_count": 2,
1135
+ "completed_request_count": 0,
1136
+ "request_samples": [
1137
+ {
1138
+ "ttft": 0.6008273230399936,
1139
+ "time_to_second_token": 0.01194550096988678,
1140
+ "latency": 0.0,
1141
+ "inter_token_latency_avg": 0.007806093966972579,
1142
+ "chunk_inter_token_latency_avg": 0.021451006201617922,
1143
+ "input_tokens": 32768,
1144
+ "output_tokens": 3065,
1145
+ "output_tps_per_user": 128.1050425771172,
1146
+ "e2e_output_tps_per_user": 0.0,
1147
+ "completed": false
1148
+ },
1149
+ {
1150
+ "ttft": 1.3023468609899282,
1151
+ "time_to_second_token": 0.02031848393380642,
1152
+ "latency": 0.0,
1153
+ "inter_token_latency_avg": 0.008072237041730289,
1154
+ "chunk_inter_token_latency_avg": 0.021009731884177655,
1155
+ "input_tokens": 32768,
1156
+ "output_tokens": 2877,
1157
+ "output_tps_per_user": 123.88139679625283,
1158
+ "e2e_output_tps_per_user": 0.0,
1159
+ "completed": false
1160
+ }
1161
+ ],
1162
+ "total_tokens": 5050,
1163
+ "wall_time": 25.579677499830723,
1164
+ "num_completed": 2,
1165
+ "num_errors": 0,
1166
+ "server_gen_throughput": 252.43712151026958,
1167
+ "server_utilization": 0.012030798845043322,
1168
+ "server_spec_accept_rate": 0.5173611111111112,
1169
+ "server_spec_accept_length": 0.0,
1170
+ "avg_running_reqs": 2,
1171
+ "max_running_reqs": 2,
1172
+ "effective_concurrency": 2,
1173
+ "avg_queue_reqs": 0,
1174
+ "max_queue_reqs": 0,
1175
+ "queue_fraction": 0.0,
1176
+ "underfilled": false,
1177
+ "warmup_timed_out": false,
1178
+ "warmup_duration": 5.549,
1179
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1180
+ "timeout_reason": "",
1181
+ "capacity_limited": false,
1182
+ "hardware_summary": {
1183
+ "samples": 8,
1184
+ "duration_seconds": 16.873,
1185
+ "gpu_count": 4,
1186
+ "cpu_util_avg_pct": 10.84,
1187
+ "cpu_temp_max_c": 77.25,
1188
+ "gpu_util_avg_pct": 100.0,
1189
+ "gpu_util_max_pct": 100.0,
1190
+ "mem_util_avg_pct": 45.06,
1191
+ "mem_util_max_pct": 49.0,
1192
+ "temp_avg_c": 68.59,
1193
+ "temp_max_c": 84.0,
1194
+ "power_total_avg_w": 1178.44,
1195
+ "power_total_max_w": 1178.81,
1196
+ "power_limit_total_w": 1200.0,
1197
+ "vram_used_avg_mb": 380174.0,
1198
+ "vram_used_max_mb": 380174.0,
1199
+ "vram_total_mb": 391548.0,
1200
+ "vram_used_avg_pct": 97.1,
1201
+ "vram_used_max_pct": 97.1,
1202
+ "pcie_rx_avg_mb_s": 11056.12,
1203
+ "pcie_rx_max_mb_s": 11191.0,
1204
+ "pcie_tx_avg_mb_s": 11029.38,
1205
+ "pcie_tx_max_mb_s": 11153.0
1206
+ }
1207
+ },
1208
+ {
1209
+ "concurrency": 4,
1210
+ "context_tokens": 32768,
1211
+ "benchmark_mode": "duration",
1212
+ "request_count_target": 0,
1213
+ "warmup_request_count": 0,
1214
+ "measurement_seconds": 19.966272,
1215
+ "measurement_wall_seconds": 20.000472,
1216
+ "client_output_tokens": 5944,
1217
+ "server_output_tokens": 5956,
1218
+ "aggregate_source": "openai_continuous_usage",
1219
+ "aggregate_tps": 297.7020479777527,
1220
+ "per_request_avg_tps": 74.42551199443818,
1221
+ "ttft_avg": 2.7296407557441853,
1222
+ "ttft_p50": 3.4367186579620466,
1223
+ "ttft_p90": 3.437307026120834,
1224
+ "ttft_p99": 3.4374052726081574,
1225
+ "time_to_second_token_avg": 0.02786868193652481,
1226
+ "time_to_second_token_p50": 0.03361363697331399,
1227
+ "time_to_second_token_p90": 0.033683925354853275,
1228
+ "time_to_second_token_p99": 0.03370359269436449,
1229
+ "request_latency_avg": 0.0,
1230
+ "request_latency_p50": 0.0,
1231
+ "request_latency_p90": 0.0,
1232
+ "request_latency_p99": 0.0,
1233
+ "inter_token_latency_avg": 0.013227158501728438,
1234
+ "inter_token_latency_p50": 0.013280406418363698,
1235
+ "inter_token_latency_p90": 0.01328424798952288,
1236
+ "inter_token_latency_p99": 0.013284256693709504,
1237
+ "output_tps_per_user_avg": 75.60591881108353,
1238
+ "output_tps_per_user_p50": 75.29890661417849,
1239
+ "output_tps_per_user_p90": 76.18032211148439,
1240
+ "output_tps_per_user_p99": 76.51194460773044,
1241
+ "e2e_output_tps_per_user_avg": 0.0,
1242
+ "e2e_output_tps_per_user_p50": 0.0,
1243
+ "e2e_output_tps_per_user_p90": 0.0,
1244
+ "e2e_output_tps_per_user_p99": 0.0,
1245
+ "chunk_inter_token_latency_avg": 0.03458492849846504,
1246
+ "chunk_inter_token_latency_p50": 0.0345171884634303,
1247
+ "chunk_inter_token_latency_p90": 0.03476874527228148,
1248
+ "chunk_inter_token_latency_p99": 0.034855800227265865,
1249
+ "input_seq_len_avg": 32768.0,
1250
+ "output_seq_len_avg": 1799.75,
1251
+ "output_seq_len_p50": 1738.5,
1252
+ "output_seq_len_p90": 1910.5,
1253
+ "output_seq_len_p99": 1976.6499999999999,
1254
+ "request_count": 4,
1255
+ "completed_request_count": 0,
1256
+ "request_samples": [
1257
+ {
1258
+ "ttft": 0.6077095181681216,
1259
+ "time_to_second_token": 0.01054167584516108,
1260
+ "latency": 0.0,
1261
+ "inter_token_latency_avg": 0.013063563509345002,
1262
+ "chunk_inter_token_latency_avg": 0.03486547300004191,
1263
+ "input_tokens": 32768,
1264
+ "output_tokens": 1984,
1265
+ "output_tps_per_user": 76.54879155175779,
1266
+ "e2e_output_tps_per_user": 0.0,
1267
+ "completed": false
1268
+ },
1269
+ {
1270
+ "ttft": 3.4374161888845265,
1271
+ "time_to_second_token": 0.03363293595612049,
1272
+ "latency": 0.0,
1273
+ "inter_token_latency_avg": 0.013284225423113112,
1274
+ "chunk_inter_token_latency_avg": 0.034491329686020145,
1275
+ "input_tokens": 32768,
1276
+ "output_tokens": 1738,
1277
+ "output_tps_per_user": 75.27725314417718,
1278
+ "e2e_output_tps_per_user": 0.0,
1279
+ "completed": false
1280
+ },
1281
+ {
1282
+ "ttft": 3.4370523130055517,
1283
+ "time_to_second_token": 0.033594337990507483,
1284
+ "latency": 0.0,
1285
+ "inter_token_latency_avg": 0.013276587413614283,
1286
+ "chunk_inter_token_latency_avg": 0.03443986406695765,
1287
+ "input_tokens": 32768,
1288
+ "output_tokens": 1739,
1289
+ "output_tps_per_user": 75.3205600841798,
1290
+ "e2e_output_tps_per_user": 0.0,
1291
+ "completed": false
1292
+ },
1293
+ {
1294
+ "ttft": 3.4363850029185414,
1295
+ "time_to_second_token": 0.03370577795431018,
1296
+ "latency": 0.0,
1297
+ "inter_token_latency_avg": 0.013284257660841351,
1298
+ "chunk_inter_token_latency_avg": 0.03454304724084046,
1299
+ "input_tokens": 32768,
1300
+ "output_tokens": 1738,
1301
+ "output_tps_per_user": 75.27707046421935,
1302
+ "e2e_output_tps_per_user": 0.0,
1303
+ "completed": false
1304
+ }
1305
+ ],
1306
+ "total_tokens": 5944,
1307
+ "wall_time": 27.572524111019447,
1308
+ "num_completed": 4,
1309
+ "num_errors": 0,
1310
+ "server_gen_throughput": 297.6989316734227,
1311
+ "server_utilization": 0.04379210779595766,
1312
+ "server_spec_accept_rate": 0.5028735632183908,
1313
+ "server_spec_accept_length": 0.0,
1314
+ "avg_running_reqs": 4,
1315
+ "max_running_reqs": 4,
1316
+ "effective_concurrency": 4,
1317
+ "avg_queue_reqs": 0,
1318
+ "max_queue_reqs": 0,
1319
+ "queue_fraction": 0.0,
1320
+ "underfilled": false,
1321
+ "warmup_timed_out": false,
1322
+ "warmup_duration": 7.566,
1323
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1324
+ "timeout_reason": "",
1325
+ "capacity_limited": false,
1326
+ "hardware_summary": {
1327
+ "samples": 9,
1328
+ "duration_seconds": 19.263,
1329
+ "gpu_count": 4,
1330
+ "cpu_util_avg_pct": 11.06,
1331
+ "cpu_temp_max_c": 76.75,
1332
+ "gpu_util_avg_pct": 100.0,
1333
+ "gpu_util_max_pct": 100.0,
1334
+ "mem_util_avg_pct": 38.81,
1335
+ "mem_util_max_pct": 43.0,
1336
+ "temp_avg_c": 69.0,
1337
+ "temp_max_c": 84.0,
1338
+ "power_total_avg_w": 1173.28,
1339
+ "power_total_max_w": 1174.27,
1340
+ "power_limit_total_w": 1200.0,
1341
+ "vram_used_avg_mb": 380174.0,
1342
+ "vram_used_max_mb": 380174.0,
1343
+ "vram_total_mb": 391548.0,
1344
+ "vram_used_avg_pct": 97.1,
1345
+ "vram_used_max_pct": 97.1,
1346
+ "pcie_rx_avg_mb_s": 7830.56,
1347
+ "pcie_rx_max_mb_s": 8017.0,
1348
+ "pcie_tx_avg_mb_s": 7803.78,
1349
+ "pcie_tx_max_mb_s": 8098.0
1350
+ }
1351
+ }
1352
+ ],
1353
+ "summary_table": {
1354
+ "0": {
1355
+ "1": 167.19542416324822,
1356
+ "2": 244.64143701284553,
1357
+ "4": 292.85565306969767
1358
+ },
1359
+ "8192": {
1360
+ "1": 172.95726809626717,
1361
+ "2": 245.98233342981337,
1362
+ "4": 298.40065698894307
1363
+ },
1364
+ "32768": {
1365
+ "1": 186.35728993398365,
1366
+ "2": 252.6460844518971,
1367
+ "4": 297.7020479777527
1368
+ }
1369
+ },
1370
+ "burst_results": [],
1371
+ "burst_summary_table": {},
1372
+ "methodology": {
1373
+ "prefill": {
1374
+ "name": "Prefill",
1375
+ "present": false,
1376
+ "mode": "skipped",
1377
+ "formula": "prompt_tokens / TTFT",
1378
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
1379
+ },
1380
+ "sustained_decode": {
1381
+ "name": "Sustained Decode",
1382
+ "present": true,
1383
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
1384
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
1385
+ },
1386
+ "burst_e2e_decode": {
1387
+ "name": "Burst / E2E Decode",
1388
+ "present": false,
1389
+ "status": "not run; use --run-burst",
1390
+ "formula": "sum(completion_tokens) / profiling_wall_time",
1391
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
1392
+ }
1393
+ }
1394
+ }
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.log ADDED
@@ -0,0 +1,123 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0', '8k', '32k'] │
16
+ │ Duration: 20.0s per decode test | Max tokens: 8192 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill: skipped | Sustained decode: 9 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 17,031,168 tokens (2079 blocks × 2048; local
23
+ 4,257,792 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: skipped
26
+ Calibrating padding text (run=wwbkdgkvesfu, up to 32k)...
27
+ 8k: 50,544 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,140 chars (32,768 prompt tokens via /tokenize)
29
+ Token targeting: /tokenize exact
30
+ Done.
31
+
32
+
33
+
34
+ llm-decode-bench v0.4.29
35
+ ╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
36
+ │ Sustained Decode │
37
+ │ Steady-state decode throughput after the engine has admitted the requested │
38
+ │ concurrency and passed warmup. Use this as the main tuning/regression signal │
39
+ │ for kernels, NCCL, DCP, MTP, and scheduler changes. │
40
+ ╰──────────────────────────────────────────────────────────────────────────────╯
41
+ Aggregate tok/s + TTFT/ITL
42
+ ╭────────────┬─────────────┬─────────────┬──────────────╮
43
+ │ ctx \ conc │ 1 │ 2 │ 4 │
44
+ ├────────────┼─────────────┼─────────────┼──────────────┤
45
+ │ 0 │ 167.2 71/6 │ 244.6 114/8 │ 292.9 183/13 │
46
+ │ 8k │ 173.0 579/6 │ 246.0 943/8 │ 298.4 3k/13 │
47
+ │ 32k │ 186.4 598/5 │ 252.6 952/8 │ 297.7 3k/13 │
48
+ ╰────────────┴─────────────┴─────────────┴──────────────╯
49
+ Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
50
+ (continuous completion_tokens when the server supports it). Prometheus is kept
51
+ as validation/scheduler data.
52
+ Aggregate source(s): openai_continuous_usage
53
+ Per-Request tok/s
54
+ ╭────────────┬───────┬───────┬──────╮
55
+ │ ctx \ conc │ 1 │ 2 │ 4 │
56
+ ├────────────┼───────┼───────┼──────┤
57
+ │ 0 │ 167.2 │ 122.3 │ 73.2 │
58
+ │ 8k │ 173.0 │ 123.0 │ 74.6 │
59
+ │ 32k │ 186.4 │ 126.3 │ 74.4 │
60
+ ╰────────────┴───────┴───────┴──────╯
61
+ Client request latency: p50 /
62
+ p90 ms
63
+ ╭────────────┬─────┬─────┬─────╮
64
+ │ ctx \ conc │ 1 │ 2 │ 4 │
65
+ ├────────────┼─────┼─────┼─────┤
66
+ │ 0 │ —/— │ —/— │ —/— │
67
+ │ 8k │ —/— │ —/— │ —/— │
68
+ │ 32k │ —/— │ —/— │ —/— │
69
+ ╰────────────┴─────┴─────┴─────╯
70
+ Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
71
+ coordinate. ITL is computed from observed generated tokens, including streams
72
+ stopped at the measurement boundary; a missing ITL means no stream produced at
73
+ least two measured output tokens. Per-request tok/s and request latency are
74
+ shown in separate per-cell matrices. Completion/sample counts and full
75
+ request-level distributions remain in JSON under request_samples.
76
+ Sustained mode: client latency metrics explain request UX variance; aggregate
77
+ tok/s remains the primary throughput signal.
78
+ ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
79
+ Hardware Summary
80
+ ╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
81
+ │ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
82
+ ├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
83
+ │ 0 │ │ sust… │ 99/99% │ 48% │ 1151/1… │ 80C │ 76C │ 97… │ 8407/8352 │
84
+ │ … │ │ sust… │ 99/99% │ 49% │ 1154/1… │ 81C │ 75C │ 97… │ 8274/8292 │
85
+ │ … │ │ sust… │ 99/100% │ 48% │ 1154/1… │ 82C │ 76C │ 97… │ 8368/8179 │
86
+ │ 0 │ │ sust… │ 100/100% │ 45% │ 1177/1… │ 83C │ 76C │ 97… │ 11281/11144 │
87
+ │ 0 │ │ sust… │ 100/100% │ 40% │ 1173/1… │ 84C │ 77C │ 97… │ 7881/7872 │
88
+ │ … │ │ sust… │ 100/100% │ 45% │ 1178/1… │ 84C │ 76C │ 97… │ 11234/11152 │
89
+ │ … │ │ sust… │ 100/100% │ 39% │ 1174/1… │ 84C │ 77C │ 97… │ 7879/7780 │
90
+ │ … │ │ sust… │ 100/100% │ 45% │ 1178/1… │ 84C │ 77C │ 97… │ 11056/11029 │
91
+ │ … │ │ sust… │ 100/100% │ 39% │ 1173/1… │ 84C │ 77C │ 97… │ 7831/7804 │
92
+ ╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
93
+ ╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
94
+ │ avg 1,101 W | max 1,179 W | limit 1,200 W | over 4m 27s | 112 samples │
95
+ ╰───────────────────────────────────────────────────────────────────────╯
96
+ Hardware summary is sampled from nvidia-smi during the measured part of each
97
+ cell. Whole-run GPU power is the sampled sum of GPU power draw across the
98
+ complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
99
+ a coarse live diagnostic, not a per-kernel NCCL profiler.
100
+
101
+ ╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
102
+ │ Burst / E2E Decode │
103
+ │ Not run. Re-run with --run-burst to append a finite client-facing request │
104
+ │ burst after Sustained Decode. This is intentionally disabled by default │
105
+ │ because it adds another full decode matrix. │
106
+ ╰──────────────────────────────────────────────────────────────────────────────╯
107
+
108
+ ╭────────────────────────────── Primary Summary ───────────────────────────────╮
109
+ │ Primary matrices repeated last so the important numbers are visible without │
110
+ │ scrolling back through diagnostics. │
111
+ ╰─────────────────────────────────────��────────────────────────────────────────╯
112
+ Aggregate decode tok/s
113
+ ╭────────────┬───────┬───────┬───────╮
114
+ │ ctx \ conc │ 1 │ 2 │ 4 │
115
+ ├────────────┼───────┼───────┼───────┤
116
+ │ 0 │ 167.2 │ 244.6 │ 292.9 │
117
+ │ 8k │ 173.0 │ 246.0 │ 298.4 │
118
+ │ 32k │ 186.4 │ 252.6 │ 297.7 │
119
+ ╰────────────┴───────┴───────┴───────╯
120
+
121
+ Results saved to
122
+ <campaign>/batch16384-speed-w
123
+ indow-01/results-01/batch16384/rep-1/decode-cap8192.json
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-command.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json",
20
+ "--contexts",
21
+ "0",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--prefill-only",
25
+ "--prefill-contexts",
26
+ "8k,32k,64k,128k",
27
+ "--prefill-duration",
28
+ "20",
29
+ "--cell-warmup-timeout-seconds",
30
+ "180"
31
+ ]
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "0068157e646ee939ef2c7a0c5c860e6ec4582eecfe6284929bdeaa7272046797"
5
+ }
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json ADDED
@@ -0,0 +1,396 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T07:13:30.053751",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "standalone_cold",
15
+ "standalone_prefill": true,
16
+ "prefill_only": true,
17
+ "skip_prefill": false,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 0,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 17031168,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0
42
+ ],
43
+ "startup_diagnostics_available": true,
44
+ "nvidia_p2p_override_effective": true,
45
+ "p2pmark_status": "not_run",
46
+ "amd_fabric_status": "not_run"
47
+ },
48
+ "startup_diagnostics": {
49
+ "version": "0.4.29",
50
+ "server_url": "http://127.0.0.1:8001",
51
+ "hostname": "<host>",
52
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
53
+ "env": {},
54
+ "args": {
55
+ "concurrency": "1,2,4",
56
+ "contexts": "0",
57
+ "max_tokens": 8192,
58
+ "duration": 20.0,
59
+ "request_count": 0,
60
+ "run_burst": false,
61
+ "standalone_prefill": true,
62
+ "prefill_only": true,
63
+ "skip_prefill": false,
64
+ "prefill_contexts": "8k,32k,64k,128k",
65
+ "prefill_metric": "client",
66
+ "dcp_size": 0,
67
+ "kv_budget": 0
68
+ },
69
+ "nvidia_p2p_override": {
70
+ "effective": true,
71
+ "configured": true,
72
+ "params_path": "/proc/driver/nvidia/params",
73
+ "params_available": true,
74
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
75
+ "modprobe_available": true,
76
+ "runtime": {
77
+ "ForceP2P": "0x11",
78
+ "RMForceP2PType": "1",
79
+ "RMPcieP2PType": "2",
80
+ "GrdmaPciTopoCheckOverride": "1",
81
+ "EnableResizableBar": "1",
82
+ "DmaRemapPeerMmio": "1"
83
+ },
84
+ "expected": {
85
+ "ForceP2P": "0x11",
86
+ "RMForceP2PType": "1",
87
+ "RMPcieP2PType": "2",
88
+ "GrdmaPciTopoCheckOverride": "1",
89
+ "EnableResizableBar": "1"
90
+ },
91
+ "missing": [],
92
+ "mismatched": {},
93
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
94
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
95
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
96
+ },
97
+ "p2pmark": {
98
+ "status": "not_run"
99
+ },
100
+ "amd_fabric": {
101
+ "status": "not_run"
102
+ },
103
+ "nvidia_smi_query": {
104
+ "cmd": [
105
+ "nvidia-smi",
106
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
107
+ "--format=csv,noheader,nounits"
108
+ ],
109
+ "returncode": 0,
110
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
111
+ "stderr": ""
112
+ },
113
+ "nvidia_smi_topo": {
114
+ "cmd": [
115
+ "nvidia-smi",
116
+ "topo",
117
+ "-m"
118
+ ],
119
+ "returncode": 0,
120
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
121
+ "stderr": ""
122
+ }
123
+ },
124
+ "nvidia_p2p_override": {
125
+ "effective": true,
126
+ "configured": true,
127
+ "params_path": "/proc/driver/nvidia/params",
128
+ "params_available": true,
129
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
130
+ "modprobe_available": true,
131
+ "runtime": {
132
+ "ForceP2P": "0x11",
133
+ "RMForceP2PType": "1",
134
+ "RMPcieP2PType": "2",
135
+ "GrdmaPciTopoCheckOverride": "1",
136
+ "EnableResizableBar": "1",
137
+ "DmaRemapPeerMmio": "1"
138
+ },
139
+ "expected": {
140
+ "ForceP2P": "0x11",
141
+ "RMForceP2PType": "1",
142
+ "RMPcieP2PType": "2",
143
+ "GrdmaPciTopoCheckOverride": "1",
144
+ "EnableResizableBar": "1"
145
+ },
146
+ "missing": [],
147
+ "mismatched": {},
148
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
149
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
150
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
151
+ },
152
+ "p2pmark": {
153
+ "status": "not_run"
154
+ },
155
+ "amd_fabric": {
156
+ "status": "not_run"
157
+ },
158
+ "hardware_run_summary": {
159
+ "samples": 46,
160
+ "duration_seconds": 108.093,
161
+ "gpu_count": 4,
162
+ "cpu_util_avg_pct": 10.0,
163
+ "cpu_temp_max_c": 76.25,
164
+ "gpu_util_avg_pct": 84.77,
165
+ "gpu_util_max_pct": 100.0,
166
+ "mem_util_avg_pct": 22.67,
167
+ "mem_util_max_pct": 35.0,
168
+ "temp_avg_c": 59.3,
169
+ "temp_max_c": 80.0,
170
+ "power_total_avg_w": 1039.43,
171
+ "power_total_max_w": 1180.51,
172
+ "power_limit_total_w": 1200.0,
173
+ "vram_used_avg_mb": 376719.74,
174
+ "vram_used_max_mb": 380174.0,
175
+ "vram_total_mb": 391548.0,
176
+ "vram_used_avg_pct": 96.22,
177
+ "vram_used_max_pct": 97.1,
178
+ "pcie_rx_avg_mb_s": 42900.63,
179
+ "pcie_rx_max_mb_s": 68701.0,
180
+ "pcie_tx_avg_mb_s": 40511.74,
181
+ "pcie_tx_max_mb_s": 63363.0
182
+ },
183
+ "event_log": [],
184
+ "prefill": {
185
+ "8192": {
186
+ "ttft_seconds": 1.041,
187
+ "prefill_seconds": 1.041,
188
+ "tok_per_sec": 7874.0,
189
+ "client_ttft_seconds": 1.041,
190
+ "client_tok_per_sec": 7874.0,
191
+ "prompt_tokens": 8194,
192
+ "samples": 14,
193
+ "method": "client",
194
+ "server_validation": {
195
+ "method": "",
196
+ "tok_per_sec": 0.0,
197
+ "prefill_seconds": 0.0,
198
+ "prompt_tokens": 0,
199
+ "request_prompt_tokens": 0,
200
+ "cached_tokens": 0,
201
+ "token_source": "",
202
+ "samples": 0,
203
+ "invalid_reason": ""
204
+ },
205
+ "hardware_summary": {
206
+ "samples": 9,
207
+ "duration_seconds": 19.127,
208
+ "gpu_count": 4,
209
+ "cpu_util_avg_pct": 10.29,
210
+ "cpu_temp_max_c": 75.25,
211
+ "gpu_util_avg_pct": 65.11,
212
+ "gpu_util_max_pct": 100.0,
213
+ "mem_util_avg_pct": 18.42,
214
+ "mem_util_max_pct": 33.0,
215
+ "temp_avg_c": 52.0,
216
+ "temp_max_c": 66.0,
217
+ "power_total_avg_w": 963.89,
218
+ "power_total_max_w": 1159.22,
219
+ "power_limit_total_w": 1200.0,
220
+ "vram_used_avg_mb": 366586.0,
221
+ "vram_used_max_mb": 366586.0,
222
+ "vram_total_mb": 391548.0,
223
+ "vram_used_avg_pct": 93.62,
224
+ "vram_used_max_pct": 93.62,
225
+ "pcie_rx_avg_mb_s": 33986.56,
226
+ "pcie_rx_max_mb_s": 58738.0,
227
+ "pcie_tx_avg_mb_s": 32431.22,
228
+ "pcie_tx_max_mb_s": 56269.0
229
+ }
230
+ },
231
+ "32768": {
232
+ "ttft_seconds": 4.08,
233
+ "prefill_seconds": 4.08,
234
+ "tok_per_sec": 8033.0,
235
+ "client_ttft_seconds": 4.08,
236
+ "client_tok_per_sec": 8033.0,
237
+ "prompt_tokens": 32770,
238
+ "samples": 5,
239
+ "method": "client",
240
+ "server_validation": {
241
+ "method": "",
242
+ "tok_per_sec": 0.0,
243
+ "prefill_seconds": 0.0,
244
+ "prompt_tokens": 0,
245
+ "request_prompt_tokens": 0,
246
+ "cached_tokens": 0,
247
+ "token_source": "",
248
+ "samples": 0,
249
+ "invalid_reason": ""
250
+ },
251
+ "hardware_summary": {
252
+ "samples": 9,
253
+ "duration_seconds": 19.245,
254
+ "gpu_count": 4,
255
+ "cpu_util_avg_pct": 10.29,
256
+ "cpu_temp_max_c": 75.75,
257
+ "gpu_util_avg_pct": 97.5,
258
+ "gpu_util_max_pct": 100.0,
259
+ "mem_util_avg_pct": 25.89,
260
+ "mem_util_max_pct": 33.0,
261
+ "temp_avg_c": 58.39,
262
+ "temp_max_c": 73.0,
263
+ "power_total_avg_w": 1058.45,
264
+ "power_total_max_w": 1148.19,
265
+ "power_limit_total_w": 1200.0,
266
+ "vram_used_avg_mb": 380045.11,
267
+ "vram_used_max_mb": 380174.0,
268
+ "vram_total_mb": 391548.0,
269
+ "vram_used_avg_pct": 97.07,
270
+ "vram_used_max_pct": 97.1,
271
+ "pcie_rx_avg_mb_s": 50337.22,
272
+ "pcie_rx_max_mb_s": 61317.0,
273
+ "pcie_tx_avg_mb_s": 47808.0,
274
+ "pcie_tx_max_mb_s": 54706.0
275
+ }
276
+ },
277
+ "65536": {
278
+ "ttft_seconds": 8.239,
279
+ "prefill_seconds": 8.239,
280
+ "tok_per_sec": 7955.0,
281
+ "client_ttft_seconds": 8.239,
282
+ "client_tok_per_sec": 7955.0,
283
+ "prompt_tokens": 65538,
284
+ "samples": 3,
285
+ "method": "client",
286
+ "server_validation": {
287
+ "method": "",
288
+ "tok_per_sec": 0.0,
289
+ "prefill_seconds": 0.0,
290
+ "prompt_tokens": 0,
291
+ "request_prompt_tokens": 0,
292
+ "cached_tokens": 0,
293
+ "token_source": "",
294
+ "samples": 0,
295
+ "invalid_reason": ""
296
+ },
297
+ "hardware_summary": {
298
+ "samples": 11,
299
+ "duration_seconds": 24.061,
300
+ "gpu_count": 4,
301
+ "cpu_util_avg_pct": 10.33,
302
+ "cpu_temp_max_c": 76.0,
303
+ "gpu_util_avg_pct": 94.18,
304
+ "gpu_util_max_pct": 100.0,
305
+ "mem_util_avg_pct": 25.11,
306
+ "mem_util_max_pct": 35.0,
307
+ "temp_avg_c": 61.48,
308
+ "temp_max_c": 77.0,
309
+ "power_total_avg_w": 1093.56,
310
+ "power_total_max_w": 1180.51,
311
+ "power_limit_total_w": 1200.0,
312
+ "vram_used_avg_mb": 380174.0,
313
+ "vram_used_max_mb": 380174.0,
314
+ "vram_total_mb": 391548.0,
315
+ "vram_used_avg_pct": 97.1,
316
+ "vram_used_max_pct": 97.1,
317
+ "pcie_rx_avg_mb_s": 50410.73,
318
+ "pcie_rx_max_mb_s": 59338.0,
319
+ "pcie_tx_avg_mb_s": 47142.09,
320
+ "pcie_tx_max_mb_s": 60407.0
321
+ }
322
+ },
323
+ "131072": {
324
+ "ttft_seconds": 16.814,
325
+ "prefill_seconds": 16.814,
326
+ "tok_per_sec": 7796.0,
327
+ "client_ttft_seconds": 16.814,
328
+ "client_tok_per_sec": 7796.0,
329
+ "prompt_tokens": 131074,
330
+ "samples": 2,
331
+ "method": "client",
332
+ "server_validation": {
333
+ "method": "",
334
+ "tok_per_sec": 0.0,
335
+ "prefill_seconds": 0.0,
336
+ "prompt_tokens": 0,
337
+ "request_prompt_tokens": 0,
338
+ "cached_tokens": 0,
339
+ "token_source": "",
340
+ "samples": 0,
341
+ "invalid_reason": ""
342
+ },
343
+ "hardware_summary": {
344
+ "samples": 15,
345
+ "duration_seconds": 33.649,
346
+ "gpu_count": 4,
347
+ "cpu_util_avg_pct": 10.22,
348
+ "cpu_temp_max_c": 76.25,
349
+ "gpu_util_avg_pct": 93.33,
350
+ "gpu_util_max_pct": 100.0,
351
+ "mem_util_avg_pct": 24.52,
352
+ "mem_util_max_pct": 31.0,
353
+ "temp_avg_c": 64.23,
354
+ "temp_max_c": 80.0,
355
+ "power_total_avg_w": 1110.04,
356
+ "power_total_max_w": 1154.26,
357
+ "power_limit_total_w": 1200.0,
358
+ "vram_used_avg_mb": 380174.0,
359
+ "vram_used_max_mb": 380174.0,
360
+ "vram_total_mb": 391548.0,
361
+ "vram_used_avg_pct": 97.1,
362
+ "vram_used_max_pct": 97.1,
363
+ "pcie_rx_avg_mb_s": 43409.47,
364
+ "pcie_rx_max_mb_s": 68701.0,
365
+ "pcie_tx_avg_mb_s": 40960.33,
366
+ "pcie_tx_max_mb_s": 63363.0
367
+ }
368
+ }
369
+ },
370
+ "results": [],
371
+ "summary_table": {},
372
+ "burst_results": [],
373
+ "burst_summary_table": {},
374
+ "methodology": {
375
+ "prefill": {
376
+ "name": "Prefill",
377
+ "present": true,
378
+ "mode": "standalone_cold",
379
+ "formula": "prompt_tokens / TTFT",
380
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
381
+ },
382
+ "sustained_decode": {
383
+ "name": "Sustained Decode",
384
+ "present": false,
385
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
386
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
387
+ },
388
+ "burst_e2e_decode": {
389
+ "name": "Burst / E2E Decode",
390
+ "present": false,
391
+ "status": "not run; use --run-burst",
392
+ "formula": "sum(completion_tokens) / profiling_wall_time",
393
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
394
+ }
395
+ }
396
+ }
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.log ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0'] │
16
+ │ Decode: skipped (--prefill-only) | Max tokens: 8192 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill-only: standalone cold profile (client) | Sustained decode: 0 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 17,031,168 tokens (2079 blocks × 2048; local
23
+ 4,257,792 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: standalone cold profile ['8k', '32k', '64k', '128k']
26
+ Calibrating padding text (run=jowlqgbaqdiz, up to 128k)...
27
+ 8k: 50,558 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,152 chars (32,768 prompt tokens via /tokenize)
29
+ 64k: 411,264 chars (65,536 prompt tokens via /tokenize)
30
+ 128k: 823,408 chars (131,072 prompt tokens via /tokenize)
31
+ Token targeting: /tokenize exact
32
+ Done.
33
+
34
+
35
+
36
+ llm-decode-bench v0.4.29
37
+ Prefill Speed (C=1, client ISL / TTFT)
38
+
39
+ Client PCIe rx/tx
40
+ Context Tokens TTFT (s) tok/s Server tok/s avg N
41
+ ──────────────────────────────────────────────────────────────────────────────
42
+ 8k 8,194 1.04 7,874 — 33987/32431 14
43
+ 32k 32,770 4.08 8,033 — 50337/47808 5
44
+ 64k 65,538 8.24 7,955 — 50411/47142 3
45
+ 128k 131,074 16.81 7,796 — 43409/40960 2
46
+
47
+ Client tok/s = prompt_tokens / TTFT. Integrated scout rows come from the
48
+ prefix-cache scout request that decode needs anyway. Server tok/s is optional
49
+ Prometheus validation when the engine exports prefill counters and the exact
50
+ counter delta is uncontaminated; for vLLM this uses newly computed KV tokens,
51
+ not request prompt tokens.
52
+
53
+
54
+ Results saved to
55
+ <campaign>/batch16384-speed-w
56
+ indow-01/results-01/batch16384/rep-1/prefill.json
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/failure.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "error": "KeyboardInterrupt()"
3
+ }
results/speed-20260909/evidence/candidate-graph-analysis-02/REPORT.md ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Candidate FP8 serving graph profile
2
+
3
+ All four planned cells completed on retained image sha256:9bb99e0b47f00c4ccf77f2d4be77bb7c3ce8d6b5636169aa1f1fddda040ab45e. Runtime/source receipts match TP4/DCP4, MTP3 probabilistic, NVFP4 KV, graphs, maxseq24, batch4096, four300W GPUs. Original production and clients stayed stopped. These are instrumented diagnostic traces, not new throughput claims.
4
+
5
+ | Profile | Graph kernel events mapped | Non-graph kernel events |
6
+ |---|---:|---:|
7
+ | 8K C1 decode | 1020936 | 31624 |
8
+ | 8K C4 decode | 429400 | 16400 |
9
+ | 32K cold prefill plus setup | 214348 | 238940 |
10
+ | 64K cold prefill plus setup | 219208 | 350852 |
11
+
12
+ All1,883,892graph-associated events map via explicit process-scoped clone/original node IDs to captured topology. All6,368graph receipts succeeded. Both decode clients report zero errors and no underfill/capacity/warmup-timeout flags. Raw capture timings include start/stop RPC latency; nominal1.5seconds is the controller sleep between profiling RPCs, not an exact measured trace duration.
13
+
14
+ Target P8 direct FC1/FC2 kernels are present in both decode traces. P8 grouped FC1/FC2 kernels are additionally present in both prefill traces. The serving adapter directly dispatches P8NativeTPMoE with use_a16=False; missing target runtime raises an error. Marlin kernels are recorded separately and must retain their layer-attribution boundary. Four representative compiled target regimes were independently checked for QMMA.SF.16832.F32.E4M3.E4M3.E8; no W4A16 target replacement was made.
15
+
16
+ Do not remove waits from this evidence. Many graph edges have non-default dependency/port types, which the analyzer preserves without interpreting them as full-completion barriers. Some default-edge kernel timestamps overlap by approximately1.0-1.44microseconds in C1; keep these anomalies instead of treating them as dependency violations or deleting edges. Cross-rank, eager-prefill and external-stream dependencies remain incomplete. Summed durations are not wall time or attainable speedup fractions.
17
+
18
+ Next bounded candidate comes from actual grouped FC2 launch geometry:256CTAs,128threads, registers199(K4)/216(K5), reported dynamic shared34816/38912bytes and executed partition102400bytes where available. These resources admit2blocks per188-SM device, or376CTAs, but current grid exposes only256. Test376CTAs with unchanged task stride/ownership and unchanged FP8/Hadamard/orderedK512 math. More CTAs might reduce imbalance or hurt locality; CPU task-coverage/source-scope proof passed and incremental component test is running. This does not assume the300C1/20000prefill targets are reachable from this change.
19
+
20
+ All raw Nsight reports and graph records are indexed by graph-profile-preservation-02/INDEX.json. Exported SQLite and these analyses are additional artifacts, hashed separately here. Earlier failed profiling attempts remain preserved.
results/speed-20260909/evidence/candidate-graph-analysis-02/SUMMARY.json ADDED
@@ -0,0 +1,242 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "all-four-profiles-completed-and-graph-associated-kernels-mapped",
3
+ "profiles": [
4
+ {
5
+ "profile": "8K C1 decode",
6
+ "coverage": {
7
+ "total_kernel_events": 1052560,
8
+ "matched_graph_kernel_events": 1020936,
9
+ "non_graph_kernel_events": 31624
10
+ },
11
+ "kernel_categories": {
12
+ "other": {
13
+ "calls": 1010080,
14
+ "summed_duration_ns": 5759231895
15
+ },
16
+ "Marlin-needs-layer-attribution": {
17
+ "calls": 2832,
18
+ "summed_duration_ns": 64929376
19
+ },
20
+ "target-P8-direct-FC1": {
21
+ "calls": 19824,
22
+ "summed_duration_ns": 1332961683
23
+ },
24
+ "target-P8-direct-FC2": {
25
+ "calls": 19824,
26
+ "summed_duration_ns": 476408029
27
+ }
28
+ },
29
+ "edge_checks": {
30
+ "both_endpoints_timed": 1122416,
31
+ "nondefault_edges_not_interpreted_as_completion": 433768,
32
+ "default_completion_edges": 688648
33
+ },
34
+ "timing_anomaly_count": 299,
35
+ "grouped_fc2_resources": [],
36
+ "analysis_sha256": "8f6e2864cbd68a184a6f8b33b74adaa31da50612ce33e99e230b1205a54e0a27"
37
+ },
38
+ {
39
+ "profile": "8K C4 decode",
40
+ "coverage": {
41
+ "total_kernel_events": 445800,
42
+ "matched_graph_kernel_events": 429400,
43
+ "non_graph_kernel_events": 16400
44
+ },
45
+ "kernel_categories": {
46
+ "other": {
47
+ "calls": 427800,
48
+ "summed_duration_ns": 3763933505
49
+ },
50
+ "Marlin-needs-layer-attribution": {
51
+ "calls": 1200,
52
+ "summed_duration_ns": 66829405
53
+ },
54
+ "target-P8-direct-FC1": {
55
+ "calls": 8400,
56
+ "summed_duration_ns": 2249130432
57
+ },
58
+ "target-P8-direct-FC2": {
59
+ "calls": 8400,
60
+ "summed_duration_ns": 846519484
61
+ }
62
+ },
63
+ "edge_checks": {
64
+ "both_endpoints_timed": 473200,
65
+ "default_completion_edges": 279800,
66
+ "nondefault_edges_not_interpreted_as_completion": 193400
67
+ },
68
+ "timing_anomaly_count": 0,
69
+ "grouped_fc2_resources": [],
70
+ "analysis_sha256": "ecdbc033af4eae4984b62014ccc2e7f586a2b4e98992b3edba945b887f56fa57"
71
+ },
72
+ {
73
+ "profile": "32K cold prefill plus setup",
74
+ "coverage": {
75
+ "total_kernel_events": 453288,
76
+ "matched_graph_kernel_events": 214348,
77
+ "non_graph_kernel_events": 238940
78
+ },
79
+ "kernel_categories": {
80
+ "other": {
81
+ "calls": 438168,
82
+ "summed_duration_ns": 25529631362
83
+ },
84
+ "Marlin-needs-layer-attribution": {
85
+ "calls": 1008,
86
+ "summed_duration_ns": 21740194
87
+ },
88
+ "target-P8-direct-FC1": {
89
+ "calls": 3864,
90
+ "summed_duration_ns": 246439082
91
+ },
92
+ "target-P8-direct-FC2": {
93
+ "calls": 3864,
94
+ "summed_duration_ns": 87860594
95
+ },
96
+ "target-P8-grouped-FC1": {
97
+ "calls": 3192,
98
+ "summed_duration_ns": 4867068677
99
+ },
100
+ "target-P8-grouped-FC2": {
101
+ "calls": 3192,
102
+ "summed_duration_ns": 3764819288
103
+ }
104
+ },
105
+ "edge_checks": {
106
+ "both_endpoints_timed": 232380,
107
+ "nondefault_edges_not_interpreted_as_completion": 87884,
108
+ "default_completion_edges": 144496
109
+ },
110
+ "timing_anomaly_count": 69,
111
+ "grouped_fc2_resources": [
112
+ {
113
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
114
+ "grid_z": 256,
115
+ "block_x": 128,
116
+ "registers": 199,
117
+ "dynamic_shared": 34816,
118
+ "shared_partition": 0,
119
+ "requested_percent": null,
120
+ "calls": 1800
121
+ },
122
+ {
123
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
124
+ "grid_z": 256,
125
+ "block_x": 128,
126
+ "registers": 199,
127
+ "dynamic_shared": 34816,
128
+ "shared_partition": 102400,
129
+ "requested_percent": 68,
130
+ "calls": 100
131
+ },
132
+ {
133
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
134
+ "grid_z": 256,
135
+ "block_x": 128,
136
+ "registers": 216,
137
+ "dynamic_shared": 38912,
138
+ "shared_partition": 0,
139
+ "requested_percent": null,
140
+ "calls": 1224
141
+ },
142
+ {
143
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
144
+ "grid_z": 256,
145
+ "block_x": 128,
146
+ "registers": 216,
147
+ "dynamic_shared": 38912,
148
+ "shared_partition": 102400,
149
+ "requested_percent": 76,
150
+ "calls": 68
151
+ }
152
+ ],
153
+ "analysis_sha256": "dbc87a37bea8fdc6ce706e723c40a34d0596c40a075e8db8d9c506ead76f76cd"
154
+ },
155
+ {
156
+ "profile": "64K cold prefill plus setup",
157
+ "coverage": {
158
+ "total_kernel_events": 570060,
159
+ "matched_graph_kernel_events": 219208,
160
+ "non_graph_kernel_events": 350852
161
+ },
162
+ "kernel_categories": {
163
+ "other": {
164
+ "calls": 551700,
165
+ "summed_duration_ns": 39640395478
166
+ },
167
+ "target-P8-grouped-FC1": {
168
+ "calls": 4704,
169
+ "summed_duration_ns": 7145088363
170
+ },
171
+ "target-P8-grouped-FC2": {
172
+ "calls": 4704,
173
+ "summed_duration_ns": 5633648267
174
+ },
175
+ "Marlin-needs-layer-attribution": {
176
+ "calls": 1224,
177
+ "summed_duration_ns": 25492544
178
+ },
179
+ "target-P8-direct-FC1": {
180
+ "calls": 3864,
181
+ "summed_duration_ns": 247245351
182
+ },
183
+ "target-P8-direct-FC2": {
184
+ "calls": 3864,
185
+ "summed_duration_ns": 87240073
186
+ }
187
+ },
188
+ "edge_checks": {
189
+ "both_endpoints_timed": 237024,
190
+ "default_completion_edges": 148096,
191
+ "nondefault_edges_not_interpreted_as_completion": 88928
192
+ },
193
+ "timing_anomaly_count": 61,
194
+ "grouped_fc2_resources": [
195
+ {
196
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
197
+ "grid_z": 256,
198
+ "block_x": 128,
199
+ "registers": 199,
200
+ "dynamic_shared": 34816,
201
+ "shared_partition": 0,
202
+ "requested_percent": null,
203
+ "calls": 2700
204
+ },
205
+ {
206
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
207
+ "grid_z": 256,
208
+ "block_x": 128,
209
+ "registers": 199,
210
+ "dynamic_shared": 34816,
211
+ "shared_partition": 102400,
212
+ "requested_percent": 68,
213
+ "calls": 100
214
+ },
215
+ {
216
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
217
+ "grid_z": 256,
218
+ "block_x": 128,
219
+ "registers": 216,
220
+ "dynamic_shared": 38912,
221
+ "shared_partition": 0,
222
+ "requested_percent": null,
223
+ "calls": 1836
224
+ },
225
+ {
226
+ "kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
227
+ "grid_z": 256,
228
+ "block_x": 128,
229
+ "registers": 216,
230
+ "dynamic_shared": 38912,
231
+ "shared_partition": 102400,
232
+ "requested_percent": 76,
233
+ "calls": 68
234
+ }
235
+ ],
236
+ "analysis_sha256": "a5bb8b2ec42a9930a512254c254e23c163d31b3bfd8afc9edcaed2aa1c33c55d"
237
+ }
238
+ ],
239
+ "throughput_claim": false,
240
+ "next_candidate": "grouped-fc2-grid376-component-01",
241
+ "scope": "All graph-associated kernel events matched; eager/non-graph prefill kernels remain outside graph-topology coverage."
242
+ }
results/speed-20260909/evidence/candidate-graph-profile-01/results-01/failure.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ {
2
+ "error": "RuntimeError(\"trellismx-candidate-graph-profile-01 exited: {'Status': 'exited', 'Running': False, 'Paused': False, 'Restarting': False, 'OOMKilled': False, 'Dead': False, 'Pid': 0, 'ExitCode': 1, 'Error': '', 'StartedAt': '2026-09-09T14:00:23.959294351Z', 'FinishedAt': '2026-09-09T14:03:17.551677807Z'}\")"
3
+ }
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c1-8k/result.json ADDED
@@ -0,0 +1,345 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T10:10:50.723841",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 8192,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": 0.0,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29188096,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1
37
+ ],
38
+ "context_lengths": [
39
+ 8192
40
+ ],
41
+ "startup_diagnostics_available": true,
42
+ "nvidia_p2p_override_effective": true,
43
+ "p2pmark_status": "not_run",
44
+ "amd_fabric_status": "not_run"
45
+ },
46
+ "startup_diagnostics": {
47
+ "version": "0.4.29",
48
+ "server_url": "http://127.0.0.1:8001",
49
+ "hostname": "<host>",
50
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
51
+ "env": {},
52
+ "args": {
53
+ "concurrency": "1",
54
+ "contexts": "8k",
55
+ "max_tokens": 8192,
56
+ "duration": 20.0,
57
+ "request_count": 0,
58
+ "run_burst": false,
59
+ "standalone_prefill": false,
60
+ "prefill_only": false,
61
+ "skip_prefill": true,
62
+ "prefill_contexts": "8k,64k,128k",
63
+ "prefill_metric": "client",
64
+ "dcp_size": 0,
65
+ "kv_budget": 0
66
+ },
67
+ "nvidia_p2p_override": {
68
+ "effective": true,
69
+ "configured": true,
70
+ "params_path": "/proc/driver/nvidia/params",
71
+ "params_available": true,
72
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
73
+ "modprobe_available": true,
74
+ "runtime": {
75
+ "ForceP2P": "0x11",
76
+ "RMForceP2PType": "1",
77
+ "RMPcieP2PType": "2",
78
+ "GrdmaPciTopoCheckOverride": "1",
79
+ "EnableResizableBar": "1",
80
+ "DmaRemapPeerMmio": "1"
81
+ },
82
+ "expected": {
83
+ "ForceP2P": "0x11",
84
+ "RMForceP2PType": "1",
85
+ "RMPcieP2PType": "2",
86
+ "GrdmaPciTopoCheckOverride": "1",
87
+ "EnableResizableBar": "1"
88
+ },
89
+ "missing": [],
90
+ "mismatched": {},
91
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
92
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
93
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
94
+ },
95
+ "p2pmark": {
96
+ "status": "not_run"
97
+ },
98
+ "amd_fabric": {
99
+ "status": "not_run"
100
+ },
101
+ "nvidia_smi_query": {
102
+ "cmd": [
103
+ "nvidia-smi",
104
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
105
+ "--format=csv,noheader,nounits"
106
+ ],
107
+ "returncode": 0,
108
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
109
+ "stderr": ""
110
+ },
111
+ "nvidia_smi_topo": {
112
+ "cmd": [
113
+ "nvidia-smi",
114
+ "topo",
115
+ "-m"
116
+ ],
117
+ "returncode": 0,
118
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
119
+ "stderr": ""
120
+ }
121
+ },
122
+ "nvidia_p2p_override": {
123
+ "effective": true,
124
+ "configured": true,
125
+ "params_path": "/proc/driver/nvidia/params",
126
+ "params_available": true,
127
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
128
+ "modprobe_available": true,
129
+ "runtime": {
130
+ "ForceP2P": "0x11",
131
+ "RMForceP2PType": "1",
132
+ "RMPcieP2PType": "2",
133
+ "GrdmaPciTopoCheckOverride": "1",
134
+ "EnableResizableBar": "1",
135
+ "DmaRemapPeerMmio": "1"
136
+ },
137
+ "expected": {
138
+ "ForceP2P": "0x11",
139
+ "RMForceP2PType": "1",
140
+ "RMPcieP2PType": "2",
141
+ "GrdmaPciTopoCheckOverride": "1",
142
+ "EnableResizableBar": "1"
143
+ },
144
+ "missing": [],
145
+ "mismatched": {},
146
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
147
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
148
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
149
+ },
150
+ "p2pmark": {
151
+ "status": "not_run"
152
+ },
153
+ "amd_fabric": {
154
+ "status": "not_run"
155
+ },
156
+ "hardware_run_summary": {
157
+ "samples": 19,
158
+ "duration_seconds": 43.426,
159
+ "gpu_count": 4,
160
+ "cpu_util_avg_pct": 8.54,
161
+ "cpu_temp_max_c": 75.12,
162
+ "gpu_util_avg_pct": 64.61,
163
+ "gpu_util_max_pct": 100.0,
164
+ "mem_util_avg_pct": 35.28,
165
+ "mem_util_max_pct": 60.0,
166
+ "temp_avg_c": 48.41,
167
+ "temp_max_c": 67.0,
168
+ "power_total_avg_w": 874.72,
169
+ "power_total_max_w": 1154.84,
170
+ "power_limit_total_w": 1200.0,
171
+ "vram_used_avg_mb": 386175.47,
172
+ "vram_used_max_mb": 386578.0,
173
+ "vram_total_mb": 391548.0,
174
+ "vram_used_avg_pct": 98.63,
175
+ "vram_used_max_pct": 98.73,
176
+ "pcie_rx_avg_mb_s": 8453.16,
177
+ "pcie_rx_max_mb_s": 48004.0,
178
+ "pcie_tx_avg_mb_s": 8921.84,
179
+ "pcie_tx_max_mb_s": 47296.0
180
+ },
181
+ "event_log": [
182
+ "10:10:04 benchmark start engine=vllm",
183
+ "10:10:04 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
184
+ "10:10:04 startup decode concurrency=1 contexts=8k",
185
+ "10:10:04 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
186
+ "10:10:04 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
187
+ "10:10:04 startup KV cache budget from vLLM metrics: 29,188,096 tokens (3563 blocks x 2048; local 7,297,024 \u00d7 CP 4; CP source: local process)",
188
+ "10:10:04 startup model context length: 1,000,000 tokens",
189
+ "10:10:04 startup prefill tests: skipped",
190
+ "10:10:04 startup calibrating padding text run=tmxrepeatabc up_to=8k",
191
+ "10:10:04 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
192
+ "10:10:04 startup token targeting: /tokenize exact",
193
+ "10:10:04 startup startup preparation done",
194
+ "10:10:04 hardware monitor interval=2s",
195
+ "10:10:04 decode warmup start",
196
+ "10:10:05 decode warmup start C=1 ctx=8k 3s",
197
+ "10:10:05 cell start C=1 ctx=8k",
198
+ "10:10:10 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
199
+ "10:10:21 cell done C=1 ctx=8k 212.1 tok/s",
200
+ "10:10:21 decode warmup done C=1 ctx=8k",
201
+ "10:10:23 cell start C=1 ctx=8k",
202
+ "10:10:28 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
203
+ "10:10:48 cell done C=1 ctx=8k 215.3 tok/s"
204
+ ],
205
+ "prefill": {},
206
+ "results": [
207
+ {
208
+ "concurrency": 1,
209
+ "context_tokens": 8192,
210
+ "benchmark_mode": "duration",
211
+ "request_count_target": 0,
212
+ "warmup_request_count": 0,
213
+ "measurement_seconds": 19.989377,
214
+ "measurement_wall_seconds": 20.000509,
215
+ "client_output_tokens": 4303,
216
+ "server_output_tokens": 4306,
217
+ "aggregate_source": "openai_continuous_usage",
218
+ "aggregate_tps": 215.26434223990609,
219
+ "per_request_avg_tps": 215.26434223990609,
220
+ "ttft_avg": 0.5413568730000407,
221
+ "ttft_p50": 0.5413568730000407,
222
+ "ttft_p90": 0.5413568730000407,
223
+ "ttft_p99": 0.5413568730000407,
224
+ "time_to_second_token_avg": 0.014796818839386106,
225
+ "time_to_second_token_p50": 0.014796818839386106,
226
+ "time_to_second_token_p90": 0.014796818839386106,
227
+ "time_to_second_token_p99": 0.014796818839386106,
228
+ "request_latency_avg": 0.0,
229
+ "request_latency_p50": 0.0,
230
+ "request_latency_p90": 0.0,
231
+ "request_latency_p99": 0.0,
232
+ "inter_token_latency_avg": 0.004611418287442928,
233
+ "inter_token_latency_p50": 0.004611418287442928,
234
+ "inter_token_latency_p90": 0.004611418287442928,
235
+ "inter_token_latency_p99": 0.004611418287442928,
236
+ "output_tps_per_user_avg": 216.8530238783671,
237
+ "output_tps_per_user_p50": 216.8530238783671,
238
+ "output_tps_per_user_p90": 216.8530238783671,
239
+ "output_tps_per_user_p99": 216.8530238783671,
240
+ "e2e_output_tps_per_user_avg": 0.0,
241
+ "e2e_output_tps_per_user_p50": 0.0,
242
+ "e2e_output_tps_per_user_p90": 0.0,
243
+ "e2e_output_tps_per_user_p99": 0.0,
244
+ "chunk_inter_token_latency_avg": 0.012825661236893406,
245
+ "chunk_inter_token_latency_p50": 0.012825661236893406,
246
+ "chunk_inter_token_latency_p90": 0.012825661236893406,
247
+ "chunk_inter_token_latency_p99": 0.012825661236893406,
248
+ "input_seq_len_avg": 8192.0,
249
+ "output_seq_len_avg": 5202.0,
250
+ "output_seq_len_p50": 5202.0,
251
+ "output_seq_len_p90": 5202.0,
252
+ "output_seq_len_p99": 5202.0,
253
+ "request_count": 1,
254
+ "completed_request_count": 0,
255
+ "request_samples": [
256
+ {
257
+ "ttft": 0.5413568730000407,
258
+ "time_to_second_token": 0.014796818839386106,
259
+ "latency": 0.0,
260
+ "inter_token_latency_avg": 0.004611418287442928,
261
+ "chunk_inter_token_latency_avg": 0.012825661236893406,
262
+ "input_tokens": 8192,
263
+ "output_tokens": 5202,
264
+ "output_tps_per_user": 216.8530238783671,
265
+ "e2e_output_tps_per_user": 0.0,
266
+ "completed": false
267
+ }
268
+ ],
269
+ "total_tokens": 4303,
270
+ "wall_time": 25.557856339029968,
271
+ "num_completed": 1,
272
+ "num_errors": 0,
273
+ "server_gen_throughput": 215.2383033924078,
274
+ "server_utilization": 0.006176305446378483,
275
+ "server_spec_accept_rate": 0.3924050632911392,
276
+ "server_spec_accept_length": 0.0,
277
+ "avg_running_reqs": 1,
278
+ "max_running_reqs": 1,
279
+ "effective_concurrency": 1,
280
+ "avg_queue_reqs": 0,
281
+ "max_queue_reqs": 0,
282
+ "queue_fraction": 0.0,
283
+ "underfilled": false,
284
+ "warmup_timed_out": false,
285
+ "warmup_duration": 5.552,
286
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
287
+ "timeout_reason": "",
288
+ "capacity_limited": false,
289
+ "hardware_summary": {
290
+ "samples": 9,
291
+ "duration_seconds": 19.391,
292
+ "gpu_count": 4,
293
+ "cpu_util_avg_pct": 10.9,
294
+ "cpu_temp_max_c": 75.12,
295
+ "gpu_util_avg_pct": 99.0,
296
+ "gpu_util_max_pct": 99.0,
297
+ "mem_util_avg_pct": 56.72,
298
+ "mem_util_max_pct": 60.0,
299
+ "temp_avg_c": 53.08,
300
+ "temp_max_c": 67.0,
301
+ "power_total_avg_w": 1148.08,
302
+ "power_total_max_w": 1154.84,
303
+ "power_limit_total_w": 1200.0,
304
+ "vram_used_avg_mb": 386578.0,
305
+ "vram_used_max_mb": 386578.0,
306
+ "vram_total_mb": 391548.0,
307
+ "vram_used_avg_pct": 98.73,
308
+ "vram_used_max_pct": 98.73,
309
+ "pcie_rx_avg_mb_s": 9520.33,
310
+ "pcie_rx_max_mb_s": 9579.0,
311
+ "pcie_tx_avg_mb_s": 9139.0,
312
+ "pcie_tx_max_mb_s": 9237.0
313
+ }
314
+ }
315
+ ],
316
+ "summary_table": {
317
+ "8192": {
318
+ "1": 215.26434223990609
319
+ }
320
+ },
321
+ "burst_results": [],
322
+ "burst_summary_table": {},
323
+ "methodology": {
324
+ "prefill": {
325
+ "name": "Prefill",
326
+ "present": false,
327
+ "mode": "skipped",
328
+ "formula": "prompt_tokens / TTFT",
329
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
330
+ },
331
+ "sustained_decode": {
332
+ "name": "Sustained Decode",
333
+ "present": true,
334
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
335
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
336
+ },
337
+ "burst_e2e_decode": {
338
+ "name": "Burst / E2E Decode",
339
+ "present": false,
340
+ "status": "not run; use --run-burst",
341
+ "formula": "sum(completion_tokens) / profiling_wall_time",
342
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
343
+ }
344
+ }
345
+ }
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c4-8k/result.json ADDED
@@ -0,0 +1,381 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T10:11:44.827451",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 8192,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": 0.0,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29188096,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 4
37
+ ],
38
+ "context_lengths": [
39
+ 8192
40
+ ],
41
+ "startup_diagnostics_available": true,
42
+ "nvidia_p2p_override_effective": true,
43
+ "p2pmark_status": "not_run",
44
+ "amd_fabric_status": "not_run"
45
+ },
46
+ "startup_diagnostics": {
47
+ "version": "0.4.29",
48
+ "server_url": "http://127.0.0.1:8001",
49
+ "hostname": "<host>",
50
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
51
+ "env": {},
52
+ "args": {
53
+ "concurrency": "4",
54
+ "contexts": "8k",
55
+ "max_tokens": 8192,
56
+ "duration": 20.0,
57
+ "request_count": 0,
58
+ "run_burst": false,
59
+ "standalone_prefill": false,
60
+ "prefill_only": false,
61
+ "skip_prefill": true,
62
+ "prefill_contexts": "8k,64k,128k",
63
+ "prefill_metric": "client",
64
+ "dcp_size": 0,
65
+ "kv_budget": 0
66
+ },
67
+ "nvidia_p2p_override": {
68
+ "effective": true,
69
+ "configured": true,
70
+ "params_path": "/proc/driver/nvidia/params",
71
+ "params_available": true,
72
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
73
+ "modprobe_available": true,
74
+ "runtime": {
75
+ "ForceP2P": "0x11",
76
+ "RMForceP2PType": "1",
77
+ "RMPcieP2PType": "2",
78
+ "GrdmaPciTopoCheckOverride": "1",
79
+ "EnableResizableBar": "1",
80
+ "DmaRemapPeerMmio": "1"
81
+ },
82
+ "expected": {
83
+ "ForceP2P": "0x11",
84
+ "RMForceP2PType": "1",
85
+ "RMPcieP2PType": "2",
86
+ "GrdmaPciTopoCheckOverride": "1",
87
+ "EnableResizableBar": "1"
88
+ },
89
+ "missing": [],
90
+ "mismatched": {},
91
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
92
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
93
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
94
+ },
95
+ "p2pmark": {
96
+ "status": "not_run"
97
+ },
98
+ "amd_fabric": {
99
+ "status": "not_run"
100
+ },
101
+ "nvidia_smi_query": {
102
+ "cmd": [
103
+ "nvidia-smi",
104
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
105
+ "--format=csv,noheader,nounits"
106
+ ],
107
+ "returncode": 0,
108
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
109
+ "stderr": ""
110
+ },
111
+ "nvidia_smi_topo": {
112
+ "cmd": [
113
+ "nvidia-smi",
114
+ "topo",
115
+ "-m"
116
+ ],
117
+ "returncode": 0,
118
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
119
+ "stderr": ""
120
+ }
121
+ },
122
+ "nvidia_p2p_override": {
123
+ "effective": true,
124
+ "configured": true,
125
+ "params_path": "/proc/driver/nvidia/params",
126
+ "params_available": true,
127
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
128
+ "modprobe_available": true,
129
+ "runtime": {
130
+ "ForceP2P": "0x11",
131
+ "RMForceP2PType": "1",
132
+ "RMPcieP2PType": "2",
133
+ "GrdmaPciTopoCheckOverride": "1",
134
+ "EnableResizableBar": "1",
135
+ "DmaRemapPeerMmio": "1"
136
+ },
137
+ "expected": {
138
+ "ForceP2P": "0x11",
139
+ "RMForceP2PType": "1",
140
+ "RMPcieP2PType": "2",
141
+ "GrdmaPciTopoCheckOverride": "1",
142
+ "EnableResizableBar": "1"
143
+ },
144
+ "missing": [],
145
+ "mismatched": {},
146
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
147
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
148
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
149
+ },
150
+ "p2pmark": {
151
+ "status": "not_run"
152
+ },
153
+ "amd_fabric": {
154
+ "status": "not_run"
155
+ },
156
+ "hardware_run_summary": {
157
+ "samples": 21,
158
+ "duration_seconds": 48.105,
159
+ "gpu_count": 4,
160
+ "cpu_util_avg_pct": 9.28,
161
+ "cpu_temp_max_c": 76.0,
162
+ "gpu_util_avg_pct": 76.04,
163
+ "gpu_util_max_pct": 100.0,
164
+ "mem_util_avg_pct": 35.81,
165
+ "mem_util_max_pct": 59.0,
166
+ "temp_avg_c": 57.21,
167
+ "temp_max_c": 75.0,
168
+ "power_total_avg_w": 970.98,
169
+ "power_total_max_w": 1182.0,
170
+ "power_limit_total_w": 1200.0,
171
+ "vram_used_avg_mb": 386578.0,
172
+ "vram_used_max_mb": 386578.0,
173
+ "vram_total_mb": 391548.0,
174
+ "vram_used_avg_pct": 98.73,
175
+ "vram_used_max_pct": 98.73,
176
+ "pcie_rx_avg_mb_s": 9912.24,
177
+ "pcie_rx_max_mb_s": 43706.0,
178
+ "pcie_tx_avg_mb_s": 10658.48,
179
+ "pcie_tx_max_mb_s": 52315.0
180
+ },
181
+ "event_log": [
182
+ "10:10:54 benchmark start engine=vllm",
183
+ "10:10:54 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
184
+ "10:10:54 startup decode concurrency=4 contexts=8k",
185
+ "10:10:54 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
186
+ "10:10:54 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
187
+ "10:10:54 startup KV cache budget from vLLM metrics: 29,188,096 tokens (3563 blocks x 2048; local 7,297,024 \u00d7 CP 4; CP source: local process)",
188
+ "10:10:54 startup model context length: 1,000,000 tokens",
189
+ "10:10:54 startup prefill tests: skipped",
190
+ "10:10:54 startup calibrating padding text run=tmxrepeatabc up_to=8k",
191
+ "10:10:54 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
192
+ "10:10:54 startup token targeting: /tokenize exact",
193
+ "10:10:54 startup startup preparation done",
194
+ "10:10:54 hardware monitor interval=2s",
195
+ "10:10:54 decode warmup start",
196
+ "10:10:55 decode warmup start C=1 ctx=8k 3s",
197
+ "10:10:55 cell start C=1 ctx=8k",
198
+ "10:11:00 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
199
+ "10:11:03 cell done C=1 ctx=8k 219.4 tok/s",
200
+ "10:11:03 decode warmup done C=1 ctx=8k",
201
+ "10:11:05 cell start C=4 ctx=8k",
202
+ "10:11:22 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
203
+ "10:11:42 cell done C=4 ctx=8k 335.7 tok/s"
204
+ ],
205
+ "prefill": {},
206
+ "results": [
207
+ {
208
+ "concurrency": 4,
209
+ "context_tokens": 8192,
210
+ "benchmark_mode": "duration",
211
+ "request_count_target": 0,
212
+ "warmup_request_count": 0,
213
+ "measurement_seconds": 19.991854,
214
+ "measurement_wall_seconds": 20.00195,
215
+ "client_output_tokens": 6712,
216
+ "server_output_tokens": 6712,
217
+ "aggregate_source": "openai_continuous_usage",
218
+ "aggregate_tps": 335.7367410580083,
219
+ "per_request_avg_tps": 83.93418526450208,
220
+ "ttft_avg": 4.347738385258708,
221
+ "ttft_p50": 1.9870852205203846,
222
+ "ttft_p90": 9.619254515157083,
223
+ "ttft_p99": 12.563032215915152,
224
+ "time_to_second_token_avg": 0.031753852672409266,
225
+ "time_to_second_token_p50": 0.03671444847714156,
226
+ "time_to_second_token_p90": 0.036814691172912715,
227
+ "time_to_second_token_p99": 0.036851499965414404,
228
+ "request_latency_avg": 0.0,
229
+ "request_latency_p50": 0.0,
230
+ "request_latency_p90": 0.0,
231
+ "request_latency_p99": 0.0,
232
+ "inter_token_latency_avg": 0.01416472446590232,
233
+ "inter_token_latency_p50": 0.014673835583566086,
234
+ "inter_token_latency_p90": 0.015258161326212282,
235
+ "inter_token_latency_p99": 0.015378115281519295,
236
+ "output_tps_per_user_avg": 71.30240784736792,
237
+ "output_tps_per_user_p50": 68.1721620239172,
238
+ "output_tps_per_user_p90": 79.55850802561741,
239
+ "output_tps_per_user_p99": 83.46057979574924,
240
+ "e2e_output_tps_per_user_avg": 0.0,
241
+ "e2e_output_tps_per_user_p50": 0.0,
242
+ "e2e_output_tps_per_user_p90": 0.0,
243
+ "e2e_output_tps_per_user_p99": 0.0,
244
+ "chunk_inter_token_latency_avg": 0.03829464632355565,
245
+ "chunk_inter_token_latency_p50": 0.04061762081634814,
246
+ "chunk_inter_token_latency_p90": 0.04061850702094191,
247
+ "chunk_inter_token_latency_p99": 0.04061852844490954,
248
+ "input_seq_len_avg": 8192.0,
249
+ "output_seq_len_avg": 2266.5,
250
+ "output_seq_len_p50": 2333.0,
251
+ "output_seq_len_p90": 2389.0,
252
+ "output_seq_len_p99": 2405.2,
253
+ "request_count": 4,
254
+ "completed_request_count": 0,
255
+ "request_samples": [
256
+ {
257
+ "ttft": 0.5266644728835672,
258
+ "time_to_second_token": 0.01673092390410602,
259
+ "latency": 0.0,
260
+ "inter_token_latency_avg": 0.01539144349877563,
261
+ "chunk_inter_token_latency_avg": 0.04061679015537416,
262
+ "input_tokens": 8192,
263
+ "output_tokens": 2347,
264
+ "output_tps_per_user": 64.97116401587341,
265
+ "e2e_output_tps_per_user": 0.0,
266
+ "completed": false
267
+ },
268
+ {
269
+ "ttft": 1.9872382539324462,
270
+ "time_to_second_token": 0.036709635984152555,
271
+ "latency": 0.0,
272
+ "inter_token_latency_avg": 0.014947169590231138,
273
+ "chunk_inter_token_latency_avg": 0.040618451477322126,
274
+ "input_tokens": 8192,
275
+ "output_tokens": 2319,
276
+ "output_tps_per_user": 66.90229838922544,
277
+ "e2e_output_tps_per_user": 0.0,
278
+ "completed": false
279
+ },
280
+ {
281
+ "ttft": 1.986932187108323,
282
+ "time_to_second_token": 0.03671926097013056,
283
+ "latency": 0.0,
284
+ "inter_token_latency_avg": 0.014400501576901033,
285
+ "chunk_inter_token_latency_avg": 0.04061853082535039,
286
+ "input_tokens": 8192,
287
+ "output_tokens": 2407,
288
+ "output_tps_per_user": 69.44202565860894,
289
+ "e2e_output_tps_per_user": 0.0,
290
+ "completed": false
291
+ },
292
+ {
293
+ "ttft": 12.890118627110496,
294
+ "time_to_second_token": 0.036855589831247926,
295
+ "latency": 0.0,
296
+ "inter_token_latency_avg": 0.011919783197701478,
297
+ "chunk_inter_token_latency_avg": 0.031324812836175914,
298
+ "input_tokens": 8192,
299
+ "output_tokens": 1993,
300
+ "output_tps_per_user": 83.89414332576389,
301
+ "e2e_output_tps_per_user": 0.0,
302
+ "completed": false
303
+ }
304
+ ],
305
+ "total_tokens": 6712,
306
+ "wall_time": 37.68367512291297,
307
+ "num_completed": 4,
308
+ "num_errors": 0,
309
+ "server_gen_throughput": 335.4828005774738,
310
+ "server_utilization": 0.02470522178551371,
311
+ "server_spec_accept_rate": 0.5364583333333334,
312
+ "server_spec_accept_length": 0.0,
313
+ "avg_running_reqs": 4,
314
+ "max_running_reqs": 4,
315
+ "effective_concurrency": 4,
316
+ "avg_queue_reqs": 0,
317
+ "max_queue_reqs": 0,
318
+ "queue_fraction": 0.0,
319
+ "underfilled": false,
320
+ "warmup_timed_out": false,
321
+ "warmup_duration": 17.659,
322
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
323
+ "timeout_reason": "",
324
+ "capacity_limited": false,
325
+ "hardware_summary": {
326
+ "samples": 8,
327
+ "duration_seconds": 16.884,
328
+ "gpu_count": 4,
329
+ "cpu_util_avg_pct": 10.94,
330
+ "cpu_temp_max_c": 75.75,
331
+ "gpu_util_avg_pct": 100.0,
332
+ "gpu_util_max_pct": 100.0,
333
+ "mem_util_avg_pct": 45.19,
334
+ "mem_util_max_pct": 48.0,
335
+ "temp_avg_c": 60.38,
336
+ "temp_max_c": 75.0,
337
+ "power_total_avg_w": 1179.07,
338
+ "power_total_max_w": 1180.51,
339
+ "power_limit_total_w": 1200.0,
340
+ "vram_used_avg_mb": 386578.0,
341
+ "vram_used_max_mb": 386578.0,
342
+ "vram_total_mb": 391548.0,
343
+ "vram_used_avg_pct": 98.73,
344
+ "vram_used_max_pct": 98.73,
345
+ "pcie_rx_avg_mb_s": 8712.12,
346
+ "pcie_rx_max_mb_s": 9059.0,
347
+ "pcie_tx_avg_mb_s": 8716.25,
348
+ "pcie_tx_max_mb_s": 9186.0
349
+ }
350
+ }
351
+ ],
352
+ "summary_table": {
353
+ "8192": {
354
+ "4": 335.7367410580083
355
+ }
356
+ },
357
+ "burst_results": [],
358
+ "burst_summary_table": {},
359
+ "methodology": {
360
+ "prefill": {
361
+ "name": "Prefill",
362
+ "present": false,
363
+ "mode": "skipped",
364
+ "formula": "prompt_tokens / TTFT",
365
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
366
+ },
367
+ "sustained_decode": {
368
+ "name": "Sustained Decode",
369
+ "present": true,
370
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
371
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
372
+ },
373
+ "burst_e2e_decode": {
374
+ "name": "Burst / E2E Decode",
375
+ "present": false,
376
+ "status": "not run; use --run-burst",
377
+ "formula": "sum(completion_tokens) / profiling_wall_time",
378
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
379
+ }
380
+ }
381
+ }
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-32k/result.json ADDED
@@ -0,0 +1,257 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T10:11:59.320922",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "standalone_cold",
15
+ "standalone_prefill": true,
16
+ "prefill_only": true,
17
+ "skip_prefill": false,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 0,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": 0.0,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29188096,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 4
38
+ ],
39
+ "context_lengths": [
40
+ 8192
41
+ ],
42
+ "startup_diagnostics_available": true,
43
+ "nvidia_p2p_override_effective": true,
44
+ "p2pmark_status": "not_run",
45
+ "amd_fabric_status": "not_run"
46
+ },
47
+ "startup_diagnostics": {
48
+ "version": "0.4.29",
49
+ "server_url": "http://127.0.0.1:8001",
50
+ "hostname": "<host>",
51
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
52
+ "env": {},
53
+ "args": {
54
+ "concurrency": "1,4",
55
+ "contexts": "8k",
56
+ "max_tokens": 8192,
57
+ "duration": 20.0,
58
+ "request_count": 0,
59
+ "run_burst": false,
60
+ "standalone_prefill": true,
61
+ "prefill_only": true,
62
+ "skip_prefill": false,
63
+ "prefill_contexts": "32k",
64
+ "prefill_metric": "client",
65
+ "dcp_size": 0,
66
+ "kv_budget": 0
67
+ },
68
+ "nvidia_p2p_override": {
69
+ "effective": true,
70
+ "configured": true,
71
+ "params_path": "/proc/driver/nvidia/params",
72
+ "params_available": true,
73
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
74
+ "modprobe_available": true,
75
+ "runtime": {
76
+ "ForceP2P": "0x11",
77
+ "RMForceP2PType": "1",
78
+ "RMPcieP2PType": "2",
79
+ "GrdmaPciTopoCheckOverride": "1",
80
+ "EnableResizableBar": "1",
81
+ "DmaRemapPeerMmio": "1"
82
+ },
83
+ "expected": {
84
+ "ForceP2P": "0x11",
85
+ "RMForceP2PType": "1",
86
+ "RMPcieP2PType": "2",
87
+ "GrdmaPciTopoCheckOverride": "1",
88
+ "EnableResizableBar": "1"
89
+ },
90
+ "missing": [],
91
+ "mismatched": {},
92
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
93
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
94
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
95
+ },
96
+ "p2pmark": {
97
+ "status": "not_run"
98
+ },
99
+ "amd_fabric": {
100
+ "status": "not_run"
101
+ },
102
+ "nvidia_smi_query": {
103
+ "cmd": [
104
+ "nvidia-smi",
105
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
106
+ "--format=csv,noheader,nounits"
107
+ ],
108
+ "returncode": 0,
109
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
110
+ "stderr": ""
111
+ },
112
+ "nvidia_smi_topo": {
113
+ "cmd": [
114
+ "nvidia-smi",
115
+ "topo",
116
+ "-m"
117
+ ],
118
+ "returncode": 0,
119
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
120
+ "stderr": ""
121
+ }
122
+ },
123
+ "nvidia_p2p_override": {
124
+ "effective": true,
125
+ "configured": true,
126
+ "params_path": "/proc/driver/nvidia/params",
127
+ "params_available": true,
128
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
129
+ "modprobe_available": true,
130
+ "runtime": {
131
+ "ForceP2P": "0x11",
132
+ "RMForceP2PType": "1",
133
+ "RMPcieP2PType": "2",
134
+ "GrdmaPciTopoCheckOverride": "1",
135
+ "EnableResizableBar": "1",
136
+ "DmaRemapPeerMmio": "1"
137
+ },
138
+ "expected": {
139
+ "ForceP2P": "0x11",
140
+ "RMForceP2PType": "1",
141
+ "RMPcieP2PType": "2",
142
+ "GrdmaPciTopoCheckOverride": "1",
143
+ "EnableResizableBar": "1"
144
+ },
145
+ "missing": [],
146
+ "mismatched": {},
147
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
148
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
149
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
150
+ },
151
+ "p2pmark": {
152
+ "status": "not_run"
153
+ },
154
+ "amd_fabric": {
155
+ "status": "not_run"
156
+ },
157
+ "hardware_run_summary": {
158
+ "samples": 4,
159
+ "duration_seconds": 7.229,
160
+ "gpu_count": 4,
161
+ "cpu_util_avg_pct": 8.88,
162
+ "cpu_temp_max_c": 73.88,
163
+ "gpu_util_avg_pct": 72.19,
164
+ "gpu_util_max_pct": 100.0,
165
+ "mem_util_avg_pct": 19.81,
166
+ "mem_util_max_pct": 32.0,
167
+ "temp_avg_c": 59.81,
168
+ "temp_max_c": 74.0,
169
+ "power_total_avg_w": 936.62,
170
+ "power_total_max_w": 1134.55,
171
+ "power_limit_total_w": 1200.0,
172
+ "vram_used_avg_mb": 386578.0,
173
+ "vram_used_max_mb": 386578.0,
174
+ "vram_total_mb": 391548.0,
175
+ "vram_used_avg_pct": 98.73,
176
+ "vram_used_max_pct": 98.73,
177
+ "pcie_rx_avg_mb_s": 43008.75,
178
+ "pcie_rx_max_mb_s": 58512.0,
179
+ "pcie_tx_avg_mb_s": 40414.0,
180
+ "pcie_tx_max_mb_s": 46717.0
181
+ },
182
+ "event_log": [],
183
+ "prefill": {
184
+ "32768": {
185
+ "ttft_seconds": 4.13,
186
+ "prefill_seconds": 4.13,
187
+ "tok_per_sec": 7935.0,
188
+ "client_ttft_seconds": 4.13,
189
+ "client_tok_per_sec": 7935.0,
190
+ "prompt_tokens": 32770,
191
+ "samples": 1,
192
+ "method": "client",
193
+ "server_validation": {
194
+ "method": "",
195
+ "tok_per_sec": 0.0,
196
+ "prefill_seconds": 0.0,
197
+ "prompt_tokens": 0,
198
+ "request_prompt_tokens": 0,
199
+ "cached_tokens": 0,
200
+ "token_source": "",
201
+ "samples": 0,
202
+ "invalid_reason": ""
203
+ },
204
+ "hardware_summary": {
205
+ "samples": 2,
206
+ "duration_seconds": 2.407,
207
+ "gpu_count": 4,
208
+ "cpu_util_avg_pct": 12.25,
209
+ "cpu_temp_max_c": 73.88,
210
+ "gpu_util_avg_pct": 94.5,
211
+ "gpu_util_max_pct": 100.0,
212
+ "mem_util_avg_pct": 26.25,
213
+ "mem_util_max_pct": 32.0,
214
+ "temp_avg_c": 61.75,
215
+ "temp_max_c": 74.0,
216
+ "power_total_avg_w": 1127.2,
217
+ "power_total_max_w": 1133.91,
218
+ "power_limit_total_w": 1200.0,
219
+ "vram_used_avg_mb": 386578.0,
220
+ "vram_used_max_mb": 386578.0,
221
+ "vram_total_mb": 391548.0,
222
+ "vram_used_avg_pct": 98.73,
223
+ "vram_used_max_pct": 98.73,
224
+ "pcie_rx_avg_mb_s": 54350.5,
225
+ "pcie_rx_max_mb_s": 58512.0,
226
+ "pcie_tx_avg_mb_s": 45629.0,
227
+ "pcie_tx_max_mb_s": 46717.0
228
+ }
229
+ }
230
+ },
231
+ "results": [],
232
+ "summary_table": {},
233
+ "burst_results": [],
234
+ "burst_summary_table": {},
235
+ "methodology": {
236
+ "prefill": {
237
+ "name": "Prefill",
238
+ "present": true,
239
+ "mode": "standalone_cold",
240
+ "formula": "prompt_tokens / TTFT",
241
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
242
+ },
243
+ "sustained_decode": {
244
+ "name": "Sustained Decode",
245
+ "present": false,
246
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
247
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
248
+ },
249
+ "burst_e2e_decode": {
250
+ "name": "Burst / E2E Decode",
251
+ "present": false,
252
+ "status": "not run; use --run-burst",
253
+ "formula": "sum(completion_tokens) / profiling_wall_time",
254
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
255
+ }
256
+ }
257
+ }
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-64k/result.json ADDED
@@ -0,0 +1,257 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T10:12:28.058934",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "standalone_cold",
15
+ "standalone_prefill": true,
16
+ "prefill_only": true,
17
+ "skip_prefill": false,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 0,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": 0.0,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29188096,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 4
38
+ ],
39
+ "context_lengths": [
40
+ 8192
41
+ ],
42
+ "startup_diagnostics_available": true,
43
+ "nvidia_p2p_override_effective": true,
44
+ "p2pmark_status": "not_run",
45
+ "amd_fabric_status": "not_run"
46
+ },
47
+ "startup_diagnostics": {
48
+ "version": "0.4.29",
49
+ "server_url": "http://127.0.0.1:8001",
50
+ "hostname": "<host>",
51
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
52
+ "env": {},
53
+ "args": {
54
+ "concurrency": "1,4",
55
+ "contexts": "8k",
56
+ "max_tokens": 8192,
57
+ "duration": 20.0,
58
+ "request_count": 0,
59
+ "run_burst": false,
60
+ "standalone_prefill": true,
61
+ "prefill_only": true,
62
+ "skip_prefill": false,
63
+ "prefill_contexts": "64k",
64
+ "prefill_metric": "client",
65
+ "dcp_size": 0,
66
+ "kv_budget": 0
67
+ },
68
+ "nvidia_p2p_override": {
69
+ "effective": true,
70
+ "configured": true,
71
+ "params_path": "/proc/driver/nvidia/params",
72
+ "params_available": true,
73
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
74
+ "modprobe_available": true,
75
+ "runtime": {
76
+ "ForceP2P": "0x11",
77
+ "RMForceP2PType": "1",
78
+ "RMPcieP2PType": "2",
79
+ "GrdmaPciTopoCheckOverride": "1",
80
+ "EnableResizableBar": "1",
81
+ "DmaRemapPeerMmio": "1"
82
+ },
83
+ "expected": {
84
+ "ForceP2P": "0x11",
85
+ "RMForceP2PType": "1",
86
+ "RMPcieP2PType": "2",
87
+ "GrdmaPciTopoCheckOverride": "1",
88
+ "EnableResizableBar": "1"
89
+ },
90
+ "missing": [],
91
+ "mismatched": {},
92
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
93
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
94
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
95
+ },
96
+ "p2pmark": {
97
+ "status": "not_run"
98
+ },
99
+ "amd_fabric": {
100
+ "status": "not_run"
101
+ },
102
+ "nvidia_smi_query": {
103
+ "cmd": [
104
+ "nvidia-smi",
105
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
106
+ "--format=csv,noheader,nounits"
107
+ ],
108
+ "returncode": 0,
109
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
110
+ "stderr": ""
111
+ },
112
+ "nvidia_smi_topo": {
113
+ "cmd": [
114
+ "nvidia-smi",
115
+ "topo",
116
+ "-m"
117
+ ],
118
+ "returncode": 0,
119
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
120
+ "stderr": ""
121
+ }
122
+ },
123
+ "nvidia_p2p_override": {
124
+ "effective": true,
125
+ "configured": true,
126
+ "params_path": "/proc/driver/nvidia/params",
127
+ "params_available": true,
128
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
129
+ "modprobe_available": true,
130
+ "runtime": {
131
+ "ForceP2P": "0x11",
132
+ "RMForceP2PType": "1",
133
+ "RMPcieP2PType": "2",
134
+ "GrdmaPciTopoCheckOverride": "1",
135
+ "EnableResizableBar": "1",
136
+ "DmaRemapPeerMmio": "1"
137
+ },
138
+ "expected": {
139
+ "ForceP2P": "0x11",
140
+ "RMForceP2PType": "1",
141
+ "RMPcieP2PType": "2",
142
+ "GrdmaPciTopoCheckOverride": "1",
143
+ "EnableResizableBar": "1"
144
+ },
145
+ "missing": [],
146
+ "mismatched": {},
147
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
148
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
149
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
150
+ },
151
+ "p2pmark": {
152
+ "status": "not_run"
153
+ },
154
+ "amd_fabric": {
155
+ "status": "not_run"
156
+ },
157
+ "hardware_run_summary": {
158
+ "samples": 6,
159
+ "duration_seconds": 12.079,
160
+ "gpu_count": 4,
161
+ "cpu_util_avg_pct": 9.23,
162
+ "cpu_temp_max_c": 74.12,
163
+ "gpu_util_avg_pct": 86.92,
164
+ "gpu_util_max_pct": 100.0,
165
+ "mem_util_avg_pct": 23.62,
166
+ "mem_util_max_pct": 37.0,
167
+ "temp_avg_c": 59.46,
168
+ "temp_max_c": 74.0,
169
+ "power_total_avg_w": 1002.06,
170
+ "power_total_max_w": 1132.26,
171
+ "power_limit_total_w": 1200.0,
172
+ "vram_used_avg_mb": 386844.67,
173
+ "vram_used_max_mb": 386978.0,
174
+ "vram_total_mb": 391548.0,
175
+ "vram_used_avg_pct": 98.8,
176
+ "vram_used_max_pct": 98.83,
177
+ "pcie_rx_avg_mb_s": 43749.5,
178
+ "pcie_rx_max_mb_s": 56738.0,
179
+ "pcie_tx_avg_mb_s": 46172.67,
180
+ "pcie_tx_max_mb_s": 58712.0
181
+ },
182
+ "event_log": [],
183
+ "prefill": {
184
+ "65536": {
185
+ "ttft_seconds": 8.202,
186
+ "prefill_seconds": 8.202,
187
+ "tok_per_sec": 7991.0,
188
+ "client_ttft_seconds": 8.202,
189
+ "client_tok_per_sec": 7991.0,
190
+ "prompt_tokens": 65538,
191
+ "samples": 1,
192
+ "method": "client",
193
+ "server_validation": {
194
+ "method": "",
195
+ "tok_per_sec": 0.0,
196
+ "prefill_seconds": 0.0,
197
+ "prompt_tokens": 0,
198
+ "request_prompt_tokens": 0,
199
+ "cached_tokens": 0,
200
+ "token_source": "",
201
+ "samples": 0,
202
+ "invalid_reason": ""
203
+ },
204
+ "hardware_summary": {
205
+ "samples": 4,
206
+ "duration_seconds": 7.26,
207
+ "gpu_count": 4,
208
+ "cpu_util_avg_pct": 11.1,
209
+ "cpu_temp_max_c": 74.12,
210
+ "gpu_util_avg_pct": 99.75,
211
+ "gpu_util_max_pct": 100.0,
212
+ "mem_util_avg_pct": 26.44,
213
+ "mem_util_max_pct": 29.0,
214
+ "temp_avg_c": 61.12,
215
+ "temp_max_c": 74.0,
216
+ "power_total_avg_w": 1131.9,
217
+ "power_total_max_w": 1132.26,
218
+ "power_limit_total_w": 1200.0,
219
+ "vram_used_avg_mb": 386928.0,
220
+ "vram_used_max_mb": 386978.0,
221
+ "vram_total_mb": 391548.0,
222
+ "vram_used_avg_pct": 98.82,
223
+ "vram_used_max_pct": 98.83,
224
+ "pcie_rx_avg_mb_s": 53121.5,
225
+ "pcie_rx_max_mb_s": 56738.0,
226
+ "pcie_tx_avg_mb_s": 52848.5,
227
+ "pcie_tx_max_mb_s": 57089.0
228
+ }
229
+ }
230
+ },
231
+ "results": [],
232
+ "summary_table": {},
233
+ "burst_results": [],
234
+ "burst_summary_table": {},
235
+ "methodology": {
236
+ "prefill": {
237
+ "name": "Prefill",
238
+ "present": true,
239
+ "mode": "standalone_cold",
240
+ "formula": "prompt_tokens / TTFT",
241
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
242
+ },
243
+ "sustained_decode": {
244
+ "name": "Sustained Decode",
245
+ "present": false,
246
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
247
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
248
+ },
249
+ "burst_e2e_decode": {
250
+ "name": "Burst / E2E Decode",
251
+ "present": false,
252
+ "status": "not run; use --run-burst",
253
+ "formula": "sum(completion_tokens) / profiling_wall_time",
254
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
255
+ }
256
+ }
257
+ }
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/result.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "status": "four-profile-cells-completed",
3
+ "graph_join": "pending",
4
+ "target_p8_dispatch": "pending Nsight classification",
5
+ "throughput_claim": false,
6
+ "production_restarted": false
7
+ }
results/speed-20260909/evidence/candidate-kernel-window-02/results-01/result.json ADDED
@@ -0,0 +1,23 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "scope": "representative component equality",
3
+ "results": [
4
+ {
5
+ "rank": 0,
6
+ "status": "passed"
7
+ },
8
+ {
9
+ "rank": 1,
10
+ "status": "passed"
11
+ },
12
+ {
13
+ "rank": 2,
14
+ "status": "passed"
15
+ },
16
+ {
17
+ "rank": 3,
18
+ "status": "passed"
19
+ }
20
+ ],
21
+ "full_model": "NOT TESTED",
22
+ "throughput": "NOT MEASURED"
23
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-command.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "512",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json",
20
+ "--contexts",
21
+ "0,8k,32k",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--skip-prefill",
25
+ "--cell-warmup-timeout-seconds",
26
+ "180"
27
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "e084ee41fd496dbfcf36f6f66715a1e622256de26889bd9d64e4c6c0f7335b4d"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json ADDED
@@ -0,0 +1,2366 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:26:52.004095",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 32768,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 512,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0,
42
+ 8192,
43
+ 32768
44
+ ],
45
+ "startup_diagnostics_available": true,
46
+ "nvidia_p2p_override_effective": true,
47
+ "p2pmark_status": "not_run",
48
+ "amd_fabric_status": "not_run"
49
+ },
50
+ "startup_diagnostics": {
51
+ "version": "0.4.29",
52
+ "server_url": "http://127.0.0.1:8001",
53
+ "hostname": "<host>",
54
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
55
+ "env": {},
56
+ "args": {
57
+ "concurrency": "1,2,4",
58
+ "contexts": "0,8k,32k",
59
+ "max_tokens": 512,
60
+ "duration": 20.0,
61
+ "request_count": 0,
62
+ "run_burst": false,
63
+ "standalone_prefill": false,
64
+ "prefill_only": false,
65
+ "skip_prefill": true,
66
+ "prefill_contexts": "8k,64k,128k",
67
+ "prefill_metric": "client",
68
+ "dcp_size": 0,
69
+ "kv_budget": 0
70
+ },
71
+ "nvidia_p2p_override": {
72
+ "effective": true,
73
+ "configured": true,
74
+ "params_path": "/proc/driver/nvidia/params",
75
+ "params_available": true,
76
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
77
+ "modprobe_available": true,
78
+ "runtime": {
79
+ "ForceP2P": "0x11",
80
+ "RMForceP2PType": "1",
81
+ "RMPcieP2PType": "2",
82
+ "GrdmaPciTopoCheckOverride": "1",
83
+ "EnableResizableBar": "1",
84
+ "DmaRemapPeerMmio": "1"
85
+ },
86
+ "expected": {
87
+ "ForceP2P": "0x11",
88
+ "RMForceP2PType": "1",
89
+ "RMPcieP2PType": "2",
90
+ "GrdmaPciTopoCheckOverride": "1",
91
+ "EnableResizableBar": "1"
92
+ },
93
+ "missing": [],
94
+ "mismatched": {},
95
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
96
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
97
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
98
+ },
99
+ "p2pmark": {
100
+ "status": "not_run"
101
+ },
102
+ "amd_fabric": {
103
+ "status": "not_run"
104
+ },
105
+ "nvidia_smi_query": {
106
+ "cmd": [
107
+ "nvidia-smi",
108
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
109
+ "--format=csv,noheader,nounits"
110
+ ],
111
+ "returncode": 0,
112
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
113
+ "stderr": ""
114
+ },
115
+ "nvidia_smi_topo": {
116
+ "cmd": [
117
+ "nvidia-smi",
118
+ "topo",
119
+ "-m"
120
+ ],
121
+ "returncode": 0,
122
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
123
+ "stderr": ""
124
+ }
125
+ },
126
+ "nvidia_p2p_override": {
127
+ "effective": true,
128
+ "configured": true,
129
+ "params_path": "/proc/driver/nvidia/params",
130
+ "params_available": true,
131
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
132
+ "modprobe_available": true,
133
+ "runtime": {
134
+ "ForceP2P": "0x11",
135
+ "RMForceP2PType": "1",
136
+ "RMPcieP2PType": "2",
137
+ "GrdmaPciTopoCheckOverride": "1",
138
+ "EnableResizableBar": "1",
139
+ "DmaRemapPeerMmio": "1"
140
+ },
141
+ "expected": {
142
+ "ForceP2P": "0x11",
143
+ "RMForceP2PType": "1",
144
+ "RMPcieP2PType": "2",
145
+ "GrdmaPciTopoCheckOverride": "1",
146
+ "EnableResizableBar": "1"
147
+ },
148
+ "missing": [],
149
+ "mismatched": {},
150
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
151
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
152
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
153
+ },
154
+ "p2pmark": {
155
+ "status": "not_run"
156
+ },
157
+ "amd_fabric": {
158
+ "status": "not_run"
159
+ },
160
+ "hardware_run_summary": {
161
+ "samples": 113,
162
+ "duration_seconds": 270.016,
163
+ "gpu_count": 4,
164
+ "cpu_util_avg_pct": 10.99,
165
+ "cpu_temp_max_c": 77.12,
166
+ "gpu_util_avg_pct": 91.62,
167
+ "gpu_util_max_pct": 100.0,
168
+ "mem_util_avg_pct": 33.39,
169
+ "mem_util_max_pct": 56.0,
170
+ "temp_avg_c": 67.84,
171
+ "temp_max_c": 84.0,
172
+ "power_total_avg_w": 1098.07,
173
+ "power_total_max_w": 1177.71,
174
+ "power_limit_total_w": 1200.0,
175
+ "vram_used_avg_mb": 384778.0,
176
+ "vram_used_max_mb": 384778.0,
177
+ "vram_total_mb": 391548.0,
178
+ "vram_used_avg_pct": 98.27,
179
+ "vram_used_max_pct": 98.27,
180
+ "pcie_rx_avg_mb_s": 16074.05,
181
+ "pcie_rx_max_mb_s": 74971.0,
182
+ "pcie_tx_avg_mb_s": 16658.8,
183
+ "pcie_tx_max_mb_s": 78099.0
184
+ },
185
+ "event_log": [
186
+ "02:22:19 benchmark start engine=vllm",
187
+ "02:22:19 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
188
+ "02:22:19 startup decode concurrency=1,2,4 contexts=0,8k,32k",
189
+ "02:22:19 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
190
+ "02:22:19 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
191
+ "02:22:19 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
192
+ "02:22:19 startup model context length: 1,000,000 tokens",
193
+ "02:22:19 startup prefill tests: skipped",
194
+ "02:22:19 startup calibrating padding text run=ftbfeppyynkm up_to=32k",
195
+ "02:22:19 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
196
+ "02:22:19 startup context 32k: 205,152 chars (32,768 prompt tokens via /tokenize)",
197
+ "02:22:19 startup token targeting: /tokenize exact",
198
+ "02:22:19 startup startup preparation done",
199
+ "02:22:19 hardware monitor interval=2s",
200
+ "02:22:19 decode warmup start",
201
+ "02:22:19 decode warmup start C=1 ctx=32k 3s",
202
+ "02:22:19 cell start C=1 ctx=32k",
203
+ "02:22:27 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
204
+ "02:22:30 cell done C=1 ctx=32k 172.8 tok/s",
205
+ "02:22:30 decode warmup done C=1 ctx=32k",
206
+ "02:22:32 cell start C=1 ctx=0",
207
+ "02:22:38 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
208
+ "02:22:58 cell done C=1 ctx=0 181.4 tok/s",
209
+ "02:23:00 cell start C=1 ctx=8k",
210
+ "02:23:06 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
211
+ "02:23:26 cell done C=1 ctx=8k 161.5 tok/s",
212
+ "02:23:28 cell start C=1 ctx=32k",
213
+ "02:23:37 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
214
+ "02:23:57 cell done C=1 ctx=32k 162.9 tok/s",
215
+ "02:23:59 cell start C=2 ctx=0",
216
+ "02:24:05 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
217
+ "02:24:25 cell done C=2 ctx=0 257.3 tok/s",
218
+ "02:24:27 cell start C=4 ctx=0",
219
+ "02:24:32 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
220
+ "02:24:52 cell done C=4 ctx=0 309.0 tok/s",
221
+ "02:24:54 cell start C=2 ctx=8k",
222
+ "02:25:00 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
223
+ "02:25:20 cell done C=2 ctx=8k 179.5 tok/s",
224
+ "02:25:22 cell start C=4 ctx=8k",
225
+ "02:25:31 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
226
+ "02:25:51 cell done C=4 ctx=8k 217.9 tok/s",
227
+ "02:25:53 cell start C=2 ctx=32k",
228
+ "02:25:58 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
229
+ "02:26:18 cell done C=2 ctx=32k 187.5 tok/s",
230
+ "02:26:20 cell start C=4 ctx=32k",
231
+ "02:26:29 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
232
+ "02:26:49 cell done C=4 ctx=32k 217.2 tok/s"
233
+ ],
234
+ "prefill": {},
235
+ "results": [
236
+ {
237
+ "concurrency": 1,
238
+ "context_tokens": 0,
239
+ "benchmark_mode": "duration",
240
+ "request_count_target": 0,
241
+ "warmup_request_count": 0,
242
+ "measurement_seconds": 19.997167,
243
+ "measurement_wall_seconds": 20.00031,
244
+ "client_output_tokens": 3627,
245
+ "server_output_tokens": 3627,
246
+ "aggregate_source": "openai_continuous_usage",
247
+ "aggregate_tps": 181.37569616773123,
248
+ "per_request_avg_tps": 181.37569616773123,
249
+ "ttft_avg": 0.08000486893579364,
250
+ "ttft_p50": 0.08374336501583457,
251
+ "ttft_p90": 0.08476541170384735,
252
+ "ttft_p99": 0.08516110226279125,
253
+ "time_to_second_token_avg": 0.013620631862431764,
254
+ "time_to_second_token_p50": 0.013637091615237296,
255
+ "time_to_second_token_p90": 0.01392088970169425,
256
+ "time_to_second_token_p99": 0.014014782523736358,
257
+ "request_latency_avg": 2.806835822191917,
258
+ "request_latency_p50": 2.8223699629306793,
259
+ "request_latency_p90": 2.8927299182862045,
260
+ "request_latency_p99": 2.9908720326423643,
261
+ "inter_token_latency_avg": 0.0052722589844875455,
262
+ "inter_token_latency_p50": 0.00534923418599077,
263
+ "inter_token_latency_p90": 0.005470371673862338,
264
+ "inter_token_latency_p99": 0.005685154040414228,
265
+ "output_tps_per_user_avg": 190.17533606228403,
266
+ "output_tps_per_user_p50": 186.94466223296394,
267
+ "output_tps_per_user_p90": 203.41232058199725,
268
+ "output_tps_per_user_p99": 211.44483943260568,
269
+ "e2e_output_tps_per_user_avg": 182.67236866069015,
270
+ "e2e_output_tps_per_user_p50": 181.4078263036614,
271
+ "e2e_output_tps_per_user_p90": 189.76977705704618,
272
+ "e2e_output_tps_per_user_p99": 196.55013284253087,
273
+ "chunk_inter_token_latency_avg": 0.014190152055605577,
274
+ "chunk_inter_token_latency_p50": 0.01418408028440155,
275
+ "chunk_inter_token_latency_p90": 0.014293591843861054,
276
+ "chunk_inter_token_latency_p99": 0.01437531164027474,
277
+ "input_seq_len_avg": 78.0,
278
+ "output_seq_len_avg": 512.0,
279
+ "output_seq_len_p50": 512.0,
280
+ "output_seq_len_p90": 512.0,
281
+ "output_seq_len_p99": 512.0,
282
+ "request_count": 10,
283
+ "completed_request_count": 9,
284
+ "request_samples": [
285
+ {
286
+ "ttft": 0.07054085680283606,
287
+ "time_to_second_token": 0.013252795208245516,
288
+ "latency": 2.5949868359602988,
289
+ "inter_token_latency_avg": 0.004940207395611473,
290
+ "chunk_inter_token_latency_avg": 0.014024699884208127,
291
+ "input_tokens": 78,
292
+ "output_tokens": 512,
293
+ "output_tps_per_user": 202.42065158809498,
294
+ "e2e_output_tps_per_user": 197.30350570758472,
295
+ "completed": true
296
+ },
297
+ {
298
+ "ttft": 0.08365814504213631,
299
+ "time_to_second_token": 0.013909297995269299,
300
+ "latency": 2.8654682198539376,
301
+ "inter_token_latency_avg": 0.0054438553323127225,
302
+ "chunk_inter_token_latency_avg": 0.014120863323917774,
303
+ "input_tokens": 78,
304
+ "output_tokens": 512,
305
+ "output_tps_per_user": 183.69334579197354,
306
+ "e2e_output_tps_per_user": 178.67935035974622,
307
+ "completed": true
308
+ },
309
+ {
310
+ "ttft": 0.08446813188493252,
311
+ "time_to_second_token": 0.013657593168318272,
312
+ "latency": 3.001776712015271,
313
+ "inter_token_latency_avg": 0.005709018747808882,
314
+ "chunk_inter_token_latency_avg": 0.014230773561611409,
315
+ "input_tokens": 78,
316
+ "output_tokens": 512,
317
+ "output_tps_per_user": 175.16144965959333,
318
+ "e2e_output_tps_per_user": 170.56565131930282,
319
+ "completed": true
320
+ },
321
+ {
322
+ "ttft": 0.07495116395875812,
323
+ "time_to_second_token": 0.01324723707512021,
324
+ "latency": 2.8223699629306793,
325
+ "inter_token_latency_avg": 0.005376553422645638,
326
+ "chunk_inter_token_latency_avg": 0.014384391617654037,
327
+ "input_tokens": 78,
328
+ "output_tokens": 512,
329
+ "output_tps_per_user": 185.99275807212763,
330
+ "e2e_output_tps_per_user": 181.4078263036614,
331
+ "completed": true
332
+ },
333
+ {
334
+ "ttft": 0.08520506788045168,
335
+ "time_to_second_token": 0.013890709029510617,
336
+ "latency": 2.7519469358958304,
337
+ "inter_token_latency_avg": 0.0052186729315369445,
338
+ "chunk_inter_token_latency_avg": 0.014184797170294567,
339
+ "input_tokens": 78,
340
+ "output_tokens": 512,
341
+ "output_tps_per_user": 191.61959623047144,
342
+ "e2e_output_tps_per_user": 186.05009904863252,
343
+ "completed": true
344
+ },
345
+ {
346
+ "ttft": 0.08382858498953283,
347
+ "time_to_second_token": 0.014025215059518814,
348
+ "latency": 2.826261157169938,
349
+ "inter_token_latency_avg": 0.005366795640274766,
350
+ "chunk_inter_token_latency_avg": 0.014283502980106277,
351
+ "input_tokens": 78,
352
+ "output_tokens": 512,
353
+ "output_tps_per_user": 186.3309257568083,
354
+ "e2e_output_tps_per_user": 181.158064144606,
355
+ "completed": true
356
+ },
357
+ {
358
+ "ttft": 0.07493811403401196,
359
+ "time_to_second_token": 0.013330142945051193,
360
+ "latency": 2.725051681045443,
361
+ "inter_token_latency_avg": 0.005186132225071294,
362
+ "chunk_inter_token_latency_avg": 0.0140963487606991,
363
+ "input_tokens": 78,
364
+ "output_tokens": 512,
365
+ "output_tps_per_user": 192.8219252038552,
366
+ "e2e_output_tps_per_user": 187.88634489441154,
367
+ "completed": true
368
+ },
369
+ {
370
+ "ttft": 0.08471656101755798,
371
+ "time_to_second_token": 0.013534185010939837,
372
+ "latency": 2.8092013269197196,
373
+ "inter_token_latency_avg": 0.005331672731706774,
374
+ "chunk_inter_token_latency_avg": 0.014264318146084616,
375
+ "input_tokens": 78,
376
+ "output_tokens": 512,
377
+ "output_tps_per_user": 187.5583987091196,
378
+ "e2e_output_tps_per_user": 182.25820808699618,
379
+ "completed": true
380
+ },
381
+ {
382
+ "ttft": 0.08452034182846546,
383
+ "time_to_second_token": 0.01361659006215632,
384
+ "latency": 2.8644595679361373,
385
+ "inter_token_latency_avg": 0.005440194180249847,
386
+ "chunk_inter_token_latency_avg": 0.01418336339850853,
387
+ "input_tokens": 78,
388
+ "output_tokens": 512,
389
+ "output_tps_per_user": 183.81696808367857,
390
+ "e2e_output_tps_per_user": 178.74226808127003,
391
+ "completed": true
392
+ },
393
+ {
394
+ "ttft": 0.07322172191925347,
395
+ "time_to_second_token": 0.013742553070187569,
396
+ "latency": 0.0,
397
+ "inter_token_latency_avg": 0.00470948723765711,
398
+ "chunk_inter_token_latency_avg": 0.01412846171297133,
399
+ "input_tokens": 78,
400
+ "output_tokens": 43,
401
+ "output_tps_per_user": 212.33734152711773,
402
+ "e2e_output_tps_per_user": 0.0,
403
+ "completed": false
404
+ }
405
+ ],
406
+ "total_tokens": 3627,
407
+ "wall_time": 25.549715572968125,
408
+ "num_completed": 1,
409
+ "num_errors": 0,
410
+ "server_gen_throughput": 181.29852055793228,
411
+ "server_utilization": 0.005862646566164198,
412
+ "server_spec_accept_rate": 0.5046296296296297,
413
+ "server_spec_accept_length": 0.0,
414
+ "avg_running_reqs": 1,
415
+ "max_running_reqs": 1,
416
+ "effective_concurrency": 1,
417
+ "avg_queue_reqs": 0,
418
+ "max_queue_reqs": 0,
419
+ "queue_fraction": 0.0,
420
+ "underfilled": false,
421
+ "warmup_timed_out": false,
422
+ "warmup_duration": 5.538,
423
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
424
+ "timeout_reason": "",
425
+ "capacity_limited": false,
426
+ "hardware_summary": {
427
+ "samples": 8,
428
+ "duration_seconds": 16.928,
429
+ "gpu_count": 4,
430
+ "cpu_util_avg_pct": 11.61,
431
+ "cpu_temp_max_c": 76.0,
432
+ "gpu_util_avg_pct": 99.0,
433
+ "gpu_util_max_pct": 99.0,
434
+ "mem_util_avg_pct": 44.03,
435
+ "mem_util_max_pct": 56.0,
436
+ "temp_avg_c": 67.31,
437
+ "temp_max_c": 82.0,
438
+ "power_total_avg_w": 1151.08,
439
+ "power_total_max_w": 1152.57,
440
+ "power_limit_total_w": 1200.0,
441
+ "vram_used_avg_mb": 384778.0,
442
+ "vram_used_max_mb": 384778.0,
443
+ "vram_total_mb": 391548.0,
444
+ "vram_used_avg_pct": 98.27,
445
+ "vram_used_max_pct": 98.27,
446
+ "pcie_rx_avg_mb_s": 8368.88,
447
+ "pcie_rx_max_mb_s": 8659.0,
448
+ "pcie_tx_avg_mb_s": 8253.75,
449
+ "pcie_tx_max_mb_s": 8495.0
450
+ }
451
+ },
452
+ {
453
+ "concurrency": 1,
454
+ "context_tokens": 8192,
455
+ "benchmark_mode": "duration",
456
+ "request_count_target": 0,
457
+ "warmup_request_count": 0,
458
+ "measurement_seconds": 19.997908,
459
+ "measurement_wall_seconds": 20.001023,
460
+ "client_output_tokens": 3229,
461
+ "server_output_tokens": 3229,
462
+ "aggregate_source": "openai_continuous_usage",
463
+ "aggregate_tps": 161.46688548018994,
464
+ "per_request_avg_tps": 161.46688548018994,
465
+ "ttft_avg": 0.6216339806560427,
466
+ "ttft_p50": 0.6253726300783455,
467
+ "ttft_p90": 0.6310847522690892,
468
+ "ttft_p99": 0.631135935941711,
469
+ "time_to_second_token_avg": 0.017570865049492568,
470
+ "time_to_second_token_p50": 0.01762134348973632,
471
+ "time_to_second_token_p90": 0.018190144654363395,
472
+ "time_to_second_token_p99": 0.0183273093868047,
473
+ "request_latency_avg": 3.1911156099023565,
474
+ "request_latency_p50": 3.198385320138186,
475
+ "request_latency_p90": 3.315028171055019,
476
+ "request_latency_p99": 3.317255131038837,
477
+ "inter_token_latency_avg": 0.00510038525208451,
478
+ "inter_token_latency_p50": 0.0050694173082997274,
479
+ "inter_token_latency_p90": 0.005415639657450936,
480
+ "inter_token_latency_p99": 0.005569194878906556,
481
+ "output_tps_per_user_avg": 196.5708670855281,
482
+ "output_tps_per_user_p50": 197.2764163282376,
483
+ "output_tps_per_user_p90": 207.7094766010595,
484
+ "output_tps_per_user_p99": 208.07124550938897,
485
+ "e2e_output_tps_per_user_avg": 160.58636690583256,
486
+ "e2e_output_tps_per_user_p50": 160.0807747510169,
487
+ "e2e_output_tps_per_user_p90": 165.8642163873898,
488
+ "e2e_output_tps_per_user_p99": 166.33389003015049,
489
+ "chunk_inter_token_latency_avg": 0.014235832994823009,
490
+ "chunk_inter_token_latency_p50": 0.014211497073928412,
491
+ "chunk_inter_token_latency_p90": 0.014303687557771127,
492
+ "chunk_inter_token_latency_p99": 0.014314021112869283,
493
+ "input_seq_len_avg": 8192.0,
494
+ "output_seq_len_avg": 512.0,
495
+ "output_seq_len_p50": 512.0,
496
+ "output_seq_len_p90": 512.0,
497
+ "output_seq_len_p99": 512.0,
498
+ "request_count": 8,
499
+ "completed_request_count": 7,
500
+ "request_samples": [
501
+ {
502
+ "ttft": 0.5874758099671453,
503
+ "time_to_second_token": 0.01834254991263151,
504
+ "latency": 3.317502571037039,
505
+ "inter_token_latency_avg": 0.00534251812342445,
506
+ "chunk_inter_token_latency_avg": 0.014218889380572364,
507
+ "input_tokens": 8192,
508
+ "output_tokens": 512,
509
+ "output_tps_per_user": 187.17765235375927,
510
+ "e2e_output_tps_per_user": 154.33296253330428,
511
+ "completed": true
512
+ },
513
+ {
514
+ "ttft": 0.6288027700502425,
515
+ "time_to_second_token": 0.017373620066791773,
516
+ "latency": 3.3133785710670054,
517
+ "inter_token_latency_avg": 0.005253572996118909,
518
+ "chunk_inter_token_latency_avg": 0.01420410476728446,
519
+ "input_tokens": 8192,
520
+ "output_tokens": 512,
521
+ "output_tps_per_user": 190.34664612802612,
522
+ "e2e_output_tps_per_user": 154.52505321030097,
523
+ "completed": true
524
+ },
525
+ {
526
+ "ttft": 0.6219424901064485,
527
+ "time_to_second_token": 0.017552112927660346,
528
+ "latency": 3.1045689159072936,
529
+ "inter_token_latency_avg": 0.004858368739336292,
530
+ "chunk_inter_token_latency_avg": 0.014267967964372673,
531
+ "input_tokens": 8192,
532
+ "output_tokens": 512,
533
+ "output_tps_per_user": 205.8304039179643,
534
+ "e2e_output_tps_per_user": 164.91822661001254,
535
+ "completed": true
536
+ },
537
+ {
538
+ "ttft": 0.6217653600033373,
539
+ "time_to_second_token": 0.01812482811510563,
540
+ "latency": 3.077180569060147,
541
+ "inter_token_latency_avg": 0.004805117825942876,
542
+ "chunk_inter_token_latency_avg": 0.014193151497438206,
543
+ "input_tokens": 8192,
544
+ "output_tokens": 512,
545
+ "output_tps_per_user": 208.11144205475892,
546
+ "e2e_output_tps_per_user": 166.38607599045721,
547
+ "completed": true
548
+ },
549
+ {
550
+ "ttft": 0.6305665450636297,
551
+ "time_to_second_token": 0.016708664130419493,
552
+ "latency": 3.198385320138186,
553
+ "inter_token_latency_avg": 0.005025085665507938,
554
+ "chunk_inter_token_latency_avg": 0.014186844061185394,
555
+ "input_tokens": 8192,
556
+ "output_tokens": 512,
557
+ "output_tps_per_user": 199.00158257280566,
558
+ "e2e_output_tps_per_user": 160.0807747510169,
559
+ "completed": true
560
+ },
561
+ {
562
+ "ttft": 0.6203168679494411,
563
+ "time_to_second_token": 0.01790391909889877,
564
+ "latency": 3.233442581957206,
565
+ "inter_token_latency_avg": 0.005113748951091517,
566
+ "chunk_inter_token_latency_avg": 0.01420177018482481,
567
+ "input_tokens": 8192,
568
+ "output_tokens": 512,
569
+ "output_tps_per_user": 195.55125008366954,
570
+ "e2e_output_tps_per_user": 158.34516526039127,
571
+ "completed": true
572
+ },
573
+ {
574
+ "ttft": 0.6311416230164468,
575
+ "time_to_second_token": 0.01687065209262073,
576
+ "latency": 3.093350740149617,
577
+ "inter_token_latency_avg": 0.004818413145074698,
578
+ "chunk_inter_token_latency_avg": 0.014315169285657967,
579
+ "input_tokens": 8192,
580
+ "output_tokens": 512,
581
+ "output_tps_per_user": 207.53720569233118,
582
+ "e2e_output_tps_per_user": 165.51630998534486,
583
+ "completed": true
584
+ },
585
+ {
586
+ "ttft": 0.6310603790916502,
587
+ "time_to_second_token": 0.01769057405181229,
588
+ "latency": 0.0,
589
+ "inter_token_latency_avg": 0.005586256570179402,
590
+ "chunk_inter_token_latency_avg": 0.014298766817248195,
591
+ "input_tokens": 8192,
592
+ "output_tokens": 280,
593
+ "output_tps_per_user": 179.01075388090973,
594
+ "e2e_output_tps_per_user": 0.0,
595
+ "completed": false
596
+ }
597
+ ],
598
+ "total_tokens": 3229,
599
+ "wall_time": 26.068293957971036,
600
+ "num_completed": 1,
601
+ "num_errors": 0,
602
+ "server_gen_throughput": 161.40326963780146,
603
+ "server_utilization": 0.006141820212172022,
604
+ "server_spec_accept_rate": 0.5918367346938775,
605
+ "server_spec_accept_length": 0.0,
606
+ "avg_running_reqs": 1,
607
+ "max_running_reqs": 1,
608
+ "effective_concurrency": 1,
609
+ "avg_queue_reqs": 0,
610
+ "max_queue_reqs": 0,
611
+ "queue_fraction": 0.0,
612
+ "underfilled": false,
613
+ "warmup_timed_out": false,
614
+ "warmup_duration": 6.056,
615
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
616
+ "timeout_reason": "",
617
+ "capacity_limited": false,
618
+ "hardware_summary": {
619
+ "samples": 8,
620
+ "duration_seconds": 16.899,
621
+ "gpu_count": 4,
622
+ "cpu_util_avg_pct": 11.6,
623
+ "cpu_temp_max_c": 75.88,
624
+ "gpu_util_avg_pct": 99.12,
625
+ "gpu_util_max_pct": 100.0,
626
+ "mem_util_avg_pct": 40.22,
627
+ "mem_util_max_pct": 55.0,
628
+ "temp_avg_c": 67.81,
629
+ "temp_max_c": 83.0,
630
+ "power_total_avg_w": 1153.3,
631
+ "power_total_max_w": 1156.82,
632
+ "power_limit_total_w": 1200.0,
633
+ "vram_used_avg_mb": 384778.0,
634
+ "vram_used_max_mb": 384778.0,
635
+ "vram_total_mb": 391548.0,
636
+ "vram_used_avg_pct": 98.27,
637
+ "vram_used_max_pct": 98.27,
638
+ "pcie_rx_avg_mb_s": 12913.25,
639
+ "pcie_rx_max_mb_s": 39230.0,
640
+ "pcie_tx_avg_mb_s": 11280.75,
641
+ "pcie_tx_max_mb_s": 30681.0
642
+ }
643
+ },
644
+ {
645
+ "concurrency": 1,
646
+ "context_tokens": 32768,
647
+ "benchmark_mode": "duration",
648
+ "request_count_target": 0,
649
+ "warmup_request_count": 0,
650
+ "measurement_seconds": 19.99527,
651
+ "measurement_wall_seconds": 20.000365,
652
+ "client_output_tokens": 3258,
653
+ "server_output_tokens": 3258,
654
+ "aggregate_source": "openai_continuous_usage",
655
+ "aggregate_tps": 162.9385363574482,
656
+ "per_request_avg_tps": 162.9385363574482,
657
+ "ttft_avg": 0.6264620197180193,
658
+ "ttft_p50": 0.6282767069060355,
659
+ "ttft_p90": 0.6339533890830353,
660
+ "ttft_p99": 0.6355810678633861,
661
+ "time_to_second_token_avg": 0.011372056760592386,
662
+ "time_to_second_token_p50": 0.011673877947032452,
663
+ "time_to_second_token_p90": 0.013470914796926081,
664
+ "time_to_second_token_p99": 0.013671491749119014,
665
+ "request_latency_avg": 3.179016282981528,
666
+ "request_latency_p50": 3.1919217659160495,
667
+ "request_latency_p90": 3.2463170818053184,
668
+ "request_latency_p99": 3.250220402767882,
669
+ "inter_token_latency_avg": 0.004995384074006483,
670
+ "inter_token_latency_p50": 0.005010807180026302,
671
+ "inter_token_latency_p90": 0.005126464669228168,
672
+ "inter_token_latency_p99": 0.00513858911541697,
673
+ "output_tps_per_user_avg": 200.31161173007573,
674
+ "output_tps_per_user_p50": 199.57010444464703,
675
+ "output_tps_per_user_p90": 206.65625468620667,
676
+ "output_tps_per_user_p99": 210.60956431602483,
677
+ "e2e_output_tps_per_user_avg": 161.12331942575418,
678
+ "e2e_output_tps_per_user_p50": 160.40493393893104,
679
+ "e2e_output_tps_per_user_p90": 165.31501161627682,
680
+ "e2e_output_tps_per_user_p99": 167.56189180026823,
681
+ "chunk_inter_token_latency_avg": 0.014215011037926982,
682
+ "chunk_inter_token_latency_p50": 0.014209193651253612,
683
+ "chunk_inter_token_latency_p90": 0.014249588821529074,
684
+ "chunk_inter_token_latency_p99": 0.014264281110023109,
685
+ "input_seq_len_avg": 32768.0,
686
+ "output_seq_len_avg": 512.0,
687
+ "output_seq_len_p50": 512.0,
688
+ "output_seq_len_p90": 512.0,
689
+ "output_seq_len_p99": 512.0,
690
+ "request_count": 8,
691
+ "completed_request_count": 7,
692
+ "request_samples": [
693
+ {
694
+ "ttft": 0.6073537350166589,
695
+ "time_to_second_token": 0.01270562899298966,
696
+ "latency": 3.2036298899911344,
697
+ "inter_token_latency_avg": 0.005080775254353181,
698
+ "chunk_inter_token_latency_avg": 0.014187301393303145,
699
+ "input_tokens": 32768,
700
+ "output_tokens": 512,
701
+ "output_tps_per_user": 196.82035711837585,
702
+ "e2e_output_tps_per_user": 159.81871114375727,
703
+ "completed": true
704
+ },
705
+ {
706
+ "ttft": 0.624475359916687,
707
+ "time_to_second_token": 0.013375401962548494,
708
+ "latency": 3.1919217659160495,
709
+ "inter_token_latency_avg": 0.005024356958902862,
710
+ "chunk_inter_token_latency_avg": 0.014184786773477141,
711
+ "input_tokens": 32768,
712
+ "output_tokens": 512,
713
+ "output_tps_per_user": 199.03044472747092,
714
+ "e2e_output_tps_per_user": 160.40493393893104,
715
+ "completed": true
716
+ },
717
+ {
718
+ "ttft": 0.6302267559804022,
719
+ "time_to_second_token": 0.01065197098068893,
720
+ "latency": 3.1838252879679203,
721
+ "inter_token_latency_avg": 0.004997257401149742,
722
+ "chunk_inter_token_latency_avg": 0.014265913586522447,
723
+ "input_tokens": 32768,
724
+ "output_tokens": 512,
725
+ "output_tps_per_user": 200.10976416182314,
726
+ "e2e_output_tps_per_user": 160.81284420188285,
727
+ "completed": true
728
+ },
729
+ {
730
+ "ttft": 0.6267525688745081,
731
+ "time_to_second_token": 0.010681336978450418,
732
+ "latency": 3.2434257329441607,
733
+ "inter_token_latency_avg": 0.005120691123423978,
734
+ "chunk_inter_token_latency_avg": 0.014221049804726372,
735
+ "input_tokens": 32768,
736
+ "output_tokens": 512,
737
+ "output_tps_per_user": 195.28613929194475,
738
+ "e2e_output_tps_per_user": 157.8577843788769,
739
+ "completed": true
740
+ },
741
+ {
742
+ "ttft": 0.6241466680075973,
743
+ "time_to_second_token": 0.012666418915614486,
744
+ "latency": 3.2506541050970554,
745
+ "inter_token_latency_avg": 0.005139936276104614,
746
+ "chunk_inter_token_latency_avg": 0.014197337497780854,
747
+ "input_tokens": 32768,
748
+ "output_tokens": 512,
749
+ "output_tps_per_user": 194.5549412059767,
750
+ "e2e_output_tps_per_user": 157.50676123835487,
751
+ "completed": true
752
+ },
753
+ {
754
+ "ttft": 0.6298008449375629,
755
+ "time_to_second_token": 0.01369377807714045,
756
+ "latency": 3.0510415688622743,
757
+ "inter_token_latency_avg": 0.004738240164236226,
758
+ "chunk_inter_token_latency_avg": 0.014242592493674773,
759
+ "input_tokens": 32768,
760
+ "output_tokens": 512,
761
+ "output_tps_per_user": 211.0488209415602,
762
+ "e2e_output_tps_per_user": 167.81154515404506,
763
+ "completed": true
764
+ },
765
+ {
766
+ "ttft": 0.6331783039495349,
767
+ "time_to_second_token": 0.00986177520826459,
768
+ "latency": 3.1286156300920993,
769
+ "inter_token_latency_avg": 0.00488343899440815,
770
+ "chunk_inter_token_latency_avg": 0.01417862117126457,
771
+ "input_tokens": 32768,
772
+ "output_tokens": 512,
773
+ "output_tps_per_user": 204.77372629105514,
774
+ "e2e_output_tps_per_user": 163.6506559244313,
775
+ "completed": true
776
+ },
777
+ {
778
+ "ttft": 0.6357619210612029,
779
+ "time_to_second_token": 0.007340142969042063,
780
+ "latency": 0.0,
781
+ "inter_token_latency_avg": 0.004978376419473111,
782
+ "chunk_inter_token_latency_avg": 0.014242485582666553,
783
+ "input_tokens": 32768,
784
+ "output_tokens": 330,
785
+ "output_tps_per_user": 200.86870010239915,
786
+ "e2e_output_tps_per_user": 0.0,
787
+ "completed": false
788
+ }
789
+ ],
790
+ "total_tokens": 3258,
791
+ "wall_time": 29.116754403105006,
792
+ "num_completed": 1,
793
+ "num_errors": 0,
794
+ "server_gen_throughput": 162.8569571510368,
795
+ "server_utilization": 0.006979341150195384,
796
+ "server_spec_accept_rate": 0.6607142857142857,
797
+ "server_spec_accept_length": 0.0,
798
+ "avg_running_reqs": 0.9,
799
+ "max_running_reqs": 1,
800
+ "effective_concurrency": 0.9,
801
+ "avg_queue_reqs": 0,
802
+ "max_queue_reqs": 0,
803
+ "queue_fraction": 0.0,
804
+ "underfilled": false,
805
+ "warmup_timed_out": false,
806
+ "warmup_duration": 9.107,
807
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
808
+ "timeout_reason": "",
809
+ "capacity_limited": false,
810
+ "hardware_summary": {
811
+ "samples": 8,
812
+ "duration_seconds": 16.912,
813
+ "gpu_count": 4,
814
+ "cpu_util_avg_pct": 11.53,
815
+ "cpu_temp_max_c": 76.5,
816
+ "gpu_util_avg_pct": 97.97,
817
+ "gpu_util_max_pct": 100.0,
818
+ "mem_util_avg_pct": 38.0,
819
+ "mem_util_max_pct": 56.0,
820
+ "temp_avg_c": 68.16,
821
+ "temp_max_c": 84.0,
822
+ "power_total_avg_w": 1149.9,
823
+ "power_total_max_w": 1155.94,
824
+ "power_limit_total_w": 1200.0,
825
+ "vram_used_avg_mb": 384778.0,
826
+ "vram_used_max_mb": 384778.0,
827
+ "vram_total_mb": 391548.0,
828
+ "vram_used_avg_pct": 98.27,
829
+ "vram_used_max_pct": 98.27,
830
+ "pcie_rx_avg_mb_s": 8327.0,
831
+ "pcie_rx_max_mb_s": 8635.0,
832
+ "pcie_tx_avg_mb_s": 8132.62,
833
+ "pcie_tx_max_mb_s": 8389.0
834
+ }
835
+ },
836
+ {
837
+ "concurrency": 2,
838
+ "context_tokens": 0,
839
+ "benchmark_mode": "duration",
840
+ "request_count_target": 0,
841
+ "warmup_request_count": 0,
842
+ "measurement_seconds": 19.997106,
843
+ "measurement_wall_seconds": 20.0002,
844
+ "client_output_tokens": 5145,
845
+ "server_output_tokens": 5145,
846
+ "aggregate_source": "openai_continuous_usage",
847
+ "aggregate_tps": 257.2872334123046,
848
+ "per_request_avg_tps": 128.6436167061523,
849
+ "ttft_avg": 0.12196867128035851,
850
+ "ttft_p50": 0.1225746686104685,
851
+ "ttft_p90": 0.14186890744604172,
852
+ "ttft_p99": 0.16953610270284114,
853
+ "time_to_second_token_avg": 0.01940517939094986,
854
+ "time_to_second_token_p50": 0.02044149092398584,
855
+ "time_to_second_token_p90": 0.020944337057881058,
856
+ "time_to_second_token_p99": 0.021031919752713294,
857
+ "request_latency_avg": 3.9966156358908242,
858
+ "request_latency_p50": 3.9488427881151438,
859
+ "request_latency_p90": 4.166194258979521,
860
+ "request_latency_p99": 4.237951860819012,
861
+ "inter_token_latency_avg": 0.007504969642701137,
862
+ "inter_token_latency_p50": 0.00753824443737712,
863
+ "inter_token_latency_p90": 0.007882004104182853,
864
+ "inter_token_latency_p99": 0.008046831693049199,
865
+ "output_tps_per_user_avg": 133.69662528358077,
866
+ "output_tps_per_user_p50": 132.65955785556645,
867
+ "output_tps_per_user_p90": 139.45345815466882,
868
+ "output_tps_per_user_p99": 155.8549841566213,
869
+ "e2e_output_tps_per_user_avg": 128.2813132485478,
870
+ "e2e_output_tps_per_user_p50": 129.6588243765429,
871
+ "e2e_output_tps_per_user_p90": 134.02921814320612,
872
+ "e2e_output_tps_per_user_p99": 135.85115936191062,
873
+ "chunk_inter_token_latency_avg": 0.02096237495683361,
874
+ "chunk_inter_token_latency_p50": 0.02092384556948202,
875
+ "chunk_inter_token_latency_p90": 0.021176504729596514,
876
+ "chunk_inter_token_latency_p99": 0.021264344922454187,
877
+ "input_seq_len_avg": 78.0,
878
+ "output_seq_len_avg": 512.0,
879
+ "output_seq_len_p50": 512.0,
880
+ "output_seq_len_p90": 512.0,
881
+ "output_seq_len_p99": 512.0,
882
+ "request_count": 14,
883
+ "completed_request_count": 12,
884
+ "request_samples": [
885
+ {
886
+ "ttft": 0.07110429811291397,
887
+ "time_to_second_token": 0.013481322908774018,
888
+ "latency": 3.940448695095256,
889
+ "inter_token_latency_avg": 0.007572102538125914,
890
+ "chunk_inter_token_latency_avg": 0.020915375118823472,
891
+ "input_tokens": 78,
892
+ "output_tokens": 512,
893
+ "output_tps_per_user": 132.06371611648814,
894
+ "e2e_output_tps_per_user": 129.93444138412337,
895
+ "completed": true
896
+ },
897
+ {
898
+ "ttft": 0.17250841204077005,
899
+ "time_to_second_token": 0.020715914899483323,
900
+ "latency": 4.167202410986647,
901
+ "inter_token_latency_avg": 0.007817405085999759,
902
+ "chunk_inter_token_latency_avg": 0.021135947084369718,
903
+ "input_tokens": 78,
904
+ "output_tokens": 512,
905
+ "output_tps_per_user": 127.91968549652191,
906
+ "e2e_output_tps_per_user": 122.86420228835883,
907
+ "completed": true
908
+ },
909
+ {
910
+ "ttft": 0.12372587202116847,
911
+ "time_to_second_token": 0.020451861899346113,
912
+ "latency": 4.157120890915394,
913
+ "inter_token_latency_avg": 0.007893140937170695,
914
+ "chunk_inter_token_latency_avg": 0.02089841978701671,
915
+ "input_tokens": 78,
916
+ "output_tokens": 512,
917
+ "output_tps_per_user": 126.69227725185547,
918
+ "e2e_output_tps_per_user": 123.16216281294098,
919
+ "completed": true
920
+ },
921
+ {
922
+ "ttft": 0.11786476895213127,
923
+ "time_to_second_token": 0.02033530385233462,
924
+ "latency": 3.869324626866728,
925
+ "inter_token_latency_avg": 0.007341408723903321,
926
+ "chunk_inter_token_latency_avg": 0.020841443655081095,
927
+ "input_tokens": 78,
928
+ "output_tokens": 512,
929
+ "output_tps_per_user": 136.21363931748436,
930
+ "e2e_output_tps_per_user": 132.32283392427672,
931
+ "completed": true
932
+ },
933
+ {
934
+ "ttft": 0.12338922801427543,
935
+ "time_to_second_token": 0.020431119948625565,
936
+ "latency": 4.137814508052543,
937
+ "inter_token_latency_avg": 0.007856018160544554,
938
+ "chunk_inter_token_latency_avg": 0.02090846500019931,
939
+ "input_tokens": 78,
940
+ "output_tokens": 512,
941
+ "output_tps_per_user": 127.2909481068057,
942
+ "e2e_output_tps_per_user": 123.73681783066978,
943
+ "completed": true
944
+ },
945
+ {
946
+ "ttft": 0.12249546311795712,
947
+ "time_to_second_token": 0.02092183893546462,
948
+ "latency": 3.9572368811350316,
949
+ "inter_token_latency_avg": 0.007504386336628326,
950
+ "chunk_inter_token_latency_avg": 0.02107000779130261,
951
+ "input_tokens": 78,
952
+ "output_tokens": 512,
953
+ "output_tps_per_user": 133.25539959464479,
954
+ "e2e_output_tps_per_user": 129.38320736896245,
955
+ "completed": true
956
+ },
957
+ {
958
+ "ttft": 0.1226538741029799,
959
+ "time_to_second_token": 0.020244625862687826,
960
+ "latency": 0.0,
961
+ "inter_token_latency_avg": 0.0063211789832794095,
962
+ "chunk_inter_token_latency_avg": 0.020737902980232446,
963
+ "input_tokens": 78,
964
+ "output_tokens": 188,
965
+ "output_tps_per_user": 158.198336520002,
966
+ "e2e_output_tps_per_user": 0.0,
967
+ "completed": false
968
+ },
969
+ {
970
+ "ttft": 0.14964449405670166,
971
+ "time_to_second_token": 0.017545362003147602,
972
+ "latency": 3.9196609100326896,
973
+ "inter_token_latency_avg": 0.007377722927545964,
974
+ "chunk_inter_token_latency_avg": 0.020714375911955976,
975
+ "input_tokens": 78,
976
+ "output_tokens": 512,
977
+ "output_tps_per_user": 135.54317637306931,
978
+ "e2e_output_tps_per_user": 130.62354416666363,
979
+ "completed": true
980
+ },
981
+ {
982
+ "ttft": 0.10573618090711534,
983
+ "time_to_second_token": 0.013638033997267485,
984
+ "latency": 3.814666331978515,
985
+ "inter_token_latency_avg": 0.0072581803347776894,
986
+ "chunk_inter_token_latency_avg": 0.021193886577550853,
987
+ "input_tokens": 78,
988
+ "output_tokens": 512,
989
+ "output_tps_per_user": 137.77557926033936,
990
+ "e2e_output_tps_per_user": 134.21881638975384,
991
+ "completed": true
992
+ },
993
+ {
994
+ "ttft": 0.12302991887554526,
995
+ "time_to_second_token": 0.020953979110345244,
996
+ "latency": 4.246696174843237,
997
+ "inter_token_latency_avg": 0.008069796978410355,
998
+ "chunk_inter_token_latency_avg": 0.020932316020140566,
999
+ "input_tokens": 78,
1000
+ "output_tokens": 512,
1001
+ "output_tps_per_user": 123.91885479589686,
1002
+ "e2e_output_tps_per_user": 120.56431138940616,
1003
+ "completed": true
1004
+ },
1005
+ {
1006
+ "ttft": 0.11646403884515166,
1007
+ "time_to_second_token": 0.021043566055595875,
1008
+ "latency": 3.930458223912865,
1009
+ "inter_token_latency_avg": 0.00746378509797987,
1010
+ "chunk_inter_token_latency_avg": 0.020841498279058544,
1011
+ "input_tokens": 78,
1012
+ "output_tokens": 512,
1013
+ "output_tps_per_user": 133.98027768386012,
1014
+ "e2e_output_tps_per_user": 130.26470982059993,
1015
+ "completed": true
1016
+ },
1017
+ {
1018
+ "ttft": 0.11775453994050622,
1019
+ "time_to_second_token": 0.020746542140841484,
1020
+ "latency": 4.055516151012853,
1021
+ "inter_token_latency_avg": 0.007705991411100482,
1022
+ "chunk_inter_token_latency_avg": 0.021057548722312015,
1023
+ "input_tokens": 78,
1024
+ "output_tokens": 512,
1025
+ "output_tps_per_user": 129.7691557973319,
1026
+ "e2e_output_tps_per_user": 126.2478019899217,
1027
+ "completed": true
1028
+ },
1029
+ {
1030
+ "ttft": 0.11773488996550441,
1031
+ "time_to_second_token": 0.020856699906289577,
1032
+ "latency": 3.763241825858131,
1033
+ "inter_token_latency_avg": 0.007134064453801618,
1034
+ "chunk_inter_token_latency_avg": 0.020951189286739235,
1035
+ "input_tokens": 78,
1036
+ "output_tokens": 512,
1037
+ "output_tps_per_user": 140.17254910938146,
1038
+ "e2e_output_tps_per_user": 136.05290961689627,
1039
+ "completed": true
1040
+ },
1041
+ {
1042
+ "ttft": 0.1234554189722985,
1043
+ "time_to_second_token": 0.02030633995309472,
1044
+ "latency": 0.0,
1045
+ "inter_token_latency_avg": 0.00775439302854797,
1046
+ "chunk_inter_token_latency_avg": 0.02127487318088802,
1047
+ "input_tokens": 78,
1048
+ "output_tokens": 215,
1049
+ "output_tps_per_user": 128.95915854644946,
1050
+ "e2e_output_tps_per_user": 0.0,
1051
+ "completed": false
1052
+ }
1053
+ ],
1054
+ "total_tokens": 5145,
1055
+ "wall_time": 25.557628852082416,
1056
+ "num_completed": 2,
1057
+ "num_errors": 0,
1058
+ "server_gen_throughput": 257.1817350436559,
1059
+ "server_utilization": 0.011725293132328285,
1060
+ "server_spec_accept_rate": 0.7541666666666667,
1061
+ "server_spec_accept_length": 0.0,
1062
+ "avg_running_reqs": 2,
1063
+ "max_running_reqs": 2,
1064
+ "effective_concurrency": 2,
1065
+ "avg_queue_reqs": 0,
1066
+ "max_queue_reqs": 0,
1067
+ "queue_fraction": 0.0,
1068
+ "underfilled": false,
1069
+ "warmup_timed_out": false,
1070
+ "warmup_duration": 5.539,
1071
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1072
+ "timeout_reason": "",
1073
+ "capacity_limited": false,
1074
+ "hardware_summary": {
1075
+ "samples": 8,
1076
+ "duration_seconds": 16.896,
1077
+ "gpu_count": 4,
1078
+ "cpu_util_avg_pct": 11.54,
1079
+ "cpu_temp_max_c": 76.38,
1080
+ "gpu_util_avg_pct": 100.0,
1081
+ "gpu_util_max_pct": 100.0,
1082
+ "mem_util_avg_pct": 40.28,
1083
+ "mem_util_max_pct": 50.0,
1084
+ "temp_avg_c": 68.16,
1085
+ "temp_max_c": 84.0,
1086
+ "power_total_avg_w": 1172.61,
1087
+ "power_total_max_w": 1175.0,
1088
+ "power_limit_total_w": 1200.0,
1089
+ "vram_used_avg_mb": 384778.0,
1090
+ "vram_used_max_mb": 384778.0,
1091
+ "vram_total_mb": 391548.0,
1092
+ "vram_used_avg_pct": 98.27,
1093
+ "vram_used_max_pct": 98.27,
1094
+ "pcie_rx_avg_mb_s": 11187.12,
1095
+ "pcie_rx_max_mb_s": 11336.0,
1096
+ "pcie_tx_avg_mb_s": 11117.38,
1097
+ "pcie_tx_max_mb_s": 11239.0
1098
+ }
1099
+ },
1100
+ {
1101
+ "concurrency": 4,
1102
+ "context_tokens": 0,
1103
+ "benchmark_mode": "duration",
1104
+ "request_count_target": 0,
1105
+ "warmup_request_count": 0,
1106
+ "measurement_seconds": 19.982047,
1107
+ "measurement_wall_seconds": 20.00121,
1108
+ "client_output_tokens": 6174,
1109
+ "server_output_tokens": 6174,
1110
+ "aggregate_source": "openai_continuous_usage",
1111
+ "aggregate_tps": 308.9773476463454,
1112
+ "per_request_avg_tps": 77.24433691158634,
1113
+ "ttft_avg": 0.17778973274535553,
1114
+ "ttft_p50": 0.1807985259220004,
1115
+ "ttft_p90": 0.1864776147995144,
1116
+ "ttft_p99": 0.21677800566889344,
1117
+ "time_to_second_token_avg": 0.03174264398694504,
1118
+ "time_to_second_token_p50": 0.03366913797799498,
1119
+ "time_to_second_token_p90": 0.03429299045819789,
1120
+ "time_to_second_token_p99": 0.03469313668319955,
1121
+ "request_latency_avg": 6.506094520983215,
1122
+ "request_latency_p50": 6.382211998803541,
1123
+ "request_latency_p90": 6.8716935135424135,
1124
+ "request_latency_p99": 7.440296533070504,
1125
+ "inter_token_latency_avg": 0.01244116428728022,
1126
+ "inter_token_latency_p50": 0.012306649429466642,
1127
+ "inter_token_latency_p90": 0.01313604974973618,
1128
+ "inter_token_latency_p99": 0.014173560375473144,
1129
+ "output_tps_per_user_avg": 80.63692147986394,
1130
+ "output_tps_per_user_p50": 81.27236798491623,
1131
+ "output_tps_per_user_p90": 85.68934634961276,
1132
+ "output_tps_per_user_p99": 87.43239080011531,
1133
+ "e2e_output_tps_per_user_avg": 78.95186391949123,
1134
+ "e2e_output_tps_per_user_p50": 80.2229697314949,
1135
+ "e2e_output_tps_per_user_p90": 83.75147031670433,
1136
+ "e2e_output_tps_per_user_p99": 84.93555512907336,
1137
+ "chunk_inter_token_latency_avg": 0.03454775971555825,
1138
+ "chunk_inter_token_latency_p50": 0.034643082201032946,
1139
+ "chunk_inter_token_latency_p90": 0.035144349182500076,
1140
+ "chunk_inter_token_latency_p99": 0.035562162856296174,
1141
+ "input_seq_len_avg": 78.0,
1142
+ "output_seq_len_avg": 512.0,
1143
+ "output_seq_len_p50": 512.0,
1144
+ "output_seq_len_p90": 512.0,
1145
+ "output_seq_len_p99": 512.0,
1146
+ "request_count": 17,
1147
+ "completed_request_count": 13,
1148
+ "request_samples": [
1149
+ {
1150
+ "ttft": 0.0723429499194026,
1151
+ "time_to_second_token": 0.013155136024579406,
1152
+ "latency": 6.662627859041095,
1153
+ "inter_token_latency_avg": 0.012896839352488634,
1154
+ "chunk_inter_token_latency_avg": 0.03450410947184132,
1155
+ "input_tokens": 78,
1156
+ "output_tokens": 512,
1157
+ "output_tps_per_user": 77.53837763413215,
1158
+ "e2e_output_tps_per_user": 76.84655526801231,
1159
+ "completed": true
1160
+ },
1161
+ {
1162
+ "ttft": 0.1807985259220004,
1163
+ "time_to_second_token": 0.03360512410290539,
1164
+ "latency": 6.19808776397258,
1165
+ "inter_token_latency_avg": 0.011775517099903288,
1166
+ "chunk_inter_token_latency_avg": 0.03399598439576599,
1167
+ "input_tokens": 78,
1168
+ "output_tokens": 512,
1169
+ "output_tps_per_user": 84.92196066771564,
1170
+ "e2e_output_tps_per_user": 82.60612296845576,
1171
+ "completed": true
1172
+ },
1173
+ {
1174
+ "ttft": 0.18202503910288215,
1175
+ "time_to_second_token": 0.033671696903184056,
1176
+ "latency": 6.092495953198522,
1177
+ "inter_token_latency_avg": 0.011566479283944501,
1178
+ "chunk_inter_token_latency_avg": 0.03456415739237217,
1179
+ "input_tokens": 78,
1180
+ "output_tokens": 512,
1181
+ "output_tps_per_user": 86.45673203150989,
1182
+ "e2e_output_tps_per_user": 84.03780715376647,
1183
+ "completed": true
1184
+ },
1185
+ {
1186
+ "ttft": 0.18031293293461204,
1187
+ "time_to_second_token": 0.03369922889396548,
1188
+ "latency": 6.382211998803541,
1189
+ "inter_token_latency_avg": 0.012136788778608472,
1190
+ "chunk_inter_token_latency_avg": 0.03503897777327079,
1191
+ "input_tokens": 78,
1192
+ "output_tokens": 512,
1193
+ "output_tps_per_user": 82.39411744254264,
1194
+ "e2e_output_tps_per_user": 80.2229697314949,
1195
+ "completed": true
1196
+ },
1197
+ {
1198
+ "ttft": 0.1809463920071721,
1199
+ "time_to_second_token": 0.0,
1200
+ "latency": 0.0,
1201
+ "inter_token_latency_avg": 0.0,
1202
+ "chunk_inter_token_latency_avg": 0.0,
1203
+ "input_tokens": 78,
1204
+ "output_tokens": 1,
1205
+ "output_tps_per_user": 0.0,
1206
+ "e2e_output_tps_per_user": 0.0,
1207
+ "completed": false
1208
+ },
1209
+ {
1210
+ "ttft": 0.18660231097601354,
1211
+ "time_to_second_token": 0.02942012087441981,
1212
+ "latency": 6.2684185188263655,
1213
+ "inter_token_latency_avg": 0.01190179297035294,
1214
+ "chunk_inter_token_latency_avg": 0.03378786782139084,
1215
+ "input_tokens": 78,
1216
+ "output_tokens": 512,
1217
+ "output_tps_per_user": 84.0209540269247,
1218
+ "e2e_output_tps_per_user": 81.67929414130147,
1219
+ "completed": true
1220
+ },
1221
+ {
1222
+ "ttft": 0.17968551511876285,
1223
+ "time_to_second_token": 0.03475157590582967,
1224
+ "latency": 6.310488264076412,
1225
+ "inter_token_latency_avg": 0.01199765704296996,
1226
+ "chunk_inter_token_latency_avg": 0.034637303666427394,
1227
+ "input_tokens": 78,
1228
+ "output_tokens": 512,
1229
+ "output_tps_per_user": 83.34960704564823,
1230
+ "e2e_output_tps_per_user": 81.13476779834168,
1231
+ "completed": true
1232
+ },
1233
+ {
1234
+ "ttft": 0.17955420492216945,
1235
+ "time_to_second_token": 0.033856919035315514,
1236
+ "latency": 6.5550508559681475,
1237
+ "inter_token_latency_avg": 0.01247651008032481,
1238
+ "chunk_inter_token_latency_avg": 0.03561729972651385,
1239
+ "input_tokens": 78,
1240
+ "output_tokens": 512,
1241
+ "output_tps_per_user": 80.15061852728982,
1242
+ "e2e_output_tps_per_user": 78.10770827717403,
1243
+ "completed": true
1244
+ },
1245
+ {
1246
+ "ttft": 0.17957319412380457,
1247
+ "time_to_second_token": 0.034223999828100204,
1248
+ "latency": 0.0,
1249
+ "inter_token_latency_avg": 0.012843022881727473,
1250
+ "chunk_inter_token_latency_avg": 0.03484932613412567,
1251
+ "input_tokens": 78,
1252
+ "output_tokens": 484,
1253
+ "output_tps_per_user": 77.86328882297322,
1254
+ "e2e_output_tps_per_user": 0.0,
1255
+ "completed": false
1256
+ },
1257
+ {
1258
+ "ttft": 0.18639448401518166,
1259
+ "time_to_second_token": 0.029375833924859762,
1260
+ "latency": 6.019423788879067,
1261
+ "inter_token_latency_avg": 0.011414930146504668,
1262
+ "chunk_inter_token_latency_avg": 0.03352315692450509,
1263
+ "input_tokens": 78,
1264
+ "output_tokens": 512,
1265
+ "output_tps_per_user": 87.60456587692804,
1266
+ "e2e_output_tps_per_user": 85.0579753075243,
1267
+ "completed": true
1268
+ },
1269
+ {
1270
+ "ttft": 0.1805800509173423,
1271
+ "time_to_second_token": 0.0336665790528059,
1272
+ "latency": 7.51252193399705,
1273
+ "inter_token_latency_avg": 0.014348222863169682,
1274
+ "chunk_inter_token_latency_avg": 0.035249720591729365,
1275
+ "input_tokens": 78,
1276
+ "output_tokens": 512,
1277
+ "output_tps_per_user": 69.69504234331978,
1278
+ "e2e_output_tps_per_user": 68.15287921929428,
1279
+ "completed": true
1280
+ },
1281
+ {
1282
+ "ttft": 0.18062497000209987,
1283
+ "time_to_second_token": 0.03364452510140836,
1284
+ "latency": 6.614234033040702,
1285
+ "inter_token_latency_avg": 0.012590233000075543,
1286
+ "chunk_inter_token_latency_avg": 0.03477626520561407,
1287
+ "input_tokens": 78,
1288
+ "output_tokens": 512,
1289
+ "output_tps_per_user": 79.42664762391608,
1290
+ "e2e_output_tps_per_user": 77.40881218329416,
1291
+ "completed": true
1292
+ },
1293
+ {
1294
+ "ttft": 0.18063039891421795,
1295
+ "time_to_second_token": 0.033818941097706556,
1296
+ "latency": 0.0,
1297
+ "inter_token_latency_avg": 0.013183806278526097,
1298
+ "chunk_inter_token_latency_avg": 0.03436578836602469,
1299
+ "input_tokens": 78,
1300
+ "output_tokens": 392,
1301
+ "output_tps_per_user": 75.8506290879599,
1302
+ "e2e_output_tps_per_user": 0.0,
1303
+ "completed": false
1304
+ },
1305
+ {
1306
+ "ttft": 0.18626655801199377,
1307
+ "time_to_second_token": 0.029395153978839517,
1308
+ "latency": 6.337131014093757,
1309
+ "inter_token_latency_avg": 0.012036916743799928,
1310
+ "chunk_inter_token_latency_avg": 0.03379595854989979,
1311
+ "input_tokens": 78,
1312
+ "output_tokens": 512,
1313
+ "output_tps_per_user": 83.07775332209455,
1314
+ "e2e_output_tps_per_user": 80.79365865425756,
1315
+ "completed": true
1316
+ },
1317
+ {
1318
+ "ttft": 0.22252575703896582,
1319
+ "time_to_second_token": 0.03383295307867229,
1320
+ "latency": 6.910643592942506,
1321
+ "inter_token_latency_avg": 0.013088293220946262,
1322
+ "chunk_inter_token_latency_avg": 0.03465346028965565,
1323
+ "input_tokens": 78,
1324
+ "output_tokens": 512,
1325
+ "output_tps_per_user": 76.40415622715561,
1326
+ "e2e_output_tps_per_user": 74.0886131825522,
1327
+ "completed": true
1328
+ },
1329
+ {
1330
+ "ttft": 0.18178053596056998,
1331
+ "time_to_second_token": 0.03340253490023315,
1332
+ "latency": 6.715893195942044,
1333
+ "inter_token_latency_avg": 0.012786913228926564,
1334
+ "chunk_inter_token_latency_avg": 0.03475591840415678,
1335
+ "input_tokens": 78,
1336
+ "output_tokens": 512,
1337
+ "output_tps_per_user": 78.20495706014485,
1338
+ "e2e_output_tps_per_user": 76.23706706791684,
1339
+ "completed": true
1340
+ },
1341
+ {
1342
+ "ttft": 0.18178163678385317,
1343
+ "time_to_second_token": 0.03436198108829558,
1344
+ "latency": 0.0,
1345
+ "inter_token_latency_avg": 0.012014705624214687,
1346
+ "chunk_inter_token_latency_avg": 0.03464886073563849,
1347
+ "input_tokens": 78,
1348
+ "output_tokens": 448,
1349
+ "output_tps_per_user": 83.23133593756798,
1350
+ "e2e_output_tps_per_user": 0.0,
1351
+ "completed": false
1352
+ }
1353
+ ],
1354
+ "total_tokens": 6174,
1355
+ "wall_time": 25.552133698016405,
1356
+ "num_completed": 4,
1357
+ "num_errors": 0,
1358
+ "server_gen_throughput": 308.60564141811994,
1359
+ "server_utilization": 0.02345058626465657,
1360
+ "server_spec_accept_rate": 0.603448275862069,
1361
+ "server_spec_accept_length": 0.0,
1362
+ "avg_running_reqs": 3.9,
1363
+ "max_running_reqs": 4,
1364
+ "effective_concurrency": 3.9,
1365
+ "avg_queue_reqs": 0,
1366
+ "max_queue_reqs": 0,
1367
+ "queue_fraction": 0.0,
1368
+ "underfilled": true,
1369
+ "warmup_timed_out": false,
1370
+ "warmup_duration": 5.536,
1371
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1372
+ "timeout_reason": "",
1373
+ "capacity_limited": false,
1374
+ "hardware_summary": {
1375
+ "samples": 8,
1376
+ "duration_seconds": 16.827,
1377
+ "gpu_count": 4,
1378
+ "cpu_util_avg_pct": 11.47,
1379
+ "cpu_temp_max_c": 77.12,
1380
+ "gpu_util_avg_pct": 100.0,
1381
+ "gpu_util_max_pct": 100.0,
1382
+ "mem_util_avg_pct": 35.47,
1383
+ "mem_util_max_pct": 45.0,
1384
+ "temp_avg_c": 68.72,
1385
+ "temp_max_c": 84.0,
1386
+ "power_total_avg_w": 1170.25,
1387
+ "power_total_max_w": 1172.14,
1388
+ "power_limit_total_w": 1200.0,
1389
+ "vram_used_avg_mb": 384778.0,
1390
+ "vram_used_max_mb": 384778.0,
1391
+ "vram_total_mb": 391548.0,
1392
+ "vram_used_avg_pct": 98.27,
1393
+ "vram_used_max_pct": 98.27,
1394
+ "pcie_rx_avg_mb_s": 7813.12,
1395
+ "pcie_rx_max_mb_s": 8161.0,
1396
+ "pcie_tx_avg_mb_s": 8053.88,
1397
+ "pcie_tx_max_mb_s": 8437.0
1398
+ }
1399
+ },
1400
+ {
1401
+ "concurrency": 2,
1402
+ "context_tokens": 8192,
1403
+ "benchmark_mode": "duration",
1404
+ "request_count_target": 0,
1405
+ "warmup_request_count": 0,
1406
+ "measurement_seconds": 19.985498,
1407
+ "measurement_wall_seconds": 20.000656,
1408
+ "client_output_tokens": 3587,
1409
+ "server_output_tokens": 3587,
1410
+ "aggregate_source": "openai_continuous_usage",
1411
+ "aggregate_tps": 179.48013878635928,
1412
+ "per_request_avg_tps": 89.74006939317964,
1413
+ "ttft_avg": 1.1894093764014542,
1414
+ "ttft_p50": 1.2212169540580362,
1415
+ "ttft_p90": 1.3684029981726777,
1416
+ "ttft_p99": 1.9125716026616284,
1417
+ "time_to_second_token_avg": 0.025312776444479823,
1418
+ "time_to_second_token_p50": 0.027226490550674498,
1419
+ "time_to_second_token_p90": 0.02769280921202153,
1420
+ "time_to_second_token_p99": 0.0277038907376118,
1421
+ "request_latency_avg": 5.413165779871633,
1422
+ "request_latency_p50": 5.438335195998661,
1423
+ "request_latency_p90": 5.877898134151473,
1424
+ "request_latency_p99": 6.255758887748234,
1425
+ "inter_token_latency_avg": 0.00830825860560843,
1426
+ "inter_token_latency_p50": 0.008347879003080923,
1427
+ "inter_token_latency_p90": 0.008825519827072598,
1428
+ "inter_token_latency_p99": 0.009230215194777226,
1429
+ "output_tps_per_user_avg": 120.7893245972912,
1430
+ "output_tps_per_user_p50": 119.79434916732279,
1431
+ "output_tps_per_user_p90": 129.7361418336343,
1432
+ "output_tps_per_user_p99": 131.83454344392672,
1433
+ "e2e_output_tps_per_user_avg": 95.20339521438798,
1434
+ "e2e_output_tps_per_user_p50": 94.14836874853962,
1435
+ "e2e_output_tps_per_user_p90": 103.85715540642657,
1436
+ "e2e_output_tps_per_user_p99": 108.00400096911441,
1437
+ "chunk_inter_token_latency_avg": 0.023168869852685122,
1438
+ "chunk_inter_token_latency_p50": 0.023156017746843782,
1439
+ "chunk_inter_token_latency_p90": 0.024457862475522812,
1440
+ "chunk_inter_token_latency_p99": 0.025339872482635258,
1441
+ "input_seq_len_avg": 8192.0,
1442
+ "output_seq_len_avg": 512.0,
1443
+ "output_seq_len_p50": 512.0,
1444
+ "output_seq_len_p90": 512.0,
1445
+ "output_seq_len_p99": 512.0,
1446
+ "request_count": 10,
1447
+ "completed_request_count": 8,
1448
+ "request_samples": [
1449
+ {
1450
+ "ttft": 0.5938855609856546,
1451
+ "time_to_second_token": 0.01751627796329558,
1452
+ "latency": 5.02539852890186,
1453
+ "inter_token_latency_avg": 0.008672236727820363,
1454
+ "chunk_inter_token_latency_avg": 0.024348972351187943,
1455
+ "input_tokens": 8192,
1456
+ "output_tokens": 512,
1457
+ "output_tps_per_user": 115.31050539614768,
1458
+ "e2e_output_tps_per_user": 101.88246704324189,
1459
+ "completed": true
1460
+ },
1461
+ {
1462
+ "ttft": 0.7053975961171091,
1463
+ "time_to_second_token": 0.01768357283435762,
1464
+ "latency": 4.720427100080997,
1465
+ "inter_token_latency_avg": 0.007857200594841268,
1466
+ "chunk_inter_token_latency_avg": 0.02206060167013125,
1467
+ "input_tokens": 8192,
1468
+ "output_tokens": 512,
1469
+ "output_tps_per_user": 127.27179202431984,
1470
+ "e2e_output_tps_per_user": 108.46476158719085,
1471
+ "completed": true
1472
+ },
1473
+ {
1474
+ "ttft": 1.2253350547980517,
1475
+ "time_to_second_token": 0.02671587117947638,
1476
+ "latency": 5.5139663990121335,
1477
+ "inter_token_latency_avg": 0.008392624939753585,
1478
+ "chunk_inter_token_latency_avg": 0.02330777904464175,
1479
+ "input_tokens": 8192,
1480
+ "output_tokens": 512,
1481
+ "output_tps_per_user": 119.1522327255769,
1482
+ "e2e_output_tps_per_user": 92.85511788605177,
1483
+ "completed": true
1484
+ },
1485
+ {
1486
+ "ttft": 1.2199293570593,
1487
+ "time_to_second_token": 0.027305281022563577,
1488
+ "latency": 5.4628303539939225,
1489
+ "inter_token_latency_avg": 0.008303133066408263,
1490
+ "chunk_inter_token_latency_avg": 0.023185251349369523,
1491
+ "input_tokens": 8192,
1492
+ "output_tokens": 512,
1493
+ "output_tps_per_user": 120.4364656090687,
1494
+ "e2e_output_tps_per_user": 93.72430897944183,
1495
+ "completed": true
1496
+ },
1497
+ {
1498
+ "ttft": 1.2225045510567725,
1499
+ "time_to_second_token": 0.02714770007878542,
1500
+ "latency": 0.0,
1501
+ "inter_token_latency_avg": 0.009275181346744406,
1502
+ "chunk_inter_token_latency_avg": 0.02543787359453664,
1503
+ "input_tokens": 8192,
1504
+ "output_tokens": 278,
1505
+ "output_tps_per_user": 107.81460357656518,
1506
+ "e2e_output_tps_per_user": 0.0,
1507
+ "completed": false
1508
+ },
1509
+ {
1510
+ "ttft": 1.3012216889765114,
1511
+ "time_to_second_token": 0.02770512201823294,
1512
+ "latency": 5.4138400380034,
1513
+ "inter_token_latency_avg": 0.008048176808271797,
1514
+ "chunk_inter_token_latency_avg": 0.022596804115532356,
1515
+ "input_tokens": 8192,
1516
+ "output_tokens": 512,
1517
+ "output_tps_per_user": 124.25174344731278,
1518
+ "e2e_output_tps_per_user": 94.57242851763742,
1519
+ "completed": true
1520
+ },
1521
+ {
1522
+ "ttft": 1.9730347809381783,
1523
+ "time_to_second_token": 0.02644847217015922,
1524
+ "latency": 6.297743415925652,
1525
+ "inter_token_latency_avg": 0.008463226291560613,
1526
+ "chunk_inter_token_latency_avg": 0.02312678414431804,
1527
+ "input_tokens": 8192,
1528
+ "output_tokens": 512,
1529
+ "output_tps_per_user": 118.15824905889414,
1530
+ "e2e_output_tps_per_user": 81.2989615780886,
1531
+ "completed": true
1532
+ },
1533
+ {
1534
+ "ttft": 1.213654592167586,
1535
+ "time_to_second_token": 0.027514670975506306,
1536
+ "latency": 5.69796444196254,
1537
+ "inter_token_latency_avg": 0.008775557435997953,
1538
+ "chunk_inter_token_latency_avg": 0.023114999225747185,
1539
+ "input_tokens": 8192,
1540
+ "output_tokens": 512,
1541
+ "output_tps_per_user": 113.9528750501854,
1542
+ "e2e_output_tps_per_user": 89.85665060128959,
1543
+ "completed": true
1544
+ },
1545
+ {
1546
+ "ttft": 1.2265115019399673,
1547
+ "time_to_second_token": 0.027691441122442484,
1548
+ "latency": 5.1731559610925615,
1549
+ "inter_token_latency_avg": 0.00772337467544539,
1550
+ "chunk_inter_token_latency_avg": 0.023352925793802333,
1551
+ "input_tokens": 8192,
1552
+ "output_tokens": 512,
1553
+ "output_tps_per_user": 129.4770799064377,
1554
+ "e2e_output_tps_per_user": 98.97246552216193,
1555
+ "completed": true
1556
+ },
1557
+ {
1558
+ "ttft": 1.2126190799754113,
1559
+ "time_to_second_token": 0.027399355079978704,
1560
+ "latency": 0.0,
1561
+ "inter_token_latency_avg": 0.007571874169240656,
1562
+ "chunk_inter_token_latency_avg": 0.02115670723758419,
1563
+ "input_tokens": 8192,
1564
+ "output_tokens": 96,
1565
+ "output_tps_per_user": 132.06769917840364,
1566
+ "e2e_output_tps_per_user": 0.0,
1567
+ "completed": false
1568
+ }
1569
+ ],
1570
+ "total_tokens": 3587,
1571
+ "wall_time": 25.556549122091383,
1572
+ "num_completed": 2,
1573
+ "num_errors": 0,
1574
+ "server_gen_throughput": 179.29840638216805,
1575
+ "server_utilization": 0.012283640424343933,
1576
+ "server_spec_accept_rate": 0.6568627450980392,
1577
+ "server_spec_accept_length": 0.0,
1578
+ "avg_running_reqs": 1.9,
1579
+ "max_running_reqs": 2,
1580
+ "effective_concurrency": 1.9,
1581
+ "avg_queue_reqs": 0,
1582
+ "max_queue_reqs": 0,
1583
+ "queue_fraction": 0.0,
1584
+ "underfilled": true,
1585
+ "warmup_timed_out": false,
1586
+ "warmup_duration": 5.55,
1587
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1588
+ "timeout_reason": "",
1589
+ "capacity_limited": false,
1590
+ "hardware_summary": {
1591
+ "samples": 8,
1592
+ "duration_seconds": 16.921,
1593
+ "gpu_count": 4,
1594
+ "cpu_util_avg_pct": 11.51,
1595
+ "cpu_temp_max_c": 76.38,
1596
+ "gpu_util_avg_pct": 99.88,
1597
+ "gpu_util_max_pct": 100.0,
1598
+ "mem_util_avg_pct": 36.44,
1599
+ "mem_util_max_pct": 55.0,
1600
+ "temp_avg_c": 68.5,
1601
+ "temp_max_c": 84.0,
1602
+ "power_total_avg_w": 1162.41,
1603
+ "power_total_max_w": 1177.27,
1604
+ "power_limit_total_w": 1200.0,
1605
+ "vram_used_avg_mb": 384778.0,
1606
+ "vram_used_max_mb": 384778.0,
1607
+ "vram_total_mb": 391548.0,
1608
+ "vram_used_avg_pct": 98.27,
1609
+ "vram_used_max_pct": 98.27,
1610
+ "pcie_rx_avg_mb_s": 29258.88,
1611
+ "pcie_rx_max_mb_s": 70866.0,
1612
+ "pcie_tx_avg_mb_s": 28010.62,
1613
+ "pcie_tx_max_mb_s": 69989.0
1614
+ }
1615
+ },
1616
+ {
1617
+ "concurrency": 4,
1618
+ "context_tokens": 8192,
1619
+ "benchmark_mode": "duration",
1620
+ "request_count_target": 0,
1621
+ "warmup_request_count": 0,
1622
+ "measurement_seconds": 19.930302,
1623
+ "measurement_wall_seconds": 20.000584,
1624
+ "client_output_tokens": 4342,
1625
+ "server_output_tokens": 4342,
1626
+ "aggregate_source": "openai_continuous_usage",
1627
+ "aggregate_tps": 217.85922059414168,
1628
+ "per_request_avg_tps": 54.46480514853542,
1629
+ "ttft_avg": 1.6845163684144306,
1630
+ "ttft_p50": 1.2621239490108564,
1631
+ "ttft_p90": 2.2127018794883044,
1632
+ "ttft_p99": 4.023498326067348,
1633
+ "time_to_second_token_avg": 0.03870494751026854,
1634
+ "time_to_second_token_p50": 0.04049360693898052,
1635
+ "time_to_second_token_p90": 0.041569333686493334,
1636
+ "time_to_second_token_p99": 0.044618300595320765,
1637
+ "request_latency_avg": 9.701175137112537,
1638
+ "request_latency_p50": 9.277128694113344,
1639
+ "request_latency_p90": 11.318623020872474,
1640
+ "request_latency_p99": 12.65666121724993,
1641
+ "inter_token_latency_avg": 0.015277014782169696,
1642
+ "inter_token_latency_p50": 0.015349018777003027,
1643
+ "inter_token_latency_p90": 0.016694702791212282,
1644
+ "inter_token_latency_p99": 0.017054498783689094,
1645
+ "output_tps_per_user_avg": 65.75017254999644,
1646
+ "output_tps_per_user_p50": 65.15961146932285,
1647
+ "output_tps_per_user_p90": 71.0132044249438,
1648
+ "output_tps_per_user_p99": 72.7397198355044,
1649
+ "e2e_output_tps_per_user_avg": 53.62687357686466,
1650
+ "e2e_output_tps_per_user_p50": 55.18948986068087,
1651
+ "e2e_output_tps_per_user_p90": 60.13891332324956,
1652
+ "e2e_output_tps_per_user_p99": 62.607782503263365,
1653
+ "chunk_inter_token_latency_avg": 0.04311462391925306,
1654
+ "chunk_inter_token_latency_p50": 0.04431330813678551,
1655
+ "chunk_inter_token_latency_p90": 0.04551846648306814,
1656
+ "chunk_inter_token_latency_p99": 0.0457929443786719,
1657
+ "input_seq_len_avg": 8192.0,
1658
+ "output_seq_len_avg": 512.0,
1659
+ "output_seq_len_p50": 512.0,
1660
+ "output_seq_len_p90": 512.0,
1661
+ "output_seq_len_p99": 512.0,
1662
+ "request_count": 12,
1663
+ "completed_request_count": 9,
1664
+ "request_samples": [
1665
+ {
1666
+ "ttft": 0.5955398578662425,
1667
+ "time_to_second_token": 0.016524713020771742,
1668
+ "latency": 8.142221544869244,
1669
+ "inter_token_latency_avg": 0.01476845731311742,
1670
+ "chunk_inter_token_latency_avg": 0.04035658656151338,
1671
+ "input_tokens": 8192,
1672
+ "output_tokens": 512,
1673
+ "output_tps_per_user": 67.71187936547678,
1674
+ "e2e_output_tps_per_user": 62.882101301042674,
1675
+ "completed": true
1676
+ },
1677
+ {
1678
+ "ttft": 1.2590207229368389,
1679
+ "time_to_second_token": 0.0395864921156317,
1680
+ "latency": 9.238898925017565,
1681
+ "inter_token_latency_avg": 0.015616200004071872,
1682
+ "chunk_inter_token_latency_avg": 0.045599304011889864,
1683
+ "input_tokens": 8192,
1684
+ "output_tokens": 512,
1685
+ "output_tps_per_user": 64.03606509517381,
1686
+ "e2e_output_tps_per_user": 55.41785922276735,
1687
+ "completed": true
1688
+ },
1689
+ {
1690
+ "ttft": 1.2618258679285645,
1691
+ "time_to_second_token": 0.03996979701332748,
1692
+ "latency": 9.196670216973871,
1693
+ "inter_token_latency_avg": 0.01552807113316107,
1694
+ "chunk_inter_token_latency_avg": 0.044577777241827564,
1695
+ "input_tokens": 8192,
1696
+ "output_tokens": 512,
1697
+ "output_tps_per_user": 64.39949890907208,
1698
+ "e2e_output_tps_per_user": 55.67232356065407,
1699
+ "completed": true
1700
+ },
1701
+ {
1702
+ "ttft": 2.2127146429847926,
1703
+ "time_to_second_token": 0.040582613088190556,
1704
+ "latency": 10.946945744100958,
1705
+ "inter_token_latency_avg": 0.017092428769307565,
1706
+ "chunk_inter_token_latency_avg": 0.04479092872367264,
1707
+ "input_tokens": 8192,
1708
+ "output_tokens": 512,
1709
+ "output_tps_per_user": 58.50543614934785,
1710
+ "e2e_output_tps_per_user": 46.771036594924595,
1711
+ "completed": true
1712
+ },
1713
+ {
1714
+ "ttft": 1.2621886630076915,
1715
+ "time_to_second_token": 0.04047246486879885,
1716
+ "latency": 8.611827800050378,
1717
+ "inter_token_latency_avg": 0.014382855454095277,
1718
+ "chunk_inter_token_latency_avg": 0.0445432674972284,
1719
+ "input_tokens": 8192,
1720
+ "output_tokens": 512,
1721
+ "output_tps_per_user": 69.52722310195134,
1722
+ "e2e_output_tps_per_user": 59.453116328801286,
1723
+ "completed": true
1724
+ },
1725
+ {
1726
+ "ttft": 1.923778808210045,
1727
+ "time_to_second_token": 0.04072440997697413,
1728
+ "latency": 0.0,
1729
+ "inter_token_latency_avg": 0.014053212739941631,
1730
+ "chunk_inter_token_latency_avg": 0.04110083452024025,
1731
+ "input_tokens": 8192,
1732
+ "output_tokens": 428,
1733
+ "output_tps_per_user": 71.15810587267559,
1734
+ "e2e_output_tps_per_user": 0.0,
1735
+ "completed": false
1736
+ },
1737
+ {
1738
+ "ttft": 2.2125870080199093,
1739
+ "time_to_second_token": 0.04051474900916219,
1740
+ "latency": 9.543051223969087,
1741
+ "inter_token_latency_avg": 0.01434533114667158,
1742
+ "chunk_inter_token_latency_avg": 0.04188836694828101,
1743
+ "input_tokens": 8192,
1744
+ "output_tokens": 512,
1745
+ "output_tps_per_user": 69.70909139535765,
1746
+ "e2e_output_tps_per_user": 53.65160345299416,
1747
+ "completed": true
1748
+ },
1749
+ {
1750
+ "ttft": 1.2591751390136778,
1751
+ "time_to_second_token": 0.040605568094179034,
1752
+ "latency": 9.277128694113344,
1753
+ "inter_token_latency_avg": 0.015690711458120676,
1754
+ "chunk_inter_token_latency_avg": 0.04581687745771238,
1755
+ "input_tokens": 8192,
1756
+ "output_tokens": 512,
1757
+ "output_tps_per_user": 63.7319730637487,
1758
+ "e2e_output_tps_per_user": 55.18948986068087,
1759
+ "completed": true
1760
+ },
1761
+ {
1762
+ "ttft": 1.4571730380412191,
1763
+ "time_to_second_token": 0.04498353600502014,
1764
+ "latency": 0.0,
1765
+ "inter_token_latency_avg": 0.015169966420844982,
1766
+ "chunk_inter_token_latency_avg": 0.042138795613458284,
1767
+ "input_tokens": 8192,
1768
+ "output_tokens": 476,
1769
+ "output_tps_per_user": 65.91972402957363,
1770
+ "e2e_output_tps_per_user": 0.0,
1771
+ "completed": false
1772
+ },
1773
+ {
1774
+ "ttft": 4.247303050942719,
1775
+ "time_to_second_token": 0.039115041960030794,
1776
+ "latency": 12.805332127958536,
1777
+ "inter_token_latency_avg": 0.01674761071823056,
1778
+ "chunk_inter_token_latency_avg": 0.04457306810945738,
1779
+ "input_tokens": 8192,
1780
+ "output_tokens": 512,
1781
+ "output_tps_per_user": 59.71000979330461,
1782
+ "e2e_output_tps_per_user": 39.98334403854502,
1783
+ "completed": true
1784
+ },
1785
+ {
1786
+ "ttft": 1.260830387007445,
1787
+ "time_to_second_token": 0.03971677087247372,
1788
+ "latency": 9.548499956959859,
1789
+ "inter_token_latency_avg": 0.016218531448047777,
1790
+ "chunk_inter_token_latency_avg": 0.044083348776342623,
1791
+ "input_tokens": 8192,
1792
+ "output_tokens": 512,
1793
+ "output_tps_per_user": 61.65786361134256,
1794
+ "e2e_output_tps_per_user": 53.620987831371934,
1795
+ "completed": true
1796
+ },
1797
+ {
1798
+ "ttft": 1.2620592350140214,
1799
+ "time_to_second_token": 0.04166321409866214,
1800
+ "latency": 0.0,
1801
+ "inter_token_latency_avg": 0.013710800780425945,
1802
+ "chunk_inter_token_latency_avg": 0.03790633156941291,
1803
+ "input_tokens": 8192,
1804
+ "output_tokens": 283,
1805
+ "output_tps_per_user": 72.93520021293268,
1806
+ "e2e_output_tps_per_user": 0.0,
1807
+ "completed": false
1808
+ }
1809
+ ],
1810
+ "total_tokens": 4342,
1811
+ "wall_time": 28.853528001811355,
1812
+ "num_completed": 4,
1813
+ "num_errors": 0,
1814
+ "server_gen_throughput": 217.03898889784327,
1815
+ "server_utilization": 0.027359017308765998,
1816
+ "server_spec_accept_rate": 0.6283185840707964,
1817
+ "server_spec_accept_length": 0.0,
1818
+ "avg_running_reqs": 3.9,
1819
+ "max_running_reqs": 4,
1820
+ "effective_concurrency": 3.9,
1821
+ "avg_queue_reqs": 0.1,
1822
+ "max_queue_reqs": 1,
1823
+ "queue_fraction": 0.05,
1824
+ "underfilled": true,
1825
+ "warmup_timed_out": false,
1826
+ "warmup_duration": 8.573,
1827
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1828
+ "timeout_reason": "",
1829
+ "capacity_limited": true,
1830
+ "hardware_summary": {
1831
+ "samples": 8,
1832
+ "duration_seconds": 16.883,
1833
+ "gpu_count": 4,
1834
+ "cpu_util_avg_pct": 11.5,
1835
+ "cpu_temp_max_c": 76.25,
1836
+ "gpu_util_avg_pct": 100.0,
1837
+ "gpu_util_max_pct": 100.0,
1838
+ "mem_util_avg_pct": 34.25,
1839
+ "mem_util_max_pct": 46.0,
1840
+ "temp_avg_c": 68.66,
1841
+ "temp_max_c": 84.0,
1842
+ "power_total_avg_w": 1159.22,
1843
+ "power_total_max_w": 1172.44,
1844
+ "power_limit_total_w": 1200.0,
1845
+ "vram_used_avg_mb": 384778.0,
1846
+ "vram_used_max_mb": 384778.0,
1847
+ "vram_total_mb": 391548.0,
1848
+ "vram_used_avg_pct": 98.27,
1849
+ "vram_used_max_pct": 98.27,
1850
+ "pcie_rx_avg_mb_s": 27724.5,
1851
+ "pcie_rx_max_mb_s": 67581.0,
1852
+ "pcie_tx_avg_mb_s": 27474.12,
1853
+ "pcie_tx_max_mb_s": 67821.0
1854
+ }
1855
+ },
1856
+ {
1857
+ "concurrency": 2,
1858
+ "context_tokens": 32768,
1859
+ "benchmark_mode": "duration",
1860
+ "request_count_target": 0,
1861
+ "warmup_request_count": 0,
1862
+ "measurement_seconds": 19.997661,
1863
+ "measurement_wall_seconds": 20.00078,
1864
+ "client_output_tokens": 3749,
1865
+ "server_output_tokens": 3749,
1866
+ "aggregate_source": "openai_continuous_usage",
1867
+ "aggregate_tps": 187.47192708215627,
1868
+ "per_request_avg_tps": 93.73596354107814,
1869
+ "ttft_avg": 1.1601904219947756,
1870
+ "ttft_p50": 1.2142878150334582,
1871
+ "ttft_p90": 1.3290542589034884,
1872
+ "ttft_p99": 1.4677723066555337,
1873
+ "time_to_second_token_avg": 0.017825737223029138,
1874
+ "time_to_second_token_p50": 0.019167354563251138,
1875
+ "time_to_second_token_p90": 0.020459615555591882,
1876
+ "time_to_second_token_p99": 0.021644204219337552,
1877
+ "request_latency_avg": 5.313494802772766,
1878
+ "request_latency_p50": 5.375003230990842,
1879
+ "request_latency_p90": 5.586414952063933,
1880
+ "request_latency_p99": 5.701329016662203,
1881
+ "inter_token_latency_avg": 0.008015108251874231,
1882
+ "inter_token_latency_p50": 0.008146798233829671,
1883
+ "inter_token_latency_p90": 0.008546021327498477,
1884
+ "inter_token_latency_p99": 0.009447031634392257,
1885
+ "output_tps_per_user_avg": 125.91902973995667,
1886
+ "output_tps_per_user_p50": 122.74874938988106,
1887
+ "output_tps_per_user_p90": 143.33476272223038,
1888
+ "output_tps_per_user_p99": 150.46171413409994,
1889
+ "e2e_output_tps_per_user_avg": 96.63340093514198,
1890
+ "e2e_output_tps_per_user_p50": 95.25646775312458,
1891
+ "e2e_output_tps_per_user_p90": 104.75414728581389,
1892
+ "e2e_output_tps_per_user_p99": 105.26579246796483,
1893
+ "chunk_inter_token_latency_avg": 0.023044757916286292,
1894
+ "chunk_inter_token_latency_p50": 0.02316345668564709,
1895
+ "chunk_inter_token_latency_p90": 0.02447127468927797,
1896
+ "chunk_inter_token_latency_p99": 0.025079763939748784,
1897
+ "input_seq_len_avg": 32768.0,
1898
+ "output_seq_len_avg": 512.0,
1899
+ "output_seq_len_p50": 512.0,
1900
+ "output_seq_len_p90": 512.0,
1901
+ "output_seq_len_p99": 512.0,
1902
+ "request_count": 10,
1903
+ "completed_request_count": 8,
1904
+ "request_samples": [
1905
+ {
1906
+ "ttft": 0.6090631559491158,
1907
+ "time_to_second_token": 0.01044343295507133,
1908
+ "latency": 5.4876536841038615,
1909
+ "inter_token_latency_avg": 0.009547143890713788,
1910
+ "chunk_inter_token_latency_avg": 0.025147373856467762,
1911
+ "input_tokens": 32768,
1912
+ "output_tokens": 512,
1913
+ "output_tps_per_user": 104.74336738264405,
1914
+ "e2e_output_tps_per_user": 93.30034828602892,
1915
+ "completed": true
1916
+ },
1917
+ {
1918
+ "ttft": 1.4831854230724275,
1919
+ "time_to_second_token": 0.016494586830958724,
1920
+ "latency": 5.7140972460620105,
1921
+ "inter_token_latency_avg": 0.00827967088647668,
1922
+ "chunk_inter_token_latency_avg": 0.023119736737647993,
1923
+ "input_tokens": 32768,
1924
+ "output_tokens": 512,
1925
+ "output_tps_per_user": 120.77774753502777,
1926
+ "e2e_output_tps_per_user": 89.60295527921852,
1927
+ "completed": true
1928
+ },
1929
+ {
1930
+ "ttft": 1.2215185849927366,
1931
+ "time_to_second_token": 0.01897827093489468,
1932
+ "latency": 5.5316939689219,
1933
+ "inter_token_latency_avg": 0.00843478548714122,
1934
+ "chunk_inter_token_latency_avg": 0.023049066224220125,
1935
+ "input_tokens": 32768,
1936
+ "output_tokens": 512,
1937
+ "output_tps_per_user": 118.55666057239915,
1938
+ "e2e_output_tps_per_user": 92.55754256770396,
1939
+ "completed": true
1940
+ },
1941
+ {
1942
+ "ttft": 1.213982854038477,
1943
+ "time_to_second_token": 0.017720089061185718,
1944
+ "latency": 5.389688584022224,
1945
+ "inter_token_latency_avg": 0.008171635479420248,
1946
+ "chunk_inter_token_latency_avg": 0.02332796497197624,
1947
+ "input_tokens": 32768,
1948
+ "output_tokens": 512,
1949
+ "output_tps_per_user": 122.37452374355627,
1950
+ "e2e_output_tps_per_user": 94.99621212213043,
1951
+ "completed": true
1952
+ },
1953
+ {
1954
+ "ttft": 1.2081716759130359,
1955
+ "time_to_second_token": 0.02177582518197596,
1956
+ "latency": 0.0,
1957
+ "inter_token_latency_avg": 0.0066114129892226245,
1958
+ "chunk_inter_token_latency_avg": 0.020994135983320967,
1959
+ "input_tokens": 32768,
1960
+ "output_tokens": 182,
1961
+ "output_tps_per_user": 151.25359762430767,
1962
+ "e2e_output_tps_per_user": 0.0,
1963
+ "completed": false
1964
+ },
1965
+ {
1966
+ "ttft": 1.3119285739958286,
1967
+ "time_to_second_token": 0.020313370041549206,
1968
+ "latency": 4.8990289689973,
1969
+ "inter_token_latency_avg": 0.0070197659393375165,
1970
+ "chunk_inter_token_latency_avg": 0.020734684364170353,
1971
+ "input_tokens": 32768,
1972
+ "output_tokens": 512,
1973
+ "output_tps_per_user": 142.45489217755514,
1974
+ "e2e_output_tps_per_user": 104.51050672288487,
1975
+ "completed": true
1976
+ },
1977
+ {
1978
+ "ttft": 0.9148407790344208,
1979
+ "time_to_second_token": 0.013460942078381777,
1980
+ "latency": 4.861252913950011,
1981
+ "inter_token_latency_avg": 0.007722920029189022,
1982
+ "chunk_inter_token_latency_avg": 0.02335155109417509,
1983
+ "input_tokens": 32768,
1984
+ "output_tokens": 512,
1985
+ "output_tps_per_user": 129.48470218783416,
1986
+ "e2e_output_tps_per_user": 105.32264193264827,
1987
+ "completed": true
1988
+ },
1989
+ {
1990
+ "ttft": 1.2099958129692823,
1991
+ "time_to_second_token": 0.02024669898673892,
1992
+ "latency": 5.36031787795946,
1993
+ "inter_token_latency_avg": 0.008121960988239096,
1994
+ "chunk_inter_token_latency_avg": 0.023186156787654625,
1995
+ "input_tokens": 32768,
1996
+ "output_tokens": 512,
1997
+ "output_tps_per_user": 123.12297503620586,
1998
+ "e2e_output_tps_per_user": 95.51672338411872,
1999
+ "completed": true
2000
+ },
2001
+ {
2002
+ "ttft": 1.2145927760284394,
2003
+ "time_to_second_token": 0.019467717967927456,
2004
+ "latency": 5.264225178165361,
2005
+ "inter_token_latency_avg": 0.007924916638232724,
2006
+ "chunk_inter_token_latency_avg": 0.023140756583639555,
2007
+ "input_tokens": 32768,
2008
+ "output_tokens": 512,
2009
+ "output_tps_per_user": 126.18429261143653,
2010
+ "e2e_output_tps_per_user": 97.26027718640209,
2011
+ "completed": true
2012
+ },
2013
+ {
2014
+ "ttft": 1.2146245839539915,
2015
+ "time_to_second_token": 0.019356438191607594,
2016
+ "latency": 0.0,
2017
+ "inter_token_latency_avg": 0.008316870190769392,
2018
+ "chunk_inter_token_latency_avg": 0.024396152559590215,
2019
+ "input_tokens": 32768,
2020
+ "output_tokens": 353,
2021
+ "output_tps_per_user": 120.23753852860004,
2022
+ "e2e_output_tps_per_user": 0.0,
2023
+ "completed": false
2024
+ }
2025
+ ],
2026
+ "total_tokens": 3749,
2027
+ "wall_time": 25.56879776297137,
2028
+ "num_completed": 2,
2029
+ "num_errors": 0,
2030
+ "server_gen_throughput": 187.39629296317187,
2031
+ "server_utilization": 0.013121161362367406,
2032
+ "server_spec_accept_rate": 0.6424242424242425,
2033
+ "server_spec_accept_length": 0.0,
2034
+ "avg_running_reqs": 1.9,
2035
+ "max_running_reqs": 2,
2036
+ "effective_concurrency": 1.9,
2037
+ "avg_queue_reqs": 0,
2038
+ "max_queue_reqs": 0,
2039
+ "queue_fraction": 0.0,
2040
+ "underfilled": true,
2041
+ "warmup_timed_out": false,
2042
+ "warmup_duration": 5.55,
2043
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
2044
+ "timeout_reason": "",
2045
+ "capacity_limited": false,
2046
+ "hardware_summary": {
2047
+ "samples": 9,
2048
+ "duration_seconds": 19.323,
2049
+ "gpu_count": 4,
2050
+ "cpu_util_avg_pct": 11.59,
2051
+ "cpu_temp_max_c": 75.75,
2052
+ "gpu_util_avg_pct": 99.92,
2053
+ "gpu_util_max_pct": 100.0,
2054
+ "mem_util_avg_pct": 36.47,
2055
+ "mem_util_max_pct": 55.0,
2056
+ "temp_avg_c": 68.5,
2057
+ "temp_max_c": 84.0,
2058
+ "power_total_avg_w": 1160.0,
2059
+ "power_total_max_w": 1177.71,
2060
+ "power_limit_total_w": 1200.0,
2061
+ "vram_used_avg_mb": 384778.0,
2062
+ "vram_used_max_mb": 384778.0,
2063
+ "vram_total_mb": 391548.0,
2064
+ "vram_used_avg_pct": 98.27,
2065
+ "vram_used_max_pct": 98.27,
2066
+ "pcie_rx_avg_mb_s": 22504.67,
2067
+ "pcie_rx_max_mb_s": 74746.0,
2068
+ "pcie_tx_avg_mb_s": 23453.89,
2069
+ "pcie_tx_max_mb_s": 70866.0
2070
+ }
2071
+ },
2072
+ {
2073
+ "concurrency": 4,
2074
+ "context_tokens": 32768,
2075
+ "benchmark_mode": "duration",
2076
+ "request_count_target": 0,
2077
+ "warmup_request_count": 0,
2078
+ "measurement_seconds": 19.777601,
2079
+ "measurement_wall_seconds": 20.001801,
2080
+ "client_output_tokens": 4295,
2081
+ "server_output_tokens": 4295,
2082
+ "aggregate_source": "openai_continuous_usage",
2083
+ "aggregate_tps": 217.16486009639328,
2084
+ "per_request_avg_tps": 54.29121502409832,
2085
+ "ttft_avg": 1.7167683002503158,
2086
+ "ttft_p50": 1.2517880250234157,
2087
+ "ttft_p90": 2.2701463484205306,
2088
+ "ttft_p99": 3.9549784664250893,
2089
+ "time_to_second_token_avg": 0.03188446167713174,
2090
+ "time_to_second_token_p50": 0.03449397184886038,
2091
+ "time_to_second_token_p90": 0.03738179220817983,
2092
+ "time_to_second_token_p99": 0.039646569741889834,
2093
+ "request_latency_avg": 9.444077699189075,
2094
+ "request_latency_p50": 9.303670058376156,
2095
+ "request_latency_p90": 10.359732999047264,
2096
+ "request_latency_p99": 11.179385469323025,
2097
+ "inter_token_latency_avg": 0.014419010797828693,
2098
+ "inter_token_latency_p50": 0.014763854274074246,
2099
+ "inter_token_latency_p90": 0.015734248097034612,
2100
+ "inter_token_latency_p99": 0.01580286538425228,
2101
+ "output_tps_per_user_avg": 70.02365221779164,
2102
+ "output_tps_per_user_p50": 67.73299041267488,
2103
+ "output_tps_per_user_p90": 73.71836215116325,
2104
+ "output_tps_per_user_p99": 90.54101765838234,
2105
+ "e2e_output_tps_per_user_avg": 54.73059414564044,
2106
+ "e2e_output_tps_per_user_p50": 55.03247278818422,
2107
+ "e2e_output_tps_per_user_p90": 60.046589541051475,
2108
+ "e2e_output_tps_per_user_p99": 65.10818536189939,
2109
+ "chunk_inter_token_latency_avg": 0.042334877835426166,
2110
+ "chunk_inter_token_latency_p50": 0.042865508716204204,
2111
+ "chunk_inter_token_latency_p90": 0.04529354933856685,
2112
+ "chunk_inter_token_latency_p99": 0.0453712748584173,
2113
+ "input_seq_len_avg": 32768.0,
2114
+ "output_seq_len_avg": 512.0,
2115
+ "output_seq_len_p50": 512.0,
2116
+ "output_seq_len_p90": 512.0,
2117
+ "output_seq_len_p99": 512.0,
2118
+ "request_count": 13,
2119
+ "completed_request_count": 10,
2120
+ "request_samples": [
2121
+ {
2122
+ "ttft": 0.6077568109612912,
2123
+ "time_to_second_token": 0.010776221984997392,
2124
+ "latency": 7.796489109983668,
2125
+ "inter_token_latency_avg": 0.014067969274016393,
2126
+ "chunk_inter_token_latency_avg": 0.04016051563699652,
2127
+ "input_tokens": 32768,
2128
+ "output_tokens": 512,
2129
+ "output_tps_per_user": 71.08346489261992,
2130
+ "e2e_output_tps_per_user": 65.67058489754916,
2131
+ "completed": true
2132
+ },
2133
+ {
2134
+ "ttft": 1.2430980298668146,
2135
+ "time_to_second_token": 0.032302224077284336,
2136
+ "latency": 8.787427563918754,
2137
+ "inter_token_latency_avg": 0.014763854274074246,
2138
+ "chunk_inter_token_latency_avg": 0.042865508716204204,
2139
+ "input_tokens": 32768,
2140
+ "output_tokens": 512,
2141
+ "output_tps_per_user": 67.73299041267488,
2142
+ "e2e_output_tps_per_user": 58.2650606535041,
2143
+ "completed": true
2144
+ },
2145
+ {
2146
+ "ttft": 1.24380440893583,
2147
+ "time_to_second_token": 0.03317554807290435,
2148
+ "latency": 9.216824874980375,
2149
+ "inter_token_latency_avg": 0.015602779776995196,
2150
+ "chunk_inter_token_latency_avg": 0.04530125264798037,
2151
+ "input_tokens": 32768,
2152
+ "output_tokens": 512,
2153
+ "output_tps_per_user": 64.09114364828787,
2154
+ "e2e_output_tps_per_user": 55.55058351926104,
2155
+ "completed": true
2156
+ },
2157
+ {
2158
+ "ttft": 1.2160316940862685,
2159
+ "time_to_second_token": 0.03449397184886038,
2160
+ "latency": 0.0,
2161
+ "inter_token_latency_avg": 0.010777328099156248,
2162
+ "chunk_inter_token_latency_avg": 0.03472694609728125,
2163
+ "input_tokens": 32768,
2164
+ "output_tokens": 30,
2165
+ "output_tps_per_user": 92.78737649995917,
2166
+ "e2e_output_tps_per_user": 0.0,
2167
+ "completed": false
2168
+ },
2169
+ {
2170
+ "ttft": 2.2123095341958106,
2171
+ "time_to_second_token": 0.026011183857917786,
2172
+ "latency": 9.824777495115995,
2173
+ "inter_token_latency_avg": 0.014897197575186271,
2174
+ "chunk_inter_token_latency_avg": 0.04349981691954392,
2175
+ "input_tokens": 32768,
2176
+ "output_tokens": 512,
2177
+ "output_tps_per_user": 67.1267193009284,
2178
+ "e2e_output_tps_per_user": 52.1131394837716,
2179
+ "completed": true
2180
+ },
2181
+ {
2182
+ "ttft": 2.2846055519767106,
2183
+ "time_to_second_token": 0.03989882906898856,
2184
+ "latency": 10.258541336050257,
2185
+ "inter_token_latency_avg": 0.015604571006014768,
2186
+ "chunk_inter_token_latency_avg": 0.04333660752213884,
2187
+ "input_tokens": 32768,
2188
+ "output_tokens": 512,
2189
+ "output_tps_per_user": 64.08378670676373,
2190
+ "e2e_output_tps_per_user": 49.909629763906594,
2191
+ "completed": true
2192
+ },
2193
+ {
2194
+ "ttft": 1.214527043979615,
2195
+ "time_to_second_token": 0.03522572107613087,
2196
+ "latency": 0.0,
2197
+ "inter_token_latency_avg": 0.014891711289040101,
2198
+ "chunk_inter_token_latency_avg": 0.0437039353047916,
2199
+ "input_tokens": 32768,
2200
+ "output_tokens": 406,
2201
+ "output_tps_per_user": 67.15144959437758,
2202
+ "e2e_output_tps_per_user": 0.0,
2203
+ "completed": false
2204
+ },
2205
+ {
2206
+ "ttft": 2.212038089055568,
2207
+ "time_to_second_token": 0.025989510817453265,
2208
+ "latency": 9.277765536913648,
2209
+ "inter_token_latency_avg": 0.013827255279565714,
2210
+ "chunk_inter_token_latency_avg": 0.04180903815300639,
2211
+ "input_tokens": 32768,
2212
+ "output_tokens": 512,
2213
+ "output_tps_per_user": 72.32093280853985,
2214
+ "e2e_output_tps_per_user": 55.185701553126606,
2215
+ "completed": true
2216
+ },
2217
+ {
2218
+ "ttft": 1.4273314119782299,
2219
+ "time_to_second_token": 0.0377966680098325,
2220
+ "latency": 8.616380715044215,
2221
+ "inter_token_latency_avg": 0.01406858963418001,
2222
+ "chunk_inter_token_latency_avg": 0.0425387532725798,
2223
+ "input_tokens": 32768,
2224
+ "output_tokens": 512,
2225
+ "output_tps_per_user": 71.08033043841677,
2226
+ "e2e_output_tps_per_user": 59.42170116810729,
2227
+ "completed": true
2228
+ },
2229
+ {
2230
+ "ttft": 1.2517880250234157,
2231
+ "time_to_second_token": 0.03299012500792742,
2232
+ "latency": 9.329574579838663,
2233
+ "inter_token_latency_avg": 0.015807801477133558,
2234
+ "chunk_inter_token_latency_avg": 0.045380823341658695,
2235
+ "input_tokens": 32768,
2236
+ "output_tokens": 512,
2237
+ "output_tps_per_user": 63.25990375363259,
2238
+ "e2e_output_tps_per_user": 54.879244023241846,
2239
+ "completed": true
2240
+ },
2241
+ {
2242
+ "ttft": 4.1827565911225975,
2243
+ "time_to_second_token": 0.03485451405867934,
2244
+ "latency": 11.27045796602033,
2245
+ "inter_token_latency_avg": 0.013870257093733334,
2246
+ "chunk_inter_token_latency_avg": 0.041939061389927416,
2247
+ "input_tokens": 32768,
2248
+ "output_tokens": 512,
2249
+ "output_tps_per_user": 72.09671697086321,
2250
+ "e2e_output_tps_per_user": 45.42850002578825,
2251
+ "completed": true
2252
+ },
2253
+ {
2254
+ "ttft": 2.005770788062364,
2255
+ "time_to_second_token": 0.03572228900156915,
2256
+ "latency": 10.062537814024836,
2257
+ "inter_token_latency_avg": 0.015766667369789572,
2258
+ "chunk_inter_token_latency_avg": 0.04526273610091276,
2259
+ "input_tokens": 32768,
2260
+ "output_tokens": 512,
2261
+ "output_tps_per_user": 63.424944317408176,
2262
+ "e2e_output_tps_per_user": 50.88179636814792,
2263
+ "completed": true
2264
+ },
2265
+ {
2266
+ "ttft": 1.2161699240095913,
2267
+ "time_to_second_token": 0.03526119492016733,
2268
+ "latency": 0.0,
2269
+ "inter_token_latency_avg": 0.013501158222887602,
2270
+ "chunk_inter_token_latency_avg": 0.03982841675751843,
2271
+ "input_tokens": 32768,
2272
+ "output_tokens": 355,
2273
+ "output_tps_per_user": 74.0677194868191,
2274
+ "e2e_output_tps_per_user": 0.0,
2275
+ "completed": false
2276
+ }
2277
+ ],
2278
+ "total_tokens": 4295,
2279
+ "wall_time": 29.076672724913806,
2280
+ "num_completed": 4,
2281
+ "num_errors": 0,
2282
+ "server_gen_throughput": 214.67688879215112,
2283
+ "server_utilization": 0.009771077610273626,
2284
+ "server_spec_accept_rate": 0.625,
2285
+ "server_spec_accept_length": 0.0,
2286
+ "avg_running_reqs": 4.0,
2287
+ "max_running_reqs": 4,
2288
+ "effective_concurrency": 4.0,
2289
+ "avg_queue_reqs": 0,
2290
+ "max_queue_reqs": 0,
2291
+ "queue_fraction": 0.0,
2292
+ "underfilled": false,
2293
+ "warmup_timed_out": false,
2294
+ "warmup_duration": 8.573,
2295
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
2296
+ "timeout_reason": "",
2297
+ "capacity_limited": false,
2298
+ "hardware_summary": {
2299
+ "samples": 9,
2300
+ "duration_seconds": 19.242,
2301
+ "gpu_count": 4,
2302
+ "cpu_util_avg_pct": 11.54,
2303
+ "cpu_temp_max_c": 76.75,
2304
+ "gpu_util_avg_pct": 100.0,
2305
+ "gpu_util_max_pct": 100.0,
2306
+ "mem_util_avg_pct": 31.81,
2307
+ "mem_util_max_pct": 46.0,
2308
+ "temp_avg_c": 68.78,
2309
+ "temp_max_c": 84.0,
2310
+ "power_total_avg_w": 1160.18,
2311
+ "power_total_max_w": 1172.54,
2312
+ "power_limit_total_w": 1200.0,
2313
+ "vram_used_avg_mb": 384778.0,
2314
+ "vram_used_max_mb": 384778.0,
2315
+ "vram_total_mb": 391548.0,
2316
+ "vram_used_avg_pct": 98.27,
2317
+ "vram_used_max_pct": 98.27,
2318
+ "pcie_rx_avg_mb_s": 19024.67,
2319
+ "pcie_rx_max_mb_s": 74971.0,
2320
+ "pcie_tx_avg_mb_s": 20175.22,
2321
+ "pcie_tx_max_mb_s": 67212.0
2322
+ }
2323
+ }
2324
+ ],
2325
+ "summary_table": {
2326
+ "0": {
2327
+ "1": 181.37569616773123,
2328
+ "2": 257.2872334123046,
2329
+ "4": 308.9773476463454
2330
+ },
2331
+ "8192": {
2332
+ "1": 161.46688548018994,
2333
+ "2": 179.48013878635928,
2334
+ "4": 217.85922059414168
2335
+ },
2336
+ "32768": {
2337
+ "1": 162.9385363574482,
2338
+ "2": 187.47192708215627,
2339
+ "4": 217.16486009639328
2340
+ }
2341
+ },
2342
+ "burst_results": [],
2343
+ "burst_summary_table": {},
2344
+ "methodology": {
2345
+ "prefill": {
2346
+ "name": "Prefill",
2347
+ "present": false,
2348
+ "mode": "skipped",
2349
+ "formula": "prompt_tokens / TTFT",
2350
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
2351
+ },
2352
+ "sustained_decode": {
2353
+ "name": "Sustained Decode",
2354
+ "present": true,
2355
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
2356
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
2357
+ },
2358
+ "burst_e2e_decode": {
2359
+ "name": "Burst / E2E Decode",
2360
+ "present": false,
2361
+ "status": "not run; use --run-burst",
2362
+ "formula": "sum(completion_tokens) / profiling_wall_time",
2363
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
2364
+ }
2365
+ }
2366
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.log ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0', '8k', '32k'] │
16
+ │ Duration: 20.0s per decode test | Max tokens: 512 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill: skipped | Sustained decode: 9 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
23
+ 7,337,984 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: skipped
26
+ Calibrating padding text (run=ftbfeppyynkm, up to 32k)...
27
+ 8k: 50,558 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,152 chars (32,768 prompt tokens via /tokenize)
29
+ Token targeting: /tokenize exact
30
+ Done.
31
+
32
+
33
+
34
+ llm-decode-bench v0.4.29
35
+ ╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
36
+ │ Sustained Decode │
37
+ │ Steady-state decode throughput after the engine has admitted the requested │
38
+ │ concurrency and passed warmup. Use this as the main tuning/regression signal │
39
+ │ for kernels, NCCL, DCP, MTP, and scheduler changes. │
40
+ ╰──────────────────────────────────────────────────────────────────────────────╯
41
+ Aggregate tok/s + TTFT/ITL
42
+ ╭────────────┬─────────────┬──────────────────┬────────────────────╮
43
+ │ ctx \ conc │ 1 │ 2 │ 4 │
44
+ ├────────────┼─────────────┼──────────────────┼────────────────────┤
45
+ │ 0 │ 181.4 84/5 │ 257.3 123/8 │ 309.0 (4/4) 181/12 │
46
+ │ 8k │ 161.5 625/5 │ 179.5 (2/2) 1k/8 │ ∅ (4/4)* 1k/15 │
47
+ │ 32k │ 162.9 628/5 │ 187.5 (2/2) 1k/8 │ 217.2 1k/15 │
48
+ ╰────────────┴─────────────┴──────────────────┴────────────────────╯
49
+ Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
50
+ (continuous completion_tokens when the server supports it). Prometheus is kept
51
+ as validation/scheduler data.
52
+ Aggregate source(s): openai_continuous_usage
53
+ ∅ = skipped/hidden because the cell does not fit in KV cache; exact deficit is
54
+ kept in JSON timeout_reason
55
+ (X/Y) = avg running / requested concurrency from Prometheus; * =
56
+ capacity-limited or warmup timed out
57
+ Per-Request tok/s
58
+ ╭────────────┬───────┬────────────┬────────────╮
59
+ │ ctx \ conc │ 1 │ 2 │ 4 │
60
+ ├────────────┼───────┼──���─────────┼────────────┤
61
+ │ 0 │ 181.4 │ 128.6 │ 77.2 (4/4) │
62
+ │ 8k │ 161.5 │ 89.7 (2/2) │ ∅ (4/4)* │
63
+ │ 32k │ 162.9 │ 93.7 (2/2) │ 54.3 │
64
+ ╰────────────┴───────┴────────────┴────────────╯
65
+ Client request latency: p50 / p90 ms
66
+ ╭────────────┬───────────┬───────────┬────────────╮
67
+ │ ctx \ conc │ 1 │ 2 │ 4 │
68
+ ├────────────┼───────────┼───────────┼────────────┤
69
+ │ 0 │ 2.8k/2.9k │ 3.9k/4.2k │ 6.4k/6.9k │
70
+ │ 8k │ 3.2k/3.3k │ 5.4k/5.9k │ 9.3k/11.3k │
71
+ │ 32k │ 3.2k/3.2k │ 5.4k/5.6k │ 9.3k/10.4k │
72
+ ╰────────────┴───────────┴───────────┴────────────╯
73
+ Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
74
+ coordinate. ITL is computed from observed generated tokens, including streams
75
+ stopped at the measurement boundary; a missing ITL means no stream produced at
76
+ least two measured output tokens. Per-request tok/s and request latency are
77
+ shown in separate per-cell matrices. Completion/sample counts and full
78
+ request-level distributions remain in JSON under request_samples.
79
+ Sustained mode: client latency metrics explain request UX variance; aggregate
80
+ tok/s remains the primary throughput signal.
81
+ ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
82
+ Hardware Summary
83
+ ╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
84
+ │ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
85
+ ├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
86
+ │ 0 │ │ sust… │ 99/99% │ 44% │ 1151/1… │ 82C │ 76C │ 98… │ 8369/8254 │
87
+ │ … │ │ sust… │ 99/100% │ 40% │ 1153/1… │ 83C │ 76C │ 98… │ 12913/11281 │
88
+ │ … │ │ sust… │ 98/100% │ 38% │ 1150/1… │ 84C │ 76C │ 98… │ 8327/8133 │
89
+ │ 0 │ │ sust… │ 100/100% │ 40% │ 1173/1… │ 84C │ 76C │ 98… │ 11187/11117 │
90
+ │ 0 │ │ sust… │ 100/100% │ 35% │ 1170/1… │ 84C │ 77C │ 98… │ 7813/8054 │
91
+ │ … │ │ sust… │ 100/100% │ 36% │ 1162/1… │ 84C │ 76C │ 98… │ 29259/28011 │
92
+ │ … │ │ sust… │ 100/100% │ 34% │ 1159/1… │ 84C │ 76C │ 98… │ 27724/27474 │
93
+ │ … │ │ sust… │ 100/100% │ 36% │ 1160/1… │ 84C │ 76C │ 98… │ 22505/23454 │
94
+ │ … │ │ sust… │ 100/100% │ 32% │ 1160/1… │ 84C │ 77C │ 98… │ 19025/20175 │
95
+ ╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
96
+ ╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
97
+ │ avg 1,098 W | max 1,178 W | limit 1,200 W | over 4m 30s | 113 samples │
98
+ ╰───────────────────────────────────────────────────────────────────────╯
99
+ Hardware summary is sampled from nvidia-smi during the measured part of each
100
+ cell. Whole-run GPU power is the sampled sum of GPU power draw across the
101
+ complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
102
+ a coarse live diagnostic, not a per-kernel NCCL profiler.
103
+
104
+ ╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
105
+ │ Burst / E2E Decode │
106
+ │ Not run. Re-run with --run-burst to append a finite client-facing request │
107
+ │ burst after Sustained Decode. This is intentionally disabled by default │
108
+ │ because it adds another full decode matrix. │
109
+ ╰──────────────────────────────────────────────────────────────────────────────╯
110
+
111
+ ╭────────────────────────────── Primary Summary ───────────────────────────────╮
112
+ │ Primary matrices repeated last so the important numbers are visible without │
113
+ │ scrolling back through diagnostics. │
114
+ ╰──────────────────────────────────────────────────────────────────────────────╯
115
+ Aggregate decode tok/s
116
+ ╭────────────┬───────┬─────────────┬─────────────╮
117
+ │ ctx \ conc │ 1 │ 2 │ 4 │
118
+ ├────────────┼───────┼─────────────┼─────────────┤
119
+ │ 0 │ 181.4 │ 257.3 │ 309.0 (4/4) │
120
+ │ 8k │ 161.5 │ 179.5 (2/2) │ ∅ (4/4)* │
121
+ │ 32k │ 162.9 │ 187.5 (2/2) │ 217.2 │
122
+ ╰────────────┴───────┴─────────────┴─────────────╯
123
+
124
+ Results saved to
125
+ <campaign>/candidate-speed-wi
126
+ ndow-01/results-01/decode-warp-quant/rep-1/decode-cap512.json
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-command.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json",
20
+ "--contexts",
21
+ "0,8k,32k",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--skip-prefill",
25
+ "--cell-warmup-timeout-seconds",
26
+ "180"
27
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "b236cf123d887ede3ee9c5ba2a6ef16597fbf3e99450397dcb46488a24dd99b3"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json ADDED
@@ -0,0 +1,1394 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:22:18.084779",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 32768,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0,
42
+ 8192,
43
+ 32768
44
+ ],
45
+ "startup_diagnostics_available": true,
46
+ "nvidia_p2p_override_effective": true,
47
+ "p2pmark_status": "not_run",
48
+ "amd_fabric_status": "not_run"
49
+ },
50
+ "startup_diagnostics": {
51
+ "version": "0.4.29",
52
+ "server_url": "http://127.0.0.1:8001",
53
+ "hostname": "<host>",
54
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
55
+ "env": {},
56
+ "args": {
57
+ "concurrency": "1,2,4",
58
+ "contexts": "0,8k,32k",
59
+ "max_tokens": 8192,
60
+ "duration": 20.0,
61
+ "request_count": 0,
62
+ "run_burst": false,
63
+ "standalone_prefill": false,
64
+ "prefill_only": false,
65
+ "skip_prefill": true,
66
+ "prefill_contexts": "8k,64k,128k",
67
+ "prefill_metric": "client",
68
+ "dcp_size": 0,
69
+ "kv_budget": 0
70
+ },
71
+ "nvidia_p2p_override": {
72
+ "effective": true,
73
+ "configured": true,
74
+ "params_path": "/proc/driver/nvidia/params",
75
+ "params_available": true,
76
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
77
+ "modprobe_available": true,
78
+ "runtime": {
79
+ "ForceP2P": "0x11",
80
+ "RMForceP2PType": "1",
81
+ "RMPcieP2PType": "2",
82
+ "GrdmaPciTopoCheckOverride": "1",
83
+ "EnableResizableBar": "1",
84
+ "DmaRemapPeerMmio": "1"
85
+ },
86
+ "expected": {
87
+ "ForceP2P": "0x11",
88
+ "RMForceP2PType": "1",
89
+ "RMPcieP2PType": "2",
90
+ "GrdmaPciTopoCheckOverride": "1",
91
+ "EnableResizableBar": "1"
92
+ },
93
+ "missing": [],
94
+ "mismatched": {},
95
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
96
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
97
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
98
+ },
99
+ "p2pmark": {
100
+ "status": "not_run"
101
+ },
102
+ "amd_fabric": {
103
+ "status": "not_run"
104
+ },
105
+ "nvidia_smi_query": {
106
+ "cmd": [
107
+ "nvidia-smi",
108
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
109
+ "--format=csv,noheader,nounits"
110
+ ],
111
+ "returncode": 0,
112
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
113
+ "stderr": ""
114
+ },
115
+ "nvidia_smi_topo": {
116
+ "cmd": [
117
+ "nvidia-smi",
118
+ "topo",
119
+ "-m"
120
+ ],
121
+ "returncode": 0,
122
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
123
+ "stderr": ""
124
+ }
125
+ },
126
+ "nvidia_p2p_override": {
127
+ "effective": true,
128
+ "configured": true,
129
+ "params_path": "/proc/driver/nvidia/params",
130
+ "params_available": true,
131
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
132
+ "modprobe_available": true,
133
+ "runtime": {
134
+ "ForceP2P": "0x11",
135
+ "RMForceP2PType": "1",
136
+ "RMPcieP2PType": "2",
137
+ "GrdmaPciTopoCheckOverride": "1",
138
+ "EnableResizableBar": "1",
139
+ "DmaRemapPeerMmio": "1"
140
+ },
141
+ "expected": {
142
+ "ForceP2P": "0x11",
143
+ "RMForceP2PType": "1",
144
+ "RMPcieP2PType": "2",
145
+ "GrdmaPciTopoCheckOverride": "1",
146
+ "EnableResizableBar": "1"
147
+ },
148
+ "missing": [],
149
+ "mismatched": {},
150
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
151
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
152
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
153
+ },
154
+ "p2pmark": {
155
+ "status": "not_run"
156
+ },
157
+ "amd_fabric": {
158
+ "status": "not_run"
159
+ },
160
+ "hardware_run_summary": {
161
+ "samples": 114,
162
+ "duration_seconds": 272.443,
163
+ "gpu_count": 4,
164
+ "cpu_util_avg_pct": 10.97,
165
+ "cpu_temp_max_c": 76.25,
166
+ "gpu_util_avg_pct": 90.62,
167
+ "gpu_util_max_pct": 100.0,
168
+ "mem_util_avg_pct": 35.29,
169
+ "mem_util_max_pct": 57.0,
170
+ "temp_avg_c": 66.66,
171
+ "temp_max_c": 84.0,
172
+ "power_total_avg_w": 1094.77,
173
+ "power_total_max_w": 1178.23,
174
+ "power_limit_total_w": 1200.0,
175
+ "vram_used_avg_mb": 384774.21,
176
+ "vram_used_max_mb": 384778.0,
177
+ "vram_total_mb": 391548.0,
178
+ "vram_used_avg_pct": 98.27,
179
+ "vram_used_max_pct": 98.27,
180
+ "pcie_rx_avg_mb_s": 12350.85,
181
+ "pcie_rx_max_mb_s": 60894.0,
182
+ "pcie_tx_avg_mb_s": 12097.24,
183
+ "pcie_tx_max_mb_s": 58766.0
184
+ },
185
+ "event_log": [
186
+ "02:17:43 benchmark start engine=vllm",
187
+ "02:17:43 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
188
+ "02:17:43 startup decode concurrency=1,2,4 contexts=0,8k,32k",
189
+ "02:17:43 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
190
+ "02:17:43 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
191
+ "02:17:43 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
192
+ "02:17:43 startup model context length: 1,000,000 tokens",
193
+ "02:17:43 startup prefill tests: skipped",
194
+ "02:17:43 startup calibrating padding text run=rumxqbvzafby up_to=32k",
195
+ "02:17:43 startup context 8k: 50,540 chars (8,192 prompt tokens via /tokenize)",
196
+ "02:17:43 startup context 32k: 205,139 chars (32,768 prompt tokens via /tokenize)",
197
+ "02:17:43 startup token targeting: /tokenize exact",
198
+ "02:17:43 startup startup preparation done",
199
+ "02:17:43 hardware monitor interval=2s",
200
+ "02:17:43 decode warmup start",
201
+ "02:17:43 decode warmup start C=1 ctx=32k 3s",
202
+ "02:17:43 cell start C=1 ctx=32k",
203
+ "02:17:51 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
204
+ "02:17:54 cell done C=1 ctx=32k 187.9 tok/s",
205
+ "02:17:54 decode warmup done C=1 ctx=32k",
206
+ "02:17:56 cell start C=1 ctx=0",
207
+ "02:18:02 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
208
+ "02:18:22 cell done C=1 ctx=0 185.7 tok/s",
209
+ "02:18:24 cell start C=1 ctx=8k",
210
+ "02:18:30 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
211
+ "02:18:50 cell done C=1 ctx=8k 166.9 tok/s",
212
+ "02:18:52 cell start C=1 ctx=32k",
213
+ "02:19:01 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
214
+ "02:19:21 cell done C=1 ctx=32k 173.8 tok/s",
215
+ "02:19:23 cell start C=2 ctx=0",
216
+ "02:19:28 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
217
+ "02:19:49 cell done C=2 ctx=0 246.2 tok/s",
218
+ "02:19:51 cell start C=4 ctx=0",
219
+ "02:19:56 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
220
+ "02:20:16 cell done C=4 ctx=0 303.4 tok/s",
221
+ "02:20:18 cell start C=2 ctx=8k",
222
+ "02:20:24 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
223
+ "02:20:44 cell done C=2 ctx=8k 236.9 tok/s",
224
+ "02:20:46 cell start C=4 ctx=8k",
225
+ "02:20:57 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
226
+ "02:21:17 cell done C=4 ctx=8k 291.1 tok/s",
227
+ "02:21:19 cell start C=2 ctx=32k",
228
+ "02:21:25 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
229
+ "02:21:45 cell done C=2 ctx=32k 231.8 tok/s",
230
+ "02:21:47 cell start C=4 ctx=32k",
231
+ "02:21:56 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
232
+ "02:22:16 cell done C=4 ctx=32k 303.9 tok/s"
233
+ ],
234
+ "prefill": {},
235
+ "results": [
236
+ {
237
+ "concurrency": 1,
238
+ "context_tokens": 0,
239
+ "benchmark_mode": "duration",
240
+ "request_count_target": 0,
241
+ "warmup_request_count": 0,
242
+ "measurement_seconds": 19.990511,
243
+ "measurement_wall_seconds": 20.000618,
244
+ "client_output_tokens": 3712,
245
+ "server_output_tokens": 3712,
246
+ "aggregate_source": "openai_continuous_usage",
247
+ "aggregate_tps": 185.68810232786203,
248
+ "per_request_avg_tps": 185.68810232786203,
249
+ "ttft_avg": 0.07161088101565838,
250
+ "ttft_p50": 0.07161088101565838,
251
+ "ttft_p90": 0.07161088101565838,
252
+ "ttft_p99": 0.07161088101565838,
253
+ "time_to_second_token_avg": 0.01281460584141314,
254
+ "time_to_second_token_p50": 0.01281460584141314,
255
+ "time_to_second_token_p90": 0.01281460584141314,
256
+ "time_to_second_token_p99": 0.01281460584141314,
257
+ "request_latency_avg": 0.0,
258
+ "request_latency_p50": 0.0,
259
+ "request_latency_p90": 0.0,
260
+ "request_latency_p99": 0.0,
261
+ "inter_token_latency_avg": 0.005329198802730078,
262
+ "inter_token_latency_p50": 0.005329198802730078,
263
+ "inter_token_latency_p90": 0.005329198802730078,
264
+ "inter_token_latency_p99": 0.005329198802730078,
265
+ "output_tps_per_user_avg": 187.64546736138146,
266
+ "output_tps_per_user_p50": 187.64546736138146,
267
+ "output_tps_per_user_p90": 187.64546736138146,
268
+ "output_tps_per_user_p99": 187.64546736138146,
269
+ "e2e_output_tps_per_user_avg": 0.0,
270
+ "e2e_output_tps_per_user_p50": 0.0,
271
+ "e2e_output_tps_per_user_p90": 0.0,
272
+ "e2e_output_tps_per_user_p99": 0.0,
273
+ "chunk_inter_token_latency_avg": 0.014062018498253509,
274
+ "chunk_inter_token_latency_p50": 0.014062018498253509,
275
+ "chunk_inter_token_latency_p90": 0.014062018498253509,
276
+ "chunk_inter_token_latency_p99": 0.014062018498253509,
277
+ "input_seq_len_avg": 78.0,
278
+ "output_seq_len_avg": 4777.0,
279
+ "output_seq_len_p50": 4777.0,
280
+ "output_seq_len_p90": 4777.0,
281
+ "output_seq_len_p99": 4777.0,
282
+ "request_count": 1,
283
+ "completed_request_count": 0,
284
+ "request_samples": [
285
+ {
286
+ "ttft": 0.07161088101565838,
287
+ "time_to_second_token": 0.01281460584141314,
288
+ "latency": 0.0,
289
+ "inter_token_latency_avg": 0.005329198802730078,
290
+ "chunk_inter_token_latency_avg": 0.014062018498253509,
291
+ "input_tokens": 78,
292
+ "output_tokens": 4777,
293
+ "output_tps_per_user": 187.64546736138146,
294
+ "e2e_output_tps_per_user": 0.0,
295
+ "completed": false
296
+ }
297
+ ],
298
+ "total_tokens": 3712,
299
+ "wall_time": 25.54038013308309,
300
+ "num_completed": 1,
301
+ "num_errors": 0,
302
+ "server_gen_throughput": 185.54062187903287,
303
+ "server_utilization": 0.005862646566164198,
304
+ "server_spec_accept_rate": 0.5648148148148148,
305
+ "server_spec_accept_length": 0.0,
306
+ "avg_running_reqs": 1,
307
+ "max_running_reqs": 1,
308
+ "effective_concurrency": 1,
309
+ "avg_queue_reqs": 0,
310
+ "max_queue_reqs": 0,
311
+ "queue_fraction": 0.0,
312
+ "underfilled": false,
313
+ "warmup_timed_out": false,
314
+ "warmup_duration": 5.533,
315
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
316
+ "timeout_reason": "",
317
+ "capacity_limited": false,
318
+ "hardware_summary": {
319
+ "samples": 9,
320
+ "duration_seconds": 19.316,
321
+ "gpu_count": 4,
322
+ "cpu_util_avg_pct": 11.53,
323
+ "cpu_temp_max_c": 75.25,
324
+ "gpu_util_avg_pct": 99.0,
325
+ "gpu_util_max_pct": 99.0,
326
+ "mem_util_avg_pct": 43.92,
327
+ "mem_util_max_pct": 56.0,
328
+ "temp_avg_c": 65.25,
329
+ "temp_max_c": 80.0,
330
+ "power_total_avg_w": 1149.49,
331
+ "power_total_max_w": 1153.43,
332
+ "power_limit_total_w": 1200.0,
333
+ "vram_used_avg_mb": 384770.0,
334
+ "vram_used_max_mb": 384770.0,
335
+ "vram_total_mb": 391548.0,
336
+ "vram_used_avg_pct": 98.27,
337
+ "vram_used_max_pct": 98.27,
338
+ "pcie_rx_avg_mb_s": 8501.89,
339
+ "pcie_rx_max_mb_s": 8783.0,
340
+ "pcie_tx_avg_mb_s": 8302.22,
341
+ "pcie_tx_max_mb_s": 8578.0
342
+ }
343
+ },
344
+ {
345
+ "concurrency": 1,
346
+ "context_tokens": 8192,
347
+ "benchmark_mode": "duration",
348
+ "request_count_target": 0,
349
+ "warmup_request_count": 0,
350
+ "measurement_seconds": 20.000286,
351
+ "measurement_wall_seconds": 20.000336,
352
+ "client_output_tokens": 3338,
353
+ "server_output_tokens": 3338,
354
+ "aggregate_source": "openai_continuous_usage",
355
+ "aggregate_tps": 166.89761277085296,
356
+ "per_request_avg_tps": 166.89761277085296,
357
+ "ttft_avg": 0.5802665068767965,
358
+ "ttft_p50": 0.5802665068767965,
359
+ "ttft_p90": 0.5802665068767965,
360
+ "ttft_p99": 0.5802665068767965,
361
+ "time_to_second_token_avg": 0.016474336152896285,
362
+ "time_to_second_token_p50": 0.016474336152896285,
363
+ "time_to_second_token_p90": 0.016474336152896285,
364
+ "time_to_second_token_p99": 0.016474336152896285,
365
+ "request_latency_avg": 0.0,
366
+ "request_latency_p50": 0.0,
367
+ "request_latency_p90": 0.0,
368
+ "request_latency_p99": 0.0,
369
+ "inter_token_latency_avg": 0.0059050409250968015,
370
+ "inter_token_latency_p50": 0.0059050409250968015,
371
+ "inter_token_latency_p90": 0.0059050409250968015,
372
+ "inter_token_latency_p99": 0.0059050409250968015,
373
+ "output_tps_per_user_avg": 169.34683648845447,
374
+ "output_tps_per_user_p50": 169.34683648845447,
375
+ "output_tps_per_user_p90": 169.34683648845447,
376
+ "output_tps_per_user_p99": 169.34683648845447,
377
+ "e2e_output_tps_per_user_avg": 0.0,
378
+ "e2e_output_tps_per_user_p50": 0.0,
379
+ "e2e_output_tps_per_user_p90": 0.0,
380
+ "e2e_output_tps_per_user_p99": 0.0,
381
+ "chunk_inter_token_latency_avg": 0.014231043370286765,
382
+ "chunk_inter_token_latency_p50": 0.014231043370286765,
383
+ "chunk_inter_token_latency_p90": 0.014231043370286765,
384
+ "chunk_inter_token_latency_p99": 0.014231043370286765,
385
+ "input_seq_len_avg": 8192.0,
386
+ "output_seq_len_avg": 4057.0,
387
+ "output_seq_len_p50": 4057.0,
388
+ "output_seq_len_p90": 4057.0,
389
+ "output_seq_len_p99": 4057.0,
390
+ "request_count": 1,
391
+ "completed_request_count": 0,
392
+ "request_samples": [
393
+ {
394
+ "ttft": 0.5802665068767965,
395
+ "time_to_second_token": 0.016474336152896285,
396
+ "latency": 0.0,
397
+ "inter_token_latency_avg": 0.0059050409250968015,
398
+ "chunk_inter_token_latency_avg": 0.014231043370286765,
399
+ "input_tokens": 8192,
400
+ "output_tokens": 4057,
401
+ "output_tps_per_user": 169.34683648845447,
402
+ "e2e_output_tps_per_user": 0.0,
403
+ "completed": false
404
+ }
405
+ ],
406
+ "total_tokens": 3338,
407
+ "wall_time": 26.075741773936898,
408
+ "num_completed": 1,
409
+ "num_errors": 0,
410
+ "server_gen_throughput": 166.8524947323479,
411
+ "server_utilization": 0.006141820212172022,
412
+ "server_spec_accept_rate": 0.3286384976525822,
413
+ "server_spec_accept_length": 0.0,
414
+ "avg_running_reqs": 1,
415
+ "max_running_reqs": 1,
416
+ "effective_concurrency": 1,
417
+ "avg_queue_reqs": 0,
418
+ "max_queue_reqs": 0,
419
+ "queue_fraction": 0.0,
420
+ "underfilled": false,
421
+ "warmup_timed_out": false,
422
+ "warmup_duration": 6.06,
423
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
424
+ "timeout_reason": "",
425
+ "capacity_limited": false,
426
+ "hardware_summary": {
427
+ "samples": 8,
428
+ "duration_seconds": 16.922,
429
+ "gpu_count": 4,
430
+ "cpu_util_avg_pct": 11.59,
431
+ "cpu_temp_max_c": 75.0,
432
+ "gpu_util_avg_pct": 99.0,
433
+ "gpu_util_max_pct": 99.0,
434
+ "mem_util_avg_pct": 44.31,
435
+ "mem_util_max_pct": 56.0,
436
+ "temp_avg_c": 65.97,
437
+ "temp_max_c": 81.0,
438
+ "power_total_avg_w": 1152.41,
439
+ "power_total_max_w": 1153.89,
440
+ "power_limit_total_w": 1200.0,
441
+ "vram_used_avg_mb": 384770.0,
442
+ "vram_used_max_mb": 384770.0,
443
+ "vram_total_mb": 391548.0,
444
+ "vram_used_avg_pct": 98.27,
445
+ "vram_used_max_pct": 98.27,
446
+ "pcie_rx_avg_mb_s": 8303.25,
447
+ "pcie_rx_max_mb_s": 8596.0,
448
+ "pcie_tx_avg_mb_s": 8386.88,
449
+ "pcie_tx_max_mb_s": 8513.0
450
+ }
451
+ },
452
+ {
453
+ "concurrency": 1,
454
+ "context_tokens": 32768,
455
+ "benchmark_mode": "duration",
456
+ "request_count_target": 0,
457
+ "warmup_request_count": 0,
458
+ "measurement_seconds": 19.989714,
459
+ "measurement_wall_seconds": 20.000821,
460
+ "client_output_tokens": 3475,
461
+ "server_output_tokens": 3475,
462
+ "aggregate_source": "openai_continuous_usage",
463
+ "aggregate_tps": 173.8394059270179,
464
+ "per_request_avg_tps": 173.8394059270179,
465
+ "ttft_avg": 0.5945898578502238,
466
+ "ttft_p50": 0.5945898578502238,
467
+ "ttft_p90": 0.5945898578502238,
468
+ "ttft_p99": 0.5945898578502238,
469
+ "time_to_second_token_avg": 0.011255914112553,
470
+ "time_to_second_token_p50": 0.011255914112553,
471
+ "time_to_second_token_p90": 0.011255914112553,
472
+ "time_to_second_token_p99": 0.011255914112553,
473
+ "request_latency_avg": 0.0,
474
+ "request_latency_p50": 0.0,
475
+ "request_latency_p90": 0.0,
476
+ "request_latency_p99": 0.0,
477
+ "inter_token_latency_avg": 0.005599262254130903,
478
+ "inter_token_latency_p50": 0.005599262254130903,
479
+ "inter_token_latency_p90": 0.005599262254130903,
480
+ "inter_token_latency_p99": 0.005599262254130903,
481
+ "output_tps_per_user_avg": 178.59495673063742,
482
+ "output_tps_per_user_p50": 178.59495673063742,
483
+ "output_tps_per_user_p90": 178.59495673063742,
484
+ "output_tps_per_user_p99": 178.59495673063742,
485
+ "e2e_output_tps_per_user_avg": 0.0,
486
+ "e2e_output_tps_per_user_p50": 0.0,
487
+ "e2e_output_tps_per_user_p90": 0.0,
488
+ "e2e_output_tps_per_user_p99": 0.0,
489
+ "chunk_inter_token_latency_avg": 0.014177278953883576,
490
+ "chunk_inter_token_latency_p50": 0.014177278953883576,
491
+ "chunk_inter_token_latency_p90": 0.014177278953883576,
492
+ "chunk_inter_token_latency_p99": 0.014177278953883576,
493
+ "input_seq_len_avg": 32768.0,
494
+ "output_seq_len_avg": 4275.0,
495
+ "output_seq_len_p50": 4275.0,
496
+ "output_seq_len_p90": 4275.0,
497
+ "output_seq_len_p99": 4275.0,
498
+ "request_count": 1,
499
+ "completed_request_count": 0,
500
+ "request_samples": [
501
+ {
502
+ "ttft": 0.5945898578502238,
503
+ "time_to_second_token": 0.011255914112553,
504
+ "latency": 0.0,
505
+ "inter_token_latency_avg": 0.005599262254130903,
506
+ "chunk_inter_token_latency_avg": 0.014177278953883576,
507
+ "input_tokens": 32768,
508
+ "output_tokens": 4275,
509
+ "output_tps_per_user": 178.59495673063742,
510
+ "e2e_output_tps_per_user": 0.0,
511
+ "completed": false
512
+ }
513
+ ],
514
+ "total_tokens": 3475,
515
+ "wall_time": 29.124992428114638,
516
+ "num_completed": 1,
517
+ "num_errors": 0,
518
+ "server_gen_throughput": 173.70034429070287,
519
+ "server_utilization": 0.006979341150195384,
520
+ "server_spec_accept_rate": 0.38967136150234744,
521
+ "server_spec_accept_length": 0.0,
522
+ "avg_running_reqs": 1,
523
+ "max_running_reqs": 1,
524
+ "effective_concurrency": 1,
525
+ "avg_queue_reqs": 0,
526
+ "max_queue_reqs": 0,
527
+ "queue_fraction": 0.0,
528
+ "underfilled": false,
529
+ "warmup_timed_out": false,
530
+ "warmup_duration": 9.119,
531
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
532
+ "timeout_reason": "",
533
+ "capacity_limited": false,
534
+ "hardware_summary": {
535
+ "samples": 8,
536
+ "duration_seconds": 16.869,
537
+ "gpu_count": 4,
538
+ "cpu_util_avg_pct": 11.56,
539
+ "cpu_temp_max_c": 74.88,
540
+ "gpu_util_avg_pct": 99.0,
541
+ "gpu_util_max_pct": 99.0,
542
+ "mem_util_avg_pct": 43.78,
543
+ "mem_util_max_pct": 55.0,
544
+ "temp_avg_c": 66.56,
545
+ "temp_max_c": 82.0,
546
+ "power_total_avg_w": 1153.2,
547
+ "power_total_max_w": 1153.93,
548
+ "power_limit_total_w": 1200.0,
549
+ "vram_used_avg_mb": 384770.0,
550
+ "vram_used_max_mb": 384770.0,
551
+ "vram_total_mb": 391548.0,
552
+ "vram_used_avg_pct": 98.27,
553
+ "vram_used_max_pct": 98.27,
554
+ "pcie_rx_avg_mb_s": 8358.0,
555
+ "pcie_rx_max_mb_s": 8664.0,
556
+ "pcie_tx_avg_mb_s": 8305.5,
557
+ "pcie_tx_max_mb_s": 8516.0
558
+ }
559
+ },
560
+ {
561
+ "concurrency": 2,
562
+ "context_tokens": 0,
563
+ "benchmark_mode": "duration",
564
+ "request_count_target": 0,
565
+ "warmup_request_count": 0,
566
+ "measurement_seconds": 19.998344,
567
+ "measurement_wall_seconds": 20.000475,
568
+ "client_output_tokens": 4924,
569
+ "server_output_tokens": 4924,
570
+ "aggregate_source": "openai_continuous_usage",
571
+ "aggregate_tps": 246.22038564672286,
572
+ "per_request_avg_tps": 123.11019282336143,
573
+ "ttft_avg": 0.11326407559681684,
574
+ "ttft_p50": 0.11326407559681684,
575
+ "ttft_p90": 0.14774754794780165,
576
+ "ttft_p99": 0.15550632922677324,
577
+ "time_to_second_token_avg": 0.015398422023281455,
578
+ "time_to_second_token_p50": 0.015398422023281455,
579
+ "time_to_second_token_p90": 0.017018200410529972,
580
+ "time_to_second_token_p99": 0.01738265054766089,
581
+ "request_latency_avg": 0.0,
582
+ "request_latency_p50": 0.0,
583
+ "request_latency_p90": 0.0,
584
+ "request_latency_p99": 0.0,
585
+ "inter_token_latency_avg": 0.007996202209365196,
586
+ "inter_token_latency_p50": 0.007996202209365196,
587
+ "inter_token_latency_p90": 0.008062418200969368,
588
+ "inter_token_latency_p99": 0.008077316799080308,
589
+ "output_tps_per_user_avg": 125.07276978039832,
590
+ "output_tps_per_user_p50": 125.07276978039832,
591
+ "output_tps_per_user_p90": 126.10848864503505,
592
+ "output_tps_per_user_p99": 126.34152538957832,
593
+ "e2e_output_tps_per_user_avg": 0.0,
594
+ "e2e_output_tps_per_user_p50": 0.0,
595
+ "e2e_output_tps_per_user_p90": 0.0,
596
+ "e2e_output_tps_per_user_p99": 0.0,
597
+ "chunk_inter_token_latency_avg": 0.02072697419690292,
598
+ "chunk_inter_token_latency_p50": 0.02072697419690292,
599
+ "chunk_inter_token_latency_p90": 0.020734919510937286,
600
+ "chunk_inter_token_latency_p99": 0.02073670720659502,
601
+ "input_seq_len_avg": 78.0,
602
+ "output_seq_len_avg": 3180.5,
603
+ "output_seq_len_p50": 3180.5,
604
+ "output_seq_len_p90": 3202.5,
605
+ "output_seq_len_p99": 3207.45,
606
+ "request_count": 2,
607
+ "completed_request_count": 0,
608
+ "request_samples": [
609
+ {
610
+ "ttft": 0.07015973515808582,
611
+ "time_to_second_token": 0.01337369903922081,
612
+ "latency": 0.0,
613
+ "inter_token_latency_avg": 0.008078972198870412,
614
+ "chunk_inter_token_latency_avg": 0.020736905839445877,
615
+ "input_tokens": 78,
616
+ "output_tokens": 3153,
617
+ "output_tps_per_user": 123.77812119960238,
618
+ "e2e_output_tps_per_user": 0.0,
619
+ "completed": false
620
+ },
621
+ {
622
+ "ttft": 0.15636841603554785,
623
+ "time_to_second_token": 0.0174231450073421,
624
+ "latency": 0.0,
625
+ "inter_token_latency_avg": 0.007913432219859979,
626
+ "chunk_inter_token_latency_avg": 0.02071704255435996,
627
+ "input_tokens": 78,
628
+ "output_tokens": 3208,
629
+ "output_tps_per_user": 126.36741836119424,
630
+ "e2e_output_tps_per_user": 0.0,
631
+ "completed": false
632
+ }
633
+ ],
634
+ "total_tokens": 4924,
635
+ "wall_time": 25.577398049877957,
636
+ "num_completed": 2,
637
+ "num_errors": 0,
638
+ "server_gen_throughput": 246.13329155338783,
639
+ "server_utilization": 0.005862646566164198,
640
+ "server_spec_accept_rate": 0.47959183673469385,
641
+ "server_spec_accept_length": 0.0,
642
+ "avg_running_reqs": 2,
643
+ "max_running_reqs": 2,
644
+ "effective_concurrency": 2,
645
+ "avg_queue_reqs": 0,
646
+ "max_queue_reqs": 0,
647
+ "queue_fraction": 0.0,
648
+ "underfilled": false,
649
+ "warmup_timed_out": false,
650
+ "warmup_duration": 5.537,
651
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
652
+ "timeout_reason": "",
653
+ "capacity_limited": false,
654
+ "hardware_summary": {
655
+ "samples": 9,
656
+ "duration_seconds": 19.327,
657
+ "gpu_count": 4,
658
+ "cpu_util_avg_pct": 11.58,
659
+ "cpu_temp_max_c": 75.75,
660
+ "gpu_util_avg_pct": 100.0,
661
+ "gpu_util_max_pct": 100.0,
662
+ "mem_util_avg_pct": 40.89,
663
+ "mem_util_max_pct": 50.0,
664
+ "temp_avg_c": 67.36,
665
+ "temp_max_c": 83.0,
666
+ "power_total_avg_w": 1176.5,
667
+ "power_total_max_w": 1178.11,
668
+ "power_limit_total_w": 1200.0,
669
+ "vram_used_avg_mb": 384770.0,
670
+ "vram_used_max_mb": 384770.0,
671
+ "vram_total_mb": 391548.0,
672
+ "vram_used_avg_pct": 98.27,
673
+ "vram_used_max_pct": 98.27,
674
+ "pcie_rx_avg_mb_s": 11309.78,
675
+ "pcie_rx_max_mb_s": 11397.0,
676
+ "pcie_tx_avg_mb_s": 11179.56,
677
+ "pcie_tx_max_mb_s": 11268.0
678
+ }
679
+ },
680
+ {
681
+ "concurrency": 4,
682
+ "context_tokens": 0,
683
+ "benchmark_mode": "duration",
684
+ "request_count_target": 0,
685
+ "warmup_request_count": 0,
686
+ "measurement_seconds": 19.993715,
687
+ "measurement_wall_seconds": 20.000812,
688
+ "client_output_tokens": 6067,
689
+ "server_output_tokens": 6067,
690
+ "aggregate_source": "openai_continuous_usage",
691
+ "aggregate_tps": 303.4453614548168,
692
+ "per_request_avg_tps": 75.8613403637042,
693
+ "ttft_avg": 0.15646358660887927,
694
+ "ttft_p50": 0.18395527265965939,
695
+ "ttft_p90": 0.18397697466425597,
696
+ "ttft_p99": 0.1839802248729393,
697
+ "time_to_second_token_avg": 0.025640497857239097,
698
+ "time_to_second_token_p50": 0.029807523358613253,
699
+ "time_to_second_token_p90": 0.029867575969547033,
700
+ "time_to_second_token_p99": 0.029868205869570376,
701
+ "request_latency_avg": 0.0,
702
+ "request_latency_p50": 0.0,
703
+ "request_latency_p90": 0.0,
704
+ "request_latency_p99": 0.0,
705
+ "inter_token_latency_avg": 0.012890471746641244,
706
+ "inter_token_latency_p50": 0.01291707436819987,
707
+ "inter_token_latency_p90": 0.013106307107593229,
708
+ "inter_token_latency_p99": 0.013131838856624441,
709
+ "output_tps_per_user_avg": 77.59775588315449,
710
+ "output_tps_per_user_p50": 77.42393606470978,
711
+ "output_tps_per_user_p90": 79.03458812531339,
712
+ "output_tps_per_user_p99": 79.37137995360604,
713
+ "e2e_output_tps_per_user_avg": 0.0,
714
+ "e2e_output_tps_per_user_p50": 0.0,
715
+ "e2e_output_tps_per_user_p90": 0.0,
716
+ "e2e_output_tps_per_user_p99": 0.0,
717
+ "chunk_inter_token_latency_avg": 0.033961205786559916,
718
+ "chunk_inter_token_latency_p50": 0.03394124054211881,
719
+ "chunk_inter_token_latency_p90": 0.03400282599744794,
720
+ "chunk_inter_token_latency_p99": 0.034024420723019345,
721
+ "input_seq_len_avg": 78.0,
722
+ "output_seq_len_avg": 1970.25,
723
+ "output_seq_len_p50": 1968.0,
724
+ "output_seq_len_p90": 2007.1,
725
+ "output_seq_len_p99": 2013.31,
726
+ "request_count": 4,
727
+ "completed_request_count": 0,
728
+ "request_samples": [
729
+ {
730
+ "ttft": 0.07396321510896087,
731
+ "time_to_second_token": 0.013078668853268027,
732
+ "latency": 0.0,
733
+ "inter_token_latency_avg": 0.012794035052220763,
734
+ "chunk_inter_token_latency_avg": 0.03394683967189242,
735
+ "input_tokens": 78,
736
+ "output_tokens": 1991,
737
+ "output_tps_per_user": 78.16142412603614,
738
+ "e2e_output_tps_per_user": 0.0,
739
+ "completed": false
740
+ },
741
+ {
742
+ "ttft": 0.18394199712201953,
743
+ "time_to_second_token": 0.029868275858461857,
744
+ "latency": 0.0,
745
+ "inter_token_latency_avg": 0.013040113684178978,
746
+ "chunk_inter_token_latency_avg": 0.034026820136971725,
747
+ "input_tokens": 78,
748
+ "output_tokens": 1945,
749
+ "output_tps_per_user": 76.68644800338343,
750
+ "e2e_output_tps_per_user": 0.0,
751
+ "completed": false
752
+ },
753
+ {
754
+ "ttft": 0.18398058600723743,
755
+ "time_to_second_token": 0.029865942895412445,
756
+ "latency": 0.0,
757
+ "inter_token_latency_avg": 0.013134675717627909,
758
+ "chunk_inter_token_latency_avg": 0.0339356414123452,
759
+ "input_tokens": 78,
760
+ "output_tokens": 1931,
761
+ "output_tps_per_user": 76.13435013533761,
762
+ "e2e_output_tps_per_user": 0.0,
763
+ "completed": false
764
+ },
765
+ {
766
+ "ttft": 0.18396854819729924,
767
+ "time_to_second_token": 0.02974910382181406,
768
+ "latency": 0.0,
769
+ "inter_token_latency_avg": 0.012593062532537325,
770
+ "chunk_inter_token_latency_avg": 0.033935521925030306,
771
+ "input_tokens": 78,
772
+ "output_tokens": 2014,
773
+ "output_tps_per_user": 79.40880126786078,
774
+ "e2e_output_tps_per_user": 0.0,
775
+ "completed": false
776
+ }
777
+ ],
778
+ "total_tokens": 6067,
779
+ "wall_time": 25.569467527093366,
780
+ "num_completed": 4,
781
+ "num_errors": 0,
782
+ "server_gen_throughput": 303.24736236685254,
783
+ "server_utilization": 0.02345058626465657,
784
+ "server_spec_accept_rate": 0.5431034482758621,
785
+ "server_spec_accept_length": 0.0,
786
+ "avg_running_reqs": 4,
787
+ "max_running_reqs": 4,
788
+ "effective_concurrency": 4,
789
+ "avg_queue_reqs": 0,
790
+ "max_queue_reqs": 0,
791
+ "queue_fraction": 0.0,
792
+ "underfilled": false,
793
+ "warmup_timed_out": false,
794
+ "warmup_duration": 5.541,
795
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
796
+ "timeout_reason": "",
797
+ "capacity_limited": false,
798
+ "hardware_summary": {
799
+ "samples": 8,
800
+ "duration_seconds": 16.826,
801
+ "gpu_count": 4,
802
+ "cpu_util_avg_pct": 11.47,
803
+ "cpu_temp_max_c": 75.88,
804
+ "gpu_util_avg_pct": 100.0,
805
+ "gpu_util_max_pct": 100.0,
806
+ "mem_util_avg_pct": 35.31,
807
+ "mem_util_max_pct": 43.0,
808
+ "temp_avg_c": 68.16,
809
+ "temp_max_c": 83.0,
810
+ "power_total_avg_w": 1172.73,
811
+ "power_total_max_w": 1173.45,
812
+ "power_limit_total_w": 1200.0,
813
+ "vram_used_avg_mb": 384778.0,
814
+ "vram_used_max_mb": 384778.0,
815
+ "vram_total_mb": 391548.0,
816
+ "vram_used_avg_pct": 98.27,
817
+ "vram_used_max_pct": 98.27,
818
+ "pcie_rx_avg_mb_s": 7893.62,
819
+ "pcie_rx_max_mb_s": 8279.0,
820
+ "pcie_tx_avg_mb_s": 7755.88,
821
+ "pcie_tx_max_mb_s": 8153.0
822
+ }
823
+ },
824
+ {
825
+ "concurrency": 2,
826
+ "context_tokens": 8192,
827
+ "benchmark_mode": "duration",
828
+ "request_count_target": 0,
829
+ "warmup_request_count": 0,
830
+ "measurement_seconds": 19.988454,
831
+ "measurement_wall_seconds": 20.00057,
832
+ "client_output_tokens": 4735,
833
+ "server_output_tokens": 4735,
834
+ "aggregate_source": "openai_continuous_usage",
835
+ "aggregate_tps": 236.8867590379722,
836
+ "per_request_avg_tps": 118.4433795189861,
837
+ "ttft_avg": 0.9413391120033339,
838
+ "ttft_p50": 0.9413391120033339,
839
+ "ttft_p90": 1.2234342327574268,
840
+ "ttft_p99": 1.2869056349270978,
841
+ "time_to_second_token_avg": 0.021329568000510335,
842
+ "time_to_second_token_p50": 0.021329568000510335,
843
+ "time_to_second_token_p90": 0.02503214799799025,
844
+ "time_to_second_token_p99": 0.02586522849742323,
845
+ "request_latency_avg": 0.0,
846
+ "request_latency_p50": 0.0,
847
+ "request_latency_p90": 0.0,
848
+ "request_latency_p99": 0.0,
849
+ "inter_token_latency_avg": 0.008409807194832745,
850
+ "inter_token_latency_p50": 0.008409807194832745,
851
+ "inter_token_latency_p90": 0.00844567690517398,
852
+ "inter_token_latency_p99": 0.00845374759000076,
853
+ "output_tps_per_user_avg": 118.91217038040975,
854
+ "output_tps_per_user_p50": 118.91217038040975,
855
+ "output_tps_per_user_p90": 119.41935740726734,
856
+ "output_tps_per_user_p99": 119.5334744883103,
857
+ "e2e_output_tps_per_user_avg": 0.0,
858
+ "e2e_output_tps_per_user_p50": 0.0,
859
+ "e2e_output_tps_per_user_p90": 0.0,
860
+ "e2e_output_tps_per_user_p99": 0.0,
861
+ "chunk_inter_token_latency_avg": 0.02114018666727937,
862
+ "chunk_inter_token_latency_p50": 0.02114018666727937,
863
+ "chunk_inter_token_latency_p90": 0.02137045316589387,
864
+ "chunk_inter_token_latency_p99": 0.021422263128082132,
865
+ "input_seq_len_avg": 8192.0,
866
+ "output_seq_len_avg": 2805.0,
867
+ "output_seq_len_p50": 2805.0,
868
+ "output_seq_len_p90": 2826.6,
869
+ "output_seq_len_p99": 2831.46,
870
+ "request_count": 2,
871
+ "completed_request_count": 0,
872
+ "request_samples": [
873
+ {
874
+ "ttft": 0.5887202110607177,
875
+ "time_to_second_token": 0.01670134300366044,
876
+ "latency": 0.0,
877
+ "inter_token_latency_avg": 0.00845464433275929,
878
+ "chunk_inter_token_latency_avg": 0.021428019790547495,
879
+ "input_tokens": 8192,
880
+ "output_tokens": 2832,
881
+ "output_tps_per_user": 118.27818659683774,
882
+ "e2e_output_tps_per_user": 0.0,
883
+ "completed": false
884
+ },
885
+ {
886
+ "ttft": 1.29395801294595,
887
+ "time_to_second_token": 0.02595779299736023,
888
+ "latency": 0.0,
889
+ "inter_token_latency_avg": 0.008364970056906203,
890
+ "chunk_inter_token_latency_avg": 0.020852353544011243,
891
+ "input_tokens": 8192,
892
+ "output_tokens": 2778,
893
+ "output_tps_per_user": 119.54615416398174,
894
+ "e2e_output_tps_per_user": 0.0,
895
+ "completed": false
896
+ }
897
+ ],
898
+ "total_tokens": 4735,
899
+ "wall_time": 25.56432215613313,
900
+ "num_completed": 2,
901
+ "num_errors": 0,
902
+ "server_gen_throughput": 236.6879632349728,
903
+ "server_utilization": 0.012283640424343933,
904
+ "server_spec_accept_rate": 0.4791666666666667,
905
+ "server_spec_accept_length": 0.0,
906
+ "avg_running_reqs": 2,
907
+ "max_running_reqs": 2,
908
+ "effective_concurrency": 2,
909
+ "avg_queue_reqs": 0,
910
+ "max_queue_reqs": 0,
911
+ "queue_fraction": 0.0,
912
+ "underfilled": false,
913
+ "warmup_timed_out": false,
914
+ "warmup_duration": 5.554,
915
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
916
+ "timeout_reason": "",
917
+ "capacity_limited": false,
918
+ "hardware_summary": {
919
+ "samples": 8,
920
+ "duration_seconds": 16.882,
921
+ "gpu_count": 4,
922
+ "cpu_util_avg_pct": 11.54,
923
+ "cpu_temp_max_c": 75.62,
924
+ "gpu_util_avg_pct": 100.0,
925
+ "gpu_util_max_pct": 100.0,
926
+ "mem_util_avg_pct": 40.84,
927
+ "mem_util_max_pct": 50.0,
928
+ "temp_avg_c": 67.75,
929
+ "temp_max_c": 83.0,
930
+ "power_total_avg_w": 1177.72,
931
+ "power_total_max_w": 1178.14,
932
+ "power_limit_total_w": 1200.0,
933
+ "vram_used_avg_mb": 384778.0,
934
+ "vram_used_max_mb": 384778.0,
935
+ "vram_total_mb": 391548.0,
936
+ "vram_used_avg_pct": 98.27,
937
+ "vram_used_max_pct": 98.27,
938
+ "pcie_rx_avg_mb_s": 11234.38,
939
+ "pcie_rx_max_mb_s": 11286.0,
940
+ "pcie_tx_avg_mb_s": 11126.88,
941
+ "pcie_tx_max_mb_s": 11190.0
942
+ }
943
+ },
944
+ {
945
+ "concurrency": 4,
946
+ "context_tokens": 8192,
947
+ "benchmark_mode": "duration",
948
+ "request_count_target": 0,
949
+ "warmup_request_count": 0,
950
+ "measurement_seconds": 19.988009,
951
+ "measurement_wall_seconds": 20.000154,
952
+ "client_output_tokens": 5819,
953
+ "server_output_tokens": 5819,
954
+ "aggregate_source": "openai_continuous_usage",
955
+ "aggregate_tps": 291.1245424388467,
956
+ "per_request_avg_tps": 72.78113560971167,
957
+ "ttft_avg": 3.0831757867126726,
958
+ "ttft_p50": 2.1997361014364287,
959
+ "ttft_p90": 5.800122947827914,
960
+ "ttft_p99": 7.188826314958277,
961
+ "time_to_second_token_avg": 0.03507792699383572,
962
+ "time_to_second_token_p50": 0.0406363804358989,
963
+ "time_to_second_token_p90": 0.0417290014680475,
964
+ "time_to_second_token_p99": 0.041733411899767814,
965
+ "request_latency_avg": 0.0,
966
+ "request_latency_p50": 0.0,
967
+ "request_latency_p90": 0.0,
968
+ "request_latency_p99": 0.0,
969
+ "inter_token_latency_avg": 0.014156019616279113,
970
+ "inter_token_latency_p50": 0.014055135648646407,
971
+ "inter_token_latency_p90": 0.01461799139621679,
972
+ "inter_token_latency_p99": 0.014785406939632685,
973
+ "output_tps_per_user_avg": 70.69962462194486,
974
+ "output_tps_per_user_p50": 71.15434738255996,
975
+ "output_tps_per_user_p90": 72.6003158823885,
976
+ "output_tps_per_user_p99": 72.90651063203018,
977
+ "e2e_output_tps_per_user_avg": 0.0,
978
+ "e2e_output_tps_per_user_p50": 0.0,
979
+ "e2e_output_tps_per_user_p90": 0.0,
980
+ "e2e_output_tps_per_user_p99": 0.0,
981
+ "chunk_inter_token_latency_avg": 0.03618710908102364,
982
+ "chunk_inter_token_latency_p50": 0.03679464568694915,
983
+ "chunk_inter_token_latency_p90": 0.03703470265231493,
984
+ "chunk_inter_token_latency_p99": 0.03705366284251491,
985
+ "input_seq_len_avg": 8192.0,
986
+ "output_seq_len_avg": 1940.0,
987
+ "output_seq_len_p50": 2013.5,
988
+ "output_seq_len_p90": 2034.4,
989
+ "output_seq_len_p99": 2037.64,
990
+ "request_count": 4,
991
+ "completed_request_count": 0,
992
+ "request_samples": [
993
+ {
994
+ "ttft": 0.5901042548939586,
995
+ "time_to_second_token": 0.01730504515580833,
996
+ "latency": 0.0,
997
+ "inter_token_latency_avg": 0.014804008666678895,
998
+ "chunk_inter_token_latency_avg": 0.03705576953031491,
999
+ "input_tokens": 8192,
1000
+ "output_tokens": 2026,
1001
+ "output_tps_per_user": 67.54927145178024,
1002
+ "e2e_output_tps_per_user": 0.0,
1003
+ "completed": false
1004
+ },
1005
+ {
1006
+ "ttft": 2.1997808848973364,
1007
+ "time_to_second_token": 0.04173390194773674,
1008
+ "latency": 0.0,
1009
+ "inter_token_latency_avg": 0.01418395109847188,
1010
+ "chunk_inter_token_latency_avg": 0.03660374477025001,
1011
+ "input_tokens": 8192,
1012
+ "output_tokens": 2001,
1013
+ "output_tps_per_user": 70.50221712254323,
1014
+ "e2e_output_tps_per_user": 0.0,
1015
+ "completed": false
1016
+ },
1017
+ {
1018
+ "ttft": 2.199691317975521,
1019
+ "time_to_second_token": 0.04171756701543927,
1020
+ "latency": 0.0,
1021
+ "inter_token_latency_avg": 0.013926320198820936,
1022
+ "chunk_inter_token_latency_avg": 0.0369855466036483,
1023
+ "input_tokens": 8192,
1024
+ "output_tokens": 2038,
1025
+ "output_tps_per_user": 71.80647764257671,
1026
+ "e2e_output_tps_per_user": 0.0,
1027
+ "completed": false
1028
+ },
1029
+ {
1030
+ "ttft": 7.343126689083874,
1031
+ "time_to_second_token": 0.03955519385635853,
1032
+ "latency": 0.0,
1033
+ "inter_token_latency_avg": 0.013709798501144739,
1034
+ "chunk_inter_token_latency_avg": 0.03410337541988133,
1035
+ "input_tokens": 8192,
1036
+ "output_tokens": 1695,
1037
+ "output_tps_per_user": 72.94053227087926,
1038
+ "e2e_output_tps_per_user": 0.0,
1039
+ "completed": false
1040
+ }
1041
+ ],
1042
+ "total_tokens": 5819,
1043
+ "wall_time": 31.621210746001452,
1044
+ "num_completed": 4,
1045
+ "num_errors": 0,
1046
+ "server_gen_throughput": 290.8514927893891,
1047
+ "server_utilization": 0.024567280848687867,
1048
+ "server_spec_accept_rate": 0.5,
1049
+ "server_spec_accept_length": 0.0,
1050
+ "avg_running_reqs": 4,
1051
+ "max_running_reqs": 4,
1052
+ "effective_concurrency": 4,
1053
+ "avg_queue_reqs": 0,
1054
+ "max_queue_reqs": 0,
1055
+ "queue_fraction": 0.0,
1056
+ "underfilled": false,
1057
+ "warmup_timed_out": false,
1058
+ "warmup_duration": 11.597,
1059
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1060
+ "timeout_reason": "",
1061
+ "capacity_limited": false,
1062
+ "hardware_summary": {
1063
+ "samples": 8,
1064
+ "duration_seconds": 16.872,
1065
+ "gpu_count": 4,
1066
+ "cpu_util_avg_pct": 11.55,
1067
+ "cpu_temp_max_c": 75.88,
1068
+ "gpu_util_avg_pct": 100.0,
1069
+ "gpu_util_max_pct": 100.0,
1070
+ "mem_util_avg_pct": 34.75,
1071
+ "mem_util_max_pct": 43.0,
1072
+ "temp_avg_c": 68.12,
1073
+ "temp_max_c": 83.0,
1074
+ "power_total_avg_w": 1173.04,
1075
+ "power_total_max_w": 1173.65,
1076
+ "power_limit_total_w": 1200.0,
1077
+ "vram_used_avg_mb": 384778.0,
1078
+ "vram_used_max_mb": 384778.0,
1079
+ "vram_total_mb": 391548.0,
1080
+ "vram_used_avg_pct": 98.27,
1081
+ "vram_used_max_pct": 98.27,
1082
+ "pcie_rx_avg_mb_s": 8010.0,
1083
+ "pcie_rx_max_mb_s": 8332.0,
1084
+ "pcie_tx_avg_mb_s": 7806.5,
1085
+ "pcie_tx_max_mb_s": 8129.0
1086
+ }
1087
+ },
1088
+ {
1089
+ "concurrency": 2,
1090
+ "context_tokens": 32768,
1091
+ "benchmark_mode": "duration",
1092
+ "request_count_target": 0,
1093
+ "warmup_request_count": 0,
1094
+ "measurement_seconds": 19.989635,
1095
+ "measurement_wall_seconds": 20.000757,
1096
+ "client_output_tokens": 4633,
1097
+ "server_output_tokens": 4633,
1098
+ "aggregate_source": "openai_continuous_usage",
1099
+ "aggregate_tps": 231.7701102106262,
1100
+ "per_request_avg_tps": 115.8850551053131,
1101
+ "ttft_avg": 0.9621059750206769,
1102
+ "ttft_p50": 0.9621059750206769,
1103
+ "ttft_p90": 1.2434888477437198,
1104
+ "ttft_p99": 1.3067999941064046,
1105
+ "time_to_second_token_avg": 0.01357436552643776,
1106
+ "time_to_second_token_p50": 0.01357436552643776,
1107
+ "time_to_second_token_p90": 0.01572811957448721,
1108
+ "time_to_second_token_p99": 0.016212714235298336,
1109
+ "request_latency_avg": 0.0,
1110
+ "request_latency_p50": 0.0,
1111
+ "request_latency_p90": 0.0,
1112
+ "request_latency_p99": 0.0,
1113
+ "inter_token_latency_avg": 0.008537162788289372,
1114
+ "inter_token_latency_p50": 0.008537162788289372,
1115
+ "inter_token_latency_p90": 0.00858358805811946,
1116
+ "inter_token_latency_p99": 0.00859403374383123,
1117
+ "output_tps_per_user_avg": 117.14034665810209,
1118
+ "output_tps_per_user_p50": 117.14034665810209,
1119
+ "output_tps_per_user_p90": 117.77735831366668,
1120
+ "output_tps_per_user_p99": 117.92068593616872,
1121
+ "e2e_output_tps_per_user_avg": 0.0,
1122
+ "e2e_output_tps_per_user_p50": 0.0,
1123
+ "e2e_output_tps_per_user_p90": 0.0,
1124
+ "e2e_output_tps_per_user_p99": 0.0,
1125
+ "chunk_inter_token_latency_avg": 0.021224602006348896,
1126
+ "chunk_inter_token_latency_p50": 0.021224602006348896,
1127
+ "chunk_inter_token_latency_p90": 0.021463160367322105,
1128
+ "chunk_inter_token_latency_p99": 0.021516835998541078,
1129
+ "input_seq_len_avg": 32768.0,
1130
+ "output_seq_len_avg": 2760.5,
1131
+ "output_seq_len_p50": 2760.5,
1132
+ "output_seq_len_p90": 2778.5,
1133
+ "output_seq_len_p99": 2782.55,
1134
+ "request_count": 2,
1135
+ "completed_request_count": 0,
1136
+ "request_samples": [
1137
+ {
1138
+ "ttft": 0.6103773841168731,
1139
+ "time_to_second_token": 0.010882172966375947,
1140
+ "latency": 0.0,
1141
+ "inter_token_latency_avg": 0.008595194375576983,
1142
+ "chunk_inter_token_latency_avg": 0.021522799957565408,
1143
+ "input_tokens": 32768,
1144
+ "output_tokens": 2783,
1145
+ "output_tps_per_user": 116.34408208864636,
1146
+ "e2e_output_tps_per_user": 0.0,
1147
+ "completed": false
1148
+ },
1149
+ {
1150
+ "ttft": 1.3138345659244806,
1151
+ "time_to_second_token": 0.016266558086499572,
1152
+ "latency": 0.0,
1153
+ "inter_token_latency_avg": 0.00847913120100176,
1154
+ "chunk_inter_token_latency_avg": 0.020926404055132387,
1155
+ "input_tokens": 32768,
1156
+ "output_tokens": 2738,
1157
+ "output_tps_per_user": 117.93661122755783,
1158
+ "e2e_output_tps_per_user": 0.0,
1159
+ "completed": false
1160
+ }
1161
+ ],
1162
+ "total_tokens": 4633,
1163
+ "wall_time": 25.564056790899485,
1164
+ "num_completed": 2,
1165
+ "num_errors": 0,
1166
+ "server_gen_throughput": 231.5835038411289,
1167
+ "server_utilization": 0.013121161362367406,
1168
+ "server_spec_accept_rate": 0.5,
1169
+ "server_spec_accept_length": 0.0,
1170
+ "avg_running_reqs": 2,
1171
+ "max_running_reqs": 2,
1172
+ "effective_concurrency": 2,
1173
+ "avg_queue_reqs": 0,
1174
+ "max_queue_reqs": 0,
1175
+ "queue_fraction": 0.0,
1176
+ "underfilled": false,
1177
+ "warmup_timed_out": false,
1178
+ "warmup_duration": 5.553,
1179
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1180
+ "timeout_reason": "",
1181
+ "capacity_limited": false,
1182
+ "hardware_summary": {
1183
+ "samples": 8,
1184
+ "duration_seconds": 16.876,
1185
+ "gpu_count": 4,
1186
+ "cpu_util_avg_pct": 11.51,
1187
+ "cpu_temp_max_c": 75.25,
1188
+ "gpu_util_avg_pct": 100.0,
1189
+ "gpu_util_max_pct": 100.0,
1190
+ "mem_util_avg_pct": 40.84,
1191
+ "mem_util_max_pct": 51.0,
1192
+ "temp_avg_c": 67.97,
1193
+ "temp_max_c": 83.0,
1194
+ "power_total_avg_w": 1177.56,
1195
+ "power_total_max_w": 1178.23,
1196
+ "power_limit_total_w": 1200.0,
1197
+ "vram_used_avg_mb": 384778.0,
1198
+ "vram_used_max_mb": 384778.0,
1199
+ "vram_total_mb": 391548.0,
1200
+ "vram_used_avg_pct": 98.27,
1201
+ "vram_used_max_pct": 98.27,
1202
+ "pcie_rx_avg_mb_s": 11116.0,
1203
+ "pcie_rx_max_mb_s": 11361.0,
1204
+ "pcie_tx_avg_mb_s": 11002.38,
1205
+ "pcie_tx_max_mb_s": 11184.0
1206
+ }
1207
+ },
1208
+ {
1209
+ "concurrency": 4,
1210
+ "context_tokens": 32768,
1211
+ "benchmark_mode": "duration",
1212
+ "request_count_target": 0,
1213
+ "warmup_request_count": 0,
1214
+ "measurement_seconds": 19.997266,
1215
+ "measurement_wall_seconds": 20.000362,
1216
+ "client_output_tokens": 6078,
1217
+ "server_output_tokens": 6078,
1218
+ "aggregate_source": "openai_continuous_usage",
1219
+ "aggregate_tps": 303.94154153181165,
1220
+ "per_request_avg_tps": 75.98538538295291,
1221
+ "ttft_avg": 2.3015168351703323,
1222
+ "ttft_p50": 2.210984098375775,
1223
+ "ttft_p90": 3.5852746270596985,
1224
+ "ttft_p99": 4.115192952086217,
1225
+ "time_to_second_token_avg": 0.023911484284326434,
1226
+ "time_to_second_token_p50": 0.024445773102343082,
1227
+ "time_to_second_token_p90": 0.03160055840853602,
1228
+ "time_to_second_token_p99": 0.03434951798757538,
1229
+ "request_latency_avg": 0.0,
1230
+ "request_latency_p50": 0.0,
1231
+ "request_latency_p90": 0.0,
1232
+ "request_latency_p99": 0.0,
1233
+ "inter_token_latency_avg": 0.013270635558348034,
1234
+ "inter_token_latency_p50": 0.01317594037834293,
1235
+ "inter_token_latency_p90": 0.013953309896230092,
1236
+ "inter_token_latency_p99": 0.014080148007312025,
1237
+ "output_tps_per_user_avg": 75.5138067713092,
1238
+ "output_tps_per_user_p50": 75.98396361855555,
1239
+ "output_tps_per_user_p90": 78.96660843355386,
1240
+ "output_tps_per_user_p99": 79.11936279225931,
1241
+ "e2e_output_tps_per_user_avg": 0.0,
1242
+ "e2e_output_tps_per_user_p50": 0.0,
1243
+ "e2e_output_tps_per_user_p90": 0.0,
1244
+ "e2e_output_tps_per_user_p99": 0.0,
1245
+ "chunk_inter_token_latency_avg": 0.03530397237485847,
1246
+ "chunk_inter_token_latency_p50": 0.03529481439149047,
1247
+ "chunk_inter_token_latency_p90": 0.03563218072342417,
1248
+ "chunk_inter_token_latency_p99": 0.03572440514104369,
1249
+ "input_seq_len_avg": 32768.0,
1250
+ "output_seq_len_avg": 1907.25,
1251
+ "output_seq_len_p50": 1856.0,
1252
+ "output_seq_len_p90": 2040.9,
1253
+ "output_seq_len_p99": 2110.29,
1254
+ "request_count": 4,
1255
+ "completed_request_count": 0,
1256
+ "request_samples": [
1257
+ {
1258
+ "ttft": 0.6100263779517263,
1259
+ "time_to_second_token": 0.012099432991817594,
1260
+ "latency": 0.0,
1261
+ "inter_token_latency_avg": 0.012727410407705224,
1262
+ "chunk_inter_token_latency_avg": 0.03573465229855697,
1263
+ "input_tokens": 32768,
1264
+ "output_tokens": 2118,
1265
+ "output_tps_per_user": 78.57057861468788,
1266
+ "e2e_output_tps_per_user": 0.0,
1267
+ "completed": false
1268
+ },
1269
+ {
1270
+ "ttft": 2.2114123029168695,
1271
+ "time_to_second_token": 0.024417920038104057,
1272
+ "latency": 0.0,
1273
+ "inter_token_latency_avg": 0.014094241130765572,
1274
+ "chunk_inter_token_latency_avg": 0.035393080381447624,
1275
+ "input_tokens": 32768,
1276
+ "output_tokens": 1799,
1277
+ "output_tps_per_user": 70.95096434934358,
1278
+ "e2e_output_tps_per_user": 0.0,
1279
+ "completed": false
1280
+ },
1281
+ {
1282
+ "ttft": 2.2105558938346803,
1283
+ "time_to_second_token": 0.024473626166582108,
1284
+ "latency": 0.0,
1285
+ "inter_token_latency_avg": 0.013624470348980637,
1286
+ "chunk_inter_token_latency_avg": 0.035196548401533315,
1287
+ "input_tokens": 32768,
1288
+ "output_tokens": 1861,
1289
+ "output_tps_per_user": 73.39734862242322,
1290
+ "e2e_output_tps_per_user": 0.0,
1291
+ "completed": false
1292
+ },
1293
+ {
1294
+ "ttft": 4.174072765978053,
1295
+ "time_to_second_token": 0.03465495794080198,
1296
+ "latency": 0.0,
1297
+ "inter_token_latency_avg": 0.012636420345940702,
1298
+ "chunk_inter_token_latency_avg": 0.03489160841789597,
1299
+ "input_tokens": 32768,
1300
+ "output_tokens": 1851,
1301
+ "output_tps_per_user": 79.13633549878213,
1302
+ "e2e_output_tps_per_user": 0.0,
1303
+ "completed": false
1304
+ }
1305
+ ],
1306
+ "total_tokens": 6078,
1307
+ "wall_time": 28.609053145861253,
1308
+ "num_completed": 4,
1309
+ "num_errors": 0,
1310
+ "server_gen_throughput": 303.824900318235,
1311
+ "server_utilization": 0.02540480178671134,
1312
+ "server_spec_accept_rate": 0.5316091954022989,
1313
+ "server_spec_accept_length": 0.0,
1314
+ "avg_running_reqs": 4,
1315
+ "max_running_reqs": 4,
1316
+ "effective_concurrency": 4,
1317
+ "avg_queue_reqs": 0,
1318
+ "max_queue_reqs": 0,
1319
+ "queue_fraction": 0.0,
1320
+ "underfilled": false,
1321
+ "warmup_timed_out": false,
1322
+ "warmup_duration": 8.576,
1323
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1324
+ "timeout_reason": "",
1325
+ "capacity_limited": false,
1326
+ "hardware_summary": {
1327
+ "samples": 9,
1328
+ "duration_seconds": 19.268,
1329
+ "gpu_count": 4,
1330
+ "cpu_util_avg_pct": 11.48,
1331
+ "cpu_temp_max_c": 76.25,
1332
+ "gpu_util_avg_pct": 100.0,
1333
+ "gpu_util_max_pct": 100.0,
1334
+ "mem_util_avg_pct": 35.08,
1335
+ "mem_util_max_pct": 44.0,
1336
+ "temp_avg_c": 68.44,
1337
+ "temp_max_c": 84.0,
1338
+ "power_total_avg_w": 1172.67,
1339
+ "power_total_max_w": 1174.13,
1340
+ "power_limit_total_w": 1200.0,
1341
+ "vram_used_avg_mb": 384778.0,
1342
+ "vram_used_max_mb": 384778.0,
1343
+ "vram_total_mb": 391548.0,
1344
+ "vram_used_avg_pct": 98.27,
1345
+ "vram_used_max_pct": 98.27,
1346
+ "pcie_rx_avg_mb_s": 7773.78,
1347
+ "pcie_rx_max_mb_s": 8029.0,
1348
+ "pcie_tx_avg_mb_s": 7899.0,
1349
+ "pcie_tx_max_mb_s": 8283.0
1350
+ }
1351
+ }
1352
+ ],
1353
+ "summary_table": {
1354
+ "0": {
1355
+ "1": 185.68810232786203,
1356
+ "2": 246.22038564672286,
1357
+ "4": 303.4453614548168
1358
+ },
1359
+ "8192": {
1360
+ "1": 166.89761277085296,
1361
+ "2": 236.8867590379722,
1362
+ "4": 291.1245424388467
1363
+ },
1364
+ "32768": {
1365
+ "1": 173.8394059270179,
1366
+ "2": 231.7701102106262,
1367
+ "4": 303.94154153181165
1368
+ }
1369
+ },
1370
+ "burst_results": [],
1371
+ "burst_summary_table": {},
1372
+ "methodology": {
1373
+ "prefill": {
1374
+ "name": "Prefill",
1375
+ "present": false,
1376
+ "mode": "skipped",
1377
+ "formula": "prompt_tokens / TTFT",
1378
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
1379
+ },
1380
+ "sustained_decode": {
1381
+ "name": "Sustained Decode",
1382
+ "present": true,
1383
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
1384
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
1385
+ },
1386
+ "burst_e2e_decode": {
1387
+ "name": "Burst / E2E Decode",
1388
+ "present": false,
1389
+ "status": "not run; use --run-burst",
1390
+ "formula": "sum(completion_tokens) / profiling_wall_time",
1391
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
1392
+ }
1393
+ }
1394
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.log ADDED
@@ -0,0 +1,123 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0', '8k', '32k'] │
16
+ │ Duration: 20.0s per decode test | Max tokens: 8192 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill: skipped | Sustained decode: 9 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
23
+ 7,337,984 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: skipped
26
+ Calibrating padding text (run=rumxqbvzafby, up to 32k)...
27
+ 8k: 50,540 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,139 chars (32,768 prompt tokens via /tokenize)
29
+ Token targeting: /tokenize exact
30
+ Done.
31
+
32
+
33
+
34
+ llm-decode-bench v0.4.29
35
+ ╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
36
+ │ Sustained Decode │
37
+ │ Steady-state decode throughput after the engine has admitted the requested │
38
+ │ concurrency and passed warmup. Use this as the main tuning/regression signal │
39
+ │ for kernels, NCCL, DCP, MTP, and scheduler changes. │
40
+ ╰──────────────────────────────────────────────────────────────────────────────╯
41
+ Aggregate tok/s + TTFT/ITL
42
+ ╭────────────┬─────────────┬─────────────┬──────────────╮
43
+ │ ctx \ conc │ 1 │ 2 │ 4 │
44
+ ├────────────┼─────────────┼─────────────┼──────────────┤
45
+ │ 0 │ 185.7 72/5 │ 246.2 113/8 │ 303.4 184/13 │
46
+ │ 8k │ 166.9 580/6 │ 236.9 941/8 │ 291.1 2k/14 │
47
+ │ 32k │ 173.8 595/6 │ 231.8 962/9 │ 303.9 2k/13 │
48
+ ╰────────────┴─────────────┴─────────────┴──────────────╯
49
+ Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
50
+ (continuous completion_tokens when the server supports it). Prometheus is kept
51
+ as validation/scheduler data.
52
+ Aggregate source(s): openai_continuous_usage
53
+ Per-Request tok/s
54
+ ╭────────────┬───────┬───────┬──────╮
55
+ │ ctx \ conc │ 1 │ 2 │ 4 │
56
+ ├────────────┼───────┼───────┼──────┤
57
+ │ 0 │ 185.7 │ 123.1 │ 75.9 │
58
+ │ 8k │ 166.9 │ 118.4 │ 72.8 │
59
+ │ 32k │ 173.8 │ 115.9 │ 76.0 │
60
+ ╰────────────┴───────┴───────┴──────╯
61
+ Client request latency: p50 /
62
+ p90 ms
63
+ ╭────────────┬─────┬─────┬─────╮
64
+ │ ctx \ conc │ 1 │ 2 │ 4 │
65
+ ├────────────┼─────┼─────┼─────┤
66
+ │ 0 │ —/— │ —/— │ —/— │
67
+ │ 8k │ —/— │ —/— │ —/— │
68
+ │ 32k │ —/— │ —/— │ —/— │
69
+ ╰────────────┴─────┴─────┴─────╯
70
+ Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
71
+ coordinate. ITL is computed from observed generated tokens, including streams
72
+ stopped at the measurement boundary; a missing ITL means no stream produced at
73
+ least two measured output tokens. Per-request tok/s and request latency are
74
+ shown in separate per-cell matrices. Completion/sample counts and full
75
+ request-level distributions remain in JSON under request_samples.
76
+ Sustained mode: client latency metrics explain request UX variance; aggregate
77
+ tok/s remains the primary throughput signal.
78
+ ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
79
+ Hardware Summary
80
+ ╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
81
+ │ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
82
+ ├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
83
+ │ 0 │ │ sust… │ 99/99% │ 44% │ 1149/1… │ 80C │ 75C │ 98… │ 8502/8302 │
84
+ │ … │ │ sust… │ 99/99% │ 44% │ 1152/1… │ 81C │ 75C │ 98… │ 8303/8387 │
85
+ │ … │ │ sust… │ 99/99% │ 44% │ 1153/1… │ 82C │ 75C │ 98… │ 8358/8306 │
86
+ │ 0 │ │ sust… │ 100/100% │ 41% │ 1176/1… │ 83C │ 76C │ 98… │ 11310/11180 │
87
+ │ 0 │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 83C │ 76C │ 98… │ 7894/7756 │
88
+ │ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 83C │ 76C │ 98… │ 11234/11127 │
89
+ │ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 83C │ 76C │ 98… │ 8010/7806 │
90
+ │ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 83C │ 75C │ 98… │ 11116/11002 │
91
+ │ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 76C │ 98… │ 7774/7899 │
92
+ ╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
93
+ ╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
94
+ │ avg 1,095 W | max 1,178 W | limit 1,200 W | over 4m 32s | 114 samples │
95
+ ╰───────────────────────────────────────────────────────────────────────╯
96
+ Hardware summary is sampled from nvidia-smi during the measured part of each
97
+ cell. Whole-run GPU power is the sampled sum of GPU power draw across the
98
+ complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
99
+ a coarse live diagnostic, not a per-kernel NCCL profiler.
100
+
101
+ ╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
102
+ │ Burst / E2E Decode │
103
+ │ Not run. Re-run with --run-burst to append a finite client-facing request │
104
+ │ burst after Sustained Decode. This is intentionally disabled by default │
105
+ │ because it adds another full decode matrix. │
106
+ ╰──────────────────────────────────────────────────────────────────────────────╯
107
+
108
+ ╭────────────────────────────── Primary Summary ───────────────────────────────╮
109
+ │ Primary matrices repeated last so the important numbers are visible without │
110
+ │ scrolling back through diagnostics. │
111
+ ╰─────────────────────────────────────��────────────────────────────────────────╯
112
+ Aggregate decode tok/s
113
+ ╭────────────┬───────┬───────┬───────╮
114
+ │ ctx \ conc │ 1 │ 2 │ 4 │
115
+ ├────────────┼───────┼───────┼───────┤
116
+ │ 0 │ 185.7 │ 246.2 │ 303.4 │
117
+ │ 8k │ 166.9 │ 236.9 │ 291.1 │
118
+ │ 32k │ 173.8 │ 231.8 │ 303.9 │
119
+ ╰────────────┴───────┴───────┴───────╯
120
+
121
+ Results saved to
122
+ <campaign>/candidate-speed-wi
123
+ ndow-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-command.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json",
20
+ "--contexts",
21
+ "0",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--prefill-only",
25
+ "--prefill-contexts",
26
+ "8k,32k,64k,128k",
27
+ "--prefill-duration",
28
+ "20",
29
+ "--cell-warmup-timeout-seconds",
30
+ "180"
31
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "67ce4838168f0c3c85e10e8484e79c3aa9878c8aa48713fceae52113000a0cf5"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json ADDED
@@ -0,0 +1,396 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:17:41.737509",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "standalone_cold",
15
+ "standalone_prefill": true,
16
+ "prefill_only": true,
17
+ "skip_prefill": false,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 0,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0
42
+ ],
43
+ "startup_diagnostics_available": true,
44
+ "nvidia_p2p_override_effective": true,
45
+ "p2pmark_status": "not_run",
46
+ "amd_fabric_status": "not_run"
47
+ },
48
+ "startup_diagnostics": {
49
+ "version": "0.4.29",
50
+ "server_url": "http://127.0.0.1:8001",
51
+ "hostname": "<host>",
52
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
53
+ "env": {},
54
+ "args": {
55
+ "concurrency": "1,2,4",
56
+ "contexts": "0",
57
+ "max_tokens": 8192,
58
+ "duration": 20.0,
59
+ "request_count": 0,
60
+ "run_burst": false,
61
+ "standalone_prefill": true,
62
+ "prefill_only": true,
63
+ "skip_prefill": false,
64
+ "prefill_contexts": "8k,32k,64k,128k",
65
+ "prefill_metric": "client",
66
+ "dcp_size": 0,
67
+ "kv_budget": 0
68
+ },
69
+ "nvidia_p2p_override": {
70
+ "effective": true,
71
+ "configured": true,
72
+ "params_path": "/proc/driver/nvidia/params",
73
+ "params_available": true,
74
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
75
+ "modprobe_available": true,
76
+ "runtime": {
77
+ "ForceP2P": "0x11",
78
+ "RMForceP2PType": "1",
79
+ "RMPcieP2PType": "2",
80
+ "GrdmaPciTopoCheckOverride": "1",
81
+ "EnableResizableBar": "1",
82
+ "DmaRemapPeerMmio": "1"
83
+ },
84
+ "expected": {
85
+ "ForceP2P": "0x11",
86
+ "RMForceP2PType": "1",
87
+ "RMPcieP2PType": "2",
88
+ "GrdmaPciTopoCheckOverride": "1",
89
+ "EnableResizableBar": "1"
90
+ },
91
+ "missing": [],
92
+ "mismatched": {},
93
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
94
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
95
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
96
+ },
97
+ "p2pmark": {
98
+ "status": "not_run"
99
+ },
100
+ "amd_fabric": {
101
+ "status": "not_run"
102
+ },
103
+ "nvidia_smi_query": {
104
+ "cmd": [
105
+ "nvidia-smi",
106
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
107
+ "--format=csv,noheader,nounits"
108
+ ],
109
+ "returncode": 0,
110
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
111
+ "stderr": ""
112
+ },
113
+ "nvidia_smi_topo": {
114
+ "cmd": [
115
+ "nvidia-smi",
116
+ "topo",
117
+ "-m"
118
+ ],
119
+ "returncode": 0,
120
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
121
+ "stderr": ""
122
+ }
123
+ },
124
+ "nvidia_p2p_override": {
125
+ "effective": true,
126
+ "configured": true,
127
+ "params_path": "/proc/driver/nvidia/params",
128
+ "params_available": true,
129
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
130
+ "modprobe_available": true,
131
+ "runtime": {
132
+ "ForceP2P": "0x11",
133
+ "RMForceP2PType": "1",
134
+ "RMPcieP2PType": "2",
135
+ "GrdmaPciTopoCheckOverride": "1",
136
+ "EnableResizableBar": "1",
137
+ "DmaRemapPeerMmio": "1"
138
+ },
139
+ "expected": {
140
+ "ForceP2P": "0x11",
141
+ "RMForceP2PType": "1",
142
+ "RMPcieP2PType": "2",
143
+ "GrdmaPciTopoCheckOverride": "1",
144
+ "EnableResizableBar": "1"
145
+ },
146
+ "missing": [],
147
+ "mismatched": {},
148
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
149
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
150
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
151
+ },
152
+ "p2pmark": {
153
+ "status": "not_run"
154
+ },
155
+ "amd_fabric": {
156
+ "status": "not_run"
157
+ },
158
+ "hardware_run_summary": {
159
+ "samples": 47,
160
+ "duration_seconds": 110.814,
161
+ "gpu_count": 4,
162
+ "cpu_util_avg_pct": 10.87,
163
+ "cpu_temp_max_c": 75.0,
164
+ "gpu_util_avg_pct": 85.39,
165
+ "gpu_util_max_pct": 100.0,
166
+ "mem_util_avg_pct": 19.4,
167
+ "mem_util_max_pct": 35.0,
168
+ "temp_avg_c": 60.56,
169
+ "temp_max_c": 81.0,
170
+ "power_total_avg_w": 1029.23,
171
+ "power_total_max_w": 1167.61,
172
+ "power_limit_total_w": 1200.0,
173
+ "vram_used_avg_mb": 384658.34,
174
+ "vram_used_max_mb": 384770.0,
175
+ "vram_total_mb": 391548.0,
176
+ "vram_used_avg_pct": 98.24,
177
+ "vram_used_max_pct": 98.27,
178
+ "pcie_rx_avg_mb_s": 43654.87,
179
+ "pcie_rx_max_mb_s": 56329.0,
180
+ "pcie_tx_avg_mb_s": 42837.81,
181
+ "pcie_tx_max_mb_s": 56583.0
182
+ },
183
+ "event_log": [],
184
+ "prefill": {
185
+ "8192": {
186
+ "ttft_seconds": 1.095,
187
+ "prefill_seconds": 1.095,
188
+ "tok_per_sec": 7485.0,
189
+ "client_ttft_seconds": 1.095,
190
+ "client_tok_per_sec": 7485.0,
191
+ "prompt_tokens": 8194,
192
+ "samples": 14,
193
+ "method": "client",
194
+ "server_validation": {
195
+ "method": "",
196
+ "tok_per_sec": 0.0,
197
+ "prefill_seconds": 0.0,
198
+ "prompt_tokens": 0,
199
+ "request_prompt_tokens": 0,
200
+ "cached_tokens": 0,
201
+ "token_source": "",
202
+ "samples": 0,
203
+ "invalid_reason": ""
204
+ },
205
+ "hardware_summary": {
206
+ "samples": 9,
207
+ "duration_seconds": 19.243,
208
+ "gpu_count": 4,
209
+ "cpu_util_avg_pct": 10.9,
210
+ "cpu_temp_max_c": 73.38,
211
+ "gpu_util_avg_pct": 74.67,
212
+ "gpu_util_max_pct": 100.0,
213
+ "mem_util_avg_pct": 18.72,
214
+ "mem_util_max_pct": 34.0,
215
+ "temp_avg_c": 54.17,
216
+ "temp_max_c": 69.0,
217
+ "power_total_avg_w": 973.23,
218
+ "power_total_max_w": 1134.69,
219
+ "power_limit_total_w": 1200.0,
220
+ "vram_used_avg_mb": 384770.0,
221
+ "vram_used_max_mb": 384770.0,
222
+ "vram_total_mb": 391548.0,
223
+ "vram_used_avg_pct": 98.27,
224
+ "vram_used_max_pct": 98.27,
225
+ "pcie_rx_avg_mb_s": 31974.11,
226
+ "pcie_rx_max_mb_s": 55296.0,
227
+ "pcie_tx_avg_mb_s": 32134.78,
228
+ "pcie_tx_max_mb_s": 49579.0
229
+ }
230
+ },
231
+ "32768": {
232
+ "ttft_seconds": 4.247,
233
+ "prefill_seconds": 4.247,
234
+ "tok_per_sec": 7716.0,
235
+ "client_ttft_seconds": 4.247,
236
+ "client_tok_per_sec": 7716.0,
237
+ "prompt_tokens": 32770,
238
+ "samples": 5,
239
+ "method": "client",
240
+ "server_validation": {
241
+ "method": "",
242
+ "tok_per_sec": 0.0,
243
+ "prefill_seconds": 0.0,
244
+ "prompt_tokens": 0,
245
+ "request_prompt_tokens": 0,
246
+ "cached_tokens": 0,
247
+ "token_source": "",
248
+ "samples": 0,
249
+ "invalid_reason": ""
250
+ },
251
+ "hardware_summary": {
252
+ "samples": 10,
253
+ "duration_seconds": 21.686,
254
+ "gpu_count": 4,
255
+ "cpu_util_avg_pct": 11.06,
256
+ "cpu_temp_max_c": 75.0,
257
+ "gpu_util_avg_pct": 86.97,
258
+ "gpu_util_max_pct": 100.0,
259
+ "mem_util_avg_pct": 19.82,
260
+ "mem_util_max_pct": 30.0,
261
+ "temp_avg_c": 58.9,
262
+ "temp_max_c": 75.0,
263
+ "power_total_avg_w": 997.14,
264
+ "power_total_max_w": 1146.5,
265
+ "power_limit_total_w": 1200.0,
266
+ "vram_used_avg_mb": 384770.0,
267
+ "vram_used_max_mb": 384770.0,
268
+ "vram_total_mb": 391548.0,
269
+ "vram_used_avg_pct": 98.27,
270
+ "vram_used_max_pct": 98.27,
271
+ "pcie_rx_avg_mb_s": 48453.9,
272
+ "pcie_rx_max_mb_s": 55948.0,
273
+ "pcie_tx_avg_mb_s": 46628.3,
274
+ "pcie_tx_max_mb_s": 53063.0
275
+ }
276
+ },
277
+ "65536": {
278
+ "ttft_seconds": 8.558,
279
+ "prefill_seconds": 8.558,
280
+ "tok_per_sec": 7659.0,
281
+ "client_ttft_seconds": 8.558,
282
+ "client_tok_per_sec": 7659.0,
283
+ "prompt_tokens": 65538,
284
+ "samples": 3,
285
+ "method": "client",
286
+ "server_validation": {
287
+ "method": "",
288
+ "tok_per_sec": 0.0,
289
+ "prefill_seconds": 0.0,
290
+ "prompt_tokens": 0,
291
+ "request_prompt_tokens": 0,
292
+ "cached_tokens": 0,
293
+ "token_source": "",
294
+ "samples": 0,
295
+ "invalid_reason": ""
296
+ },
297
+ "hardware_summary": {
298
+ "samples": 11,
299
+ "duration_seconds": 24.072,
300
+ "gpu_count": 4,
301
+ "cpu_util_avg_pct": 11.24,
302
+ "cpu_temp_max_c": 74.88,
303
+ "gpu_util_avg_pct": 97.55,
304
+ "gpu_util_max_pct": 100.0,
305
+ "mem_util_avg_pct": 21.66,
306
+ "mem_util_max_pct": 35.0,
307
+ "temp_avg_c": 62.7,
308
+ "temp_max_c": 79.0,
309
+ "power_total_avg_w": 1097.1,
310
+ "power_total_max_w": 1167.61,
311
+ "power_limit_total_w": 1200.0,
312
+ "vram_used_avg_mb": 384770.0,
313
+ "vram_used_max_mb": 384770.0,
314
+ "vram_total_mb": 391548.0,
315
+ "vram_used_avg_pct": 98.27,
316
+ "vram_used_max_pct": 98.27,
317
+ "pcie_rx_avg_mb_s": 51029.18,
318
+ "pcie_rx_max_mb_s": 56329.0,
319
+ "pcie_tx_avg_mb_s": 47833.18,
320
+ "pcie_tx_max_mb_s": 53350.0
321
+ }
322
+ },
323
+ "131072": {
324
+ "ttft_seconds": 17.398,
325
+ "prefill_seconds": 17.398,
326
+ "tok_per_sec": 7534.0,
327
+ "client_ttft_seconds": 17.398,
328
+ "client_tok_per_sec": 7534.0,
329
+ "prompt_tokens": 131074,
330
+ "samples": 2,
331
+ "method": "client",
332
+ "server_validation": {
333
+ "method": "",
334
+ "tok_per_sec": 0.0,
335
+ "prefill_seconds": 0.0,
336
+ "prompt_tokens": 0,
337
+ "request_prompt_tokens": 0,
338
+ "cached_tokens": 0,
339
+ "token_source": "",
340
+ "samples": 0,
341
+ "invalid_reason": ""
342
+ },
343
+ "hardware_summary": {
344
+ "samples": 15,
345
+ "duration_seconds": 33.754,
346
+ "gpu_count": 4,
347
+ "cpu_util_avg_pct": 11.3,
348
+ "cpu_temp_max_c": 75.0,
349
+ "gpu_util_avg_pct": 93.25,
350
+ "gpu_util_max_pct": 100.0,
351
+ "mem_util_avg_pct": 20.45,
352
+ "mem_util_max_pct": 28.0,
353
+ "temp_avg_c": 64.9,
354
+ "temp_max_c": 81.0,
355
+ "power_total_avg_w": 1108.87,
356
+ "power_total_max_w": 1149.51,
357
+ "power_limit_total_w": 1200.0,
358
+ "vram_used_avg_mb": 384770.0,
359
+ "vram_used_max_mb": 384770.0,
360
+ "vram_total_mb": 391548.0,
361
+ "vram_used_avg_pct": 98.27,
362
+ "vram_used_max_pct": 98.27,
363
+ "pcie_rx_avg_mb_s": 47286.07,
364
+ "pcie_rx_max_mb_s": 55550.0,
365
+ "pcie_tx_avg_mb_s": 48228.93,
366
+ "pcie_tx_max_mb_s": 56583.0
367
+ }
368
+ }
369
+ },
370
+ "results": [],
371
+ "summary_table": {},
372
+ "burst_results": [],
373
+ "burst_summary_table": {},
374
+ "methodology": {
375
+ "prefill": {
376
+ "name": "Prefill",
377
+ "present": true,
378
+ "mode": "standalone_cold",
379
+ "formula": "prompt_tokens / TTFT",
380
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
381
+ },
382
+ "sustained_decode": {
383
+ "name": "Sustained Decode",
384
+ "present": false,
385
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
386
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
387
+ },
388
+ "burst_e2e_decode": {
389
+ "name": "Burst / E2E Decode",
390
+ "present": false,
391
+ "status": "not run; use --run-burst",
392
+ "formula": "sum(completion_tokens) / profiling_wall_time",
393
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
394
+ }
395
+ }
396
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.log ADDED
@@ -0,0 +1,56 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0'] │
16
+ │ Decode: skipped (--prefill-only) | Max tokens: 8192 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill-only: standalone cold profile (client) | Sustained decode: 0 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
23
+ 7,337,984 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: standalone cold profile ['8k', '32k', '64k', '128k']
26
+ Calibrating padding text (run=cniaedazxusz, up to 128k)...
27
+ 8k: 50,558 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,152 chars (32,768 prompt tokens via /tokenize)
29
+ 64k: 411,264 chars (65,536 prompt tokens via /tokenize)
30
+ 128k: 823,408 chars (131,072 prompt tokens via /tokenize)
31
+ Token targeting: /tokenize exact
32
+ Done.
33
+
34
+
35
+
36
+ llm-decode-bench v0.4.29
37
+ Prefill Speed (C=1, client ISL / TTFT)
38
+
39
+ Client PCIe rx/tx
40
+ Context Tokens TTFT (s) tok/s Server tok/s avg N
41
+ ──────────────────────────────────────────────────────────────────────────────
42
+ 8k 8,194 1.09 7,485 — 31974/32135 14
43
+ 32k 32,770 4.25 7,716 — 48454/46628 5
44
+ 64k 65,538 8.56 7,659 — 51029/47833 3
45
+ 128k 131,074 17.40 7,534 — 47286/48229 2
46
+
47
+ Client tok/s = prompt_tokens / TTFT. Integrated scout rows come from the
48
+ prefix-cache scout request that decode needs anyway. Server tok/s is optional
49
+ Prometheus validation when the engine exports prefill counters and the exact
50
+ counter delta is uncontaminated; for vLLM this uses newly computed KV tokens,
51
+ not request prompt tokens.
52
+
53
+
54
+ Results saved to
55
+ <campaign>/candidate-speed-wi
56
+ ndow-01/results-01/decode-warp-quant/rep-1/prefill.json
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-command.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "512",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json",
20
+ "--contexts",
21
+ "0,8k,32k",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--skip-prefill",
25
+ "--cell-warmup-timeout-seconds",
26
+ "180"
27
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "46f539494020e5442078334b0cb382d3f3c063fdd3312f4e2fe67775555155ce"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json ADDED
@@ -0,0 +1,2342 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:37:59.150763",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 32768,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 512,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0,
42
+ 8192,
43
+ 32768
44
+ ],
45
+ "startup_diagnostics_available": true,
46
+ "nvidia_p2p_override_effective": true,
47
+ "p2pmark_status": "not_run",
48
+ "amd_fabric_status": "not_run"
49
+ },
50
+ "startup_diagnostics": {
51
+ "version": "0.4.29",
52
+ "server_url": "http://127.0.0.1:8001",
53
+ "hostname": "<host>",
54
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
55
+ "env": {},
56
+ "args": {
57
+ "concurrency": "1,2,4",
58
+ "contexts": "0,8k,32k",
59
+ "max_tokens": 512,
60
+ "duration": 20.0,
61
+ "request_count": 0,
62
+ "run_burst": false,
63
+ "standalone_prefill": false,
64
+ "prefill_only": false,
65
+ "skip_prefill": true,
66
+ "prefill_contexts": "8k,64k,128k",
67
+ "prefill_metric": "client",
68
+ "dcp_size": 0,
69
+ "kv_budget": 0
70
+ },
71
+ "nvidia_p2p_override": {
72
+ "effective": true,
73
+ "configured": true,
74
+ "params_path": "/proc/driver/nvidia/params",
75
+ "params_available": true,
76
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
77
+ "modprobe_available": true,
78
+ "runtime": {
79
+ "ForceP2P": "0x11",
80
+ "RMForceP2PType": "1",
81
+ "RMPcieP2PType": "2",
82
+ "GrdmaPciTopoCheckOverride": "1",
83
+ "EnableResizableBar": "1",
84
+ "DmaRemapPeerMmio": "1"
85
+ },
86
+ "expected": {
87
+ "ForceP2P": "0x11",
88
+ "RMForceP2PType": "1",
89
+ "RMPcieP2PType": "2",
90
+ "GrdmaPciTopoCheckOverride": "1",
91
+ "EnableResizableBar": "1"
92
+ },
93
+ "missing": [],
94
+ "mismatched": {},
95
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
96
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
97
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
98
+ },
99
+ "p2pmark": {
100
+ "status": "not_run"
101
+ },
102
+ "amd_fabric": {
103
+ "status": "not_run"
104
+ },
105
+ "nvidia_smi_query": {
106
+ "cmd": [
107
+ "nvidia-smi",
108
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
109
+ "--format=csv,noheader,nounits"
110
+ ],
111
+ "returncode": 0,
112
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
113
+ "stderr": ""
114
+ },
115
+ "nvidia_smi_topo": {
116
+ "cmd": [
117
+ "nvidia-smi",
118
+ "topo",
119
+ "-m"
120
+ ],
121
+ "returncode": 0,
122
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
123
+ "stderr": ""
124
+ }
125
+ },
126
+ "nvidia_p2p_override": {
127
+ "effective": true,
128
+ "configured": true,
129
+ "params_path": "/proc/driver/nvidia/params",
130
+ "params_available": true,
131
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
132
+ "modprobe_available": true,
133
+ "runtime": {
134
+ "ForceP2P": "0x11",
135
+ "RMForceP2PType": "1",
136
+ "RMPcieP2PType": "2",
137
+ "GrdmaPciTopoCheckOverride": "1",
138
+ "EnableResizableBar": "1",
139
+ "DmaRemapPeerMmio": "1"
140
+ },
141
+ "expected": {
142
+ "ForceP2P": "0x11",
143
+ "RMForceP2PType": "1",
144
+ "RMPcieP2PType": "2",
145
+ "GrdmaPciTopoCheckOverride": "1",
146
+ "EnableResizableBar": "1"
147
+ },
148
+ "missing": [],
149
+ "mismatched": {},
150
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
151
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
152
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
153
+ },
154
+ "p2pmark": {
155
+ "status": "not_run"
156
+ },
157
+ "amd_fabric": {
158
+ "status": "not_run"
159
+ },
160
+ "hardware_run_summary": {
161
+ "samples": 113,
162
+ "duration_seconds": 269.93,
163
+ "gpu_count": 4,
164
+ "cpu_util_avg_pct": 10.96,
165
+ "cpu_temp_max_c": 77.38,
166
+ "gpu_util_avg_pct": 91.76,
167
+ "gpu_util_max_pct": 100.0,
168
+ "mem_util_avg_pct": 33.91,
169
+ "mem_util_max_pct": 56.0,
170
+ "temp_avg_c": 68.4,
171
+ "temp_max_c": 84.0,
172
+ "power_total_avg_w": 1097.05,
173
+ "power_total_max_w": 1178.78,
174
+ "power_limit_total_w": 1200.0,
175
+ "vram_used_avg_mb": 384778.0,
176
+ "vram_used_max_mb": 384778.0,
177
+ "vram_total_mb": 391548.0,
178
+ "vram_used_avg_pct": 98.27,
179
+ "vram_used_max_pct": 98.27,
180
+ "pcie_rx_avg_mb_s": 17221.25,
181
+ "pcie_rx_max_mb_s": 74341.0,
182
+ "pcie_tx_avg_mb_s": 16568.58,
183
+ "pcie_tx_max_mb_s": 70985.0
184
+ },
185
+ "event_log": [
186
+ "02:33:26 benchmark start engine=vllm",
187
+ "02:33:26 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
188
+ "02:33:26 startup decode concurrency=1,2,4 contexts=0,8k,32k",
189
+ "02:33:26 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
190
+ "02:33:26 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
191
+ "02:33:26 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
192
+ "02:33:26 startup model context length: 1,000,000 tokens",
193
+ "02:33:26 startup prefill tests: skipped",
194
+ "02:33:26 startup calibrating padding text run=mkkehtszimwn up_to=32k",
195
+ "02:33:26 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
196
+ "02:33:26 startup context 32k: 205,152 chars (32,768 prompt tokens via /tokenize)",
197
+ "02:33:26 startup token targeting: /tokenize exact",
198
+ "02:33:26 startup startup preparation done",
199
+ "02:33:26 hardware monitor interval=2s",
200
+ "02:33:26 decode warmup start",
201
+ "02:33:26 decode warmup start C=1 ctx=32k 3s",
202
+ "02:33:26 cell start C=1 ctx=32k",
203
+ "02:33:35 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
204
+ "02:33:38 cell done C=1 ctx=32k 180.4 tok/s",
205
+ "02:33:38 decode warmup done C=1 ctx=32k",
206
+ "02:33:40 cell start C=1 ctx=0",
207
+ "02:33:45 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
208
+ "02:34:05 cell done C=1 ctx=0 191.1 tok/s",
209
+ "02:34:07 cell start C=1 ctx=8k",
210
+ "02:34:13 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
211
+ "02:34:33 cell done C=1 ctx=8k 165.2 tok/s",
212
+ "02:34:35 cell start C=1 ctx=32k",
213
+ "02:34:44 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
214
+ "02:35:04 cell done C=1 ctx=32k 162.4 tok/s",
215
+ "02:35:06 cell start C=2 ctx=0",
216
+ "02:35:12 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
217
+ "02:35:32 cell done C=2 ctx=0 257.8 tok/s",
218
+ "02:35:34 cell start C=4 ctx=0",
219
+ "02:35:39 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
220
+ "02:35:59 cell done C=4 ctx=0 305.6 tok/s",
221
+ "02:36:01 cell start C=2 ctx=8k",
222
+ "02:36:07 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
223
+ "02:36:27 cell done C=2 ctx=8k 183.5 tok/s",
224
+ "02:36:29 cell start C=4 ctx=8k",
225
+ "02:36:38 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
226
+ "02:36:58 cell done C=4 ctx=8k 214.9 tok/s",
227
+ "02:37:00 cell start C=2 ctx=32k",
228
+ "02:37:06 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
229
+ "02:37:26 cell done C=2 ctx=32k 186.4 tok/s",
230
+ "02:37:28 cell start C=4 ctx=32k",
231
+ "02:37:36 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
232
+ "02:37:57 cell done C=4 ctx=32k 213.0 tok/s"
233
+ ],
234
+ "prefill": {},
235
+ "results": [
236
+ {
237
+ "concurrency": 1,
238
+ "context_tokens": 0,
239
+ "benchmark_mode": "duration",
240
+ "request_count_target": 0,
241
+ "warmup_request_count": 0,
242
+ "measurement_seconds": 20.000282,
243
+ "measurement_wall_seconds": 20.000333,
244
+ "client_output_tokens": 3823,
245
+ "server_output_tokens": 3823,
246
+ "aggregate_source": "openai_continuous_usage",
247
+ "aggregate_tps": 191.1473049181911,
248
+ "per_request_avg_tps": 191.1473049181911,
249
+ "ttft_avg": 0.07954429925885051,
250
+ "ttft_p50": 0.08016932185273618,
251
+ "ttft_p90": 0.08546189323533326,
252
+ "ttft_p99": 0.08585712626809254,
253
+ "time_to_second_token_avg": 0.013534792908467352,
254
+ "time_to_second_token_p50": 0.013614983530715108,
255
+ "time_to_second_token_p90": 0.01384659220930189,
256
+ "time_to_second_token_p99": 0.014208161800634117,
257
+ "request_latency_avg": 2.673719661180965,
258
+ "request_latency_p50": 2.642661032034084,
259
+ "request_latency_p90": 2.810000071441755,
260
+ "request_latency_p99": 2.8247497128229586,
261
+ "inter_token_latency_avg": 0.005110297649696225,
262
+ "inter_token_latency_p50": 0.005072282791254339,
263
+ "inter_token_latency_p90": 0.00536784812823367,
264
+ "inter_token_latency_p99": 0.005399117829612438,
265
+ "output_tps_per_user_avg": 196.0011309531592,
266
+ "output_tps_per_user_p50": 197.17019837912926,
267
+ "output_tps_per_user_p90": 204.7882193380745,
268
+ "output_tps_per_user_p99": 209.5508687453994,
269
+ "e2e_output_tps_per_user_avg": 191.73898029102156,
270
+ "e2e_output_tps_per_user_p50": 193.74410633584293,
271
+ "e2e_output_tps_per_user_p90": 198.93980762708756,
272
+ "e2e_output_tps_per_user_p99": 202.87404403324115,
273
+ "chunk_inter_token_latency_avg": 0.01418378739450653,
274
+ "chunk_inter_token_latency_p50": 0.014161504070741708,
275
+ "chunk_inter_token_latency_p90": 0.014271326914315551,
276
+ "chunk_inter_token_latency_p99": 0.014287257237805287,
277
+ "input_seq_len_avg": 78.0,
278
+ "output_seq_len_avg": 512.0,
279
+ "output_seq_len_p50": 512.0,
280
+ "output_seq_len_p90": 512.0,
281
+ "output_seq_len_p99": 512.0,
282
+ "request_count": 10,
283
+ "completed_request_count": 9,
284
+ "request_samples": [
285
+ {
286
+ "ttft": 0.0710762650705874,
287
+ "time_to_second_token": 0.012838942930102348,
288
+ "latency": 2.636708308942616,
289
+ "inter_token_latency_avg": 0.0050208063480861615,
290
+ "chunk_inter_token_latency_avg": 0.014174762673326124,
291
+ "input_tokens": 78,
292
+ "output_tokens": 512,
293
+ "output_tps_per_user": 199.17119495779428,
294
+ "e2e_output_tps_per_user": 194.18150967382675,
295
+ "completed": true
296
+ },
297
+ {
298
+ "ttft": 0.07547624688595533,
299
+ "time_to_second_token": 0.013636301038786769,
300
+ "latency": 2.739387070061639,
301
+ "inter_token_latency_avg": 0.005213132726371201,
302
+ "chunk_inter_token_latency_avg": 0.014169738421147254,
303
+ "input_tokens": 78,
304
+ "output_tokens": 512,
305
+ "output_tps_per_user": 191.8232380582583,
306
+ "e2e_output_tps_per_user": 186.90312354744358,
307
+ "completed": true
308
+ },
309
+ {
310
+ "ttft": 0.07437980198301375,
311
+ "time_to_second_token": 0.013298804871737957,
312
+ "latency": 2.6143259189557284,
313
+ "inter_token_latency_avg": 0.004970540346326251,
314
+ "chunk_inter_token_latency_avg": 0.01426936020771188,
315
+ "input_tokens": 78,
316
+ "output_tokens": 512,
317
+ "output_tps_per_user": 201.18537026645492,
318
+ "e2e_output_tps_per_user": 195.84398268312097,
319
+ "completed": true
320
+ },
321
+ {
322
+ "ttft": 0.08492515003308654,
323
+ "time_to_second_token": 0.013621869031339884,
324
+ "latency": 2.642661032034084,
325
+ "inter_token_latency_avg": 0.005005353976518586,
326
+ "chunk_inter_token_latency_avg": 0.01428902727374859,
327
+ "input_tokens": 78,
328
+ "output_tokens": 512,
329
+ "output_tps_per_user": 199.78607001448037,
330
+ "e2e_output_tps_per_user": 193.74410633584293,
331
+ "completed": true
332
+ },
333
+ {
334
+ "ttft": 0.07432189281098545,
335
+ "time_to_second_token": 0.013672125991433859,
336
+ "latency": 2.8059029488358647,
337
+ "inter_token_latency_avg": 0.005345559796526182,
338
+ "chunk_inter_token_latency_avg": 0.014153269720336162,
339
+ "input_tokens": 78,
340
+ "output_tokens": 512,
341
+ "output_tps_per_user": 187.0711465335868,
342
+ "e2e_output_tps_per_user": 182.4724551547382,
343
+ "completed": true
344
+ },
345
+ {
346
+ "ttft": 0.08590104104951024,
347
+ "time_to_second_token": 0.013801953988149762,
348
+ "latency": 2.5183071410283446,
349
+ "inter_token_latency_avg": 0.004760090215222768,
350
+ "chunk_inter_token_latency_avg": 0.014141895930109503,
351
+ "input_tokens": 78,
352
+ "output_tokens": 512,
353
+ "output_tps_per_user": 210.08005201287995,
354
+ "e2e_output_tps_per_user": 203.31118141170265,
355
+ "completed": true
356
+ },
357
+ {
358
+ "ttft": 0.07369623705744743,
359
+ "time_to_second_token": 0.01322799688205123,
360
+ "latency": 2.6919372058473527,
361
+ "inter_token_latency_avg": 0.0051237592344225156,
362
+ "chunk_inter_token_latency_avg": 0.014152653885350839,
363
+ "input_tokens": 78,
364
+ "output_tokens": 512,
365
+ "output_tps_per_user": 195.1692018004642,
366
+ "e2e_output_tps_per_user": 190.19760152199967,
367
+ "completed": true
368
+ },
369
+ {
370
+ "ttft": 0.08539086184464395,
371
+ "time_to_second_token": 0.013393500121310353,
372
+ "latency": 2.826388561865315,
373
+ "inter_token_latency_avg": 0.005363987671273328,
374
+ "chunk_inter_token_latency_avg": 0.01420206062186876,
375
+ "input_tokens": 78,
376
+ "output_tokens": 512,
377
+ "output_tps_per_user": 186.42846726801207,
378
+ "e2e_output_tps_per_user": 181.14989810958562,
379
+ "completed": true
380
+ },
381
+ {
382
+ "ttft": 0.08541309903375804,
383
+ "time_to_second_token": 0.013608098030090332,
384
+ "latency": 2.5878587630577385,
385
+ "inter_token_latency_avg": 0.004897153941338514,
386
+ "chunk_inter_token_latency_avg": 0.014138111096180682,
387
+ "input_tokens": 78,
388
+ "output_tokens": 512,
389
+ "output_tps_per_user": 204.20023792976278,
390
+ "e2e_output_tps_per_user": 197.84696418093378,
391
+ "completed": true
392
+ },
393
+ {
394
+ "ttft": 0.08486239681951702,
395
+ "time_to_second_token": 0.01424833619967103,
396
+ "latency": 0.0,
397
+ "inter_token_latency_avg": 0.005402592240876745,
398
+ "chunk_inter_token_latency_avg": 0.0141469941152855,
399
+ "input_tokens": 78,
400
+ "output_tokens": 255,
401
+ "output_tps_per_user": 185.09633068989817,
402
+ "e2e_output_tps_per_user": 0.0,
403
+ "completed": false
404
+ }
405
+ ],
406
+ "total_tokens": 3823,
407
+ "wall_time": 25.55206400505267,
408
+ "num_completed": 1,
409
+ "num_errors": 0,
410
+ "server_gen_throughput": 191.09785171771475,
411
+ "server_utilization": 0.005862646566164198,
412
+ "server_spec_accept_rate": 0.6,
413
+ "server_spec_accept_length": 0.0,
414
+ "avg_running_reqs": 1,
415
+ "max_running_reqs": 1,
416
+ "effective_concurrency": 1,
417
+ "avg_queue_reqs": 0,
418
+ "max_queue_reqs": 0,
419
+ "queue_fraction": 0.0,
420
+ "underfilled": false,
421
+ "warmup_timed_out": false,
422
+ "warmup_duration": 5.537,
423
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
424
+ "timeout_reason": "",
425
+ "capacity_limited": false,
426
+ "hardware_summary": {
427
+ "samples": 9,
428
+ "duration_seconds": 19.265,
429
+ "gpu_count": 4,
430
+ "cpu_util_avg_pct": 11.57,
431
+ "cpu_temp_max_c": 75.75,
432
+ "gpu_util_avg_pct": 98.97,
433
+ "gpu_util_max_pct": 100.0,
434
+ "mem_util_avg_pct": 43.47,
435
+ "mem_util_max_pct": 55.0,
436
+ "temp_avg_c": 68.11,
437
+ "temp_max_c": 83.0,
438
+ "power_total_avg_w": 1152.88,
439
+ "power_total_max_w": 1155.06,
440
+ "power_limit_total_w": 1200.0,
441
+ "vram_used_avg_mb": 384778.0,
442
+ "vram_used_max_mb": 384778.0,
443
+ "vram_total_mb": 391548.0,
444
+ "vram_used_avg_pct": 98.27,
445
+ "vram_used_max_pct": 98.27,
446
+ "pcie_rx_avg_mb_s": 8388.89,
447
+ "pcie_rx_max_mb_s": 8661.0,
448
+ "pcie_tx_avg_mb_s": 8246.33,
449
+ "pcie_tx_max_mb_s": 8504.0
450
+ }
451
+ },
452
+ {
453
+ "concurrency": 1,
454
+ "context_tokens": 8192,
455
+ "benchmark_mode": "duration",
456
+ "request_count_target": 0,
457
+ "warmup_request_count": 0,
458
+ "measurement_seconds": 19.993423,
459
+ "measurement_wall_seconds": 20.000549,
460
+ "client_output_tokens": 3302,
461
+ "server_output_tokens": 3302,
462
+ "aggregate_source": "openai_continuous_usage",
463
+ "aggregate_tps": 165.15431387835127,
464
+ "per_request_avg_tps": 165.15431387835127,
465
+ "ttft_avg": 0.6249137402628548,
466
+ "ttft_p50": 0.6289092639926821,
467
+ "ttft_p90": 0.6339087889529764,
468
+ "ttft_p99": 0.6347594925714657,
469
+ "time_to_second_token_avg": 0.017741062707500532,
470
+ "time_to_second_token_p50": 0.01763475441839546,
471
+ "time_to_second_token_p90": 0.018197339260950685,
472
+ "time_to_second_token_p99": 0.018255747086368502,
473
+ "request_latency_avg": 3.151980838239459,
474
+ "request_latency_p50": 3.1515419960487634,
475
+ "request_latency_p90": 3.2448623499367386,
476
+ "request_latency_p99": 3.291342785186134,
477
+ "inter_token_latency_avg": 0.004930662211393978,
478
+ "inter_token_latency_p50": 0.004938842342353614,
479
+ "inter_token_latency_p90": 0.005095802539678877,
480
+ "inter_token_latency_p99": 0.0051998018340067296,
481
+ "output_tps_per_user_avg": 203.0459222201492,
482
+ "output_tps_per_user_p50": 202.51955283196384,
483
+ "output_tps_per_user_p90": 210.3772401347902,
484
+ "output_tps_per_user_p99": 215.44168752783455,
485
+ "e2e_output_tps_per_user_avg": 162.56228618550645,
486
+ "e2e_output_tps_per_user_p50": 162.46015462967605,
487
+ "e2e_output_tps_per_user_p90": 167.22336677689123,
488
+ "e2e_output_tps_per_user_p99": 170.41814081496065,
489
+ "chunk_inter_token_latency_avg": 0.01426418191410213,
490
+ "chunk_inter_token_latency_p50": 0.014250208142300946,
491
+ "chunk_inter_token_latency_p90": 0.014323315611798752,
492
+ "chunk_inter_token_latency_p99": 0.014336108877223197,
493
+ "input_seq_len_avg": 8192.0,
494
+ "output_seq_len_avg": 512.0,
495
+ "output_seq_len_p50": 512.0,
496
+ "output_seq_len_p90": 512.0,
497
+ "output_seq_len_p99": 512.0,
498
+ "request_count": 8,
499
+ "completed_request_count": 7,
500
+ "request_samples": [
501
+ {
502
+ "ttft": 0.591038762126118,
503
+ "time_to_second_token": 0.017253336030989885,
504
+ "latency": 3.1515419960487634,
505
+ "inter_token_latency_avg": 0.005010769538009091,
506
+ "chunk_inter_token_latency_avg": 0.01422501796623692,
507
+ "input_tokens": 8192,
508
+ "output_tokens": 512,
509
+ "output_tps_per_user": 199.57014434899074,
510
+ "e2e_output_tps_per_user": 162.46015462967605,
511
+ "completed": true
512
+ },
513
+ {
514
+ "ttft": 0.6249128989875317,
515
+ "time_to_second_token": 0.017716374015435576,
516
+ "latency": 3.1119065389502794,
517
+ "inter_token_latency_avg": 0.004866915146698137,
518
+ "chunk_inter_token_latency_avg": 0.014211392228358558,
519
+ "input_tokens": 8192,
520
+ "output_tokens": 512,
521
+ "output_tps_per_user": 205.46896131493693,
522
+ "e2e_output_tps_per_user": 164.52936281714614,
523
+ "completed": true
524
+ },
525
+ {
526
+ "ttft": 0.6324374838732183,
527
+ "time_to_second_token": 0.018169526010751724,
528
+ "latency": 2.998129991814494,
529
+ "inter_token_latency_avg": 0.004629535240589581,
530
+ "chunk_inter_token_latency_avg": 0.014337530351159247,
531
+ "input_tokens": 8192,
532
+ "output_tokens": 512,
533
+ "output_tps_per_user": 216.00440390483948,
534
+ "e2e_output_tps_per_user": 170.77311570807947,
535
+ "completed": true
536
+ },
537
+ {
538
+ "ttft": 0.6335036919917911,
539
+ "time_to_second_token": 0.01815098780207336,
540
+ "latency": 3.2965072779916227,
541
+ "inter_token_latency_avg": 0.005211357311154269,
542
+ "chunk_inter_token_latency_avg": 0.014317223580644255,
543
+ "input_tokens": 8192,
544
+ "output_tokens": 512,
545
+ "output_tps_per_user": 191.88858876738755,
546
+ "e2e_output_tps_per_user": 155.3159015962898,
547
+ "completed": true
548
+ },
549
+ {
550
+ "ttft": 0.6260347329080105,
551
+ "time_to_second_token": 0.017283352091908455,
552
+ "latency": 3.1057244250550866,
553
+ "inter_token_latency_avg": 0.004852621706745746,
554
+ "chunk_inter_token_latency_avg": 0.01425109018475331,
555
+ "input_tokens": 8192,
556
+ "output_tokens": 512,
557
+ "output_tps_per_user": 206.07417194912927,
558
+ "e2e_output_tps_per_user": 164.85686748943237,
559
+ "completed": true
560
+ },
561
+ {
562
+ "ttft": 0.6317837950773537,
563
+ "time_to_second_token": 0.01826223684474826,
564
+ "latency": 3.2104323979001492,
565
+ "inter_token_latency_avg": 0.005046279066189424,
566
+ "chunk_inter_token_latency_avg": 0.014246677363661853,
567
+ "input_tokens": 8192,
568
+ "output_tokens": 512,
569
+ "output_tps_per_user": 198.165814233323,
570
+ "e2e_output_tps_per_user": 159.48007512473532,
571
+ "completed": true
572
+ },
573
+ {
574
+ "ttft": 0.6247445419430733,
575
+ "time_to_second_token": 0.017539554042741656,
576
+ "latency": 3.189623239915818,
577
+ "inter_token_latency_avg": 0.005019332089966232,
578
+ "chunk_inter_token_latency_avg": 0.014249326099848582,
579
+ "input_tokens": 8192,
580
+ "output_tokens": 512,
581
+ "output_tps_per_user": 199.22969472353194,
582
+ "e2e_output_tps_per_user": 160.52052593318606,
583
+ "completed": true
584
+ },
585
+ {
586
+ "ttft": 0.6348540151957422,
587
+ "time_to_second_token": 0.017553134821355343,
588
+ "latency": 0.0,
589
+ "inter_token_latency_avg": 0.004808487591799348,
590
+ "chunk_inter_token_latency_avg": 0.014275197538154316,
591
+ "input_tokens": 8192,
592
+ "output_tokens": 381,
593
+ "output_tps_per_user": 207.96559851905482,
594
+ "e2e_output_tps_per_user": 0.0,
595
+ "completed": false
596
+ }
597
+ ],
598
+ "total_tokens": 3302,
599
+ "wall_time": 26.068795799044892,
600
+ "num_completed": 1,
601
+ "num_errors": 0,
602
+ "server_gen_throughput": 165.0529013540346,
603
+ "server_utilization": 0.006141820212172022,
604
+ "server_spec_accept_rate": 0.6956521739130435,
605
+ "server_spec_accept_length": 0.0,
606
+ "avg_running_reqs": 0.9,
607
+ "max_running_reqs": 1,
608
+ "effective_concurrency": 0.9,
609
+ "avg_queue_reqs": 0,
610
+ "max_queue_reqs": 0,
611
+ "queue_fraction": 0.0,
612
+ "underfilled": false,
613
+ "warmup_timed_out": false,
614
+ "warmup_duration": 6.061,
615
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
616
+ "timeout_reason": "",
617
+ "capacity_limited": false,
618
+ "hardware_summary": {
619
+ "samples": 8,
620
+ "duration_seconds": 16.92,
621
+ "gpu_count": 4,
622
+ "cpu_util_avg_pct": 11.54,
623
+ "cpu_temp_max_c": 76.12,
624
+ "gpu_util_avg_pct": 98.81,
625
+ "gpu_util_max_pct": 100.0,
626
+ "mem_util_avg_pct": 39.97,
627
+ "mem_util_max_pct": 56.0,
628
+ "temp_avg_c": 68.56,
629
+ "temp_max_c": 84.0,
630
+ "power_total_avg_w": 1154.06,
631
+ "power_total_max_w": 1157.95,
632
+ "power_limit_total_w": 1200.0,
633
+ "vram_used_avg_mb": 384778.0,
634
+ "vram_used_max_mb": 384778.0,
635
+ "vram_total_mb": 391548.0,
636
+ "vram_used_avg_pct": 98.27,
637
+ "vram_used_max_pct": 98.27,
638
+ "pcie_rx_avg_mb_s": 11118.0,
639
+ "pcie_rx_max_mb_s": 30706.0,
640
+ "pcie_tx_avg_mb_s": 9898.38,
641
+ "pcie_tx_max_mb_s": 21763.0
642
+ }
643
+ },
644
+ {
645
+ "concurrency": 1,
646
+ "context_tokens": 32768,
647
+ "benchmark_mode": "duration",
648
+ "request_count_target": 0,
649
+ "warmup_request_count": 0,
650
+ "measurement_seconds": 19.993472,
651
+ "measurement_wall_seconds": 20.000595,
652
+ "client_output_tokens": 3246,
653
+ "server_output_tokens": 3246,
654
+ "aggregate_source": "openai_continuous_usage",
655
+ "aggregate_tps": 162.35299164317425,
656
+ "per_request_avg_tps": 162.35299164317425,
657
+ "ttft_avg": 0.6279244296310935,
658
+ "ttft_p50": 0.629666926455684,
659
+ "ttft_p90": 0.6346158204367385,
660
+ "ttft_p99": 0.6394360246905125,
661
+ "time_to_second_token_avg": 0.011822661821497604,
662
+ "time_to_second_token_p50": 0.012242211028933525,
663
+ "time_to_second_token_p90": 0.012837472674436867,
664
+ "time_to_second_token_p99": 0.012860276207793503,
665
+ "request_latency_avg": 3.184498446295038,
666
+ "request_latency_p50": 3.200756788952276,
667
+ "request_latency_p90": 3.258722552191466,
668
+ "request_latency_p99": 3.2730169181805104,
669
+ "inter_token_latency_avg": 0.005025481984653069,
670
+ "inter_token_latency_p50": 0.005032082509756118,
671
+ "inter_token_latency_p90": 0.005174755995160009,
672
+ "inter_token_latency_p99": 0.005182216347690941,
673
+ "output_tps_per_user_avg": 199.13235326934893,
674
+ "output_tps_per_user_p50": 198.72498796426024,
675
+ "output_tps_per_user_p90": 206.07410309939215,
676
+ "output_tps_per_user_p99": 208.73558083264146,
677
+ "e2e_output_tps_per_user_avg": 160.84431793668682,
678
+ "e2e_output_tps_per_user_p50": 159.96216949916902,
679
+ "e2e_output_tps_per_user_p90": 164.8638332468683,
680
+ "e2e_output_tps_per_user_p99": 166.41261273335627,
681
+ "chunk_inter_token_latency_avg": 0.014293934351271118,
682
+ "chunk_inter_token_latency_p50": 0.014270523142799533,
683
+ "chunk_inter_token_latency_p90": 0.014406589656055605,
684
+ "chunk_inter_token_latency_p99": 0.01443256986090686,
685
+ "input_seq_len_avg": 32768.0,
686
+ "output_seq_len_avg": 512.0,
687
+ "output_seq_len_p50": 512.0,
688
+ "output_seq_len_p90": 512.0,
689
+ "output_seq_len_p99": 512.0,
690
+ "request_count": 8,
691
+ "completed_request_count": 7,
692
+ "request_samples": [
693
+ {
694
+ "ttft": 0.6056491718627512,
695
+ "time_to_second_token": 0.01198451709933579,
696
+ "latency": 3.248134132940322,
697
+ "inter_token_latency_avg": 0.005171203446335755,
698
+ "chunk_inter_token_latency_avg": 0.014283702492311194,
699
+ "input_tokens": 32768,
700
+ "output_tokens": 512,
701
+ "output_tps_per_user": 193.37858399452193,
702
+ "e2e_output_tps_per_user": 157.62895836340357,
703
+ "completed": true
704
+ },
705
+ {
706
+ "ttft": 0.6288057561032474,
707
+ "time_to_second_token": 0.012826613849028945,
708
+ "latency": 3.2020828151144087,
709
+ "inter_token_latency_avg": 0.005035767238769396,
710
+ "chunk_inter_token_latency_avg": 0.014295983661173118,
711
+ "input_tokens": 32768,
712
+ "output_tokens": 512,
713
+ "output_tps_per_user": 198.5794721211882,
714
+ "e2e_output_tps_per_user": 159.89592698329588,
715
+ "completed": true
716
+ },
717
+ {
718
+ "ttft": 0.6289016020018607,
719
+ "time_to_second_token": 0.01273016701452434,
720
+ "latency": 3.073511565104127,
721
+ "inter_token_latency_avg": 0.0047839725305328104,
722
+ "chunk_inter_token_latency_avg": 0.014212848622687594,
723
+ "input_tokens": 32768,
724
+ "output_tokens": 512,
725
+ "output_tps_per_user": 209.03130058078028,
726
+ "e2e_output_tps_per_user": 166.58469934296605,
727
+ "completed": true
728
+ },
729
+ {
730
+ "ttft": 0.6260690451599658,
731
+ "time_to_second_token": 0.011439014924690127,
732
+ "latency": 3.274605181068182,
733
+ "inter_token_latency_avg": 0.005183045275749934,
734
+ "chunk_inter_token_latency_avg": 0.014394218129935958,
735
+ "input_tokens": 32768,
736
+ "output_tokens": 512,
737
+ "output_tps_per_user": 192.9367672473805,
738
+ "e2e_output_tps_per_user": 156.35472726913133,
739
+ "completed": true
740
+ },
741
+ {
742
+ "ttft": 0.6312455229926854,
743
+ "time_to_second_token": 0.011678099865093827,
744
+ "latency": 3.200756788952276,
745
+ "inter_token_latency_avg": 0.005028397780742839,
746
+ "chunk_inter_token_latency_avg": 0.014435456550334779,
747
+ "input_tokens": 32768,
748
+ "output_tokens": 512,
749
+ "output_tps_per_user": 198.87050380733228,
750
+ "e2e_output_tps_per_user": 159.96216949916902,
751
+ "completed": true
752
+ },
753
+ {
754
+ "ttft": 0.6304322509095073,
755
+ "time_to_second_token": 0.01249990495853126,
756
+ "latency": 3.165042991982773,
757
+ "inter_token_latency_avg": 0.004960099297599346,
758
+ "chunk_inter_token_latency_avg": 0.014239386185804863,
759
+ "input_tokens": 32768,
760
+ "output_tokens": 512,
761
+ "output_tps_per_user": 201.6088670813492,
762
+ "e2e_output_tps_per_user": 161.76715491603875,
763
+ "completed": true
764
+ },
765
+ {
766
+ "ttft": 0.6323204850777984,
767
+ "time_to_second_token": 0.01286280993372202,
768
+ "latency": 3.127355648903176,
769
+ "inter_token_latency_avg": 0.004882651984002696,
770
+ "chunk_inter_token_latency_avg": 0.014257343793287873,
771
+ "input_tokens": 32768,
772
+ "output_tokens": 512,
773
+ "output_tps_per_user": 204.80673275022582,
774
+ "e2e_output_tps_per_user": 163.71658918280312,
775
+ "completed": true
776
+ },
777
+ {
778
+ "ttft": 0.6399716029409319,
779
+ "time_to_second_token": 0.008560166927054524,
780
+ "latency": 0.0,
781
+ "inter_token_latency_avg": 0.005158718323491779,
782
+ "chunk_inter_token_latency_avg": 0.014232535374633568,
783
+ "input_tokens": 32768,
784
+ "output_tokens": 310,
785
+ "output_tps_per_user": 193.84659857201325,
786
+ "e2e_output_tps_per_user": 0.0,
787
+ "completed": false
788
+ }
789
+ ],
790
+ "total_tokens": 3246,
791
+ "wall_time": 29.109418482985348,
792
+ "num_completed": 1,
793
+ "num_errors": 0,
794
+ "server_gen_throughput": 162.2569315918152,
795
+ "server_utilization": 0.006979341150195384,
796
+ "server_spec_accept_rate": 0.6163522012578616,
797
+ "server_spec_accept_length": 0.0,
798
+ "avg_running_reqs": 0.9,
799
+ "max_running_reqs": 1,
800
+ "effective_concurrency": 0.9,
801
+ "avg_queue_reqs": 0,
802
+ "max_queue_reqs": 0,
803
+ "queue_fraction": 0.0,
804
+ "underfilled": false,
805
+ "warmup_timed_out": false,
806
+ "warmup_duration": 9.102,
807
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
808
+ "timeout_reason": "",
809
+ "capacity_limited": false,
810
+ "hardware_summary": {
811
+ "samples": 8,
812
+ "duration_seconds": 16.9,
813
+ "gpu_count": 4,
814
+ "cpu_util_avg_pct": 11.49,
815
+ "cpu_temp_max_c": 76.62,
816
+ "gpu_util_avg_pct": 95.94,
817
+ "gpu_util_max_pct": 100.0,
818
+ "mem_util_avg_pct": 38.56,
819
+ "mem_util_max_pct": 55.0,
820
+ "temp_avg_c": 68.72,
821
+ "temp_max_c": 84.0,
822
+ "power_total_avg_w": 1151.98,
823
+ "power_total_max_w": 1158.51,
824
+ "power_limit_total_w": 1200.0,
825
+ "vram_used_avg_mb": 384778.0,
826
+ "vram_used_max_mb": 384778.0,
827
+ "vram_total_mb": 391548.0,
828
+ "vram_used_avg_pct": 98.27,
829
+ "vram_used_max_pct": 98.27,
830
+ "pcie_rx_avg_mb_s": 12769.0,
831
+ "pcie_rx_max_mb_s": 39144.0,
832
+ "pcie_tx_avg_mb_s": 12689.38,
833
+ "pcie_tx_max_mb_s": 43337.0
834
+ }
835
+ },
836
+ {
837
+ "concurrency": 2,
838
+ "context_tokens": 0,
839
+ "benchmark_mode": "duration",
840
+ "request_count_target": 0,
841
+ "warmup_request_count": 0,
842
+ "measurement_seconds": 19.987074,
843
+ "measurement_wall_seconds": 20.001238,
844
+ "client_output_tokens": 5152,
845
+ "server_output_tokens": 5152,
846
+ "aggregate_source": "openai_continuous_usage",
847
+ "aggregate_tps": 257.76658819245483,
848
+ "per_request_avg_tps": 128.88329409622742,
849
+ "ttft_avg": 0.12264618660057229,
850
+ "ttft_p50": 0.12349911453202367,
851
+ "ttft_p90": 0.14274162366054954,
852
+ "ttft_p99": 0.16984736640239134,
853
+ "time_to_second_token_avg": 0.019459182529577186,
854
+ "time_to_second_token_p50": 0.02056812052614987,
855
+ "time_to_second_token_p90": 0.02087695982772857,
856
+ "time_to_second_token_p99": 0.020941995084285736,
857
+ "request_latency_avg": 4.002757317425373,
858
+ "request_latency_p50": 4.017580935033038,
859
+ "request_latency_p90": 4.2898590574972335,
860
+ "request_latency_p99": 4.336852466568817,
861
+ "inter_token_latency_avg": 0.007545030826191364,
862
+ "inter_token_latency_p50": 0.00755755253999135,
863
+ "inter_token_latency_p90": 0.008070413317256143,
864
+ "inter_token_latency_p99": 0.008259788365663738,
865
+ "output_tps_per_user_avg": 132.88272402904317,
866
+ "output_tps_per_user_p50": 132.32326777905013,
867
+ "output_tps_per_user_p90": 141.37995688471477,
868
+ "output_tps_per_user_p99": 145.6980556167255,
869
+ "e2e_output_tps_per_user_avg": 128.22354203464204,
870
+ "e2e_output_tps_per_user_p50": 127.43988690683825,
871
+ "e2e_output_tps_per_user_p90": 134.35973544624332,
872
+ "e2e_output_tps_per_user_p99": 140.69704751927173,
873
+ "chunk_inter_token_latency_avg": 0.02104700313164791,
874
+ "chunk_inter_token_latency_p50": 0.02098676121404283,
875
+ "chunk_inter_token_latency_p90": 0.0212750823967716,
876
+ "chunk_inter_token_latency_p99": 0.02139749144743477,
877
+ "input_seq_len_avg": 78.0,
878
+ "output_seq_len_avg": 512.0,
879
+ "output_seq_len_p50": 512.0,
880
+ "output_seq_len_p90": 512.0,
881
+ "output_seq_len_p99": 512.0,
882
+ "request_count": 14,
883
+ "completed_request_count": 12,
884
+ "request_samples": [
885
+ {
886
+ "ttft": 0.07193002197891474,
887
+ "time_to_second_token": 0.01364690694026649,
888
+ "latency": 4.016205613967031,
889
+ "inter_token_latency_avg": 0.0077187389275697,
890
+ "chunk_inter_token_latency_avg": 0.020869183026392152,
891
+ "input_tokens": 78,
892
+ "output_tokens": 512,
893
+ "output_tps_per_user": 129.55484171490914,
894
+ "e2e_output_tps_per_user": 127.48351285089433,
895
+ "completed": true
896
+ },
897
+ {
898
+ "ttft": 0.10589846200309694,
899
+ "time_to_second_token": 0.01401821500621736,
900
+ "latency": 3.8386515881866217,
901
+ "inter_token_latency_avg": 0.007304800638323923,
902
+ "chunk_inter_token_latency_avg": 0.02108900071290127,
903
+ "input_tokens": 78,
904
+ "output_tokens": 512,
905
+ "output_tps_per_user": 136.89627541011834,
906
+ "e2e_output_tps_per_user": 133.3801696344806,
907
+ "completed": true
908
+ },
909
+ {
910
+ "ttft": 0.12403434701263905,
911
+ "time_to_second_token": 0.020574419992044568,
912
+ "latency": 3.8075810340233147,
913
+ "inter_token_latency_avg": 0.007208506236811498,
914
+ "chunk_inter_token_latency_avg": 0.021048838211489576,
915
+ "input_tokens": 78,
916
+ "output_tokens": 512,
917
+ "output_tps_per_user": 138.72499615708523,
918
+ "e2e_output_tps_per_user": 134.46857609199472,
919
+ "completed": true
920
+ },
921
+ {
922
+ "ttft": 0.11724809114821255,
923
+ "time_to_second_token": 0.020686850883066654,
924
+ "latency": 3.954718264983967,
925
+ "inter_token_latency_avg": 0.007509726367584646,
926
+ "chunk_inter_token_latency_avg": 0.02096978237068718,
927
+ "input_tokens": 78,
928
+ "output_tokens": 512,
929
+ "output_tps_per_user": 133.16064408371113,
930
+ "e2e_output_tps_per_user": 129.46560682549045,
931
+ "completed": true
932
+ },
933
+ {
934
+ "ttft": 0.11513775610364974,
935
+ "time_to_second_token": 0.020763778826221824,
936
+ "latency": 4.340469955932349,
937
+ "inter_token_latency_avg": 0.008268751858764578,
938
+ "chunk_inter_token_latency_avg": 0.021126660999143496,
939
+ "input_tokens": 78,
940
+ "output_tokens": 512,
941
+ "output_tps_per_user": 120.93723660845332,
942
+ "e2e_output_tps_per_user": 117.95957700391926,
943
+ "completed": true
944
+ },
945
+ {
946
+ "ttft": 0.12337096291594207,
947
+ "time_to_second_token": 0.020409638062119484,
948
+ "latency": 3.619222234003246,
949
+ "inter_token_latency_avg": 0.006841196225219772,
950
+ "chunk_inter_token_latency_avg": 0.02093324114423535,
951
+ "input_tokens": 78,
952
+ "output_tokens": 512,
953
+ "output_tps_per_user": 146.1732666450267,
954
+ "e2e_output_tps_per_user": 141.46685859455317,
955
+ "completed": true
956
+ },
957
+ {
958
+ "ttft": 0.12318649212829769,
959
+ "time_to_second_token": 0.02092546597123146,
960
+ "latency": 0.0,
961
+ "inter_token_latency_avg": 0.007490993097976402,
962
+ "chunk_inter_token_latency_avg": 0.021415427327156067,
963
+ "input_tokens": 78,
964
+ "output_tokens": 244,
965
+ "output_tps_per_user": 133.4936485617825,
966
+ "e2e_output_tps_per_user": 0.0,
967
+ "completed": false
968
+ },
969
+ {
970
+ "ttft": 0.15055592195130885,
971
+ "time_to_second_token": 0.018002169905230403,
972
+ "latency": 4.036904443986714,
973
+ "inter_token_latency_avg": 0.007605378712398053,
974
+ "chunk_inter_token_latency_avg": 0.02089434689266347,
975
+ "input_tokens": 78,
976
+ "output_tokens": 512,
977
+ "output_tps_per_user": 131.48589147438915,
978
+ "e2e_output_tps_per_user": 126.82985369214379,
979
+ "completed": true
980
+ },
981
+ {
982
+ "ttft": 0.17272999603301287,
983
+ "time_to_second_token": 0.020944464951753616,
984
+ "latency": 4.130337374052033,
985
+ "inter_token_latency_avg": 0.0077448285284129545,
986
+ "chunk_inter_token_latency_avg": 0.021277459021607634,
987
+ "input_tokens": 78,
988
+ "output_tokens": 512,
989
+ "output_tps_per_user": 129.11841706131574,
990
+ "e2e_output_tps_per_user": 123.96081812021731,
991
+ "completed": true
992
+ },
993
+ {
994
+ "ttft": 0.12362726614810526,
995
+ "time_to_second_token": 0.020660570822656155,
996
+ "latency": 4.093334136996418,
997
+ "inter_token_latency_avg": 0.0077685065965720414,
998
+ "chunk_inter_token_latency_avg": 0.02100374005739848,
999
+ "input_tokens": 78,
1000
+ "output_tokens": 512,
1001
+ "output_tps_per_user": 128.72486977629183,
1002
+ "e2e_output_tps_per_user": 125.08140866694363,
1003
+ "completed": true
1004
+ },
1005
+ {
1006
+ "ttft": 0.12366992980241776,
1007
+ "time_to_second_token": 0.02057293802499771,
1008
+ "latency": 3.8691232178825885,
1009
+ "inter_token_latency_avg": 0.007329654184109923,
1010
+ "chunk_inter_token_latency_avg": 0.02092432004514062,
1011
+ "input_tokens": 78,
1012
+ "output_tokens": 512,
1013
+ "output_tps_per_user": 136.43208463612325,
1014
+ "e2e_output_tps_per_user": 132.32972205010222,
1015
+ "completed": true
1016
+ },
1017
+ {
1018
+ "ttft": 0.11748491204343736,
1019
+ "time_to_second_token": 0.020382387097924948,
1020
+ "latency": 4.307583688991144,
1021
+ "inter_token_latency_avg": 0.008199801911835043,
1022
+ "chunk_inter_token_latency_avg": 0.021269536938820846,
1023
+ "input_tokens": 78,
1024
+ "output_tokens": 512,
1025
+ "output_tps_per_user": 121.9541655703496,
1026
+ "e2e_output_tps_per_user": 118.86013992218285,
1027
+ "completed": true
1028
+ },
1029
+ {
1030
+ "ttft": 0.12366419215686619,
1031
+ "time_to_second_token": 0.02027744590304792,
1032
+ "latency": 4.018956256099045,
1033
+ "inter_token_latency_avg": 0.007622880751354558,
1034
+ "chunk_inter_token_latency_avg": 0.02094243045130204,
1035
+ "input_tokens": 78,
1036
+ "output_tokens": 512,
1037
+ "output_tps_per_user": 131.18400151049244,
1038
+ "e2e_output_tps_per_user": 127.39626096278218,
1039
+ "completed": true
1040
+ },
1041
+ {
1042
+ "ttft": 0.1245082609821111,
1043
+ "time_to_second_token": 0.020563303027302027,
1044
+ "latency": 0.0,
1045
+ "inter_token_latency_avg": 0.007016667529746,
1046
+ "chunk_inter_token_latency_avg": 0.020894076644132533,
1047
+ "input_tokens": 78,
1048
+ "output_tokens": 135,
1049
+ "output_tps_per_user": 142.517797196556,
1050
+ "e2e_output_tps_per_user": 0.0,
1051
+ "completed": false
1052
+ }
1053
+ ],
1054
+ "total_tokens": 5152,
1055
+ "wall_time": 25.544181779026985,
1056
+ "num_completed": 2,
1057
+ "num_errors": 0,
1058
+ "server_gen_throughput": 257.5183083679263,
1059
+ "server_utilization": 0.011725293132328285,
1060
+ "server_spec_accept_rate": 0.5490196078431373,
1061
+ "server_spec_accept_length": 0.0,
1062
+ "avg_running_reqs": 2,
1063
+ "max_running_reqs": 2,
1064
+ "effective_concurrency": 2,
1065
+ "avg_queue_reqs": 0,
1066
+ "max_queue_reqs": 0,
1067
+ "queue_fraction": 0.0,
1068
+ "underfilled": false,
1069
+ "warmup_timed_out": false,
1070
+ "warmup_duration": 5.536,
1071
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1072
+ "timeout_reason": "",
1073
+ "capacity_limited": false,
1074
+ "hardware_summary": {
1075
+ "samples": 9,
1076
+ "duration_seconds": 19.327,
1077
+ "gpu_count": 4,
1078
+ "cpu_util_avg_pct": 11.5,
1079
+ "cpu_temp_max_c": 76.5,
1080
+ "gpu_util_avg_pct": 100.0,
1081
+ "gpu_util_max_pct": 100.0,
1082
+ "mem_util_avg_pct": 40.53,
1083
+ "mem_util_max_pct": 50.0,
1084
+ "temp_avg_c": 68.83,
1085
+ "temp_max_c": 84.0,
1086
+ "power_total_avg_w": 1175.08,
1087
+ "power_total_max_w": 1176.23,
1088
+ "power_limit_total_w": 1200.0,
1089
+ "vram_used_avg_mb": 384778.0,
1090
+ "vram_used_max_mb": 384778.0,
1091
+ "vram_total_mb": 391548.0,
1092
+ "vram_used_avg_pct": 98.27,
1093
+ "vram_used_max_pct": 98.27,
1094
+ "pcie_rx_avg_mb_s": 11424.78,
1095
+ "pcie_rx_max_mb_s": 12660.0,
1096
+ "pcie_tx_avg_mb_s": 11473.0,
1097
+ "pcie_tx_max_mb_s": 12876.0
1098
+ }
1099
+ },
1100
+ {
1101
+ "concurrency": 4,
1102
+ "context_tokens": 0,
1103
+ "benchmark_mode": "duration",
1104
+ "request_count_target": 0,
1105
+ "warmup_request_count": 0,
1106
+ "measurement_seconds": 19.982261,
1107
+ "measurement_wall_seconds": 20.001434,
1108
+ "client_output_tokens": 6107,
1109
+ "server_output_tokens": 6107,
1110
+ "aggregate_source": "openai_continuous_usage",
1111
+ "aggregate_tps": 305.62106309879454,
1112
+ "per_request_avg_tps": 76.40526577469863,
1113
+ "ttft_avg": 0.18411949259461835,
1114
+ "ttft_p50": 0.18233595008496195,
1115
+ "ttft_p90": 0.22579583001788706,
1116
+ "ttft_p99": 0.24202772771241143,
1117
+ "time_to_second_token_avg": 0.032063150822068565,
1118
+ "time_to_second_token_p50": 0.03419885353650898,
1119
+ "time_to_second_token_p90": 0.034818764543160796,
1120
+ "time_to_second_token_p99": 0.03491839349735528,
1121
+ "request_latency_avg": 6.813750679527099,
1122
+ "request_latency_p50": 6.7830935755046085,
1123
+ "request_latency_p90": 7.136929157585837,
1124
+ "request_latency_p99": 7.218653435369488,
1125
+ "inter_token_latency_avg": 0.01272961827279658,
1126
+ "inter_token_latency_p50": 0.012860421986189724,
1127
+ "inter_token_latency_p90": 0.013537522864770526,
1128
+ "inter_token_latency_p99": 0.013757404152471607,
1129
+ "output_tps_per_user_avg": 78.81967025679022,
1130
+ "output_tps_per_user_p50": 77.75794916087145,
1131
+ "output_tps_per_user_p90": 85.34425839032679,
1132
+ "output_tps_per_user_p99": 89.87592923645526,
1133
+ "e2e_output_tps_per_user_avg": 75.21735206235604,
1134
+ "e2e_output_tps_per_user_p50": 75.48202324847628,
1135
+ "e2e_output_tps_per_user_p90": 78.42642900563833,
1136
+ "e2e_output_tps_per_user_p99": 78.66327604603737,
1137
+ "chunk_inter_token_latency_avg": 0.03457812773162968,
1138
+ "chunk_inter_token_latency_p50": 0.03461816136042967,
1139
+ "chunk_inter_token_latency_p90": 0.03496178867127844,
1140
+ "chunk_inter_token_latency_p99": 0.035020915078215,
1141
+ "input_seq_len_avg": 78.0,
1142
+ "output_seq_len_avg": 512.0,
1143
+ "output_seq_len_p50": 512.0,
1144
+ "output_seq_len_p90": 512.0,
1145
+ "output_seq_len_p99": 512.0,
1146
+ "request_count": 16,
1147
+ "completed_request_count": 12,
1148
+ "request_samples": [
1149
+ {
1150
+ "ttft": 0.071592987049371,
1151
+ "time_to_second_token": 0.013179793022572994,
1152
+ "latency": 6.610689978115261,
1153
+ "inter_token_latency_avg": 0.012796667301498806,
1154
+ "chunk_inter_token_latency_avg": 0.03405779682846818,
1155
+ "input_tokens": 78,
1156
+ "output_tokens": 512,
1157
+ "output_tps_per_user": 78.14534647492752,
1158
+ "e2e_output_tps_per_user": 77.45031179725261,
1159
+ "completed": true
1160
+ },
1161
+ {
1162
+ "ttft": 0.22638577106408775,
1163
+ "time_to_second_token": 0.034233805956318974,
1164
+ "latency": 6.729027182096615,
1165
+ "inter_token_latency_avg": 0.012725325657597902,
1166
+ "chunk_inter_token_latency_avg": 0.03458851814379004,
1167
+ "input_tokens": 78,
1168
+ "output_tokens": 512,
1169
+ "output_tps_per_user": 78.5834505856383,
1170
+ "e2e_output_tps_per_user": 76.08826449122355,
1171
+ "completed": true
1172
+ },
1173
+ {
1174
+ "ttft": 0.18156861700117588,
1175
+ "time_to_second_token": 0.034193265018984675,
1176
+ "latency": 6.771002053981647,
1177
+ "inter_token_latency_avg": 0.012895173066497987,
1178
+ "chunk_inter_token_latency_avg": 0.03468122861568669,
1179
+ "input_tokens": 78,
1180
+ "output_tokens": 512,
1181
+ "output_tps_per_user": 77.5483969732851,
1182
+ "e2e_output_tps_per_user": 75.61657726848887,
1183
+ "completed": true
1184
+ },
1185
+ {
1186
+ "ttft": 0.16493634996004403,
1187
+ "time_to_second_token": 0.03424763702787459,
1188
+ "latency": 0.0,
1189
+ "inter_token_latency_avg": 0.01179988086244578,
1190
+ "chunk_inter_token_latency_avg": 0.03492764735283951,
1191
+ "input_tokens": 78,
1192
+ "output_tokens": 445,
1193
+ "output_tps_per_user": 84.74661834786765,
1194
+ "e2e_output_tps_per_user": 0.0,
1195
+ "completed": false
1196
+ },
1197
+ {
1198
+ "ttft": 0.18636853992938995,
1199
+ "time_to_second_token": 0.029447735054418445,
1200
+ "latency": 7.226989238988608,
1201
+ "inter_token_latency_avg": 0.013778122698746023,
1202
+ "chunk_inter_token_latency_avg": 0.03468286058649861,
1203
+ "input_tokens": 78,
1204
+ "output_tokens": 512,
1205
+ "output_tps_per_user": 72.57882818034507,
1206
+ "e2e_output_tps_per_user": 70.84554619755497,
1207
+ "completed": true
1208
+ },
1209
+ {
1210
+ "ttft": 0.18095432897098362,
1211
+ "time_to_second_token": 0.03420444205403328,
1212
+ "latency": 6.79518509702757,
1213
+ "inter_token_latency_avg": 0.012943700133183144,
1214
+ "chunk_inter_token_latency_avg": 0.034995929989717386,
1215
+ "input_tokens": 78,
1216
+ "output_tokens": 512,
1217
+ "output_tps_per_user": 77.25766123369529,
1218
+ "e2e_output_tps_per_user": 75.34746922846371,
1219
+ "completed": true
1220
+ },
1221
+ {
1222
+ "ttft": 0.18133209715597332,
1223
+ "time_to_second_token": 0.034592753974720836,
1224
+ "latency": 6.89423265401274,
1225
+ "inter_token_latency_avg": 0.01313679169639289,
1226
+ "chunk_inter_token_latency_avg": 0.03460258018998333,
1227
+ "input_tokens": 78,
1228
+ "output_tokens": 512,
1229
+ "output_tps_per_user": 76.12208696850851,
1230
+ "e2e_output_tps_per_user": 74.26497272354074,
1231
+ "completed": true
1232
+ },
1233
+ {
1234
+ "ttft": 0.18180311610922217,
1235
+ "time_to_second_token": 0.03340313700027764,
1236
+ "latency": 0.0,
1237
+ "inter_token_latency_avg": 0.011041162894689477,
1238
+ "chunk_inter_token_latency_avg": 0.034236164014541014,
1239
+ "input_tokens": 78,
1240
+ "output_tokens": 401,
1241
+ "output_tps_per_user": 90.57016996651457,
1242
+ "e2e_output_tps_per_user": 0.0,
1243
+ "completed": false
1244
+ },
1245
+ {
1246
+ "ttft": 0.18593904399313033,
1247
+ "time_to_second_token": 0.029689945047721267,
1248
+ "latency": 6.507442394969985,
1249
+ "inter_token_latency_avg": 0.01237084804496449,
1250
+ "chunk_inter_token_latency_avg": 0.033804830753886926,
1251
+ "input_tokens": 78,
1252
+ "output_tokens": 512,
1253
+ "output_tps_per_user": 80.83520194939638,
1254
+ "e2e_output_tps_per_user": 78.67914442020374,
1255
+ "completed": true
1256
+ },
1257
+ {
1258
+ "ttft": 0.1811696880031377,
1259
+ "time_to_second_token": 0.034721852047368884,
1260
+ "latency": 7.151209206087515,
1261
+ "inter_token_latency_avg": 0.01363999905691659,
1262
+ "chunk_inter_token_latency_avg": 0.03502532421147928,
1263
+ "input_tokens": 78,
1264
+ "output_tokens": 512,
1265
+ "output_tps_per_user": 73.31378806019188,
1266
+ "e2e_output_tps_per_user": 71.5962832641166,
1267
+ "completed": true
1268
+ },
1269
+ {
1270
+ "ttft": 0.18142080190591514,
1271
+ "time_to_second_token": 0.03491567703895271,
1272
+ "latency": 6.5193956850562245,
1273
+ "inter_token_latency_avg": 0.01240308196311215,
1274
+ "chunk_inter_token_latency_avg": 0.034633742530876005,
1275
+ "input_tokens": 78,
1276
+ "output_tokens": 512,
1277
+ "output_tps_per_user": 80.62512228606465,
1278
+ "e2e_output_tps_per_user": 78.53488647323674,
1279
+ "completed": true
1280
+ },
1281
+ {
1282
+ "ttft": 0.24478807300329208,
1283
+ "time_to_second_token": 0.03329922794364393,
1284
+ "latency": 0.0,
1285
+ "inter_token_latency_avg": 0.01343504667262446,
1286
+ "chunk_inter_token_latency_avg": 0.034875908828251166,
1287
+ "input_tokens": 78,
1288
+ "output_tokens": 380,
1289
+ "output_tps_per_user": 74.43219397500282,
1290
+ "e2e_output_tps_per_user": 0.0,
1291
+ "completed": false
1292
+ },
1293
+ {
1294
+ "ttft": 0.18580231512896717,
1295
+ "time_to_second_token": 0.029555089073255658,
1296
+ "latency": 7.008408721070737,
1297
+ "inter_token_latency_avg": 0.013351480246461388,
1298
+ "chunk_inter_token_latency_avg": 0.034457608110817016,
1299
+ "input_tokens": 78,
1300
+ "output_tokens": 512,
1301
+ "output_tps_per_user": 74.89806235267697,
1302
+ "e2e_output_tps_per_user": 73.05510000589366,
1303
+ "completed": true
1304
+ },
1305
+ {
1306
+ "ttft": 0.18286878406070173,
1307
+ "time_to_second_token": 0.0349188728723675,
1308
+ "latency": 6.753246791893616,
1309
+ "inter_token_latency_avg": 0.012857882598498854,
1310
+ "chunk_inter_token_latency_avg": 0.03458093688333113,
1311
+ "input_tokens": 78,
1312
+ "output_tokens": 512,
1313
+ "output_tps_per_user": 77.7733030566595,
1314
+ "e2e_output_tps_per_user": 75.81538418151527,
1315
+ "completed": true
1316
+ },
1317
+ {
1318
+ "ttft": 0.22520588897168636,
1319
+ "time_to_second_token": 0.03444401710294187,
1320
+ "latency": 6.798179151024669,
1321
+ "inter_token_latency_avg": 0.012862961373880592,
1322
+ "chunk_inter_token_latency_avg": 0.03477763630715864,
1323
+ "input_tokens": 78,
1324
+ "output_tokens": 512,
1325
+ "output_tps_per_user": 77.7425952650834,
1326
+ "e2e_output_tps_per_user": 75.31428469678204,
1327
+ "completed": true
1328
+ },
1329
+ {
1330
+ "ttft": 0.18377547920681536,
1331
+ "time_to_second_token": 0.033963162917643785,
1332
+ "latency": 0.0,
1333
+ "inter_token_latency_avg": 0.011635768097234754,
1334
+ "chunk_inter_token_latency_avg": 0.03432133035874999,
1335
+ "input_tokens": 78,
1336
+ "output_tokens": 411,
1337
+ "output_tps_per_user": 85.94189843278592,
1338
+ "e2e_output_tps_per_user": 0.0,
1339
+ "completed": false
1340
+ }
1341
+ ],
1342
+ "total_tokens": 6107,
1343
+ "wall_time": 25.55054651084356,
1344
+ "num_completed": 4,
1345
+ "num_errors": 0,
1346
+ "server_gen_throughput": 305.247463685858,
1347
+ "server_utilization": 0.02345058626465657,
1348
+ "server_spec_accept_rate": 0.5805555555555556,
1349
+ "server_spec_accept_length": 0.0,
1350
+ "avg_running_reqs": 3.9,
1351
+ "max_running_reqs": 4,
1352
+ "effective_concurrency": 3.9,
1353
+ "avg_queue_reqs": 0.1,
1354
+ "max_queue_reqs": 1,
1355
+ "queue_fraction": 0.05,
1356
+ "underfilled": true,
1357
+ "warmup_timed_out": false,
1358
+ "warmup_duration": 5.534,
1359
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1360
+ "timeout_reason": "",
1361
+ "capacity_limited": true,
1362
+ "hardware_summary": {
1363
+ "samples": 8,
1364
+ "duration_seconds": 16.89,
1365
+ "gpu_count": 4,
1366
+ "cpu_util_avg_pct": 11.53,
1367
+ "cpu_temp_max_c": 77.38,
1368
+ "gpu_util_avg_pct": 100.0,
1369
+ "gpu_util_max_pct": 100.0,
1370
+ "mem_util_avg_pct": 35.53,
1371
+ "mem_util_max_pct": 45.0,
1372
+ "temp_avg_c": 69.31,
1373
+ "temp_max_c": 84.0,
1374
+ "power_total_avg_w": 1171.59,
1375
+ "power_total_max_w": 1174.02,
1376
+ "power_limit_total_w": 1200.0,
1377
+ "vram_used_avg_mb": 384778.0,
1378
+ "vram_used_max_mb": 384778.0,
1379
+ "vram_total_mb": 391548.0,
1380
+ "vram_used_avg_pct": 98.27,
1381
+ "vram_used_max_pct": 98.27,
1382
+ "pcie_rx_avg_mb_s": 8123.88,
1383
+ "pcie_rx_max_mb_s": 11132.0,
1384
+ "pcie_tx_avg_mb_s": 8456.62,
1385
+ "pcie_tx_max_mb_s": 11467.0
1386
+ }
1387
+ },
1388
+ {
1389
+ "concurrency": 2,
1390
+ "context_tokens": 8192,
1391
+ "benchmark_mode": "duration",
1392
+ "request_count_target": 0,
1393
+ "warmup_request_count": 0,
1394
+ "measurement_seconds": 19.996131,
1395
+ "measurement_wall_seconds": 20.000268,
1396
+ "client_output_tokens": 3670,
1397
+ "server_output_tokens": 3670,
1398
+ "aggregate_source": "openai_continuous_usage",
1399
+ "aggregate_tps": 183.53550127953065,
1400
+ "per_request_avg_tps": 91.76775063976532,
1401
+ "ttft_avg": 1.2004322615684941,
1402
+ "ttft_p50": 1.2173257099930197,
1403
+ "ttft_p90": 1.3659410797525195,
1404
+ "ttft_p99": 1.9023066108068454,
1405
+ "time_to_second_token_avg": 0.02541845981031656,
1406
+ "time_to_second_token_p50": 0.027192589943297207,
1407
+ "time_to_second_token_p90": 0.027629395388066767,
1408
+ "time_to_second_token_p99": 0.027753254855051635,
1409
+ "request_latency_avg": 5.37285651997081,
1410
+ "request_latency_p50": 5.446895623463206,
1411
+ "request_latency_p90": 5.665212600212544,
1412
+ "request_latency_p99": 5.969511628868058,
1413
+ "inter_token_latency_avg": 0.00809208952899205,
1414
+ "inter_token_latency_p50": 0.00815642245274234,
1415
+ "inter_token_latency_p90": 0.008585907641263088,
1416
+ "inter_token_latency_p99": 0.009444491741223483,
1417
+ "output_tps_per_user_avg": 124.44614817528017,
1418
+ "output_tps_per_user_p50": 122.61098226109307,
1419
+ "output_tps_per_user_p90": 140.4927553359884,
1420
+ "output_tps_per_user_p99": 142.21413373382023,
1421
+ "e2e_output_tps_per_user_avg": 95.63015924091229,
1422
+ "e2e_output_tps_per_user_p50": 93.9996349349982,
1423
+ "e2e_output_tps_per_user_p90": 103.25204134689379,
1424
+ "e2e_output_tps_per_user_p99": 103.55866690719752,
1425
+ "chunk_inter_token_latency_avg": 0.023212736865606577,
1426
+ "chunk_inter_token_latency_p50": 0.02336988428975669,
1427
+ "chunk_inter_token_latency_p90": 0.0253701743817706,
1428
+ "chunk_inter_token_latency_p99": 0.02550767397899725,
1429
+ "input_seq_len_avg": 8192.0,
1430
+ "output_seq_len_avg": 512.0,
1431
+ "output_seq_len_p50": 512.0,
1432
+ "output_seq_len_p90": 512.0,
1433
+ "output_seq_len_p99": 512.0,
1434
+ "request_count": 10,
1435
+ "completed_request_count": 8,
1436
+ "request_samples": [
1437
+ {
1438
+ "ttft": 0.5909663320053369,
1439
+ "time_to_second_token": 0.018111236859112978,
1440
+ "latency": 5.465850109001622,
1441
+ "inter_token_latency_avg": 0.009539889974552416,
1442
+ "chunk_inter_token_latency_avg": 0.025522951712022433,
1443
+ "input_tokens": 8192,
1444
+ "output_tokens": 512,
1445
+ "output_tps_per_user": 104.82301186570206,
1446
+ "e2e_output_tps_per_user": 93.67252847947574,
1447
+ "completed": true
1448
+ },
1449
+ {
1450
+ "ttft": 1.9619027809239924,
1451
+ "time_to_second_token": 0.027767017018049955,
1452
+ "latency": 6.003322632052004,
1453
+ "inter_token_latency_avg": 0.007908845109839554,
1454
+ "chunk_inter_token_latency_avg": 0.023634034217122877,
1455
+ "input_tokens": 8192,
1456
+ "output_tokens": 512,
1457
+ "output_tps_per_user": 126.44071114199465,
1458
+ "e2e_output_tps_per_user": 85.28610427605696,
1459
+ "completed": true
1460
+ },
1461
+ {
1462
+ "ttft": 1.2156082668807358,
1463
+ "time_to_second_token": 0.026928086066618562,
1464
+ "latency": 5.520308300852776,
1465
+ "inter_token_latency_avg": 0.008424070516579334,
1466
+ "chunk_inter_token_latency_avg": 0.023395108880282824,
1467
+ "input_tokens": 8192,
1468
+ "output_tokens": 512,
1469
+ "output_tps_per_user": 118.70745835186321,
1470
+ "e2e_output_tps_per_user": 92.74844303911547,
1471
+ "completed": true
1472
+ },
1473
+ {
1474
+ "ttft": 1.2087725747842342,
1475
+ "time_to_second_token": 0.027614104095846415,
1476
+ "latency": 5.18546292395331,
1477
+ "inter_token_latency_avg": 0.007782172894655725,
1478
+ "chunk_inter_token_latency_avg": 0.0235307121252608,
1479
+ "input_tokens": 8192,
1480
+ "output_tokens": 512,
1481
+ "output_tps_per_user": 128.4988156311373,
1482
+ "e2e_output_tps_per_user": 98.73756837309712,
1483
+ "completed": true
1484
+ },
1485
+ {
1486
+ "ttft": 1.2152251708321273,
1487
+ "time_to_second_token": 0.02757256105542183,
1488
+ "latency": 0.0,
1489
+ "inter_token_latency_avg": 0.007022205717217775,
1490
+ "chunk_inter_token_latency_avg": 0.021066617151653325,
1491
+ "input_tokens": 8192,
1492
+ "output_tokens": 163,
1493
+ "output_tps_per_user": 142.40539800024598,
1494
+ "e2e_output_tps_per_user": 0.0,
1495
+ "completed": false
1496
+ },
1497
+ {
1498
+ "ttft": 1.2997231129556894,
1499
+ "time_to_second_token": 0.02719552395865321,
1500
+ "latency": 4.942431465024129,
1501
+ "inter_token_latency_avg": 0.007128587773128061,
1502
+ "chunk_inter_token_latency_avg": 0.020935105471657695,
1503
+ "input_tokens": 8192,
1504
+ "output_tokens": 512,
1505
+ "output_tps_per_user": 140.2802394844042,
1506
+ "e2e_output_tps_per_user": 103.59273641389794,
1507
+ "completed": true
1508
+ },
1509
+ {
1510
+ "ttft": 0.8319369831588119,
1511
+ "time_to_second_token": 0.017664924962446094,
1512
+ "latency": 4.9657619839999825,
1513
+ "inter_token_latency_avg": 0.008089677105364327,
1514
+ "chunk_inter_token_latency_avg": 0.022106016047278985,
1515
+ "input_tokens": 8192,
1516
+ "output_tokens": 512,
1517
+ "output_tps_per_user": 123.61432810920134,
1518
+ "e2e_output_tps_per_user": 103.10602917532059,
1519
+ "completed": true
1520
+ },
1521
+ {
1522
+ "ttft": 1.2190431531053036,
1523
+ "time_to_second_token": 0.027477154973894358,
1524
+ "latency": 5.471773606957868,
1525
+ "inter_token_latency_avg": 0.008322368794232024,
1526
+ "chunk_inter_token_latency_avg": 0.023238964228702537,
1527
+ "input_tokens": 8192,
1528
+ "output_tokens": 512,
1529
+ "output_tps_per_user": 120.15809737884591,
1530
+ "e2e_output_tps_per_user": 93.57112277981393,
1531
+ "completed": true
1532
+ },
1533
+ {
1534
+ "ttft": 1.2259023920632899,
1535
+ "time_to_second_token": 0.027189655927941203,
1536
+ "latency": 5.42794113792479,
1537
+ "inter_token_latency_avg": 0.008223167800120354,
1538
+ "chunk_inter_token_latency_avg": 0.02334465969923056,
1539
+ "input_tokens": 8192,
1540
+ "output_tokens": 512,
1541
+ "output_tps_per_user": 121.6076364129848,
1542
+ "e2e_output_tps_per_user": 94.32674139052064,
1543
+ "completed": true
1544
+ },
1545
+ {
1546
+ "ttft": 1.23524184897542,
1547
+ "time_to_second_token": 0.02666433318518102,
1548
+ "latency": 0.0,
1549
+ "inter_token_latency_avg": 0.00847990960423094,
1550
+ "chunk_inter_token_latency_avg": 0.02535319912285373,
1551
+ "input_tokens": 8192,
1552
+ "output_tokens": 294,
1553
+ "output_tps_per_user": 117.9257853764223,
1554
+ "e2e_output_tps_per_user": 0.0,
1555
+ "completed": false
1556
+ }
1557
+ ],
1558
+ "total_tokens": 3670,
1559
+ "wall_time": 25.57071568304673,
1560
+ "num_completed": 2,
1561
+ "num_errors": 0,
1562
+ "server_gen_throughput": 183.44782855657667,
1563
+ "server_utilization": 0.012283640424343933,
1564
+ "server_spec_accept_rate": 0.6352201257861635,
1565
+ "server_spec_accept_length": 0.0,
1566
+ "avg_running_reqs": 1.9,
1567
+ "max_running_reqs": 2,
1568
+ "effective_concurrency": 1.9,
1569
+ "avg_queue_reqs": 0,
1570
+ "max_queue_reqs": 0,
1571
+ "queue_fraction": 0.0,
1572
+ "underfilled": true,
1573
+ "warmup_timed_out": false,
1574
+ "warmup_duration": 5.553,
1575
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1576
+ "timeout_reason": "",
1577
+ "capacity_limited": false,
1578
+ "hardware_summary": {
1579
+ "samples": 8,
1580
+ "duration_seconds": 16.958,
1581
+ "gpu_count": 4,
1582
+ "cpu_util_avg_pct": 11.59,
1583
+ "cpu_temp_max_c": 76.75,
1584
+ "gpu_util_avg_pct": 99.81,
1585
+ "gpu_util_max_pct": 100.0,
1586
+ "mem_util_avg_pct": 37.69,
1587
+ "mem_util_max_pct": 55.0,
1588
+ "temp_avg_c": 68.81,
1589
+ "temp_max_c": 84.0,
1590
+ "power_total_avg_w": 1163.89,
1591
+ "power_total_max_w": 1178.06,
1592
+ "power_limit_total_w": 1200.0,
1593
+ "vram_used_avg_mb": 384778.0,
1594
+ "vram_used_max_mb": 384778.0,
1595
+ "vram_total_mb": 391548.0,
1596
+ "vram_used_avg_pct": 98.27,
1597
+ "vram_used_max_pct": 98.27,
1598
+ "pcie_rx_avg_mb_s": 37319.5,
1599
+ "pcie_rx_max_mb_s": 67250.0,
1600
+ "pcie_tx_avg_mb_s": 33538.0,
1601
+ "pcie_tx_max_mb_s": 67525.0
1602
+ }
1603
+ },
1604
+ {
1605
+ "concurrency": 4,
1606
+ "context_tokens": 8192,
1607
+ "benchmark_mode": "duration",
1608
+ "request_count_target": 0,
1609
+ "warmup_request_count": 0,
1610
+ "measurement_seconds": 19.987172,
1611
+ "measurement_wall_seconds": 20.001291,
1612
+ "client_output_tokens": 4295,
1613
+ "server_output_tokens": 4295,
1614
+ "aggregate_source": "openai_continuous_usage",
1615
+ "aggregate_tps": 214.8878284941506,
1616
+ "per_request_avg_tps": 53.72195712353765,
1617
+ "ttft_avg": 1.7763419252975534,
1618
+ "ttft_p50": 1.3619820600142702,
1619
+ "ttft_p90": 2.3120769894914703,
1620
+ "ttft_p99": 4.039579109621701,
1621
+ "time_to_second_token_avg": 0.03882854864544546,
1622
+ "time_to_second_token_p50": 0.04027673741802573,
1623
+ "time_to_second_token_p90": 0.043103844858706,
1624
+ "time_to_second_token_p99": 0.043685059298295525,
1625
+ "request_latency_avg": 9.80318702897057,
1626
+ "request_latency_p50": 9.880283242091537,
1627
+ "request_latency_p90": 10.619067505188287,
1628
+ "request_latency_p99": 11.727601487580687,
1629
+ "inter_token_latency_avg": 0.015157077239068617,
1630
+ "inter_token_latency_p50": 0.01507125252269121,
1631
+ "inter_token_latency_p90": 0.016277488439941666,
1632
+ "inter_token_latency_p99": 0.01681107227458143,
1633
+ "output_tps_per_user_avg": 66.20683975505182,
1634
+ "output_tps_per_user_p50": 66.35252027526646,
1635
+ "output_tps_per_user_p90": 70.24470636649606,
1636
+ "output_tps_per_user_p99": 74.73678565140237,
1637
+ "e2e_output_tps_per_user_avg": 52.8389794149083,
1638
+ "e2e_output_tps_per_user_p50": 51.82037674980822,
1639
+ "e2e_output_tps_per_user_p90": 58.62424331130355,
1640
+ "e2e_output_tps_per_user_p99": 64.67365860038261,
1641
+ "chunk_inter_token_latency_avg": 0.04252216480840284,
1642
+ "chunk_inter_token_latency_p50": 0.042666952384107254,
1643
+ "chunk_inter_token_latency_p90": 0.044934111945601436,
1644
+ "chunk_inter_token_latency_p99": 0.04507588112417429,
1645
+ "input_seq_len_avg": 8192.0,
1646
+ "output_seq_len_avg": 512.0,
1647
+ "output_seq_len_p50": 512.0,
1648
+ "output_seq_len_p90": 512.0,
1649
+ "output_seq_len_p99": 512.0,
1650
+ "request_count": 12,
1651
+ "completed_request_count": 9,
1652
+ "request_samples": [
1653
+ {
1654
+ "ttft": 0.5923555630724877,
1655
+ "time_to_second_token": 0.017299477010965347,
1656
+ "latency": 7.835237701190636,
1657
+ "inter_token_latency_avg": 0.01417393764798072,
1658
+ "chunk_inter_token_latency_avg": 0.04069034909055139,
1659
+ "input_tokens": 8192,
1660
+ "output_tokens": 512,
1661
+ "output_tps_per_user": 70.55202476797012,
1662
+ "e2e_output_tps_per_user": 65.34581585472473,
1663
+ "completed": true
1664
+ },
1665
+ {
1666
+ "ttft": 1.2594965409953147,
1667
+ "time_to_second_token": 0.03946907399222255,
1668
+ "latency": 8.991313345031813,
1669
+ "inter_token_latency_avg": 0.01513075695506164,
1670
+ "chunk_inter_token_latency_avg": 0.042717219911803855,
1671
+ "input_tokens": 8192,
1672
+ "output_tokens": 512,
1673
+ "output_tps_per_user": 66.09054675651726,
1674
+ "e2e_output_tps_per_user": 56.943850175448254,
1675
+ "completed": true
1676
+ },
1677
+ {
1678
+ "ttft": 1.2591414290945977,
1679
+ "time_to_second_token": 0.04004574380815029,
1680
+ "latency": 9.880283242091537,
1681
+ "inter_token_latency_avg": 0.0168711190078218,
1682
+ "chunk_inter_token_latency_avg": 0.044901780276025725,
1683
+ "input_tokens": 8192,
1684
+ "output_tokens": 512,
1685
+ "output_tps_per_user": 59.27289111862582,
1686
+ "e2e_output_tps_per_user": 51.82037674980822,
1687
+ "completed": true
1688
+ },
1689
+ {
1690
+ "ttft": 2.212953792186454,
1691
+ "time_to_second_token": 0.04086994403041899,
1692
+ "latency": 9.785697993123904,
1693
+ "inter_token_latency_avg": 0.014819460275807142,
1694
+ "chunk_inter_token_latency_avg": 0.04160848462053544,
1695
+ "input_tokens": 8192,
1696
+ "output_tokens": 512,
1697
+ "output_tps_per_user": 67.47884075322945,
1698
+ "e2e_output_tps_per_user": 52.32125499476542,
1699
+ "completed": true
1700
+ },
1701
+ {
1702
+ "ttft": 1.4577314089983702,
1703
+ "time_to_second_token": 0.043352056061849,
1704
+ "latency": 9.128734683152288,
1705
+ "inter_token_latency_avg": 0.015011748090320779,
1706
+ "chunk_inter_token_latency_avg": 0.04261668485641065,
1707
+ "input_tokens": 8192,
1708
+ "output_tokens": 512,
1709
+ "output_tps_per_user": 66.61449379401566,
1710
+ "e2e_output_tps_per_user": 56.08663388420428,
1711
+ "completed": true
1712
+ },
1713
+ {
1714
+ "ttft": 1.2618389299605042,
1715
+ "time_to_second_token": 0.04062130395323038,
1716
+ "latency": 0.0,
1717
+ "inter_token_latency_avg": 0.015226825442038138,
1718
+ "chunk_inter_token_latency_avg": 0.04493770435333207,
1719
+ "input_tokens": 8192,
1720
+ "output_tokens": 485,
1721
+ "output_tps_per_user": 65.67357088360686,
1722
+ "e2e_output_tps_per_user": 0.0,
1723
+ "completed": false
1724
+ },
1725
+ {
1726
+ "ttft": 2.2129524589981884,
1727
+ "time_to_second_token": 0.04079655185341835,
1728
+ "latency": 10.311141398968175,
1729
+ "inter_token_latency_avg": 0.015847727866868857,
1730
+ "chunk_inter_token_latency_avg": 0.04307547308494674,
1731
+ "input_tokens": 8192,
1732
+ "output_tokens": 512,
1733
+ "output_tps_per_user": 63.100528252418606,
1734
+ "e2e_output_tps_per_user": 49.65502655712153,
1735
+ "completed": true
1736
+ },
1737
+ {
1738
+ "ttft": 2.3230906780809164,
1739
+ "time_to_second_token": 0.04372621700167656,
1740
+ "latency": 10.149147124029696,
1741
+ "inter_token_latency_avg": 0.01531517895488998,
1742
+ "chunk_inter_token_latency_avg": 0.04372098573155743,
1743
+ "input_tokens": 8192,
1744
+ "output_tokens": 512,
1745
+ "output_tps_per_user": 65.29469900060882,
1746
+ "e2e_output_tps_per_user": 50.447588722776494,
1747
+ "completed": true
1748
+ },
1749
+ {
1750
+ "ttft": 1.2644218259956688,
1751
+ "time_to_second_token": 0.03977838600985706,
1752
+ "latency": 0.0,
1753
+ "inter_token_latency_avg": 0.015003678621124169,
1754
+ "chunk_inter_token_latency_avg": 0.04158162360711556,
1755
+ "input_tokens": 8192,
1756
+ "output_tokens": 389,
1757
+ "output_tps_per_user": 66.65032124802163,
1758
+ "e2e_output_tps_per_user": 0.0,
1759
+ "completed": false
1760
+ },
1761
+ {
1762
+ "ttft": 4.251729365205392,
1763
+ "time_to_second_token": 0.03976707882247865,
1764
+ "latency": 11.850771930068731,
1765
+ "inter_token_latency_avg": 0.014870924784468375,
1766
+ "chunk_inter_token_latency_avg": 0.04175298112562274,
1767
+ "input_tokens": 8192,
1768
+ "output_tokens": 512,
1769
+ "output_tps_per_user": 67.24531355604925,
1770
+ "e2e_output_tps_per_user": 43.203936673602875,
1771
+ "completed": true
1772
+ },
1773
+ {
1774
+ "ttft": 1.9541583999525756,
1775
+ "time_to_second_token": 0.03970902017317712,
1776
+ "latency": 10.296355843078345,
1777
+ "inter_token_latency_avg": 0.016325239614727535,
1778
+ "chunk_inter_token_latency_avg": 0.04509295915203119,
1779
+ "input_tokens": 8192,
1780
+ "output_tokens": 512,
1781
+ "output_tps_per_user": 61.254843640877844,
1782
+ "e2e_output_tps_per_user": 49.726331121722886,
1783
+ "completed": true
1784
+ },
1785
+ {
1786
+ "ttft": 1.2662327110301703,
1787
+ "time_to_second_token": 0.04050773102790117,
1788
+ "latency": 0.0,
1789
+ "inter_token_latency_avg": 0.013288329607714268,
1790
+ "chunk_inter_token_latency_avg": 0.037569731890901244,
1791
+ "input_tokens": 8192,
1792
+ "output_tokens": 312,
1793
+ "output_tps_per_user": 75.25400328868051,
1794
+ "e2e_output_tps_per_user": 0.0,
1795
+ "completed": false
1796
+ }
1797
+ ],
1798
+ "total_tokens": 4295,
1799
+ "wall_time": 28.98594278888777,
1800
+ "num_completed": 4,
1801
+ "num_errors": 0,
1802
+ "server_gen_throughput": 214.68362406405674,
1803
+ "server_utilization": 0.008933556672250154,
1804
+ "server_spec_accept_rate": 0.603448275862069,
1805
+ "server_spec_accept_length": 0.0,
1806
+ "avg_running_reqs": 3.9,
1807
+ "max_running_reqs": 4,
1808
+ "effective_concurrency": 3.9,
1809
+ "avg_queue_reqs": 0.1,
1810
+ "max_queue_reqs": 1,
1811
+ "queue_fraction": 0.05,
1812
+ "underfilled": true,
1813
+ "warmup_timed_out": false,
1814
+ "warmup_duration": 8.579,
1815
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1816
+ "timeout_reason": "",
1817
+ "capacity_limited": true,
1818
+ "hardware_summary": {
1819
+ "samples": 8,
1820
+ "duration_seconds": 16.84,
1821
+ "gpu_count": 4,
1822
+ "cpu_util_avg_pct": 11.51,
1823
+ "cpu_temp_max_c": 76.62,
1824
+ "gpu_util_avg_pct": 100.0,
1825
+ "gpu_util_max_pct": 100.0,
1826
+ "mem_util_avg_pct": 31.03,
1827
+ "mem_util_max_pct": 46.0,
1828
+ "temp_avg_c": 69.19,
1829
+ "temp_max_c": 84.0,
1830
+ "power_total_avg_w": 1159.62,
1831
+ "power_total_max_w": 1173.08,
1832
+ "power_limit_total_w": 1200.0,
1833
+ "vram_used_avg_mb": 384778.0,
1834
+ "vram_used_max_mb": 384778.0,
1835
+ "vram_total_mb": 391548.0,
1836
+ "vram_used_avg_pct": 98.27,
1837
+ "vram_used_max_pct": 98.27,
1838
+ "pcie_rx_avg_mb_s": 9737.0,
1839
+ "pcie_rx_max_mb_s": 24076.0,
1840
+ "pcie_tx_avg_mb_s": 7863.75,
1841
+ "pcie_tx_max_mb_s": 8476.0
1842
+ }
1843
+ },
1844
+ {
1845
+ "concurrency": 2,
1846
+ "context_tokens": 32768,
1847
+ "benchmark_mode": "duration",
1848
+ "request_count_target": 0,
1849
+ "warmup_request_count": 0,
1850
+ "measurement_seconds": 19.995457,
1851
+ "measurement_wall_seconds": 20.000576,
1852
+ "client_output_tokens": 3727,
1853
+ "server_output_tokens": 3727,
1854
+ "aggregate_source": "openai_continuous_usage",
1855
+ "aggregate_tps": 186.39233603677485,
1856
+ "per_request_avg_tps": 93.19616801838743,
1857
+ "ttft_avg": 1.1562239681370556,
1858
+ "ttft_p50": 1.213384915026836,
1859
+ "ttft_p90": 1.3303045589243994,
1860
+ "ttft_p99": 1.4650318499957211,
1861
+ "time_to_second_token_avg": 0.01826943955384195,
1862
+ "time_to_second_token_p50": 0.020064805983565748,
1863
+ "time_to_second_token_p90": 0.021722023980692028,
1864
+ "time_to_second_token_p99": 0.021918757599778474,
1865
+ "request_latency_avg": 5.276185235736193,
1866
+ "request_latency_p50": 5.3531620495487005,
1867
+ "request_latency_p90": 5.51592188810464,
1868
+ "request_latency_p99": 5.72075475651538,
1869
+ "inter_token_latency_avg": 0.008087649016423303,
1870
+ "inter_token_latency_p50": 0.00809587391185412,
1871
+ "inter_token_latency_p90": 0.008533725732085115,
1872
+ "inter_token_latency_p99": 0.009295477393811626,
1873
+ "output_tps_per_user_avg": 124.29876816215655,
1874
+ "output_tps_per_user_p50": 123.51971941775874,
1875
+ "output_tps_per_user_p90": 131.3783276316068,
1876
+ "output_tps_per_user_p99": 142.50807229575008,
1877
+ "e2e_output_tps_per_user_avg": 97.32706467298749,
1878
+ "e2e_output_tps_per_user_p50": 95.64440520102602,
1879
+ "e2e_output_tps_per_user_p90": 105.49679616444378,
1880
+ "e2e_output_tps_per_user_p99": 106.19561216563397,
1881
+ "chunk_inter_token_latency_avg": 0.02302849329169416,
1882
+ "chunk_inter_token_latency_p50": 0.02311558809813452,
1883
+ "chunk_inter_token_latency_p90": 0.02417786236213047,
1884
+ "chunk_inter_token_latency_p99": 0.02524273630673258,
1885
+ "input_seq_len_avg": 32768.0,
1886
+ "output_seq_len_avg": 512.0,
1887
+ "output_seq_len_p50": 512.0,
1888
+ "output_seq_len_p90": 512.0,
1889
+ "output_seq_len_p99": 512.0,
1890
+ "request_count": 10,
1891
+ "completed_request_count": 8,
1892
+ "request_samples": [
1893
+ {
1894
+ "ttft": 0.6123396621551365,
1895
+ "time_to_second_token": 0.007417730987071991,
1896
+ "latency": 5.405579176964238,
1897
+ "inter_token_latency_avg": 0.009380116467336793,
1898
+ "chunk_inter_token_latency_avg": 0.02536105563391059,
1899
+ "input_tokens": 32768,
1900
+ "output_tokens": 512,
1901
+ "output_tps_per_user": 106.60848439165707,
1902
+ "e2e_output_tps_per_user": 94.71695506410806,
1903
+ "completed": true
1904
+ },
1905
+ {
1906
+ "ttft": 1.4800015490036458,
1907
+ "time_to_second_token": 0.0212986059486866,
1908
+ "latency": 5.743513964116573,
1909
+ "inter_token_latency_avg": 0.008343468522725885,
1910
+ "chunk_inter_token_latency_avg": 0.023046013054664475,
1911
+ "input_tokens": 32768,
1912
+ "output_tokens": 512,
1913
+ "output_tps_per_user": 119.85423056085207,
1914
+ "e2e_output_tps_per_user": 89.14403328672888,
1915
+ "completed": true
1916
+ },
1917
+ {
1918
+ "ttft": 1.2117794880177826,
1919
+ "time_to_second_token": 0.018087500939145684,
1920
+ "latency": 5.4183824269566685,
1921
+ "inter_token_latency_avg": 0.00823209968481191,
1922
+ "chunk_inter_token_latency_avg": 0.02311320296120267,
1923
+ "input_tokens": 32768,
1924
+ "output_tokens": 512,
1925
+ "output_tps_per_user": 121.4756912923423,
1926
+ "e2e_output_tps_per_user": 94.49314567624086,
1927
+ "completed": true
1928
+ },
1929
+ {
1930
+ "ttft": 1.2173947461415082,
1931
+ "time_to_second_token": 0.021697735879570246,
1932
+ "latency": 5.353260674979538,
1933
+ "inter_token_latency_avg": 0.008093671093616497,
1934
+ "chunk_inter_token_latency_avg": 0.023105396250491784,
1935
+ "input_tokens": 32768,
1936
+ "output_tokens": 512,
1937
+ "output_tps_per_user": 123.55332807985033,
1938
+ "e2e_output_tps_per_user": 95.64264307042306,
1939
+ "completed": true
1940
+ },
1941
+ {
1942
+ "ttft": 1.2256406019441783,
1943
+ "time_to_second_token": 0.02125983708538115,
1944
+ "latency": 0.0,
1945
+ "inter_token_latency_avg": 0.0076920541456072695,
1946
+ "chunk_inter_token_latency_avg": 0.020978329488019826,
1947
+ "input_tokens": 32768,
1948
+ "output_tokens": 181,
1949
+ "output_tps_per_user": 130.00428508047798,
1950
+ "e2e_output_tps_per_user": 0.0,
1951
+ "completed": false
1952
+ },
1953
+ {
1954
+ "ttft": 1.3136715600267053,
1955
+ "time_to_second_token": 0.019344910979270935,
1956
+ "latency": 4.868584974901751,
1957
+ "inter_token_latency_avg": 0.006956777719912027,
1958
+ "chunk_inter_token_latency_avg": 0.020911255381617914,
1959
+ "input_tokens": 32768,
1960
+ "output_tokens": 512,
1961
+ "output_tps_per_user": 143.744710591766,
1962
+ "e2e_output_tps_per_user": 105.16402664006749,
1963
+ "completed": true
1964
+ },
1965
+ {
1966
+ "ttft": 0.8629559089895338,
1967
+ "time_to_second_token": 0.013778487918898463,
1968
+ "latency": 4.81776890787296,
1969
+ "inter_token_latency_avg": 0.007739360076092811,
1970
+ "chunk_inter_token_latency_avg": 0.02340126034842264,
1971
+ "input_tokens": 32768,
1972
+ "output_tokens": 512,
1973
+ "output_tps_per_user": 129.2096491399902,
1974
+ "e2e_output_tps_per_user": 106.27325838798845,
1975
+ "completed": true
1976
+ },
1977
+ {
1978
+ "ttft": 1.2149462150409818,
1979
+ "time_to_second_token": 0.017084267921745777,
1980
+ "latency": 5.353063424117863,
1981
+ "inter_token_latency_avg": 0.008098076730091745,
1982
+ "chunk_inter_token_latency_avg": 0.023117973235066376,
1983
+ "input_tokens": 32768,
1984
+ "output_tokens": 512,
1985
+ "output_tps_per_user": 123.48611075566714,
1986
+ "e2e_output_tps_per_user": 95.64616733162899,
1987
+ "completed": true
1988
+ },
1989
+ {
1990
+ "ttft": 1.2118236150126904,
1991
+ "time_to_second_token": 0.02194061689078808,
1992
+ "latency": 5.249328335979953,
1993
+ "inter_token_latency_avg": 0.007901183406980945,
1994
+ "chunk_inter_token_latency_avg": 0.023204050120501512,
1995
+ "input_tokens": 32768,
1996
+ "output_tokens": 512,
1997
+ "output_tps_per_user": 126.56331950432494,
1998
+ "e2e_output_tps_per_user": 97.53628792671415,
1999
+ "completed": true
2000
+ },
2001
+ {
2002
+ "ttft": 1.2116863350383937,
2003
+ "time_to_second_token": 0.02078470098786056,
2004
+ "latency": 0.0,
2005
+ "inter_token_latency_avg": 0.008439682317057152,
2006
+ "chunk_inter_token_latency_avg": 0.02404639644304379,
2007
+ "input_tokens": 32768,
2008
+ "output_tokens": 360,
2009
+ "output_tps_per_user": 118.48787222463746,
2010
+ "e2e_output_tps_per_user": 0.0,
2011
+ "completed": false
2012
+ }
2013
+ ],
2014
+ "total_tokens": 3727,
2015
+ "wall_time": 25.575839941157028,
2016
+ "num_completed": 2,
2017
+ "num_errors": 0,
2018
+ "server_gen_throughput": 186.28349244400528,
2019
+ "server_utilization": 0.013121161362367406,
2020
+ "server_spec_accept_rate": 0.55,
2021
+ "server_spec_accept_length": 0.0,
2022
+ "avg_running_reqs": 1.9,
2023
+ "max_running_reqs": 2,
2024
+ "effective_concurrency": 1.9,
2025
+ "avg_queue_reqs": 0,
2026
+ "max_queue_reqs": 0,
2027
+ "queue_fraction": 0.0,
2028
+ "underfilled": true,
2029
+ "warmup_timed_out": false,
2030
+ "warmup_duration": 5.559,
2031
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
2032
+ "timeout_reason": "",
2033
+ "capacity_limited": false,
2034
+ "hardware_summary": {
2035
+ "samples": 9,
2036
+ "duration_seconds": 19.296,
2037
+ "gpu_count": 4,
2038
+ "cpu_util_avg_pct": 11.52,
2039
+ "cpu_temp_max_c": 76.75,
2040
+ "gpu_util_avg_pct": 99.89,
2041
+ "gpu_util_max_pct": 100.0,
2042
+ "mem_util_avg_pct": 38.5,
2043
+ "mem_util_max_pct": 55.0,
2044
+ "temp_avg_c": 68.69,
2045
+ "temp_max_c": 84.0,
2046
+ "power_total_avg_w": 1159.62,
2047
+ "power_total_max_w": 1178.78,
2048
+ "power_limit_total_w": 1200.0,
2049
+ "vram_used_avg_mb": 384778.0,
2050
+ "vram_used_max_mb": 384778.0,
2051
+ "vram_total_mb": 391548.0,
2052
+ "vram_used_avg_pct": 98.27,
2053
+ "vram_used_max_pct": 98.27,
2054
+ "pcie_rx_avg_mb_s": 18465.67,
2055
+ "pcie_rx_max_mb_s": 56770.0,
2056
+ "pcie_tx_avg_mb_s": 18393.11,
2057
+ "pcie_tx_max_mb_s": 51076.0
2058
+ }
2059
+ },
2060
+ {
2061
+ "concurrency": 4,
2062
+ "context_tokens": 32768,
2063
+ "benchmark_mode": "duration",
2064
+ "request_count_target": 0,
2065
+ "warmup_request_count": 0,
2066
+ "measurement_seconds": 19.996813,
2067
+ "measurement_wall_seconds": 20.000936,
2068
+ "client_output_tokens": 4259,
2069
+ "server_output_tokens": 4259,
2070
+ "aggregate_source": "openai_continuous_usage",
2071
+ "aggregate_tps": 212.98394100240475,
2072
+ "per_request_avg_tps": 53.24598525060119,
2073
+ "ttft_avg": 1.6594751148368232,
2074
+ "ttft_p50": 1.2507346520433202,
2075
+ "ttft_p90": 2.214244986465201,
2076
+ "ttft_p99": 3.976403836400715,
2077
+ "time_to_second_token_avg": 0.03094565647188574,
2078
+ "time_to_second_token_p50": 0.03346484643407166,
2079
+ "time_to_second_token_p90": 0.03496504717040807,
2080
+ "time_to_second_token_p99": 0.035161831779405475,
2081
+ "request_latency_avg": 9.732150099524814,
2082
+ "request_latency_p50": 9.580612420104444,
2083
+ "request_latency_p90": 11.209756568027661,
2084
+ "request_latency_p99": 12.458838488040492,
2085
+ "inter_token_latency_avg": 0.015378300425214705,
2086
+ "inter_token_latency_p50": 0.01540227101955366,
2087
+ "inter_token_latency_p90": 0.016775147308305694,
2088
+ "inter_token_latency_p99": 0.016912279111026093,
2089
+ "output_tps_per_user_avg": 65.35975868873867,
2090
+ "output_tps_per_user_p50": 64.9280008387403,
2091
+ "output_tps_per_user_p90": 70.36340913782152,
2092
+ "output_tps_per_user_p99": 74.69250108569062,
2093
+ "e2e_output_tps_per_user_avg": 53.42647790415391,
2094
+ "e2e_output_tps_per_user_p50": 53.44126007285225,
2095
+ "e2e_output_tps_per_user_p90": 58.48582251783244,
2096
+ "e2e_output_tps_per_user_p99": 64.43718529934995,
2097
+ "chunk_inter_token_latency_avg": 0.043139480462465546,
2098
+ "chunk_inter_token_latency_p50": 0.04417947091825035,
2099
+ "chunk_inter_token_latency_p90": 0.04521715045374133,
2100
+ "chunk_inter_token_latency_p99": 0.04545031231990535,
2101
+ "input_seq_len_avg": 32768.0,
2102
+ "output_seq_len_avg": 512.0,
2103
+ "output_seq_len_p50": 512.0,
2104
+ "output_seq_len_p90": 512.0,
2105
+ "output_seq_len_p99": 512.0,
2106
+ "request_count": 12,
2107
+ "completed_request_count": 9,
2108
+ "request_samples": [
2109
+ {
2110
+ "ttft": 0.6141200510319322,
2111
+ "time_to_second_token": 0.011225768830627203,
2112
+ "latency": 7.865010872948915,
2113
+ "inter_token_latency_avg": 0.014189610219015622,
2114
+ "chunk_inter_token_latency_avg": 0.04028272678842768,
2115
+ "input_tokens": 32768,
2116
+ "output_tokens": 512,
2117
+ "output_tps_per_user": 70.474099327964,
2118
+ "e2e_output_tps_per_user": 65.09844783062967,
2119
+ "completed": true
2120
+ },
2121
+ {
2122
+ "ttft": 1.275223245844245,
2123
+ "time_to_second_token": 0.030873473035171628,
2124
+ "latency": 9.194723297841847,
2125
+ "inter_token_latency_avg": 0.015498043154594134,
2126
+ "chunk_inter_token_latency_avg": 0.04525428601141487,
2127
+ "input_tokens": 32768,
2128
+ "output_tokens": 512,
2129
+ "output_tps_per_user": 64.52427509879315,
2130
+ "e2e_output_tps_per_user": 55.68411179052825,
2131
+ "completed": true
2132
+ },
2133
+ {
2134
+ "ttft": 1.2505115061067045,
2135
+ "time_to_second_token": 0.035012715961784124,
2136
+ "latency": 9.598736566957086,
2137
+ "inter_token_latency_avg": 0.016337035344129905,
2138
+ "chunk_inter_token_latency_avg": 0.044882930434679474,
2139
+ "input_tokens": 32768,
2140
+ "output_tokens": 512,
2141
+ "output_tps_per_user": 61.21061618192019,
2142
+ "e2e_output_tps_per_user": 53.3403533296789,
2143
+ "completed": true
2144
+ },
2145
+ {
2146
+ "ttft": 2.2142701919656247,
2147
+ "time_to_second_token": 0.02940852497704327,
2148
+ "latency": 10.862789368024096,
2149
+ "inter_token_latency_avg": 0.016924695060779787,
2150
+ "chunk_inter_token_latency_avg": 0.044351380390043445,
2151
+ "input_tokens": 32768,
2152
+ "output_tokens": 512,
2153
+ "output_tps_per_user": 59.08525952218403,
2154
+ "e2e_output_tps_per_user": 47.13338192003727,
2155
+ "completed": true
2156
+ },
2157
+ {
2158
+ "ttft": 1.250957797979936,
2159
+ "time_to_second_token": 0.03281243494711816,
2160
+ "latency": 9.841799243818969,
2161
+ "inter_token_latency_avg": 0.01681182279029165,
2162
+ "chunk_inter_token_latency_avg": 0.044282687865149654,
2163
+ "input_tokens": 32768,
2164
+ "output_tokens": 512,
2165
+ "output_tps_per_user": 59.4819498440985,
2166
+ "e2e_output_tps_per_user": 52.023007919162325,
2167
+ "completed": true
2168
+ },
2169
+ {
2170
+ "ttft": 1.2194926510564983,
2171
+ "time_to_second_token": 0.03447063802741468,
2172
+ "latency": 0.0,
2173
+ "inter_token_latency_avg": 0.014444936651048233,
2174
+ "chunk_inter_token_latency_avg": 0.041717839432504976,
2175
+ "input_tokens": 32768,
2176
+ "output_tokens": 388,
2177
+ "output_tps_per_user": 69.22841021441465,
2178
+ "e2e_output_tps_per_user": 0.0,
2179
+ "completed": false
2180
+ },
2181
+ {
2182
+ "ttft": 2.2140181369613856,
2183
+ "time_to_second_token": 0.02934236405417323,
2184
+ "latency": 9.580612420104444,
2185
+ "inter_token_latency_avg": 0.014416035779144928,
2186
+ "chunk_inter_token_latency_avg": 0.04138536114125314,
2187
+ "input_tokens": 32768,
2188
+ "output_tokens": 512,
2189
+ "output_tps_per_user": 69.36719742653926,
2190
+ "e2e_output_tps_per_user": 53.44126007285225,
2191
+ "completed": true
2192
+ },
2193
+ {
2194
+ "ttft": 1.2175294221378863,
2195
+ "time_to_second_token": 0.029971704818308353,
2196
+ "latency": 9.039150352124125,
2197
+ "inter_token_latency_avg": 0.015306498884513187,
2198
+ "chunk_inter_token_latency_avg": 0.04547454029061766,
2199
+ "input_tokens": 32768,
2200
+ "output_tokens": 512,
2201
+ "output_tps_per_user": 65.33172657868745,
2202
+ "e2e_output_tps_per_user": 56.64249183328213,
2203
+ "completed": true
2204
+ },
2205
+ {
2206
+ "ttft": 1.2166177490726113,
2207
+ "time_to_second_token": 0.03411725792102516,
2208
+ "latency": 0.0,
2209
+ "inter_token_latency_avg": 0.015614057580944196,
2210
+ "chunk_inter_token_latency_avg": 0.044076253971351044,
2211
+ "input_tokens": 32768,
2212
+ "output_tokens": 495,
2213
+ "output_tps_per_user": 64.04485155866378,
2214
+ "e2e_output_tps_per_user": 0.0,
2215
+ "completed": false
2216
+ },
2217
+ {
2218
+ "ttft": 4.194195635151118,
2219
+ "time_to_second_token": 0.035180261824280024,
2220
+ "latency": 12.597625368041918,
2221
+ "inter_token_latency_avg": 0.016445067970432093,
2222
+ "chunk_inter_token_latency_avg": 0.0444625911793164,
2223
+ "input_tokens": 32768,
2224
+ "output_tokens": 512,
2225
+ "output_tps_per_user": 60.80850512737194,
2226
+ "e2e_output_tps_per_user": 40.64258025158129,
2227
+ "completed": true
2228
+ },
2229
+ {
2230
+ "ttft": 1.2128918368835002,
2231
+ "time_to_second_token": 0.03453602804802358,
2232
+ "latency": 9.008903405861929,
2233
+ "inter_token_latency_avg": 0.015256382718157395,
2234
+ "chunk_inter_token_latency_avg": 0.04355313725686273,
2235
+ "input_tokens": 32768,
2236
+ "output_tokens": 512,
2237
+ "output_tps_per_user": 65.54633680039039,
2238
+ "e2e_output_tps_per_user": 56.83266618963313,
2239
+ "completed": true
2240
+ },
2241
+ {
2242
+ "ttft": 2.033873153850436,
2243
+ "time_to_second_token": 0.03439670521765947,
2244
+ "latency": 0.0,
2245
+ "inter_token_latency_avg": 0.013295418949525321,
2246
+ "chunk_inter_token_latency_avg": 0.03795003078796548,
2247
+ "input_tokens": 32768,
2248
+ "output_tokens": 295,
2249
+ "output_tps_per_user": 75.21387658383661,
2250
+ "e2e_output_tps_per_user": 0.0,
2251
+ "completed": false
2252
+ }
2253
+ ],
2254
+ "total_tokens": 4259,
2255
+ "wall_time": 28.930821696063504,
2256
+ "num_completed": 4,
2257
+ "num_errors": 0,
2258
+ "server_gen_throughput": 212.86426983114268,
2259
+ "server_utilization": 0.02819653824678947,
2260
+ "server_spec_accept_rate": 0.5884057971014492,
2261
+ "server_spec_accept_length": 0.0,
2262
+ "avg_running_reqs": 3.8,
2263
+ "max_running_reqs": 4,
2264
+ "effective_concurrency": 3.8,
2265
+ "avg_queue_reqs": 0,
2266
+ "max_queue_reqs": 0,
2267
+ "queue_fraction": 0.0,
2268
+ "underfilled": true,
2269
+ "warmup_timed_out": false,
2270
+ "warmup_duration": 8.576,
2271
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
2272
+ "timeout_reason": "",
2273
+ "capacity_limited": false,
2274
+ "hardware_summary": {
2275
+ "samples": 9,
2276
+ "duration_seconds": 19.299,
2277
+ "gpu_count": 4,
2278
+ "cpu_util_avg_pct": 11.51,
2279
+ "cpu_temp_max_c": 76.75,
2280
+ "gpu_util_avg_pct": 100.0,
2281
+ "gpu_util_max_pct": 100.0,
2282
+ "mem_util_avg_pct": 32.06,
2283
+ "mem_util_max_pct": 45.0,
2284
+ "temp_avg_c": 69.14,
2285
+ "temp_max_c": 84.0,
2286
+ "power_total_avg_w": 1161.68,
2287
+ "power_total_max_w": 1174.21,
2288
+ "power_limit_total_w": 1200.0,
2289
+ "vram_used_avg_mb": 384778.0,
2290
+ "vram_used_max_mb": 384778.0,
2291
+ "vram_total_mb": 391548.0,
2292
+ "vram_used_avg_pct": 98.27,
2293
+ "vram_used_max_pct": 98.27,
2294
+ "pcie_rx_avg_mb_s": 35105.33,
2295
+ "pcie_rx_max_mb_s": 74341.0,
2296
+ "pcie_tx_avg_mb_s": 34509.67,
2297
+ "pcie_tx_max_mb_s": 70985.0
2298
+ }
2299
+ }
2300
+ ],
2301
+ "summary_table": {
2302
+ "0": {
2303
+ "1": 191.1473049181911,
2304
+ "2": 257.76658819245483,
2305
+ "4": 305.62106309879454
2306
+ },
2307
+ "8192": {
2308
+ "1": 165.15431387835127,
2309
+ "2": 183.53550127953065,
2310
+ "4": 214.8878284941506
2311
+ },
2312
+ "32768": {
2313
+ "1": 162.35299164317425,
2314
+ "2": 186.39233603677485,
2315
+ "4": 212.98394100240475
2316
+ }
2317
+ },
2318
+ "burst_results": [],
2319
+ "burst_summary_table": {},
2320
+ "methodology": {
2321
+ "prefill": {
2322
+ "name": "Prefill",
2323
+ "present": false,
2324
+ "mode": "skipped",
2325
+ "formula": "prompt_tokens / TTFT",
2326
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
2327
+ },
2328
+ "sustained_decode": {
2329
+ "name": "Sustained Decode",
2330
+ "present": true,
2331
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
2332
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
2333
+ },
2334
+ "burst_e2e_decode": {
2335
+ "name": "Burst / E2E Decode",
2336
+ "present": false,
2337
+ "status": "not run; use --run-burst",
2338
+ "formula": "sum(completion_tokens) / profiling_wall_time",
2339
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
2340
+ }
2341
+ }
2342
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.log ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0', '8k', '32k'] │
16
+ │ Duration: 20.0s per decode test | Max tokens: 512 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill: skipped | Sustained decode: 9 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
23
+ 7,337,984 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: skipped
26
+ Calibrating padding text (run=mkkehtszimwn, up to 32k)...
27
+ 8k: 50,558 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,152 chars (32,768 prompt tokens via /tokenize)
29
+ Token targeting: /tokenize exact
30
+ Done.
31
+
32
+
33
+
34
+ llm-decode-bench v0.4.29
35
+ ╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
36
+ │ Sustained Decode │
37
+ │ Steady-state decode throughput after the engine has admitted the requested │
38
+ │ concurrency and passed warmup. Use this as the main tuning/regression signal │
39
+ │ for kernels, NCCL, DCP, MTP, and scheduler changes. │
40
+ ╰──────────────────────────────────────────────────────────────────────────────╯
41
+ Aggregate tok/s + TTFT/ITL
42
+ ╭────────────┬─────────────┬──────────────────┬───────────────────╮
43
+ │ ctx \ conc │ 1 │ 2 │ 4 │
44
+ ├────────────┼─────────────┼──────────────────┼───────────────────┤
45
+ │ 0 │ 191.1 80/5 │ 257.8 123/8 │ ∅ (4/4)* 182/13 │
46
+ │ 8k │ 165.2 629/5 │ 183.5 (2/2) 1k/8 │ ∅ (4/4)* 1k/15 │
47
+ │ 32k │ 162.4 630/5 │ 186.4 (2/2) 1k/8 │ 213.0 (4/4) 1k/15 │
48
+ ╰────────────┴─────────────┴──────────────────┴───────────────────╯
49
+ Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
50
+ (continuous completion_tokens when the server supports it). Prometheus is kept
51
+ as validation/scheduler data.
52
+ Aggregate source(s): openai_continuous_usage
53
+ ∅ = skipped/hidden because the cell does not fit in KV cache; exact deficit is
54
+ kept in JSON timeout_reason
55
+ (X/Y) = avg running / requested concurrency from Prometheus; * =
56
+ capacity-limited or warmup timed out
57
+ Per-Request tok/s
58
+ ╭────────────┬───────┬────────────┬────────────╮
59
+ │ ctx \ conc │ 1 │ 2 │ 4 │
60
+ ├────────────┼───────┼──────���─────┼────────────┤
61
+ │ 0 │ 191.1 │ 128.9 │ ∅ (4/4)* │
62
+ │ 8k │ 165.2 │ 91.8 (2/2) │ ∅ (4/4)* │
63
+ │ 32k │ 162.4 │ 93.2 (2/2) │ 53.2 (4/4) │
64
+ ╰────────────┴───────┴────────────┴────────────╯
65
+ Client request latency: p50 / p90 ms
66
+ ╭────────────┬───────────┬───────────┬────────────╮
67
+ │ ctx \ conc │ 1 │ 2 │ 4 │
68
+ ├────────────┼───────────┼───────────┼────────────┤
69
+ │ 0 │ 2.6k/2.8k │ 4k/4.3k │ 6.8k/7.1k │
70
+ │ 8k │ 3.2k/3.2k │ 5.4k/5.7k │ 9.9k/10.6k │
71
+ │ 32k │ 3.2k/3.3k │ 5.4k/5.5k │ 9.6k/11.2k │
72
+ ╰────────────┴───────────┴───────────┴────────────╯
73
+ Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
74
+ coordinate. ITL is computed from observed generated tokens, including streams
75
+ stopped at the measurement boundary; a missing ITL means no stream produced at
76
+ least two measured output tokens. Per-request tok/s and request latency are
77
+ shown in separate per-cell matrices. Completion/sample counts and full
78
+ request-level distributions remain in JSON under request_samples.
79
+ Sustained mode: client latency metrics explain request UX variance; aggregate
80
+ tok/s remains the primary throughput signal.
81
+ ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
82
+ Hardware Summary
83
+ ╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
84
+ │ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
85
+ ├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
86
+ │ 0 │ │ sust… │ 99/100% │ 43% │ 1153/1… │ 83C │ 76C │ 98… │ 8389/8246 │
87
+ │ … │ │ sust… │ 99/100% │ 40% │ 1154/1… │ 84C │ 76C │ 98… │ 11118/9898 │
88
+ │ … │ │ sust… │ 96/100% │ 39% │ 1152/1… │ 84C │ 77C │ 98… │ 12769/12689 │
89
+ │ 0 │ │ sust… │ 100/100% │ 41% │ 1175/1… │ 84C │ 76C │ 98… │ 11425/11473 │
90
+ │ 0 │ │ sust… │ 100/100% │ 36% │ 1172/1… │ 84C │ 77C │ 98… │ 8124/8457 │
91
+ │ … │ │ sust… │ 100/100% │ 38% │ 1164/1… │ 84C │ 77C │ 98… │ 37320/33538 │
92
+ │ … │ │ sust… │ 100/100% │ 31% │ 1160/1… │ 84C │ 77C │ 98… │ 9737/7864 │
93
+ │ … │ │ sust… │ 100/100% │ 38% │ 1160/1… │ 84C │ 77C │ 98… │ 18466/18393 │
94
+ │ … │ │ sust… │ 100/100% │ 32% │ 1162/1… │ 84C │ 77C │ 98… │ 35105/34510 │
95
+ ╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
96
+ ╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
97
+ │ avg 1,097 W | max 1,179 W | limit 1,200 W | over 4m 29s | 113 samples │
98
+ ╰───────────────────────────────────────────────────────────────────────╯
99
+ Hardware summary is sampled from nvidia-smi during the measured part of each
100
+ cell. Whole-run GPU power is the sampled sum of GPU power draw across the
101
+ complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
102
+ a coarse live diagnostic, not a per-kernel NCCL profiler.
103
+
104
+ ╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
105
+ │ Burst / E2E Decode │
106
+ │ Not run. Re-run with --run-burst to append a finite client-facing request │
107
+ │ burst after Sustained Decode. This is intentionally disabled by default │
108
+ │ because it adds another full decode matrix. │
109
+ ╰───────���──────────────────────────────────────────────────────────────────────╯
110
+
111
+ ╭────────────────────────────── Primary Summary ───────────────────────────────╮
112
+ │ Primary matrices repeated last so the important numbers are visible without │
113
+ │ scrolling back through diagnostics. │
114
+ ╰──────────────────────────────────────────────────────────────────────────────╯
115
+ Aggregate decode tok/s
116
+ ╭────────────┬───────┬─────────────┬─────────────╮
117
+ │ ctx \ conc │ 1 │ 2 │ 4 │
118
+ ├────────────┼───────┼─────────────┼─────────────┤
119
+ │ 0 │ 191.1 │ 257.8 │ ∅ (4/4)* │
120
+ │ 8k │ 165.2 │ 183.5 (2/2) │ ∅ (4/4)* │
121
+ │ 32k │ 162.4 │ 186.4 (2/2) │ 213.0 (4/4) │
122
+ ╰────────────┴───────┴─────────────┴─────────────╯
123
+
124
+ Results saved to
125
+ <campaign>/candidate-speed-wi
126
+ ndow-01/results-01/decode-warp-quant/rep-2/decode-cap512.json
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-command.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json",
20
+ "--contexts",
21
+ "0,8k,32k",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--skip-prefill",
25
+ "--cell-warmup-timeout-seconds",
26
+ "180"
27
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "c97b1e6c81137e30eeaae4e137a8605c633dbe6eaa2b5037e53dc6f63dbeae19"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json ADDED
@@ -0,0 +1,1394 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:33:25.169970",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "skipped",
15
+ "standalone_prefill": false,
16
+ "prefill_only": false,
17
+ "skip_prefill": true,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 32768,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0,
42
+ 8192,
43
+ 32768
44
+ ],
45
+ "startup_diagnostics_available": true,
46
+ "nvidia_p2p_override_effective": true,
47
+ "p2pmark_status": "not_run",
48
+ "amd_fabric_status": "not_run"
49
+ },
50
+ "startup_diagnostics": {
51
+ "version": "0.4.29",
52
+ "server_url": "http://127.0.0.1:8001",
53
+ "hostname": "<host>",
54
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
55
+ "env": {},
56
+ "args": {
57
+ "concurrency": "1,2,4",
58
+ "contexts": "0,8k,32k",
59
+ "max_tokens": 8192,
60
+ "duration": 20.0,
61
+ "request_count": 0,
62
+ "run_burst": false,
63
+ "standalone_prefill": false,
64
+ "prefill_only": false,
65
+ "skip_prefill": true,
66
+ "prefill_contexts": "8k,64k,128k",
67
+ "prefill_metric": "client",
68
+ "dcp_size": 0,
69
+ "kv_budget": 0
70
+ },
71
+ "nvidia_p2p_override": {
72
+ "effective": true,
73
+ "configured": true,
74
+ "params_path": "/proc/driver/nvidia/params",
75
+ "params_available": true,
76
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
77
+ "modprobe_available": true,
78
+ "runtime": {
79
+ "ForceP2P": "0x11",
80
+ "RMForceP2PType": "1",
81
+ "RMPcieP2PType": "2",
82
+ "GrdmaPciTopoCheckOverride": "1",
83
+ "EnableResizableBar": "1",
84
+ "DmaRemapPeerMmio": "1"
85
+ },
86
+ "expected": {
87
+ "ForceP2P": "0x11",
88
+ "RMForceP2PType": "1",
89
+ "RMPcieP2PType": "2",
90
+ "GrdmaPciTopoCheckOverride": "1",
91
+ "EnableResizableBar": "1"
92
+ },
93
+ "missing": [],
94
+ "mismatched": {},
95
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
96
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
97
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
98
+ },
99
+ "p2pmark": {
100
+ "status": "not_run"
101
+ },
102
+ "amd_fabric": {
103
+ "status": "not_run"
104
+ },
105
+ "nvidia_smi_query": {
106
+ "cmd": [
107
+ "nvidia-smi",
108
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
109
+ "--format=csv,noheader,nounits"
110
+ ],
111
+ "returncode": 0,
112
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
113
+ "stderr": ""
114
+ },
115
+ "nvidia_smi_topo": {
116
+ "cmd": [
117
+ "nvidia-smi",
118
+ "topo",
119
+ "-m"
120
+ ],
121
+ "returncode": 0,
122
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
123
+ "stderr": ""
124
+ }
125
+ },
126
+ "nvidia_p2p_override": {
127
+ "effective": true,
128
+ "configured": true,
129
+ "params_path": "/proc/driver/nvidia/params",
130
+ "params_available": true,
131
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
132
+ "modprobe_available": true,
133
+ "runtime": {
134
+ "ForceP2P": "0x11",
135
+ "RMForceP2PType": "1",
136
+ "RMPcieP2PType": "2",
137
+ "GrdmaPciTopoCheckOverride": "1",
138
+ "EnableResizableBar": "1",
139
+ "DmaRemapPeerMmio": "1"
140
+ },
141
+ "expected": {
142
+ "ForceP2P": "0x11",
143
+ "RMForceP2PType": "1",
144
+ "RMPcieP2PType": "2",
145
+ "GrdmaPciTopoCheckOverride": "1",
146
+ "EnableResizableBar": "1"
147
+ },
148
+ "missing": [],
149
+ "mismatched": {},
150
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
151
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
152
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
153
+ },
154
+ "p2pmark": {
155
+ "status": "not_run"
156
+ },
157
+ "amd_fabric": {
158
+ "status": "not_run"
159
+ },
160
+ "hardware_run_summary": {
161
+ "samples": 113,
162
+ "duration_seconds": 269.983,
163
+ "gpu_count": 4,
164
+ "cpu_util_avg_pct": 11.42,
165
+ "cpu_temp_max_c": 77.38,
166
+ "gpu_util_avg_pct": 89.04,
167
+ "gpu_util_max_pct": 100.0,
168
+ "mem_util_avg_pct": 34.99,
169
+ "mem_util_max_pct": 56.0,
170
+ "temp_avg_c": 67.89,
171
+ "temp_max_c": 85.0,
172
+ "power_total_avg_w": 1105.19,
173
+ "power_total_max_w": 1178.45,
174
+ "power_limit_total_w": 1200.0,
175
+ "vram_used_avg_mb": 384778.0,
176
+ "vram_used_max_mb": 384778.0,
177
+ "vram_total_mb": 391548.0,
178
+ "vram_used_avg_pct": 98.27,
179
+ "vram_used_max_pct": 98.27,
180
+ "pcie_rx_avg_mb_s": 12268.5,
181
+ "pcie_rx_max_mb_s": 68298.0,
182
+ "pcie_tx_avg_mb_s": 12130.98,
183
+ "pcie_tx_max_mb_s": 63483.0
184
+ },
185
+ "event_log": [
186
+ "02:28:53 benchmark start engine=vllm",
187
+ "02:28:53 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
188
+ "02:28:53 startup decode concurrency=1,2,4 contexts=0,8k,32k",
189
+ "02:28:53 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
190
+ "02:28:53 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
191
+ "02:28:53 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
192
+ "02:28:53 startup model context length: 1,000,000 tokens",
193
+ "02:28:53 startup prefill tests: skipped",
194
+ "02:28:53 startup calibrating padding text run=lijekedoguqy up_to=32k",
195
+ "02:28:53 startup context 8k: 50,540 chars (8,192 prompt tokens via /tokenize)",
196
+ "02:28:53 startup context 32k: 205,139 chars (32,768 prompt tokens via /tokenize)",
197
+ "02:28:53 startup token targeting: /tokenize exact",
198
+ "02:28:53 startup startup preparation done",
199
+ "02:28:53 hardware monitor interval=2s",
200
+ "02:28:53 decode warmup start",
201
+ "02:28:53 decode warmup start C=1 ctx=32k 3s",
202
+ "02:28:53 cell start C=1 ctx=32k",
203
+ "02:29:01 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
204
+ "02:29:04 cell done C=1 ctx=32k 184.6 tok/s",
205
+ "02:29:04 decode warmup done C=1 ctx=32k",
206
+ "02:29:06 cell start C=1 ctx=0",
207
+ "02:29:12 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
208
+ "02:29:32 cell done C=1 ctx=0 175.2 tok/s",
209
+ "02:29:34 cell start C=1 ctx=8k",
210
+ "02:29:40 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
211
+ "02:30:00 cell done C=1 ctx=8k 174.2 tok/s",
212
+ "02:30:02 cell start C=1 ctx=32k",
213
+ "02:30:11 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
214
+ "02:30:31 cell done C=1 ctx=32k 179.3 tok/s",
215
+ "02:30:33 cell start C=2 ctx=0",
216
+ "02:30:39 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
217
+ "02:30:59 cell done C=2 ctx=0 236.4 tok/s",
218
+ "02:31:01 cell start C=4 ctx=0",
219
+ "02:31:06 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
220
+ "02:31:26 cell done C=4 ctx=0 296.7 tok/s",
221
+ "02:31:28 cell start C=2 ctx=8k",
222
+ "02:31:34 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
223
+ "02:31:54 cell done C=2 ctx=8k 246.8 tok/s",
224
+ "02:31:56 cell start C=4 ctx=8k",
225
+ "02:32:04 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
226
+ "02:32:24 cell done C=4 ctx=8k 291.4 tok/s",
227
+ "02:32:26 cell start C=2 ctx=32k",
228
+ "02:32:32 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
229
+ "02:32:52 cell done C=2 ctx=32k 239.9 tok/s",
230
+ "02:32:54 cell start C=4 ctx=32k",
231
+ "02:33:03 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
232
+ "02:33:23 cell done C=4 ctx=32k 302.2 tok/s"
233
+ ],
234
+ "prefill": {},
235
+ "results": [
236
+ {
237
+ "concurrency": 1,
238
+ "context_tokens": 0,
239
+ "benchmark_mode": "duration",
240
+ "request_count_target": 0,
241
+ "warmup_request_count": 0,
242
+ "measurement_seconds": 19.996529,
243
+ "measurement_wall_seconds": 20.000654,
244
+ "client_output_tokens": 3503,
245
+ "server_output_tokens": 3503,
246
+ "aggregate_source": "openai_continuous_usage",
247
+ "aggregate_tps": 175.1804006140207,
248
+ "per_request_avg_tps": 175.1804006140207,
249
+ "ttft_avg": 0.06969588715583086,
250
+ "ttft_p50": 0.06969588715583086,
251
+ "ttft_p90": 0.06969588715583086,
252
+ "ttft_p99": 0.06969588715583086,
253
+ "time_to_second_token_avg": 0.013065006816759706,
254
+ "time_to_second_token_p50": 0.013065006816759706,
255
+ "time_to_second_token_p90": 0.013065006816759706,
256
+ "time_to_second_token_p99": 0.013065006816759706,
257
+ "request_latency_avg": 0.0,
258
+ "request_latency_p50": 0.0,
259
+ "request_latency_p90": 0.0,
260
+ "request_latency_p99": 0.0,
261
+ "inter_token_latency_avg": 0.005635859511267837,
262
+ "inter_token_latency_p50": 0.005635859511267837,
263
+ "inter_token_latency_p90": 0.005635859511267837,
264
+ "inter_token_latency_p99": 0.005635859511267837,
265
+ "output_tps_per_user_avg": 177.43522492011888,
266
+ "output_tps_per_user_p50": 177.43522492011888,
267
+ "output_tps_per_user_p90": 177.43522492011888,
268
+ "output_tps_per_user_p99": 177.43522492011888,
269
+ "e2e_output_tps_per_user_avg": 0.0,
270
+ "e2e_output_tps_per_user_p50": 0.0,
271
+ "e2e_output_tps_per_user_p90": 0.0,
272
+ "e2e_output_tps_per_user_p99": 0.0,
273
+ "chunk_inter_token_latency_avg": 0.01424891621259546,
274
+ "chunk_inter_token_latency_p50": 0.01424891621259546,
275
+ "chunk_inter_token_latency_p90": 0.01424891621259546,
276
+ "chunk_inter_token_latency_p99": 0.01424891621259546,
277
+ "input_seq_len_avg": 78.0,
278
+ "output_seq_len_avg": 4519.0,
279
+ "output_seq_len_p50": 4519.0,
280
+ "output_seq_len_p90": 4519.0,
281
+ "output_seq_len_p99": 4519.0,
282
+ "request_count": 1,
283
+ "completed_request_count": 0,
284
+ "request_samples": [
285
+ {
286
+ "ttft": 0.06969588715583086,
287
+ "time_to_second_token": 0.013065006816759706,
288
+ "latency": 0.0,
289
+ "inter_token_latency_avg": 0.005635859511267837,
290
+ "chunk_inter_token_latency_avg": 0.01424891621259546,
291
+ "input_tokens": 78,
292
+ "output_tokens": 4519,
293
+ "output_tps_per_user": 177.43522492011888,
294
+ "e2e_output_tps_per_user": 0.0,
295
+ "completed": false
296
+ }
297
+ ],
298
+ "total_tokens": 3503,
299
+ "wall_time": 25.547412615967914,
300
+ "num_completed": 1,
301
+ "num_errors": 0,
302
+ "server_gen_throughput": 175.1028092847603,
303
+ "server_utilization": 0.005862646566164198,
304
+ "server_spec_accept_rate": 0.4835680751173709,
305
+ "server_spec_accept_length": 0.0,
306
+ "avg_running_reqs": 1,
307
+ "max_running_reqs": 1,
308
+ "effective_concurrency": 1,
309
+ "avg_queue_reqs": 0,
310
+ "max_queue_reqs": 0,
311
+ "queue_fraction": 0.0,
312
+ "underfilled": false,
313
+ "warmup_timed_out": false,
314
+ "warmup_duration": 5.536,
315
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
316
+ "timeout_reason": "",
317
+ "capacity_limited": false,
318
+ "hardware_summary": {
319
+ "samples": 8,
320
+ "duration_seconds": 16.909,
321
+ "gpu_count": 4,
322
+ "cpu_util_avg_pct": 11.56,
323
+ "cpu_temp_max_c": 75.88,
324
+ "gpu_util_avg_pct": 99.0,
325
+ "gpu_util_max_pct": 99.0,
326
+ "mem_util_avg_pct": 44.09,
327
+ "mem_util_max_pct": 56.0,
328
+ "temp_avg_c": 67.31,
329
+ "temp_max_c": 82.0,
330
+ "power_total_avg_w": 1153.01,
331
+ "power_total_max_w": 1155.24,
332
+ "power_limit_total_w": 1200.0,
333
+ "vram_used_avg_mb": 384778.0,
334
+ "vram_used_max_mb": 384778.0,
335
+ "vram_total_mb": 391548.0,
336
+ "vram_used_avg_pct": 98.27,
337
+ "vram_used_max_pct": 98.27,
338
+ "pcie_rx_avg_mb_s": 8331.0,
339
+ "pcie_rx_max_mb_s": 8554.0,
340
+ "pcie_tx_avg_mb_s": 8194.0,
341
+ "pcie_tx_max_mb_s": 8520.0
342
+ }
343
+ },
344
+ {
345
+ "concurrency": 1,
346
+ "context_tokens": 8192,
347
+ "benchmark_mode": "duration",
348
+ "request_count_target": 0,
349
+ "warmup_request_count": 0,
350
+ "measurement_seconds": 19.992808,
351
+ "measurement_wall_seconds": 20.000923,
352
+ "client_output_tokens": 3482,
353
+ "server_output_tokens": 3482,
354
+ "aggregate_source": "openai_continuous_usage",
355
+ "aggregate_tps": 174.16263081468745,
356
+ "per_request_avg_tps": 174.16263081468745,
357
+ "ttft_avg": 0.5914782441686839,
358
+ "ttft_p50": 0.5914782441686839,
359
+ "ttft_p90": 0.5914782441686839,
360
+ "ttft_p99": 0.5914782441686839,
361
+ "time_to_second_token_avg": 0.01696636783890426,
362
+ "time_to_second_token_p50": 0.01696636783890426,
363
+ "time_to_second_token_p90": 0.01696636783890426,
364
+ "time_to_second_token_p99": 0.01696636783890426,
365
+ "request_latency_avg": 0.0,
366
+ "request_latency_p50": 0.0,
367
+ "request_latency_p90": 0.0,
368
+ "request_latency_p99": 0.0,
369
+ "inter_token_latency_avg": 0.005593080112173436,
370
+ "inter_token_latency_p50": 0.005593080112173436,
371
+ "inter_token_latency_p90": 0.005593080112173436,
372
+ "inter_token_latency_p99": 0.005593080112173436,
373
+ "output_tps_per_user_avg": 178.79236126503582,
374
+ "output_tps_per_user_p50": 178.79236126503582,
375
+ "output_tps_per_user_p90": 178.79236126503582,
376
+ "output_tps_per_user_p99": 178.79236126503582,
377
+ "e2e_output_tps_per_user_avg": 0.0,
378
+ "e2e_output_tps_per_user_p50": 0.0,
379
+ "e2e_output_tps_per_user_p90": 0.0,
380
+ "e2e_output_tps_per_user_p99": 0.0,
381
+ "chunk_inter_token_latency_avg": 0.014262687604284941,
382
+ "chunk_inter_token_latency_p50": 0.014262687604284941,
383
+ "chunk_inter_token_latency_p90": 0.014262687604284941,
384
+ "chunk_inter_token_latency_p99": 0.014262687604284941,
385
+ "input_seq_len_avg": 8192.0,
386
+ "output_seq_len_avg": 4280.0,
387
+ "output_seq_len_p50": 4280.0,
388
+ "output_seq_len_p90": 4280.0,
389
+ "output_seq_len_p99": 4280.0,
390
+ "request_count": 1,
391
+ "completed_request_count": 0,
392
+ "request_samples": [
393
+ {
394
+ "ttft": 0.5914782441686839,
395
+ "time_to_second_token": 0.01696636783890426,
396
+ "latency": 0.0,
397
+ "inter_token_latency_avg": 0.005593080112173436,
398
+ "chunk_inter_token_latency_avg": 0.014262687604284941,
399
+ "input_tokens": 8192,
400
+ "output_tokens": 4280,
401
+ "output_tps_per_user": 178.79236126503582,
402
+ "e2e_output_tps_per_user": 0.0,
403
+ "completed": false
404
+ }
405
+ ],
406
+ "total_tokens": 3482,
407
+ "wall_time": 26.0623870131094,
408
+ "num_completed": 1,
409
+ "num_errors": 0,
410
+ "server_gen_throughput": 174.05206214713238,
411
+ "server_utilization": 0.006141820212172022,
412
+ "server_spec_accept_rate": 0.5047619047619047,
413
+ "server_spec_accept_length": 0.0,
414
+ "avg_running_reqs": 1,
415
+ "max_running_reqs": 1,
416
+ "effective_concurrency": 1,
417
+ "avg_queue_reqs": 0,
418
+ "max_queue_reqs": 0,
419
+ "queue_fraction": 0.0,
420
+ "underfilled": false,
421
+ "warmup_timed_out": false,
422
+ "warmup_duration": 6.055,
423
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
424
+ "timeout_reason": "",
425
+ "capacity_limited": false,
426
+ "hardware_summary": {
427
+ "samples": 8,
428
+ "duration_seconds": 16.885,
429
+ "gpu_count": 4,
430
+ "cpu_util_avg_pct": 11.53,
431
+ "cpu_temp_max_c": 76.62,
432
+ "gpu_util_avg_pct": 99.0,
433
+ "gpu_util_max_pct": 99.0,
434
+ "mem_util_avg_pct": 43.56,
435
+ "mem_util_max_pct": 55.0,
436
+ "temp_avg_c": 67.78,
437
+ "temp_max_c": 83.0,
438
+ "power_total_avg_w": 1153.83,
439
+ "power_total_max_w": 1155.28,
440
+ "power_limit_total_w": 1200.0,
441
+ "vram_used_avg_mb": 384778.0,
442
+ "vram_used_max_mb": 384778.0,
443
+ "vram_total_mb": 391548.0,
444
+ "vram_used_avg_pct": 98.27,
445
+ "vram_used_max_pct": 98.27,
446
+ "pcie_rx_avg_mb_s": 8357.0,
447
+ "pcie_rx_max_mb_s": 8584.0,
448
+ "pcie_tx_avg_mb_s": 8151.12,
449
+ "pcie_tx_max_mb_s": 8441.0
450
+ }
451
+ },
452
+ {
453
+ "concurrency": 1,
454
+ "context_tokens": 32768,
455
+ "benchmark_mode": "duration",
456
+ "request_count_target": 0,
457
+ "warmup_request_count": 0,
458
+ "measurement_seconds": 19.987041,
459
+ "measurement_wall_seconds": 20.00121,
460
+ "client_output_tokens": 3584,
461
+ "server_output_tokens": 3586,
462
+ "aggregate_source": "openai_continuous_usage",
463
+ "aggregate_tps": 179.316188111569,
464
+ "per_request_avg_tps": 179.316188111569,
465
+ "ttft_avg": 0.6033541450742632,
466
+ "ttft_p50": 0.6033541450742632,
467
+ "ttft_p90": 0.6033541450742632,
468
+ "ttft_p99": 0.6033541450742632,
469
+ "time_to_second_token_avg": 0.012713581090793014,
470
+ "time_to_second_token_p50": 0.012713581090793014,
471
+ "time_to_second_token_p90": 0.012713581090793014,
472
+ "time_to_second_token_p99": 0.012713581090793014,
473
+ "request_latency_avg": 0.0,
474
+ "request_latency_p50": 0.0,
475
+ "request_latency_p90": 0.0,
476
+ "request_latency_p99": 0.0,
477
+ "inter_token_latency_avg": 0.005507921420528074,
478
+ "inter_token_latency_p50": 0.005507921420528074,
479
+ "inter_token_latency_p90": 0.005507921420528074,
480
+ "inter_token_latency_p99": 0.005507921420528074,
481
+ "output_tps_per_user_avg": 181.55669328777836,
482
+ "output_tps_per_user_p50": 181.55669328777836,
483
+ "output_tps_per_user_p90": 181.55669328777836,
484
+ "output_tps_per_user_p99": 181.55669328777836,
485
+ "e2e_output_tps_per_user_avg": 0.0,
486
+ "e2e_output_tps_per_user_p50": 0.0,
487
+ "e2e_output_tps_per_user_p90": 0.0,
488
+ "e2e_output_tps_per_user_p99": 0.0,
489
+ "chunk_inter_token_latency_avg": 0.014389527561933152,
490
+ "chunk_inter_token_latency_p50": 0.014389527561933152,
491
+ "chunk_inter_token_latency_p90": 0.014389527561933152,
492
+ "chunk_inter_token_latency_p99": 0.014389527561933152,
493
+ "input_seq_len_avg": 32768.0,
494
+ "output_seq_len_avg": 4343.0,
495
+ "output_seq_len_p50": 4343.0,
496
+ "output_seq_len_p90": 4343.0,
497
+ "output_seq_len_p99": 4343.0,
498
+ "request_count": 1,
499
+ "completed_request_count": 0,
500
+ "request_samples": [
501
+ {
502
+ "ttft": 0.6033541450742632,
503
+ "time_to_second_token": 0.012713581090793014,
504
+ "latency": 0.0,
505
+ "inter_token_latency_avg": 0.005507921420528074,
506
+ "chunk_inter_token_latency_avg": 0.014389527561933152,
507
+ "input_tokens": 32768,
508
+ "output_tokens": 4343,
509
+ "output_tps_per_user": 181.55669328777836,
510
+ "e2e_output_tps_per_user": 0.0,
511
+ "completed": false
512
+ }
513
+ ],
514
+ "total_tokens": 3584,
515
+ "wall_time": 29.107660971814767,
516
+ "num_completed": 1,
517
+ "num_errors": 0,
518
+ "server_gen_throughput": 179.23767853872448,
519
+ "server_utilization": 0.006979341150195384,
520
+ "server_spec_accept_rate": 0.46190476190476193,
521
+ "server_spec_accept_length": 0.0,
522
+ "avg_running_reqs": 1,
523
+ "max_running_reqs": 1,
524
+ "effective_concurrency": 1,
525
+ "avg_queue_reqs": 0,
526
+ "max_queue_reqs": 0,
527
+ "queue_fraction": 0.0,
528
+ "underfilled": false,
529
+ "warmup_timed_out": false,
530
+ "warmup_duration": 9.101,
531
+ "ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
532
+ "timeout_reason": "",
533
+ "capacity_limited": false,
534
+ "hardware_summary": {
535
+ "samples": 8,
536
+ "duration_seconds": 16.891,
537
+ "gpu_count": 4,
538
+ "cpu_util_avg_pct": 11.56,
539
+ "cpu_temp_max_c": 75.88,
540
+ "gpu_util_avg_pct": 99.0,
541
+ "gpu_util_max_pct": 99.0,
542
+ "mem_util_avg_pct": 44.0,
543
+ "mem_util_max_pct": 56.0,
544
+ "temp_avg_c": 68.16,
545
+ "temp_max_c": 83.0,
546
+ "power_total_avg_w": 1155.58,
547
+ "power_total_max_w": 1156.94,
548
+ "power_limit_total_w": 1200.0,
549
+ "vram_used_avg_mb": 384778.0,
550
+ "vram_used_max_mb": 384778.0,
551
+ "vram_total_mb": 391548.0,
552
+ "vram_used_avg_pct": 98.27,
553
+ "vram_used_max_pct": 98.27,
554
+ "pcie_rx_avg_mb_s": 8340.12,
555
+ "pcie_rx_max_mb_s": 8569.0,
556
+ "pcie_tx_avg_mb_s": 8143.62,
557
+ "pcie_tx_max_mb_s": 8438.0
558
+ }
559
+ },
560
+ {
561
+ "concurrency": 2,
562
+ "context_tokens": 0,
563
+ "benchmark_mode": "duration",
564
+ "request_count_target": 0,
565
+ "warmup_request_count": 0,
566
+ "measurement_seconds": 20.00082,
567
+ "measurement_wall_seconds": 20.000837,
568
+ "client_output_tokens": 4728,
569
+ "server_output_tokens": 4728,
570
+ "aggregate_source": "openai_continuous_usage",
571
+ "aggregate_tps": 236.39031276129717,
572
+ "per_request_avg_tps": 118.19515638064858,
573
+ "ttft_avg": 0.11200506216846406,
574
+ "ttft_p50": 0.11200506216846406,
575
+ "ttft_p90": 0.1434264814015478,
576
+ "ttft_p99": 0.15049630072899162,
577
+ "time_to_second_token_avg": 0.01528907532338053,
578
+ "time_to_second_token_p50": 0.01528907532338053,
579
+ "time_to_second_token_p90": 0.017035022121854128,
580
+ "time_to_second_token_p99": 0.017427860151510686,
581
+ "request_latency_avg": 0.0,
582
+ "request_latency_p50": 0.0,
583
+ "request_latency_p90": 0.0,
584
+ "request_latency_p99": 0.0,
585
+ "inter_token_latency_avg": 0.00843379404097247,
586
+ "inter_token_latency_p50": 0.00843379404097247,
587
+ "inter_token_latency_p90": 0.008573998497973905,
588
+ "inter_token_latency_p99": 0.008605544500799228,
589
+ "output_tps_per_user_avg": 118.62182033895986,
590
+ "output_tps_per_user_p50": 118.62182033895986,
591
+ "output_tps_per_user_p90": 120.59380445765525,
592
+ "output_tps_per_user_p99": 121.03750088436172,
593
+ "e2e_output_tps_per_user_avg": 0.0,
594
+ "e2e_output_tps_per_user_p50": 0.0,
595
+ "e2e_output_tps_per_user_p90": 0.0,
596
+ "e2e_output_tps_per_user_p99": 0.0,
597
+ "chunk_inter_token_latency_avg": 0.020858210011846075,
598
+ "chunk_inter_token_latency_p50": 0.020858210011846075,
599
+ "chunk_inter_token_latency_p90": 0.02088407432588644,
600
+ "chunk_inter_token_latency_p99": 0.02088989379654552,
601
+ "input_seq_len_avg": 78.0,
602
+ "output_seq_len_avg": 3017.0,
603
+ "output_seq_len_p50": 3017.0,
604
+ "output_seq_len_p90": 3063.4,
605
+ "output_seq_len_p99": 3073.84,
606
+ "request_count": 2,
607
+ "completed_request_count": 0,
608
+ "request_samples": [
609
+ {
610
+ "ttft": 0.07272828812710941,
611
+ "time_to_second_token": 0.013106641825288534,
612
+ "latency": 0.0,
613
+ "inter_token_latency_avg": 0.008609049612224263,
614
+ "chunk_inter_token_latency_avg": 0.02089054040439653,
615
+ "input_tokens": 78,
616
+ "output_tokens": 2959,
617
+ "output_tps_per_user": 116.15684019059063,
618
+ "e2e_output_tps_per_user": 0.0,
619
+ "completed": false
620
+ },
621
+ {
622
+ "ttft": 0.15128183620981872,
623
+ "time_to_second_token": 0.017471508821472526,
624
+ "latency": 0.0,
625
+ "inter_token_latency_avg": 0.008258538469720678,
626
+ "chunk_inter_token_latency_avg": 0.020825879619295624,
627
+ "input_tokens": 78,
628
+ "output_tokens": 3075,
629
+ "output_tps_per_user": 121.0868004873291,
630
+ "e2e_output_tps_per_user": 0.0,
631
+ "completed": false
632
+ }
633
+ ],
634
+ "total_tokens": 4728,
635
+ "wall_time": 25.560176193946972,
636
+ "num_completed": 2,
637
+ "num_errors": 0,
638
+ "server_gen_throughput": 236.33660483401786,
639
+ "server_utilization": 0.011725293132328285,
640
+ "server_spec_accept_rate": 0.3958333333333333,
641
+ "server_spec_accept_length": 0.0,
642
+ "avg_running_reqs": 2,
643
+ "max_running_reqs": 2,
644
+ "effective_concurrency": 2,
645
+ "avg_queue_reqs": 0,
646
+ "max_queue_reqs": 0,
647
+ "queue_fraction": 0.0,
648
+ "underfilled": false,
649
+ "warmup_timed_out": false,
650
+ "warmup_duration": 5.538,
651
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
652
+ "timeout_reason": "",
653
+ "capacity_limited": false,
654
+ "hardware_summary": {
655
+ "samples": 8,
656
+ "duration_seconds": 16.86,
657
+ "gpu_count": 4,
658
+ "cpu_util_avg_pct": 11.55,
659
+ "cpu_temp_max_c": 77.12,
660
+ "gpu_util_avg_pct": 100.0,
661
+ "gpu_util_max_pct": 100.0,
662
+ "mem_util_avg_pct": 40.34,
663
+ "mem_util_max_pct": 50.0,
664
+ "temp_avg_c": 68.5,
665
+ "temp_max_c": 84.0,
666
+ "power_total_avg_w": 1176.31,
667
+ "power_total_max_w": 1178.12,
668
+ "power_limit_total_w": 1200.0,
669
+ "vram_used_avg_mb": 384778.0,
670
+ "vram_used_max_mb": 384778.0,
671
+ "vram_total_mb": 391548.0,
672
+ "vram_used_avg_pct": 98.27,
673
+ "vram_used_max_pct": 98.27,
674
+ "pcie_rx_avg_mb_s": 11179.25,
675
+ "pcie_rx_max_mb_s": 11225.0,
676
+ "pcie_tx_avg_mb_s": 11087.75,
677
+ "pcie_tx_max_mb_s": 11179.0
678
+ }
679
+ },
680
+ {
681
+ "concurrency": 4,
682
+ "context_tokens": 0,
683
+ "benchmark_mode": "duration",
684
+ "request_count_target": 0,
685
+ "warmup_request_count": 0,
686
+ "measurement_seconds": 19.975466,
687
+ "measurement_wall_seconds": 20.000669,
688
+ "client_output_tokens": 5926,
689
+ "server_output_tokens": 5926,
690
+ "aggregate_source": "openai_continuous_usage",
691
+ "aggregate_tps": 296.6639108412111,
692
+ "per_request_avg_tps": 74.16597771030277,
693
+ "ttft_avg": 0.1473790240706876,
694
+ "ttft_p50": 0.17257179808802903,
695
+ "ttft_p90": 0.17268561611417682,
696
+ "ttft_p99": 0.17268778499448673,
697
+ "time_to_second_token_avg": 0.025300663721282035,
698
+ "time_to_second_token_p50": 0.02924731746315956,
699
+ "time_to_second_token_p90": 0.029298453708179295,
700
+ "time_to_second_token_p99": 0.029313364170957357,
701
+ "request_latency_avg": 0.0,
702
+ "request_latency_p50": 0.0,
703
+ "request_latency_p90": 0.0,
704
+ "request_latency_p99": 0.0,
705
+ "inter_token_latency_avg": 0.013187405590978741,
706
+ "inter_token_latency_p50": 0.01321511895063593,
707
+ "inter_token_latency_p90": 0.01351998806064279,
708
+ "inter_token_latency_p99": 0.01357511053376785,
709
+ "output_tps_per_user_avg": 75.87497211232,
710
+ "output_tps_per_user_p50": 75.68227161840176,
711
+ "output_tps_per_user_p90": 77.93597879757434,
712
+ "output_tps_per_user_p99": 78.44750412533332,
713
+ "e2e_output_tps_per_user_avg": 0.0,
714
+ "e2e_output_tps_per_user_p50": 0.0,
715
+ "e2e_output_tps_per_user_p90": 0.0,
716
+ "e2e_output_tps_per_user_p99": 0.0,
717
+ "chunk_inter_token_latency_avg": 0.034238496269222964,
718
+ "chunk_inter_token_latency_p50": 0.03419213106811758,
719
+ "chunk_inter_token_latency_p90": 0.03441501889490805,
720
+ "chunk_inter_token_latency_p99": 0.0344653942706813,
721
+ "input_seq_len_avg": 78.0,
722
+ "output_seq_len_avg": 1925.25,
723
+ "output_seq_len_p50": 1918.5,
724
+ "output_seq_len_p90": 1975.6,
725
+ "output_seq_len_p99": 1988.56,
726
+ "request_count": 4,
727
+ "completed_request_count": 0,
728
+ "request_samples": [
729
+ {
730
+ "ttft": 0.0716844741255045,
731
+ "time_to_second_token": 0.01339299906976521,
732
+ "latency": 0.0,
733
+ "inter_token_latency_avg": 0.013581235253003969,
734
+ "chunk_inter_token_latency_avg": 0.03409873140600058,
735
+ "input_tokens": 78,
736
+ "output_tokens": 1874,
737
+ "output_tps_per_user": 73.63100493961437,
738
+ "e2e_output_tps_per_user": 0.0,
739
+ "completed": false
740
+ },
741
+ {
742
+ "ttft": 0.17267999309115112,
743
+ "time_to_second_token": 0.029315020889043808,
744
+ "latency": 0.0,
745
+ "inter_token_latency_avg": 0.013053159956138491,
746
+ "chunk_inter_token_latency_avg": 0.034284416068829246,
747
+ "input_tokens": 78,
748
+ "output_tokens": 1942,
749
+ "output_tps_per_user": 76.60980202190285,
750
+ "e2e_output_tps_per_user": 0.0,
751
+ "completed": false
752
+ },
753
+ {
754
+ "ttft": 0.17268802598118782,
755
+ "time_to_second_token": 0.029259796952828765,
756
+ "latency": 0.0,
757
+ "inter_token_latency_avg": 0.012738149209639133,
758
+ "chunk_inter_token_latency_avg": 0.034470991534656104,
759
+ "input_tokens": 78,
760
+ "output_tokens": 1990,
761
+ "output_tps_per_user": 78.50434027286211,
762
+ "e2e_output_tps_per_user": 0.0,
763
+ "completed": false
764
+ },
765
+ {
766
+ "ttft": 0.17246360308490694,
767
+ "time_to_second_token": 0.029234837973490357,
768
+ "latency": 0.0,
769
+ "inter_token_latency_avg": 0.01337707794513337,
770
+ "chunk_inter_token_latency_avg": 0.034099846067405924,
771
+ "input_tokens": 78,
772
+ "output_tokens": 1895,
773
+ "output_tps_per_user": 74.75474121490065,
774
+ "e2e_output_tps_per_user": 0.0,
775
+ "completed": false
776
+ }
777
+ ],
778
+ "total_tokens": 5926,
779
+ "wall_time": 25.54523406806402,
780
+ "num_completed": 4,
781
+ "num_errors": 0,
782
+ "server_gen_throughput": 296.1937254544746,
783
+ "server_utilization": 0.02345058626465657,
784
+ "server_spec_accept_rate": 0.5028735632183908,
785
+ "server_spec_accept_length": 0.0,
786
+ "avg_running_reqs": 4,
787
+ "max_running_reqs": 4,
788
+ "effective_concurrency": 4,
789
+ "avg_queue_reqs": 0,
790
+ "max_queue_reqs": 0,
791
+ "queue_fraction": 0.0,
792
+ "underfilled": false,
793
+ "warmup_timed_out": false,
794
+ "warmup_duration": 5.534,
795
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
796
+ "timeout_reason": "",
797
+ "capacity_limited": false,
798
+ "hardware_summary": {
799
+ "samples": 8,
800
+ "duration_seconds": 16.831,
801
+ "gpu_count": 4,
802
+ "cpu_util_avg_pct": 11.49,
803
+ "cpu_temp_max_c": 77.38,
804
+ "gpu_util_avg_pct": 100.0,
805
+ "gpu_util_max_pct": 100.0,
806
+ "mem_util_avg_pct": 35.0,
807
+ "mem_util_max_pct": 43.0,
808
+ "temp_avg_c": 69.0,
809
+ "temp_max_c": 84.0,
810
+ "power_total_avg_w": 1172.77,
811
+ "power_total_max_w": 1173.84,
812
+ "power_limit_total_w": 1200.0,
813
+ "vram_used_avg_mb": 384778.0,
814
+ "vram_used_max_mb": 384778.0,
815
+ "vram_total_mb": 391548.0,
816
+ "vram_used_avg_pct": 98.27,
817
+ "vram_used_max_pct": 98.27,
818
+ "pcie_rx_avg_mb_s": 8004.38,
819
+ "pcie_rx_max_mb_s": 8287.0,
820
+ "pcie_tx_avg_mb_s": 7682.5,
821
+ "pcie_tx_max_mb_s": 7820.0
822
+ }
823
+ },
824
+ {
825
+ "concurrency": 2,
826
+ "context_tokens": 8192,
827
+ "benchmark_mode": "duration",
828
+ "request_count_target": 0,
829
+ "warmup_request_count": 0,
830
+ "measurement_seconds": 19.987602,
831
+ "measurement_wall_seconds": 20.001775,
832
+ "client_output_tokens": 4932,
833
+ "server_output_tokens": 4932,
834
+ "aggregate_source": "openai_continuous_usage",
835
+ "aggregate_tps": 246.75295838278893,
836
+ "per_request_avg_tps": 123.37647919139447,
837
+ "ttft_avg": 0.9467372725484893,
838
+ "ttft_p50": 0.9467372725484893,
839
+ "ttft_p90": 1.2301197802415118,
840
+ "ttft_p99": 1.293880844472442,
841
+ "time_to_second_token_avg": 0.02229496242944151,
842
+ "time_to_second_token_p50": 0.02229496242944151,
843
+ "time_to_second_token_p90": 0.02584285002667457,
844
+ "time_to_second_token_p99": 0.026641124736052006,
845
+ "request_latency_avg": 0.0,
846
+ "request_latency_p50": 0.0,
847
+ "request_latency_p90": 0.0,
848
+ "request_latency_p99": 0.0,
849
+ "inter_token_latency_avg": 0.008086084075571213,
850
+ "inter_token_latency_p50": 0.008086084075571213,
851
+ "inter_token_latency_p90": 0.008137498278031636,
852
+ "inter_token_latency_p99": 0.008149066473585232,
853
+ "output_tps_per_user_avg": 123.67706846442331,
854
+ "output_tps_per_user_p50": 123.67706846442331,
855
+ "output_tps_per_user_p90": 124.46345131405901,
856
+ "output_tps_per_user_p99": 124.64038745522704,
857
+ "e2e_output_tps_per_user_avg": 0.0,
858
+ "e2e_output_tps_per_user_p50": 0.0,
859
+ "e2e_output_tps_per_user_p90": 0.0,
860
+ "e2e_output_tps_per_user_p99": 0.0,
861
+ "chunk_inter_token_latency_avg": 0.021277465417892594,
862
+ "chunk_inter_token_latency_p50": 0.021277465417892594,
863
+ "chunk_inter_token_latency_p90": 0.021564048664052024,
864
+ "chunk_inter_token_latency_p99": 0.021628529894437896,
865
+ "input_seq_len_avg": 8192.0,
866
+ "output_seq_len_avg": 2917.0,
867
+ "output_seq_len_p50": 2917.0,
868
+ "output_seq_len_p90": 2970.6,
869
+ "output_seq_len_p99": 2982.66,
870
+ "request_count": 2,
871
+ "completed_request_count": 0,
872
+ "request_samples": [
873
+ {
874
+ "ttft": 0.5925091379322112,
875
+ "time_to_second_token": 0.01786010293290019,
876
+ "latency": 0.0,
877
+ "inter_token_latency_avg": 0.008021816322495684,
878
+ "chunk_inter_token_latency_avg": 0.021635694475591882,
879
+ "input_tokens": 8192,
880
+ "output_tokens": 2984,
881
+ "output_tps_per_user": 124.66004702646794,
882
+ "e2e_output_tps_per_user": 0.0,
883
+ "completed": false
884
+ },
885
+ {
886
+ "ttft": 1.3009654071647674,
887
+ "time_to_second_token": 0.026729821925982833,
888
+ "latency": 0.0,
889
+ "inter_token_latency_avg": 0.008150351828646742,
890
+ "chunk_inter_token_latency_avg": 0.020919236360193307,
891
+ "input_tokens": 8192,
892
+ "output_tokens": 2850,
893
+ "output_tps_per_user": 122.6940899023787,
894
+ "e2e_output_tps_per_user": 0.0,
895
+ "completed": false
896
+ }
897
+ ],
898
+ "total_tokens": 4932,
899
+ "wall_time": 25.56168243405409,
900
+ "num_completed": 2,
901
+ "num_errors": 0,
902
+ "server_gen_throughput": 246.50722758411473,
903
+ "server_utilization": 0.012283640424343933,
904
+ "server_spec_accept_rate": 0.5763888888888888,
905
+ "server_spec_accept_length": 0.0,
906
+ "avg_running_reqs": 2,
907
+ "max_running_reqs": 2,
908
+ "effective_concurrency": 2,
909
+ "avg_queue_reqs": 0,
910
+ "max_queue_reqs": 0,
911
+ "queue_fraction": 0.0,
912
+ "underfilled": false,
913
+ "warmup_timed_out": false,
914
+ "warmup_duration": 5.552,
915
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
916
+ "timeout_reason": "",
917
+ "capacity_limited": false,
918
+ "hardware_summary": {
919
+ "samples": 8,
920
+ "duration_seconds": 16.901,
921
+ "gpu_count": 4,
922
+ "cpu_util_avg_pct": 11.55,
923
+ "cpu_temp_max_c": 77.25,
924
+ "gpu_util_avg_pct": 100.0,
925
+ "gpu_util_max_pct": 100.0,
926
+ "mem_util_avg_pct": 40.5,
927
+ "mem_util_max_pct": 50.0,
928
+ "temp_avg_c": 68.62,
929
+ "temp_max_c": 84.0,
930
+ "power_total_avg_w": 1177.05,
931
+ "power_total_max_w": 1177.91,
932
+ "power_limit_total_w": 1200.0,
933
+ "vram_used_avg_mb": 384778.0,
934
+ "vram_used_max_mb": 384778.0,
935
+ "vram_total_mb": 391548.0,
936
+ "vram_used_avg_pct": 98.27,
937
+ "vram_used_max_pct": 98.27,
938
+ "pcie_rx_avg_mb_s": 11224.88,
939
+ "pcie_rx_max_mb_s": 11341.0,
940
+ "pcie_tx_avg_mb_s": 11103.0,
941
+ "pcie_tx_max_mb_s": 11202.0
942
+ }
943
+ },
944
+ {
945
+ "concurrency": 4,
946
+ "context_tokens": 8192,
947
+ "benchmark_mode": "duration",
948
+ "request_count_target": 0,
949
+ "warmup_request_count": 0,
950
+ "measurement_seconds": 19.99956,
951
+ "measurement_wall_seconds": 20.000651,
952
+ "client_output_tokens": 5828,
953
+ "server_output_tokens": 5828,
954
+ "aggregate_source": "openai_continuous_usage",
955
+ "aggregate_tps": 291.40641309659526,
956
+ "per_request_avg_tps": 72.85160327414881,
957
+ "ttft_avg": 2.3201283884700388,
958
+ "ttft_p50": 2.213119035004638,
959
+ "ttft_p90": 3.644294320815243,
960
+ "ttft_p99": 4.196257012102286,
961
+ "time_to_second_token_avg": 0.034855086996685714,
962
+ "time_to_second_token_p50": 0.040691820555366576,
963
+ "time_to_second_token_p90": 0.041562757641077044,
964
+ "time_to_second_token_p99": 0.0415682226838544,
965
+ "request_latency_avg": 0.0,
966
+ "request_latency_p50": 0.0,
967
+ "request_latency_p90": 0.0,
968
+ "request_latency_p99": 0.0,
969
+ "inter_token_latency_avg": 0.013789638304529176,
970
+ "inter_token_latency_p50": 0.013799817255296618,
971
+ "inter_token_latency_p90": 0.013947021664650167,
972
+ "inter_token_latency_p99": 0.013999457133027752,
973
+ "output_tps_per_user_avg": 72.52803020492948,
974
+ "output_tps_per_user_p50": 72.46477597135552,
975
+ "output_tps_per_user_p90": 73.40383210307152,
976
+ "output_tps_per_user_p99": 73.74323179562586,
977
+ "e2e_output_tps_per_user_avg": 0.0,
978
+ "e2e_output_tps_per_user_p50": 0.0,
979
+ "e2e_output_tps_per_user_p90": 0.0,
980
+ "e2e_output_tps_per_user_p99": 0.0,
981
+ "chunk_inter_token_latency_avg": 0.0354126605281524,
982
+ "chunk_inter_token_latency_p50": 0.03551991527773096,
983
+ "chunk_inter_token_latency_p90": 0.035908490963787884,
984
+ "chunk_inter_token_latency_p99": 0.03602955673091524,
985
+ "input_seq_len_avg": 8192.0,
986
+ "output_seq_len_avg": 1830.25,
987
+ "output_seq_len_p50": 1837.5,
988
+ "output_seq_len_p90": 1899.9,
989
+ "output_seq_len_p99": 1923.3899999999999,
990
+ "request_count": 4,
991
+ "completed_request_count": 0,
992
+ "request_samples": [
993
+ {
994
+ "ttft": 0.5966892838478088,
995
+ "time_to_second_token": 0.01646787696518004,
996
+ "latency": 0.0,
997
+ "inter_token_latency_avg": 0.014005283296180816,
998
+ "chunk_inter_token_latency_avg": 0.03604300848281828,
999
+ "input_tokens": 8192,
1000
+ "output_tokens": 1926,
1001
+ "output_tps_per_user": 71.40162600443048,
1002
+ "e2e_output_tps_per_user": 0.0,
1003
+ "completed": false
1004
+ },
1005
+ {
1006
+ "ttft": 2.2132799359969795,
1007
+ "time_to_second_token": 0.04156882991082966,
1008
+ "latency": 0.0,
1009
+ "inter_token_latency_avg": 0.013811077857745319,
1010
+ "chunk_inter_token_latency_avg": 0.03544521380274498,
1011
+ "input_tokens": 8192,
1012
+ "output_tokens": 1836,
1013
+ "output_tps_per_user": 72.40564496848414,
1014
+ "e2e_output_tps_per_user": 0.0,
1015
+ "completed": false
1016
+ },
1017
+ {
1018
+ "ttft": 2.212958134012297,
1019
+ "time_to_second_token": 0.04154858901165426,
1020
+ "latency": 0.0,
1021
+ "inter_token_latency_avg": 0.013788556652847917,
1022
+ "chunk_inter_token_latency_avg": 0.035594616752716954,
1023
+ "input_tokens": 8192,
1024
+ "output_tokens": 1839,
1025
+ "output_tps_per_user": 72.52390697422692,
1026
+ "e2e_output_tps_per_user": 0.0,
1027
+ "completed": false
1028
+ },
1029
+ {
1030
+ "ttft": 4.25758620002307,
1031
+ "time_to_second_token": 0.039835052099078894,
1032
+ "latency": 0.0,
1033
+ "inter_token_latency_avg": 0.013553635411342652,
1034
+ "chunk_inter_token_latency_avg": 0.034567803074329405,
1035
+ "input_tokens": 8192,
1036
+ "output_tokens": 1720,
1037
+ "output_tps_per_user": 73.78094287257635,
1038
+ "e2e_output_tps_per_user": 0.0,
1039
+ "completed": false
1040
+ }
1041
+ ],
1042
+ "total_tokens": 5828,
1043
+ "wall_time": 28.611056566005573,
1044
+ "num_completed": 4,
1045
+ "num_errors": 0,
1046
+ "server_gen_throughput": 291.3185246321431,
1047
+ "server_utilization": 0.024567280848687867,
1048
+ "server_spec_accept_rate": 0.4511494252873563,
1049
+ "server_spec_accept_length": 0.0,
1050
+ "avg_running_reqs": 4,
1051
+ "max_running_reqs": 4,
1052
+ "effective_concurrency": 4,
1053
+ "avg_queue_reqs": 0,
1054
+ "max_queue_reqs": 0,
1055
+ "queue_fraction": 0.0,
1056
+ "underfilled": false,
1057
+ "warmup_timed_out": false,
1058
+ "warmup_duration": 8.576,
1059
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1060
+ "timeout_reason": "",
1061
+ "capacity_limited": false,
1062
+ "hardware_summary": {
1063
+ "samples": 8,
1064
+ "duration_seconds": 16.82,
1065
+ "gpu_count": 4,
1066
+ "cpu_util_avg_pct": 11.51,
1067
+ "cpu_temp_max_c": 76.88,
1068
+ "gpu_util_avg_pct": 100.0,
1069
+ "gpu_util_max_pct": 100.0,
1070
+ "mem_util_avg_pct": 34.94,
1071
+ "mem_util_max_pct": 43.0,
1072
+ "temp_avg_c": 69.09,
1073
+ "temp_max_c": 84.0,
1074
+ "power_total_avg_w": 1172.67,
1075
+ "power_total_max_w": 1173.43,
1076
+ "power_limit_total_w": 1200.0,
1077
+ "vram_used_avg_mb": 384778.0,
1078
+ "vram_used_max_mb": 384778.0,
1079
+ "vram_total_mb": 391548.0,
1080
+ "vram_used_avg_pct": 98.27,
1081
+ "vram_used_max_pct": 98.27,
1082
+ "pcie_rx_avg_mb_s": 7728.12,
1083
+ "pcie_rx_max_mb_s": 7936.0,
1084
+ "pcie_tx_avg_mb_s": 7832.62,
1085
+ "pcie_tx_max_mb_s": 8312.0
1086
+ }
1087
+ },
1088
+ {
1089
+ "concurrency": 2,
1090
+ "context_tokens": 32768,
1091
+ "benchmark_mode": "duration",
1092
+ "request_count_target": 0,
1093
+ "warmup_request_count": 0,
1094
+ "measurement_seconds": 19.985443,
1095
+ "measurement_wall_seconds": 20.000566,
1096
+ "client_output_tokens": 4794,
1097
+ "server_output_tokens": 4794,
1098
+ "aggregate_source": "openai_continuous_usage",
1099
+ "aggregate_tps": 239.87458866483988,
1100
+ "per_request_avg_tps": 119.93729433241994,
1101
+ "ttft_avg": 0.9611447914503515,
1102
+ "ttft_p50": 0.9611447914503515,
1103
+ "ttft_p90": 1.2436655103228986,
1104
+ "ttft_p99": 1.3072326720692218,
1105
+ "time_to_second_token_avg": 0.014196591568179429,
1106
+ "time_to_second_token_p50": 0.014196591568179429,
1107
+ "time_to_second_token_p90": 0.017423492693342268,
1108
+ "time_to_second_token_p99": 0.018149545446503906,
1109
+ "request_latency_avg": 0.0,
1110
+ "request_latency_p50": 0.0,
1111
+ "request_latency_p90": 0.0,
1112
+ "request_latency_p99": 0.0,
1113
+ "inter_token_latency_avg": 0.00824110461840657,
1114
+ "inter_token_latency_p50": 0.00824110461840657,
1115
+ "inter_token_latency_p90": 0.008593042280594048,
1116
+ "inter_token_latency_p99": 0.008672228254586231,
1117
+ "output_tps_per_user_avg": 121.68972103528682,
1118
+ "output_tps_per_user_p50": 121.68972103528682,
1119
+ "output_tps_per_user_p90": 126.88649961248755,
1120
+ "output_tps_per_user_p99": 128.0557747923577,
1121
+ "e2e_output_tps_per_user_avg": 0.0,
1122
+ "e2e_output_tps_per_user_p50": 0.0,
1123
+ "e2e_output_tps_per_user_p90": 0.0,
1124
+ "e2e_output_tps_per_user_p99": 0.0,
1125
+ "chunk_inter_token_latency_avg": 0.021374294281339835,
1126
+ "chunk_inter_token_latency_p50": 0.021374294281339835,
1127
+ "chunk_inter_token_latency_p90": 0.02164637385802789,
1128
+ "chunk_inter_token_latency_p99": 0.0217075917627827,
1129
+ "input_seq_len_avg": 32768.0,
1130
+ "output_seq_len_avg": 2865.0,
1131
+ "output_seq_len_p50": 2865.0,
1132
+ "output_seq_len_p90": 2953.0,
1133
+ "output_seq_len_p99": 2972.8,
1134
+ "request_count": 2,
1135
+ "completed_request_count": 0,
1136
+ "request_samples": [
1137
+ {
1138
+ "ttft": 0.6079938928596675,
1139
+ "time_to_second_token": 0.010162965161725879,
1140
+ "latency": 0.0,
1141
+ "inter_token_latency_avg": 0.008681026696140919,
1142
+ "chunk_inter_token_latency_avg": 0.0217143937521999,
1143
+ "input_tokens": 32768,
1144
+ "output_tokens": 2755,
1145
+ "output_tps_per_user": 115.1937478137859,
1146
+ "e2e_output_tps_per_user": 0.0,
1147
+ "completed": false
1148
+ },
1149
+ {
1150
+ "ttft": 1.3142956900410354,
1151
+ "time_to_second_token": 0.01823021797463298,
1152
+ "latency": 0.0,
1153
+ "inter_token_latency_avg": 0.007801182540672222,
1154
+ "chunk_inter_token_latency_avg": 0.021034194810479773,
1155
+ "input_tokens": 32768,
1156
+ "output_tokens": 2975,
1157
+ "output_tps_per_user": 128.18569425678774,
1158
+ "e2e_output_tps_per_user": 0.0,
1159
+ "completed": false
1160
+ }
1161
+ ],
1162
+ "total_tokens": 4794,
1163
+ "wall_time": 25.558490577852353,
1164
+ "num_completed": 2,
1165
+ "num_errors": 0,
1166
+ "server_gen_throughput": 239.6320204448516,
1167
+ "server_utilization": 0.013121161362367406,
1168
+ "server_spec_accept_rate": 0.5555555555555556,
1169
+ "server_spec_accept_length": 0.0,
1170
+ "avg_running_reqs": 2,
1171
+ "max_running_reqs": 2,
1172
+ "effective_concurrency": 2,
1173
+ "avg_queue_reqs": 0,
1174
+ "max_queue_reqs": 0,
1175
+ "queue_fraction": 0.0,
1176
+ "underfilled": false,
1177
+ "warmup_timed_out": false,
1178
+ "warmup_duration": 5.551,
1179
+ "ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
1180
+ "timeout_reason": "",
1181
+ "capacity_limited": false,
1182
+ "hardware_summary": {
1183
+ "samples": 9,
1184
+ "duration_seconds": 19.321,
1185
+ "gpu_count": 4,
1186
+ "cpu_util_avg_pct": 11.53,
1187
+ "cpu_temp_max_c": 76.88,
1188
+ "gpu_util_avg_pct": 100.0,
1189
+ "gpu_util_max_pct": 100.0,
1190
+ "mem_util_avg_pct": 40.97,
1191
+ "mem_util_max_pct": 50.0,
1192
+ "temp_avg_c": 68.61,
1193
+ "temp_max_c": 84.0,
1194
+ "power_total_avg_w": 1177.52,
1195
+ "power_total_max_w": 1178.4,
1196
+ "power_limit_total_w": 1200.0,
1197
+ "vram_used_avg_mb": 384778.0,
1198
+ "vram_used_max_mb": 384778.0,
1199
+ "vram_total_mb": 391548.0,
1200
+ "vram_used_avg_pct": 98.27,
1201
+ "vram_used_max_pct": 98.27,
1202
+ "pcie_rx_avg_mb_s": 11127.78,
1203
+ "pcie_rx_max_mb_s": 11295.0,
1204
+ "pcie_tx_avg_mb_s": 10971.33,
1205
+ "pcie_tx_max_mb_s": 11178.0
1206
+ }
1207
+ },
1208
+ {
1209
+ "concurrency": 4,
1210
+ "context_tokens": 32768,
1211
+ "benchmark_mode": "duration",
1212
+ "request_count_target": 0,
1213
+ "warmup_request_count": 0,
1214
+ "measurement_seconds": 19.994025,
1215
+ "measurement_wall_seconds": 20.000158,
1216
+ "client_output_tokens": 6042,
1217
+ "server_output_tokens": 6042,
1218
+ "aggregate_source": "openai_continuous_usage",
1219
+ "aggregate_tps": 302.19028317254777,
1220
+ "per_request_avg_tps": 75.54757079313694,
1221
+ "ttft_avg": 2.305535151215736,
1222
+ "ttft_p50": 2.2079672454856336,
1223
+ "ttft_p90": 3.5996002558851616,
1224
+ "ttft_p99": 4.136311722563113,
1225
+ "time_to_second_token_avg": 0.026320659555494785,
1226
+ "time_to_second_token_p50": 0.03037197550293058,
1227
+ "time_to_second_token_p90": 0.03217388866469264,
1228
+ "time_to_second_token_p99": 0.032832831228151914,
1229
+ "request_latency_avg": 0.0,
1230
+ "request_latency_p50": 0.0,
1231
+ "request_latency_p90": 0.0,
1232
+ "request_latency_p99": 0.0,
1233
+ "inter_token_latency_avg": 0.013299206291348701,
1234
+ "inter_token_latency_p50": 0.013299051093053463,
1235
+ "inter_token_latency_p90": 0.013574522517023369,
1236
+ "inter_token_latency_p99": 0.01365233632768284,
1237
+ "output_tps_per_user_avg": 75.22142959253767,
1238
+ "output_tps_per_user_p50": 75.19564602998912,
1239
+ "output_tps_per_user_p90": 76.78903624677658,
1240
+ "output_tps_per_user_p99": 77.24282693804788,
1241
+ "e2e_output_tps_per_user_avg": 0.0,
1242
+ "e2e_output_tps_per_user_p50": 0.0,
1243
+ "e2e_output_tps_per_user_p90": 0.0,
1244
+ "e2e_output_tps_per_user_p99": 0.0,
1245
+ "chunk_inter_token_latency_avg": 0.035561150767445995,
1246
+ "chunk_inter_token_latency_p50": 0.03564201678101365,
1247
+ "chunk_inter_token_latency_p90": 0.036069497146972274,
1248
+ "chunk_inter_token_latency_p99": 0.0361956293509813,
1249
+ "input_seq_len_avg": 32768.0,
1250
+ "output_seq_len_avg": 1899.0,
1251
+ "output_seq_len_p50": 1876.0,
1252
+ "output_seq_len_p90": 1995.4,
1253
+ "output_seq_len_p99": 2033.74,
1254
+ "request_count": 4,
1255
+ "completed_request_count": 0,
1256
+ "request_samples": [
1257
+ {
1258
+ "ttft": 0.6102597839199007,
1259
+ "time_to_second_token": 0.011632640147581697,
1260
+ "latency": 0.0,
1261
+ "inter_token_latency_avg": 0.013225319178200705,
1262
+ "chunk_inter_token_latency_avg": 0.03620964404031564,
1263
+ "input_tokens": 32768,
1264
+ "output_tokens": 2038,
1265
+ "output_tps_per_user": 75.61254186199908,
1266
+ "e2e_output_tps_per_user": 0.0,
1267
+ "completed": false
1268
+ },
1269
+ {
1270
+ "ttft": 2.2081260830163956,
1271
+ "time_to_second_token": 0.030465519055724144,
1272
+ "latency": 0.0,
1273
+ "inter_token_latency_avg": 0.013372783007906223,
1274
+ "chunk_inter_token_latency_avg": 0.03574248772917108,
1275
+ "input_tokens": 32768,
1276
+ "output_tokens": 1896,
1277
+ "output_tps_per_user": 74.77875019797916,
1278
+ "e2e_output_tps_per_user": 0.0,
1279
+ "completed": false
1280
+ },
1281
+ {
1282
+ "ttft": 2.2078084079548717,
1283
+ "time_to_second_token": 0.03027843195013702,
1284
+ "latency": 0.0,
1285
+ "inter_token_latency_avg": 0.013660982306645003,
1286
+ "chunk_inter_token_latency_avg": 0.035541545832856215,
1287
+ "input_tokens": 32768,
1288
+ "output_tokens": 1856,
1289
+ "output_tps_per_user": 73.20117818420553,
1290
+ "e2e_output_tps_per_user": 0.0,
1291
+ "completed": false
1292
+ },
1293
+ {
1294
+ "ttft": 4.195946329971775,
1295
+ "time_to_second_token": 0.03290604706853628,
1296
+ "latency": 0.0,
1297
+ "inter_token_latency_avg": 0.012937740672642875,
1298
+ "chunk_inter_token_latency_avg": 0.03475092546744106,
1299
+ "input_tokens": 32768,
1300
+ "output_tokens": 1806,
1301
+ "output_tps_per_user": 77.29324812596693,
1302
+ "e2e_output_tps_per_user": 0.0,
1303
+ "completed": false
1304
+ }
1305
+ ],
1306
+ "total_tokens": 6042,
1307
+ "wall_time": 28.607491098809987,
1308
+ "num_completed": 4,
1309
+ "num_errors": 0,
1310
+ "server_gen_throughput": 302.0169777555693,
1311
+ "server_utilization": 0.02540480178671134,
1312
+ "server_spec_accept_rate": 0.5201149425287356,
1313
+ "server_spec_accept_length": 0.0,
1314
+ "avg_running_reqs": 4,
1315
+ "max_running_reqs": 4,
1316
+ "effective_concurrency": 4,
1317
+ "avg_queue_reqs": 0,
1318
+ "max_queue_reqs": 0,
1319
+ "queue_fraction": 0.0,
1320
+ "underfilled": false,
1321
+ "warmup_timed_out": false,
1322
+ "warmup_duration": 8.578,
1323
+ "ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
1324
+ "timeout_reason": "",
1325
+ "capacity_limited": false,
1326
+ "hardware_summary": {
1327
+ "samples": 8,
1328
+ "duration_seconds": 16.85,
1329
+ "gpu_count": 4,
1330
+ "cpu_util_avg_pct": 11.53,
1331
+ "cpu_temp_max_c": 76.88,
1332
+ "gpu_util_avg_pct": 100.0,
1333
+ "gpu_util_max_pct": 100.0,
1334
+ "mem_util_avg_pct": 35.12,
1335
+ "mem_util_max_pct": 44.0,
1336
+ "temp_avg_c": 69.22,
1337
+ "temp_max_c": 84.0,
1338
+ "power_total_avg_w": 1172.46,
1339
+ "power_total_max_w": 1173.31,
1340
+ "power_limit_total_w": 1200.0,
1341
+ "vram_used_avg_mb": 384778.0,
1342
+ "vram_used_max_mb": 384778.0,
1343
+ "vram_total_mb": 391548.0,
1344
+ "vram_used_avg_pct": 98.27,
1345
+ "vram_used_max_pct": 98.27,
1346
+ "pcie_rx_avg_mb_s": 7907.88,
1347
+ "pcie_rx_max_mb_s": 8173.0,
1348
+ "pcie_tx_avg_mb_s": 7679.75,
1349
+ "pcie_tx_max_mb_s": 8296.0
1350
+ }
1351
+ }
1352
+ ],
1353
+ "summary_table": {
1354
+ "0": {
1355
+ "1": 175.1804006140207,
1356
+ "2": 236.39031276129717,
1357
+ "4": 296.6639108412111
1358
+ },
1359
+ "8192": {
1360
+ "1": 174.16263081468745,
1361
+ "2": 246.75295838278893,
1362
+ "4": 291.40641309659526
1363
+ },
1364
+ "32768": {
1365
+ "1": 179.316188111569,
1366
+ "2": 239.87458866483988,
1367
+ "4": 302.19028317254777
1368
+ }
1369
+ },
1370
+ "burst_results": [],
1371
+ "burst_summary_table": {},
1372
+ "methodology": {
1373
+ "prefill": {
1374
+ "name": "Prefill",
1375
+ "present": false,
1376
+ "mode": "skipped",
1377
+ "formula": "prompt_tokens / TTFT",
1378
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
1379
+ },
1380
+ "sustained_decode": {
1381
+ "name": "Sustained Decode",
1382
+ "present": true,
1383
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
1384
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
1385
+ },
1386
+ "burst_e2e_decode": {
1387
+ "name": "Burst / E2E Decode",
1388
+ "present": false,
1389
+ "status": "not run; use --run-burst",
1390
+ "formula": "sum(completion_tokens) / profiling_wall_time",
1391
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
1392
+ }
1393
+ }
1394
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.log ADDED
@@ -0,0 +1,123 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ New version available: v0.6.2 (current: v0.4.29)
3
+ Upgrade and restart? [Y/n]: Skipping update.
4
+
5
+ ╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
6
+ │ Effective: yes │
7
+ │ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
8
+ │ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
9
+ │ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
10
+ ╰──────────────────────────────────────────────────────────────────────────────╯
11
+ ╭─────────────────────────────── Configuration ────────────────────────────────╮
12
+ │ LLM Inference Benchmark │
13
+ │ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
14
+ │ Decode concurrency: [1, 2, 4] │
15
+ │ Decode contexts: ['0', '8k', '32k'] │
16
+ │ Duration: 20.0s per decode test | Max tokens: 8192 │
17
+ │ Pre-decode warmup: C=1 max-runnable context for 3s │
18
+ │ Prefill: skipped | Sustained decode: 9 cells │
19
+ ╰──────────────────────────────────────────────────────────────────────────────╯
20
+ Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
21
+ Models: ['glm53-flash-trellismx-p8-k45']
22
+ KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
23
+ 7,337,984 × CP 4; CP source: local process)
24
+ Model context length: 1,000,000 tokens
25
+ Prefill tests: skipped
26
+ Calibrating padding text (run=lijekedoguqy, up to 32k)...
27
+ 8k: 50,540 chars (8,192 prompt tokens via /tokenize)
28
+ 32k: 205,139 chars (32,768 prompt tokens via /tokenize)
29
+ Token targeting: /tokenize exact
30
+ Done.
31
+
32
+
33
+
34
+ llm-decode-bench v0.4.29
35
+ ╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
36
+ │ Sustained Decode │
37
+ │ Steady-state decode throughput after the engine has admitted the requested │
38
+ │ concurrency and passed warmup. Use this as the main tuning/regression signal │
39
+ │ for kernels, NCCL, DCP, MTP, and scheduler changes. │
40
+ ╰──────────────────────────────────────────────────────────────────────────────╯
41
+ Aggregate tok/s + TTFT/ITL
42
+ ╭────────────┬─────────────┬─────────────┬──────────────╮
43
+ │ ctx \ conc │ 1 │ 2 │ 4 │
44
+ ├────────────┼─────────────┼─────────────┼──────────────┤
45
+ │ 0 │ 175.2 70/6 │ 236.4 112/8 │ 296.7 173/13 │
46
+ │ 8k │ 174.2 591/6 │ 246.8 947/8 │ 291.4 2k/14 │
47
+ │ 32k │ 179.3 603/6 │ 239.9 961/8 │ 302.2 2k/13 │
48
+ ╰────────────┴─────────────┴─────────────┴──────────────╯
49
+ Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
50
+ (continuous completion_tokens when the server supports it). Prometheus is kept
51
+ as validation/scheduler data.
52
+ Aggregate source(s): openai_continuous_usage
53
+ Per-Request tok/s
54
+ ╭────────────┬───────┬───────┬──────╮
55
+ │ ctx \ conc │ 1 │ 2 │ 4 │
56
+ ├────────────┼───────┼───────┼──────┤
57
+ │ 0 │ 175.2 │ 118.2 │ 74.2 │
58
+ │ 8k │ 174.2 │ 123.4 │ 72.9 │
59
+ │ 32k │ 179.3 │ 119.9 │ 75.5 │
60
+ ╰────────────┴───────┴───────┴──────╯
61
+ Client request latency: p50 /
62
+ p90 ms
63
+ ╭────────────┬─────┬─────┬─────╮
64
+ │ ctx \ conc │ 1 │ 2 │ 4 │
65
+ ├────────────┼─────┼─────┼─────┤
66
+ │ 0 │ —/— │ —/— │ —/— │
67
+ │ 8k │ —/— │ —/— │ —/— │
68
+ │ 32k │ —/— │ —/— │ —/— │
69
+ ╰────────────┴─────┴─────┴─────╯
70
+ Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
71
+ coordinate. ITL is computed from observed generated tokens, including streams
72
+ stopped at the measurement boundary; a missing ITL means no stream produced at
73
+ least two measured output tokens. Per-request tok/s and request latency are
74
+ shown in separate per-cell matrices. Completion/sample counts and full
75
+ request-level distributions remain in JSON under request_samples.
76
+ Sustained mode: client latency metrics explain request UX variance; aggregate
77
+ tok/s remains the primary throughput signal.
78
+ ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
79
+ Hardware Summary
80
+ ╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
81
+ │ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
82
+ ├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
83
+ │ 0 │ │ sust… │ 99/99% │ 44% │ 1153/1… │ 82C │ 76C │ 98… │ 8331/8194 │
84
+ │ … │ │ sust… │ 99/99% │ 44% │ 1154/1… │ 83C │ 77C │ 98… │ 8357/8151 │
85
+ │ … │ │ sust… │ 99/99% │ 44% │ 1156/1… │ 83C │ 76C │ 98… │ 8340/8144 │
86
+ │ 0 │ │ sust… │ 100/100% │ 40% │ 1176/1… │ 84C │ 77C │ 98… │ 11179/11088 │
87
+ │ 0 │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 77C │ 98… │ 8004/7682 │
88
+ │ … │ │ sust… │ 100/100% │ 40% │ 1177/1… │ 84C │ 77C │ 98… │ 11225/11103 │
89
+ │ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 77C │ 98… │ 7728/7833 │
90
+ │ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 84C │ 77C │ 98… │ 11128/10971 │
91
+ │ … │ │ sust… │ 100/100% │ 35% │ 1172/1… │ 84C │ 77C │ 98… │ 7908/7680 │
92
+ ╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
93
+ ╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
94
+ │ avg 1,105 W | max 1,178 W | limit 1,200 W | over 4m 29s | 113 samples │
95
+ ╰───────────────────────────────────────────────────────────────────────╯
96
+ Hardware summary is sampled from nvidia-smi during the measured part of each
97
+ cell. Whole-run GPU power is the sampled sum of GPU power draw across the
98
+ complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
99
+ a coarse live diagnostic, not a per-kernel NCCL profiler.
100
+
101
+ ╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
102
+ │ Burst / E2E Decode │
103
+ │ Not run. Re-run with --run-burst to append a finite client-facing request │
104
+ │ burst after Sustained Decode. This is intentionally disabled by default │
105
+ │ because it adds another full decode matrix. │
106
+ ╰──────────────────────────────────────────────────────────────────────────────╯
107
+
108
+ ╭────────────────────────────── Primary Summary ───────────────────────────────╮
109
+ │ Primary matrices repeated last so the important numbers are visible without │
110
+ │ scrolling back through diagnostics. │
111
+ ╰─────────────────────────────────────��────────────────────────────────────────╯
112
+ Aggregate decode tok/s
113
+ ╭────────────┬───────┬───────┬───────╮
114
+ │ ctx \ conc │ 1 │ 2 │ 4 │
115
+ ├────────────┼───────┼───────┼───────┤
116
+ │ 0 │ 175.2 │ 236.4 │ 296.7 │
117
+ │ 8k │ 174.2 │ 246.8 │ 291.4 │
118
+ │ 32k │ 179.3 │ 239.9 │ 302.2 │
119
+ ╰────────────┴───────┴───────┴───────╯
120
+
121
+ Results saved to
122
+ <campaign>/candidate-speed-wi
123
+ ndow-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-command.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ "/usr/bin/python3",
3
+ "<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
4
+ "--host",
5
+ "127.0.0.1",
6
+ "--port",
7
+ "8001",
8
+ "--model",
9
+ "glm53-flash-trellismx-p8-k45",
10
+ "--duration",
11
+ "20",
12
+ "--max-tokens",
13
+ "8192",
14
+ "--token-targeting",
15
+ "exact",
16
+ "--display-mode",
17
+ "plain",
18
+ "--output",
19
+ "<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json",
20
+ "--contexts",
21
+ "0",
22
+ "--concurrency",
23
+ "1,2,4",
24
+ "--prefill-only",
25
+ "--prefill-contexts",
26
+ "8k,32k,64k,128k",
27
+ "--prefill-duration",
28
+ "20",
29
+ "--cell-warmup-timeout-seconds",
30
+ "180"
31
+ ]
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-receipt.json ADDED
@@ -0,0 +1,5 @@
 
 
 
 
 
 
1
+ {
2
+ "exit_code": 0,
3
+ "result_exists": true,
4
+ "sha256": "d0342998106c3361f6c09187c2c66554d2169d2910240fa71372ec70b061bdbb"
5
+ }
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json ADDED
@@ -0,0 +1,396 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "version": "0.4.29",
4
+ "engine": "vllm",
5
+ "model": "glm53-flash-trellismx-p8-k45",
6
+ "server": "127.0.0.1:8001",
7
+ "timestamp": "2026-09-09T02:28:52.054869",
8
+ "decode_mode": "duration",
9
+ "primary_decode_layer": "sustained_decode",
10
+ "duration_per_test": 20.0,
11
+ "request_count": 0,
12
+ "warmup_request_count": 0,
13
+ "run_burst": false,
14
+ "prefill_mode": "standalone_cold",
15
+ "standalone_prefill": true,
16
+ "prefill_only": true,
17
+ "skip_prefill": false,
18
+ "burst_e2e_status": "not_run_use_--run-burst",
19
+ "burst_request_count": 0,
20
+ "burst_warmup_request_count": 0,
21
+ "burst_requests_per_concurrency": 5,
22
+ "decode_warmup_seconds": 3.0,
23
+ "decode_warmup_context": 0,
24
+ "decode_warmup_concurrency": 1,
25
+ "cell_warmup_timeout_seconds": 180.0,
26
+ "cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
27
+ "show_capacity_limited_values": false,
28
+ "max_tokens": 8192,
29
+ "temperature": null,
30
+ "ignore_eos": true,
31
+ "max_total_tokens": 29351936,
32
+ "dcp_size": 0,
33
+ "metrics_available": true,
34
+ "metrics_warning": "",
35
+ "concurrency_levels": [
36
+ 1,
37
+ 2,
38
+ 4
39
+ ],
40
+ "context_lengths": [
41
+ 0
42
+ ],
43
+ "startup_diagnostics_available": true,
44
+ "nvidia_p2p_override_effective": true,
45
+ "p2pmark_status": "not_run",
46
+ "amd_fabric_status": "not_run"
47
+ },
48
+ "startup_diagnostics": {
49
+ "version": "0.4.29",
50
+ "server_url": "http://127.0.0.1:8001",
51
+ "hostname": "<host>",
52
+ "uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
53
+ "env": {},
54
+ "args": {
55
+ "concurrency": "1,2,4",
56
+ "contexts": "0",
57
+ "max_tokens": 8192,
58
+ "duration": 20.0,
59
+ "request_count": 0,
60
+ "run_burst": false,
61
+ "standalone_prefill": true,
62
+ "prefill_only": true,
63
+ "skip_prefill": false,
64
+ "prefill_contexts": "8k,32k,64k,128k",
65
+ "prefill_metric": "client",
66
+ "dcp_size": 0,
67
+ "kv_budget": 0
68
+ },
69
+ "nvidia_p2p_override": {
70
+ "effective": true,
71
+ "configured": true,
72
+ "params_path": "/proc/driver/nvidia/params",
73
+ "params_available": true,
74
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
75
+ "modprobe_available": true,
76
+ "runtime": {
77
+ "ForceP2P": "0x11",
78
+ "RMForceP2PType": "1",
79
+ "RMPcieP2PType": "2",
80
+ "GrdmaPciTopoCheckOverride": "1",
81
+ "EnableResizableBar": "1",
82
+ "DmaRemapPeerMmio": "1"
83
+ },
84
+ "expected": {
85
+ "ForceP2P": "0x11",
86
+ "RMForceP2PType": "1",
87
+ "RMPcieP2PType": "2",
88
+ "GrdmaPciTopoCheckOverride": "1",
89
+ "EnableResizableBar": "1"
90
+ },
91
+ "missing": [],
92
+ "mismatched": {},
93
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
94
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
95
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
96
+ },
97
+ "p2pmark": {
98
+ "status": "not_run"
99
+ },
100
+ "amd_fabric": {
101
+ "status": "not_run"
102
+ },
103
+ "nvidia_smi_query": {
104
+ "cmd": [
105
+ "nvidia-smi",
106
+ "--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
107
+ "--format=csv,noheader,nounits"
108
+ ],
109
+ "returncode": 0,
110
+ "stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
111
+ "stderr": ""
112
+ },
113
+ "nvidia_smi_topo": {
114
+ "cmd": [
115
+ "nvidia-smi",
116
+ "topo",
117
+ "-m"
118
+ ],
119
+ "returncode": 0,
120
+ "stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
121
+ "stderr": ""
122
+ }
123
+ },
124
+ "nvidia_p2p_override": {
125
+ "effective": true,
126
+ "configured": true,
127
+ "params_path": "/proc/driver/nvidia/params",
128
+ "params_available": true,
129
+ "modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
130
+ "modprobe_available": true,
131
+ "runtime": {
132
+ "ForceP2P": "0x11",
133
+ "RMForceP2PType": "1",
134
+ "RMPcieP2PType": "2",
135
+ "GrdmaPciTopoCheckOverride": "1",
136
+ "EnableResizableBar": "1",
137
+ "DmaRemapPeerMmio": "1"
138
+ },
139
+ "expected": {
140
+ "ForceP2P": "0x11",
141
+ "RMForceP2PType": "1",
142
+ "RMPcieP2PType": "2",
143
+ "GrdmaPciTopoCheckOverride": "1",
144
+ "EnableResizableBar": "1"
145
+ },
146
+ "missing": [],
147
+ "mismatched": {},
148
+ "registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
149
+ "suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
150
+ "suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
151
+ },
152
+ "p2pmark": {
153
+ "status": "not_run"
154
+ },
155
+ "amd_fabric": {
156
+ "status": "not_run"
157
+ },
158
+ "hardware_run_summary": {
159
+ "samples": 48,
160
+ "duration_seconds": 113.123,
161
+ "gpu_count": 4,
162
+ "cpu_util_avg_pct": 10.91,
163
+ "cpu_temp_max_c": 77.25,
164
+ "gpu_util_avg_pct": 88.84,
165
+ "gpu_util_max_pct": 100.0,
166
+ "mem_util_avg_pct": 19.54,
167
+ "mem_util_max_pct": 34.0,
168
+ "temp_avg_c": 67.36,
169
+ "temp_max_c": 84.0,
170
+ "power_total_avg_w": 1049.68,
171
+ "power_total_max_w": 1145.97,
172
+ "power_limit_total_w": 1200.0,
173
+ "vram_used_avg_mb": 384778.0,
174
+ "vram_used_max_mb": 384778.0,
175
+ "vram_total_mb": 391548.0,
176
+ "vram_used_avg_pct": 98.27,
177
+ "vram_used_max_pct": 98.27,
178
+ "pcie_rx_avg_mb_s": 39825.88,
179
+ "pcie_rx_max_mb_s": 55605.0,
180
+ "pcie_tx_avg_mb_s": 40385.83,
181
+ "pcie_tx_max_mb_s": 53259.0
182
+ },
183
+ "event_log": [],
184
+ "prefill": {
185
+ "8192": {
186
+ "ttft_seconds": 1.155,
187
+ "prefill_seconds": 1.155,
188
+ "tok_per_sec": 7097.0,
189
+ "client_ttft_seconds": 1.155,
190
+ "client_tok_per_sec": 7097.0,
191
+ "prompt_tokens": 8194,
192
+ "samples": 14,
193
+ "method": "client",
194
+ "server_validation": {
195
+ "method": "",
196
+ "tok_per_sec": 0.0,
197
+ "prefill_seconds": 0.0,
198
+ "prompt_tokens": 0,
199
+ "request_prompt_tokens": 0,
200
+ "cached_tokens": 0,
201
+ "token_source": "",
202
+ "samples": 0,
203
+ "invalid_reason": ""
204
+ },
205
+ "hardware_summary": {
206
+ "samples": 9,
207
+ "duration_seconds": 19.191,
208
+ "gpu_count": 4,
209
+ "cpu_util_avg_pct": 11.01,
210
+ "cpu_temp_max_c": 75.75,
211
+ "gpu_util_avg_pct": 76.69,
212
+ "gpu_util_max_pct": 100.0,
213
+ "mem_util_avg_pct": 18.22,
214
+ "mem_util_max_pct": 34.0,
215
+ "temp_avg_c": 65.86,
216
+ "temp_max_c": 80.0,
217
+ "power_total_avg_w": 987.33,
218
+ "power_total_max_w": 1139.37,
219
+ "power_limit_total_w": 1200.0,
220
+ "vram_used_avg_mb": 384778.0,
221
+ "vram_used_max_mb": 384778.0,
222
+ "vram_total_mb": 391548.0,
223
+ "vram_used_avg_pct": 98.27,
224
+ "vram_used_max_pct": 98.27,
225
+ "pcie_rx_avg_mb_s": 29085.56,
226
+ "pcie_rx_max_mb_s": 53572.0,
227
+ "pcie_tx_avg_mb_s": 29491.11,
228
+ "pcie_tx_max_mb_s": 53259.0
229
+ }
230
+ },
231
+ "32768": {
232
+ "ttft_seconds": 4.454,
233
+ "prefill_seconds": 4.454,
234
+ "tok_per_sec": 7357.0,
235
+ "client_ttft_seconds": 4.454,
236
+ "client_tok_per_sec": 7357.0,
237
+ "prompt_tokens": 32770,
238
+ "samples": 5,
239
+ "method": "client",
240
+ "server_validation": {
241
+ "method": "",
242
+ "tok_per_sec": 0.0,
243
+ "prefill_seconds": 0.0,
244
+ "prompt_tokens": 0,
245
+ "request_prompt_tokens": 0,
246
+ "cached_tokens": 0,
247
+ "token_source": "",
248
+ "samples": 0,
249
+ "invalid_reason": ""
250
+ },
251
+ "hardware_summary": {
252
+ "samples": 10,
253
+ "duration_seconds": 21.644,
254
+ "gpu_count": 4,
255
+ "cpu_util_avg_pct": 11.16,
256
+ "cpu_temp_max_c": 76.38,
257
+ "gpu_util_avg_pct": 90.4,
258
+ "gpu_util_max_pct": 100.0,
259
+ "mem_util_avg_pct": 19.95,
260
+ "mem_util_max_pct": 30.0,
261
+ "temp_avg_c": 67.0,
262
+ "temp_max_c": 83.0,
263
+ "power_total_avg_w": 1056.58,
264
+ "power_total_max_w": 1144.59,
265
+ "power_limit_total_w": 1200.0,
266
+ "vram_used_avg_mb": 384778.0,
267
+ "vram_used_max_mb": 384778.0,
268
+ "vram_total_mb": 391548.0,
269
+ "vram_used_avg_pct": 98.27,
270
+ "vram_used_max_pct": 98.27,
271
+ "pcie_rx_avg_mb_s": 45582.6,
272
+ "pcie_rx_max_mb_s": 51598.0,
273
+ "pcie_tx_avg_mb_s": 44600.2,
274
+ "pcie_tx_max_mb_s": 48299.0
275
+ }
276
+ },
277
+ "65536": {
278
+ "ttft_seconds": 8.891,
279
+ "prefill_seconds": 8.891,
280
+ "tok_per_sec": 7371.0,
281
+ "client_ttft_seconds": 8.891,
282
+ "client_tok_per_sec": 7371.0,
283
+ "prompt_tokens": 65538,
284
+ "samples": 3,
285
+ "method": "client",
286
+ "server_validation": {
287
+ "method": "",
288
+ "tok_per_sec": 0.0,
289
+ "prefill_seconds": 0.0,
290
+ "prompt_tokens": 0,
291
+ "request_prompt_tokens": 0,
292
+ "cached_tokens": 0,
293
+ "token_source": "",
294
+ "samples": 0,
295
+ "invalid_reason": ""
296
+ },
297
+ "hardware_summary": {
298
+ "samples": 12,
299
+ "duration_seconds": 26.544,
300
+ "gpu_count": 4,
301
+ "cpu_util_avg_pct": 11.23,
302
+ "cpu_temp_max_c": 76.75,
303
+ "gpu_util_avg_pct": 97.67,
304
+ "gpu_util_max_pct": 100.0,
305
+ "mem_util_avg_pct": 21.02,
306
+ "mem_util_max_pct": 29.0,
307
+ "temp_avg_c": 68.04,
308
+ "temp_max_c": 83.0,
309
+ "power_total_avg_w": 1074.72,
310
+ "power_total_max_w": 1145.97,
311
+ "power_limit_total_w": 1200.0,
312
+ "vram_used_avg_mb": 384778.0,
313
+ "vram_used_max_mb": 384778.0,
314
+ "vram_total_mb": 391548.0,
315
+ "vram_used_avg_pct": 98.27,
316
+ "vram_used_max_pct": 98.27,
317
+ "pcie_rx_avg_mb_s": 42799.33,
318
+ "pcie_rx_max_mb_s": 54328.0,
319
+ "pcie_tx_avg_mb_s": 43858.08,
320
+ "pcie_tx_max_mb_s": 53032.0
321
+ }
322
+ },
323
+ "131072": {
324
+ "ttft_seconds": 17.957,
325
+ "prefill_seconds": 17.957,
326
+ "tok_per_sec": 7299.0,
327
+ "client_ttft_seconds": 17.957,
328
+ "client_tok_per_sec": 7299.0,
329
+ "prompt_tokens": 131073,
330
+ "samples": 2,
331
+ "method": "client",
332
+ "server_validation": {
333
+ "method": "",
334
+ "tok_per_sec": 0.0,
335
+ "prefill_seconds": 0.0,
336
+ "prompt_tokens": 0,
337
+ "request_prompt_tokens": 0,
338
+ "cached_tokens": 0,
339
+ "token_source": "",
340
+ "samples": 0,
341
+ "invalid_reason": ""
342
+ },
343
+ "hardware_summary": {
344
+ "samples": 15,
345
+ "duration_seconds": 33.726,
346
+ "gpu_count": 4,
347
+ "cpu_util_avg_pct": 11.25,
348
+ "cpu_temp_max_c": 77.25,
349
+ "gpu_util_avg_pct": 99.87,
350
+ "gpu_util_max_pct": 100.0,
351
+ "mem_util_avg_pct": 21.47,
352
+ "mem_util_max_pct": 29.0,
353
+ "temp_avg_c": 68.65,
354
+ "temp_max_c": 84.0,
355
+ "power_total_avg_w": 1123.91,
356
+ "power_total_max_w": 1145.94,
357
+ "power_limit_total_w": 1200.0,
358
+ "vram_used_avg_mb": 384778.0,
359
+ "vram_used_max_mb": 384778.0,
360
+ "vram_total_mb": 391548.0,
361
+ "vram_used_avg_pct": 98.27,
362
+ "vram_used_max_pct": 98.27,
363
+ "pcie_rx_avg_mb_s": 43837.13,
364
+ "pcie_rx_max_mb_s": 55605.0,
365
+ "pcie_tx_avg_mb_s": 44351.67,
366
+ "pcie_tx_max_mb_s": 52140.0
367
+ }
368
+ }
369
+ },
370
+ "results": [],
371
+ "summary_table": {},
372
+ "burst_results": [],
373
+ "burst_summary_table": {},
374
+ "methodology": {
375
+ "prefill": {
376
+ "name": "Prefill",
377
+ "present": true,
378
+ "mode": "standalone_cold",
379
+ "formula": "prompt_tokens / TTFT",
380
+ "notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
381
+ },
382
+ "sustained_decode": {
383
+ "name": "Sustained Decode",
384
+ "present": false,
385
+ "formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
386
+ "notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
387
+ },
388
+ "burst_e2e_decode": {
389
+ "name": "Burst / E2E Decode",
390
+ "present": false,
391
+ "status": "not run; use --run-burst",
392
+ "formula": "sum(completion_tokens) / profiling_wall_time",
393
+ "notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
394
+ }
395
+ }
396
+ }