Publish selected TP4 reference, full speed evidence and matched FP8/NVFP4 cache KLD
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- README.md +181 -58
- compose.yaml +8 -3
- image-record.json +34 -126
- results/image-record-historical-20260908.json +217 -0
- results/kld-reference-20260909/README.md +33 -0
- results/kld-reference-20260909/audit.json +30 -0
- results/kld-reference-20260909/comparison.json +1053 -0
- results/speed-20260909/README.md +40 -0
- results/speed-20260909/benchmark-index.json +0 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-command.json +27 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-receipt.json +5 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json +1394 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.log +123 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-command.json +31 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-receipt.json +5 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json +396 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.log +56 -0
- results/speed-20260909/evidence/batch16384-speed-window-01/results-01/failure.json +3 -0
- results/speed-20260909/evidence/candidate-graph-analysis-02/REPORT.md +20 -0
- results/speed-20260909/evidence/candidate-graph-analysis-02/SUMMARY.json +242 -0
- results/speed-20260909/evidence/candidate-graph-profile-01/results-01/failure.json +3 -0
- results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c1-8k/result.json +345 -0
- results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c4-8k/result.json +381 -0
- results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-32k/result.json +257 -0
- results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-64k/result.json +257 -0
- results/speed-20260909/evidence/candidate-graph-profile-02/results-01/result.json +7 -0
- results/speed-20260909/evidence/candidate-kernel-window-02/results-01/result.json +23 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-command.json +27 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json +2366 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.log +126 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-command.json +27 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json +1394 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.log +123 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-command.json +31 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json +396 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.log +56 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-command.json +27 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json +2342 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.log +126 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-command.json +27 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json +1394 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.log +123 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-command.json +31 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-receipt.json +5 -0
- results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json +396 -0
README.md
CHANGED
|
@@ -18,13 +18,175 @@ tags:
|
|
| 18 |
|
| 19 |
# GLM-5.3-Flash TrellisMX-MXFP8
|
| 20 |
|
| 21 |
-
|
| 22 |
-
|
|
|
|
|
|
|
| 23 |
|
| 24 |
-
|
| 25 |
-
|
| 26 |
-
|
| 27 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 28 |
|
| 29 |
## What TrellisMX does
|
| 30 |
|
|
@@ -111,61 +273,20 @@ Existing prior art makes that broad wording inappropriate. The distinctive
|
|
| 111 |
combination above describes this implementation without asserting an
|
| 112 |
unverified first-in-history claim or general superiority over its sources.
|
| 113 |
|
| 114 |
-
|
| 115 |
|
| 116 |
-
|
| 117 |
-
bytes)** plus serving metadata, scripts, license and evaluation receipts.
|
| 118 |
-
They replace the carrier's routed experts at load time. They are not an
|
| 119 |
-
additional expert ensemble, and this is **not a standalone Transformers
|
| 120 |
-
checkpoint**. Do not point stock Transformers or unmodified vLLM at it.
|
| 121 |
|
| 122 |
-
|
| 123 |
-
|
| 124 |
-
revision `520de24eabf507659eaef7c70f14fd584527facc`. Its attention, shared
|
| 125 |
-
experts, embeddings, output head and MTP components remain part of the served
|
| 126 |
-
model. Keeping this exact dependency preserves the existing loader contract;
|
| 127 |
-
the reported logical model payload is not a claim that the two download
|
| 128 |
-
directories together occupy only that many bytes.
|
| 129 |
|
| 130 |
-
The
|
| 131 |
-
|
|
|
|
|
|
|
|
|
|
| 132 |
|
| 133 |
-
|
| 134 |
-
verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f
|
| 135 |
-
```
|
| 136 |
-
|
| 137 |
-
The image stays in the container registry; this card links its immutable
|
| 138 |
-
digest and includes `compose.yaml` and `serve.sh`. No encoder, calibration
|
| 139 |
-
corpus or teacher logits are included in this HF release.
|
| 140 |
-
|
| 141 |
-
## Serving
|
| 142 |
-
|
| 143 |
-
Requires four RTX PRO 6000 Blackwell 96GB SM120 GPUs and the NVIDIA Container
|
| 144 |
-
Toolkit. The current r27 recipe uses TP4/DCP4, probabilistic MTP3 with standard rejection, B12X_MLA_SPARSE attention,
|
| 145 |
-
NVFP4 MLA KV (`nvfp4_ds_mla`), native P8 MoE with E4M3 activations, CUDA graphs,
|
| 146 |
-
B12X PCIe collectives, prefix caching, a 4096-token scheduler batch and at most 16 sequences. GPU memory utilization is 0.97.
|
| 147 |
-
It does not alter GPU power limits, memory clocks or host services.
|
| 148 |
-
|
| 149 |
-
After `release-status.json` reports upload completion:
|
| 150 |
-
|
| 151 |
-
```bash
|
| 152 |
-
hf download brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8 --local-dir ./trellismx-p8
|
| 153 |
-
hf download local-inference-lab/GLM-5.3-Flash-NVFP4 \
|
| 154 |
-
--revision 520de24eabf507659eaef7c70f14fd584527facc --local-dir ./glm53-carrier
|
| 155 |
-
cd trellismx-p8
|
| 156 |
-
export MODEL_ROOT=/absolute/path/to/glm53-carrier
|
| 157 |
-
docker compose -f compose.yaml config --quiet
|
| 158 |
-
docker compose -f compose.yaml up -d
|
| 159 |
-
```
|
| 160 |
-
|
| 161 |
-
Default port: 8000. Default maximum model length: **1,000,000 tokens**,
|
| 162 |
-
within the model's declared 1,048,576 positions. That new launch limit is
|
| 163 |
-
**not a tested 1M-context accuracy claim**. The HF-layout launch has CPU
|
| 164 |
-
configuration checks only; a clean GPU download-to-serving test is pending.
|
| 165 |
-
Use a firewall or authenticated gateway before exposing the endpoint beyond
|
| 166 |
-
a trusted network. API credentials are not baked into the image.
|
| 167 |
-
|
| 168 |
-
## Current r27 DCP4 results
|
| 169 |
|
| 170 |
| Measurement | Result | Conditions |
|
| 171 |
| --- | --- | --- |
|
|
@@ -190,7 +311,7 @@ client snapshot and redaction hashes](results/r27-20260908/README.md). The
|
|
| 190 |
benchmark client's larger generic capacity estimate is not valid for this
|
| 191 |
split-cache layout.
|
| 192 |
|
| 193 |
-
###
|
| 194 |
|
| 195 |
| Runtime / measurement image | KV dtype | Attention backend | Mean true-decode KLD | Window BCa 95% interval |
|
| 196 |
| --- | --- | --- | ---: | --- |
|
|
@@ -301,6 +422,8 @@ not reported by this release; no values are inferred for missing metrics.
|
|
| 301 |
See the [full GitHub results and receipts](https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/blob/03527b1f092cfbf274ae3ca78cabb23077c37057/results/RC5_RELEASE_RESULTS_20260907.md),
|
| 302 |
and the included `results/` and `evidence/` files.
|
| 303 |
|
|
|
|
|
|
|
| 304 |
## Codec and limitations
|
| 305 |
|
| 306 |
Procedural MCG trellis streams decode in registers to E4M3 for native
|
|
|
|
| 18 |
|
| 19 |
# GLM-5.3-Flash TrellisMX-MXFP8
|
| 20 |
|
| 21 |
+
A compressed GLM-5.3-Flash checkpoint whose routed experts reconstruct directly
|
| 22 |
+
into **FP8 Tensor Core operands**. The September 9 reference stack is the selected
|
| 23 |
+
runtime: **222.100 tokens/s C1 decode at 8K** and **8,457 tokens/s prefill at 32K**
|
| 24 |
+
in separate `llm_decode_bench` runs on four RTX PRO 6000 Blackwell GPUs at 300W each.
|
| 25 |
|
| 26 |
+
## Recommended reference stack — September 9, 2026
|
| 27 |
+
|
| 28 |
+
[Docker Hub](https://hub.docker.com/r/verdictai/trellismx) image:
|
| 29 |
+
|
| 30 |
+
```text
|
| 31 |
+
verdictai/trellismx:glm53-flash-p8-r27-reference-20260909
|
| 32 |
+
```
|
| 33 |
+
|
| 34 |
+
Immutable image reference:
|
| 35 |
+
|
| 36 |
+
```text
|
| 37 |
+
verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf
|
| 38 |
+
```
|
| 39 |
+
|
| 40 |
+
Use the supplied [Docker Compose configuration](compose.yaml) and
|
| 41 |
+
[serving script](serve.sh). The [reference runtime recipe](runtime-reference-20260909/Dockerfile)
|
| 42 |
+
and [source manifest](runtime-reference-20260909/source-manifest.json) record the
|
| 43 |
+
inference overlays. The checkpoint has not been re-encoded for this update.
|
| 44 |
+
|
| 45 |
+
| Setting | Selected reference |
|
| 46 |
+
| --- | --- |
|
| 47 |
+
| GPUs / measured power | 4 × RTX PRO 6000 Blackwell 96GB (SM120), 300W per GPU |
|
| 48 |
+
| Parallelism | TP4 / DCP4 |
|
| 49 |
+
| Routed-expert math | E4M3 FP8, native UE8M0 scales per 32 elements |
|
| 50 |
+
| MLA KV cache | NVFP4 (`nvfp4_ds_mla`) |
|
| 51 |
+
| Speculative decoding | Probabilistic MTP3, standard rejection |
|
| 52 |
+
| Execution | CUDA graphs, B12X MLA attention and PCIe collectives |
|
| 53 |
+
| Scheduler | 24 sequence slots, 4,096-token batch |
|
| 54 |
+
| Memory / maximum request length | 0.97 GPU memory utilization / 1,000,000 tokens |
|
| 55 |
+
|
| 56 |
+
**FP8 math and NVFP4 KV cache describe different parts of the model.** Compressed
|
| 57 |
+
K4/K5 expert weights decode to FP8 operands for multiplication. NVFP4 stores the
|
| 58 |
+
MLA attention cache. Choosing FP8 KV instead changes the cache representation;
|
| 59 |
+
it does not turn this into an FP4-math checkpoint. The P4 design is separate.
|
| 60 |
+
|
| 61 |
+
Weight upload is complete: the remote inventory and all **168 sidecar SHA-256
|
| 62 |
+
identities** were verified. Local serving and measurements use the same checkpoint.
|
| 63 |
+
A separate clean download-to-GPU test of the HF directory layout has not been run;
|
| 64 |
+
see [release status](release-status.json).
|
| 65 |
+
|
| 66 |
+
## Measured speed and KV capacity
|
| 67 |
+
|
| 68 |
+
Rates below are **tokens/s from `llm_decode_bench`**. C1 is one request;
|
| 69 |
+
C4 and C8 are aggregate throughput across concurrent requests, not per user.
|
| 70 |
+
Context labels are nominal benchmark cells; raw logs retain actual prompt sizes.
|
| 71 |
+
|
| 72 |
+
| Context | Prefill | C1 decode | C4 decode, aggregate | C8 decode, aggregate |
|
| 73 |
+
| --- | ---: | ---: | ---: | ---: |
|
| 74 |
+
| 0K | — | 204.611 | 327.670 | — |
|
| 75 |
+
| 8K | — | 222.100 | 318.560 | 594.639 |
|
| 76 |
+
| 16K | — | 216.636 | 319.157 | 602.965 |
|
| 77 |
+
| 32K | 8,457 | 204.930 | 329.343 | — |
|
| 78 |
+
| 64K | 8,443 | 199.446 | 330.096 | — |
|
| 79 |
+
| 128K | 8,323 | 204.445 | 334.086 | — |
|
| 80 |
+
|
| 81 |
+
This table combines the short-context screen with the separate expanded-context
|
| 82 |
+
pass. The earlier comparison screen measured **8,407 tokens/s prefill at both
|
| 83 |
+
32K and 64K**. All runs are retained rather than selecting one figure as a repeated-run average.
|
| 84 |
+
The reference was selected for balanced use: it had the highest observed C1
|
| 85 |
+
throughput at 0K and 8K among the three finalists. Task-count performed better
|
| 86 |
+
at 16K C4 (**353.532 tokens/s**), but neither alternative was measured at C1
|
| 87 |
+
32K–128K. Reference is not established as the fastest candidate for every workload.
|
| 88 |
+
|
| 89 |
+
The engine reported **23,562,091 aggregate effective KV tokens** in the expanded
|
| 90 |
+
reference run; separate short-context sessions reported **23,568,627**. This is
|
| 91 |
+
allocated capacity across requests, not a tested single-request context length
|
| 92 |
+
or a full-capacity stress result. The configured request limit is 1,000,000 tokens.
|
| 93 |
+
Use the engine's split-cache metric, not the benchmark client's generic
|
| 94 |
+
block-count-times-DCP estimate.
|
| 95 |
+
|
| 96 |
+
These are exploratory measurements, with one server preparation per configuration.
|
| 97 |
+
Later screens used a minimum 90-second idle period and required all GPUs at or below
|
| 98 |
+
55°C for 30 continuous seconds before each cell; earlier runs retain their original
|
| 99 |
+
protocols. MTP acceptance, generated output and clocks can vary. No independently
|
| 100 |
+
repeated speed qualification or long-context accuracy claim is implied.
|
| 101 |
+
|
| 102 |
+
[All September 9 results, candidate comparison and protocol](results/speed-20260909/README.md)
|
| 103 |
+
include the full benchmark index, JSON, logs, negative results and diagnostic runs.
|
| 104 |
+
Profiled diagnostics are identified separately from serving speed measurements.
|
| 105 |
+
|
| 106 |
+
## Reference-stack KLD: matched FP8 and NVFP4 MLA cache
|
| 107 |
+
|
| 108 |
+
Lower KLD means the student's next-token distribution is closer to the BF16
|
| 109 |
+
teacher on the measured inputs. It is not a direct score for general answer quality.
|
| 110 |
+
|
| 111 |
+
| Cache on the September 9 reference | Mean true-decode KLD | Status |
|
| 112 |
+
| --- | ---: | --- |
|
| 113 |
+
| FP8 KV | **0.0319451732** | 32/32; receipt audit passed |
|
| 114 |
+
| NVFP4 MLA KV | **0.0354562238** | 32/32; receipt audit passed |
|
| 115 |
+
|
| 116 |
+
FP8 has lower observed KLD in **22/32 windows**. The paired mean difference
|
| 117 |
+
(FP8 minus NVFP4) is **−0.0035110506**, with a window BCa 95% interval
|
| 118 |
+
[−0.0081718772, −0.0013675941]. [Full results and audit](results/kld-reference-20260909/README.md).
|
| 119 |
+
|
| 120 |
+
Both arms use the **same 32 previously opened conditional-fit development windows**
|
| 121 |
+
as the prior measurement, with the same token arrays and BF16 teacher. Each window
|
| 122 |
+
contains 2,048 input tokens and 2,047 prediction rows; row zero is excluded, leaving
|
| 123 |
+
**2,046 true-decode rows per window**. Scores use CPU FP64 KL(teacher || student)
|
| 124 |
+
and the equal mean of the 32 window means.
|
| 125 |
+
|
| 126 |
+
The matched setup is TP4/DCP4, **MTP off**, one sequence, a 4,096-token batch and
|
| 127 |
+
zero prefix-cache hits. FP8 runs first, then NVFP4. The logits-capture image is
|
| 128 |
+
`sha256:0405a1c0dc128b51069798a5d00b346257bbd006c0deb7e6530a3d005d75de71`;
|
| 129 |
+
it adds the capture/warmup seam to the selected serving image and is not the public
|
| 130 |
+
serving tag. This quality measurement therefore does not measure the MTP3 speed
|
| 131 |
+
configuration's complete output behavior.
|
| 132 |
+
|
| 133 |
+
The paired result and confidence interval passed the receipt/score audit. These
|
| 134 |
+
already-opened windows provide a development comparison, **not untouched final qualification**.
|
| 135 |
+
No matched claim against EXL3, TR3 4bpw or another weight format is established by
|
| 136 |
+
this cache comparison.
|
| 137 |
+
|
| 138 |
+
The archived historical `.0341811459` cache label is under provenance review:
|
| 139 |
+
the prior card labels it NVFP4, while the owner's historical account identifies
|
| 140 |
+
FP8. The prior text and linked receipts are retained below without silently
|
| 141 |
+
changing either number. Do not use that disputed historical label as a matched
|
| 142 |
+
cache comparison. New reference results are recorded separately above.
|
| 143 |
+
|
| 144 |
+
## Run the model
|
| 145 |
+
|
| 146 |
+
Requires the custom runtime, four supported GPUs and NVIDIA Container Toolkit.
|
| 147 |
+
This is a sidecar checkpoint used with a separate carrier model; stock Transformers
|
| 148 |
+
or unmodified vLLM cannot load it as a standalone model.
|
| 149 |
+
|
| 150 |
+
```bash
|
| 151 |
+
hf download brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8 --local-dir ./trellismx-p8
|
| 152 |
+
hf download local-inference-lab/GLM-5.3-Flash-NVFP4 \
|
| 153 |
+
--revision 520de24eabf507659eaef7c70f14fd584527facc --local-dir ./glm53-carrier
|
| 154 |
+
cd trellismx-p8
|
| 155 |
+
export MODEL_ROOT=/absolute/path/to/glm53-carrier
|
| 156 |
+
docker compose -f compose.yaml config --quiet
|
| 157 |
+
docker compose -f compose.yaml up -d
|
| 158 |
+
```
|
| 159 |
+
|
| 160 |
+
The default endpoint is port 8000. The scripts do not set GPU power limits or
|
| 161 |
+
memory clocks. The measured 300W per GPU setting is part of the benchmark setup,
|
| 162 |
+
not a guarantee that a new host will reproduce its speed. The 1,000,000-token
|
| 163 |
+
launch limit is within the model's declared 1,048,576 positions; it is not a
|
| 164 |
+
measured 1M-context accuracy result. API credentials are not baked into the image.
|
| 165 |
+
|
| 166 |
+
## Required runtime and model layout
|
| 167 |
+
|
| 168 |
+
This repository contains **168 TP4 routed-expert sidecars (177,269,057,440
|
| 169 |
+
bytes)** plus serving metadata, scripts, license and evaluation receipts.
|
| 170 |
+
They replace the carrier's routed experts at load time. They are not an
|
| 171 |
+
additional expert ensemble, and this is **not a standalone Transformers
|
| 172 |
+
checkpoint**. Do not point stock Transformers or unmodified vLLM at it.
|
| 173 |
+
|
| 174 |
+
The working runtime additionally requires the stock carrier:
|
| 175 |
+
[`local-inference-lab/GLM-5.3-Flash-NVFP4`](https://huggingface.co/local-inference-lab/GLM-5.3-Flash-NVFP4/tree/520de24eabf507659eaef7c70f14fd584527facc),
|
| 176 |
+
revision `520de24eabf507659eaef7c70f14fd584527facc`. Its attention, shared
|
| 177 |
+
experts, embeddings, output head and MTP components remain part of the served
|
| 178 |
+
model. Keeping this exact dependency preserves the existing loader contract;
|
| 179 |
+
the reported logical model payload is not a claim that the two download
|
| 180 |
+
directories together occupy only that many bytes.
|
| 181 |
+
|
| 182 |
+
## Checkpoint and implementation details
|
| 183 |
+
|
| 184 |
+
This is the **17-K5 / 25-K4 coupled checkpoint**, covering all 42 routed layers
|
| 185 |
+
(3–44). Routed tensors occupy **4.6587417643 bits/weight including metadata**;
|
| 186 |
+
this is not a uniform 4.25 bpw model.
|
| 187 |
+
|
| 188 |
+
<details>
|
| 189 |
+
<summary>How the compressed weights, FP8 math and prior work fit together</summary>
|
| 190 |
|
| 191 |
## What TrellisMX does
|
| 192 |
|
|
|
|
| 273 |
combination above describes this implementation without asserting an
|
| 274 |
unverified first-in-history claim or general superiority over its sources.
|
| 275 |
|
| 276 |
+
</details>
|
| 277 |
|
| 278 |
+
## Earlier measurements and provenance
|
|
|
|
|
|
|
|
|
|
|
|
|
| 279 |
|
| 280 |
+
<details>
|
| 281 |
+
<summary>September 8, RC5 and DCP1 results — archived runtime settings and receipts</summary>
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 282 |
|
| 283 |
+
The text below preserves the prior release's results and links. References to
|
| 284 |
+
“current” or “production” inside this archive describe the earlier release,
|
| 285 |
+
not the September 9 reference above. In particular, the old 16-sequence recipe
|
| 286 |
+
and old image digest are superseded. The historical `.0341811459` cache label
|
| 287 |
+
is disputed as described above; retaining the original text is not a resolution.
|
| 288 |
|
| 289 |
+
## September 8 r27 DCP4 results (superseded runtime)
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 290 |
|
| 291 |
| Measurement | Result | Conditions |
|
| 292 |
| --- | --- | --- |
|
|
|
|
| 311 |
benchmark client's larger generic capacity estimate is not valid for this
|
| 312 |
split-cache layout.
|
| 313 |
|
| 314 |
+
### September 8 r27 KV-cache KLD comparison
|
| 315 |
|
| 316 |
| Runtime / measurement image | KV dtype | Attention backend | Mean true-decode KLD | Window BCa 95% interval |
|
| 317 |
| --- | --- | --- | ---: | --- |
|
|
|
|
| 422 |
See the [full GitHub results and receipts](https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/blob/03527b1f092cfbf274ae3ca78cabb23077c37057/results/RC5_RELEASE_RESULTS_20260907.md),
|
| 423 |
and the included `results/` and `evidence/` files.
|
| 424 |
|
| 425 |
+
</details>
|
| 426 |
+
|
| 427 |
## Codec and limitations
|
| 428 |
|
| 429 |
Procedural MCG trellis streams decode in registers to E4M3 for native
|
compose.yaml
CHANGED
|
@@ -1,12 +1,12 @@
|
|
| 1 |
services:
|
| 2 |
trellismx:
|
| 3 |
-
image: verdictai/trellismx:glm53-flash-p8-r27-
|
| 4 |
restart: "no"
|
| 5 |
network_mode: host
|
| 6 |
ipc: host
|
| 7 |
gpus: all
|
| 8 |
working_dir: /
|
| 9 |
-
entrypoint: ["/bin/bash", "/
|
| 10 |
environment:
|
| 11 |
PORT: "8000"
|
| 12 |
SERVED_MODEL_NAME: "glm53-flash-trellismx-p8-k45"
|
|
@@ -19,11 +19,16 @@ services:
|
|
| 19 |
NUM_SPECULATIVE_TOKENS: "3"
|
| 20 |
KV_CACHE_DTYPE: "nvfp4_ds_mla"
|
| 21 |
MAX_MODEL_LEN: "1000000"
|
| 22 |
-
MAX_NUM_SEQS: "
|
| 23 |
MAX_NUM_BATCHED_TOKENS: "4096"
|
| 24 |
GPU_MEMORY_UTILIZATION: "0.97"
|
| 25 |
CP_KV_CACHE_INTERLEAVE_SIZE: "4"
|
| 26 |
DCP_CKV_GATHER: "auto"
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 27 |
VLLM_NO_USAGE_STATS: "1"
|
| 28 |
volumes:
|
| 29 |
- ${MODEL_ROOT:?Set MODEL_ROOT to the pinned carrier directory}:/model:ro
|
|
|
|
| 1 |
services:
|
| 2 |
trellismx:
|
| 3 |
+
image: verdictai/trellismx:glm53-flash-p8-r27-reference-20260909@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf
|
| 4 |
restart: "no"
|
| 5 |
network_mode: host
|
| 6 |
ipc: host
|
| 7 |
gpus: all
|
| 8 |
working_dir: /
|
| 9 |
+
entrypoint: ["/bin/bash", "/release/serve-r27-production.sh"]
|
| 10 |
environment:
|
| 11 |
PORT: "8000"
|
| 12 |
SERVED_MODEL_NAME: "glm53-flash-trellismx-p8-k45"
|
|
|
|
| 19 |
NUM_SPECULATIVE_TOKENS: "3"
|
| 20 |
KV_CACHE_DTYPE: "nvfp4_ds_mla"
|
| 21 |
MAX_MODEL_LEN: "1000000"
|
| 22 |
+
MAX_NUM_SEQS: "24"
|
| 23 |
MAX_NUM_BATCHED_TOKENS: "4096"
|
| 24 |
GPU_MEMORY_UTILIZATION: "0.97"
|
| 25 |
CP_KV_CACHE_INTERLEAVE_SIZE: "4"
|
| 26 |
DCP_CKV_GATHER: "auto"
|
| 27 |
+
NCCL_MIN_NCHANNELS: "8"
|
| 28 |
+
NCCL_MAX_NCHANNELS: "8"
|
| 29 |
+
VLLM_PCIE_ONESHOT_ALLREDUCE_MAX_SIZE: "131072"
|
| 30 |
+
VLLM_PCIE_ONESHOT_FUSED_ADD_RMS_NORM_MAX_SIZE: "86016"
|
| 31 |
+
VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD: "4096"
|
| 32 |
VLLM_NO_USAGE_STATS: "1"
|
| 33 |
volumes:
|
| 34 |
- ${MODEL_ROOT:?Set MODEL_ROOT to the pinned carrier directory}:/model:ro
|
image-record.json
CHANGED
|
@@ -1,8 +1,8 @@
|
|
| 1 |
{
|
| 2 |
"schema_version": 4,
|
| 3 |
-
"record_id": "trellismx-
|
| 4 |
-
"title": "TrellisMX
|
| 5 |
-
"summary": "
|
| 6 |
"model_family": "GLM-5.3-Flash",
|
| 7 |
"release_class": "experimental",
|
| 8 |
"distribution_role": "custom",
|
|
@@ -10,9 +10,9 @@
|
|
| 10 |
"maintenance_status": "ephemeral",
|
| 11 |
"image": {
|
| 12 |
"repository": "verdictai/trellismx",
|
| 13 |
-
"tag": "glm53-flash-p8-r27-
|
| 14 |
-
"digest": "sha256:
|
| 15 |
-
"reference": "verdictai/trellismx:glm53-flash-p8-r27-
|
| 16 |
},
|
| 17 |
"base_image": {
|
| 18 |
"repository": "voipmonitor/vllm",
|
|
@@ -33,69 +33,29 @@
|
|
| 33 |
"relationship": "Community source contract; does not qualify this custom P8 runtime."
|
| 34 |
},
|
| 35 |
"build": {
|
| 36 |
-
"recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/
|
| 37 |
-
"recipe_commit": "
|
| 38 |
-
"build_command": "docker build --pull=false -t trellismx-
|
| 39 |
-
"
|
| 40 |
-
|
| 41 |
-
|
| 42 |
-
|
| 43 |
-
"commit": "c88fb8847dc8bc18bca56640759f137198750085",
|
| 44 |
-
"release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
|
| 45 |
-
"pull_requests": [],
|
| 46 |
-
"patches": [
|
| 47 |
-
{
|
| 48 |
-
"path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
|
| 49 |
-
"sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
|
| 50 |
-
"purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
|
| 51 |
-
"authors": [
|
| 52 |
-
"Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
|
| 53 |
-
]
|
| 54 |
-
}
|
| 55 |
-
],
|
| 56 |
-
"overlays": []
|
| 57 |
-
},
|
| 58 |
-
{
|
| 59 |
-
"name": "b12x",
|
| 60 |
-
"repository": "https://github.com/local-inference-lab/b12x",
|
| 61 |
-
"commit": "7093ad77849cf181bcf0c30b897c54fd32dac40e",
|
| 62 |
-
"release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
|
| 63 |
-
"pull_requests": [],
|
| 64 |
-
"patches": [
|
| 65 |
-
{
|
| 66 |
-
"path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
|
| 67 |
-
"sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
|
| 68 |
-
"purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
|
| 69 |
-
"authors": [
|
| 70 |
-
"Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
|
| 71 |
-
]
|
| 72 |
-
}
|
| 73 |
-
],
|
| 74 |
-
"overlays": []
|
| 75 |
-
}
|
| 76 |
-
],
|
| 77 |
-
"package_changes": [
|
| 78 |
-
"No dependency installs; r27 native extension binaries and packages inherited."
|
| 79 |
-
],
|
| 80 |
-
"build_arguments": [
|
| 81 |
-
"JOVIAN_IMAGE=voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512"
|
| 82 |
-
],
|
| 83 |
"environment_defaults": [
|
| 84 |
-
"
|
|
|
|
|
|
|
| 85 |
],
|
| 86 |
"entrypoint_changes": [
|
| 87 |
-
"
|
| 88 |
-
]
|
| 89 |
-
"result_tree": "N/A",
|
| 90 |
-
"integration_patch_sha256": "N/A"
|
| 91 |
},
|
| 92 |
"changes": {
|
| 93 |
"inherited": [
|
| 94 |
"r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
|
| 95 |
],
|
| 96 |
"introduced": [
|
| 97 |
-
"
|
| 98 |
-
"
|
| 99 |
],
|
| 100 |
"compatibility_impact": [
|
| 101 |
"TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
|
|
@@ -103,7 +63,7 @@
|
|
| 103 |
},
|
| 104 |
"tested_configurations": [
|
| 105 |
{
|
| 106 |
-
"name": "
|
| 107 |
"hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
|
| 108 |
"topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
|
| 109 |
"power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
|
|
@@ -111,86 +71,34 @@
|
|
| 111 |
"cuda_runtime": "13.3",
|
| 112 |
"pytorch": "2.13.0",
|
| 113 |
"nccl": "2.31.2",
|
| 114 |
-
"engine_source": "
|
| 115 |
"model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
|
| 116 |
"quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
|
| 117 |
"parallelism": "TP4/DCP4, EP off",
|
| 118 |
-
"kv_cache": "nvfp4_ds_mla;
|
| 119 |
"speculative_mode": "MTP3 probabilistic draft plus standard rejection",
|
| 120 |
"graph_mode": "FULL_AND_PIECEWISE",
|
| 121 |
-
"scheduler_limits": "
|
| 122 |
"cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
|
| 123 |
"launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
|
| 124 |
}
|
| 125 |
],
|
| 126 |
"validation": {
|
| 127 |
-
"
|
| 128 |
-
|
| 129 |
-
|
| 130 |
-
|
| 131 |
-
|
| 132 |
-
],
|
| 133 |
-
"results": [
|
| 134 |
-
{
|
| 135 |
-
"name": "Deployed inference source parity",
|
| 136 |
-
"status": "passed",
|
| 137 |
-
"conditions": "Current local image57a9967b75e6 vs public build context",
|
| 138 |
-
"measurement": "SHA256 every included deployed inference file",
|
| 139 |
-
"result": "413 files matched",
|
| 140 |
-
"conclusion": "Source parity only; rebuilt image is separately identified and is not a new GPU benchmark.",
|
| 141 |
-
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/runtime-parity.json",
|
| 142 |
-
"evidence_sha256": "sha256:033767e89483ccf6290f427decbbef96911511d704358da970cee40b512207fa",
|
| 143 |
-
"category": "correctness"
|
| 144 |
-
},
|
| 145 |
-
{
|
| 146 |
-
"name": "Public image CPU overlay import and manifest inventory",
|
| 147 |
-
"status": "passed",
|
| 148 |
-
"conditions": "Public image, no GPUs passed, local HF-layout checkpoint read-only",
|
| 149 |
-
"measurement": "Import native overlay validator; verify168 file inventory, sizes, and design identities and launcher SHA",
|
| 150 |
-
"result": "168 records loaded",
|
| 151 |
-
"conclusion": "CPU manifest/inventory gate only, no complete sidecar rehash or fresh GPU launch claim.",
|
| 152 |
-
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/public-image-cpu-check.txt",
|
| 153 |
-
"evidence_sha256": "sha256:ea5d2725e24abafef7edfd7266efbde829f61153a62426e93df67274eb454d62",
|
| 154 |
-
"category": "smoke"
|
| 155 |
-
},
|
| 156 |
-
{
|
| 157 |
-
"name": "Recorded current-server speed observations",
|
| 158 |
-
"status": "passed",
|
| 159 |
-
"conditions": "Four SM120 GPUs at300W; TP4/DCP4 MTP3 NVFP4 MLA KV; one run each, no matched baseline",
|
| 160 |
-
"measurement": "C1/C2 20s streaming cells; separate10s cold-prefill targets; full JSON/logs retained",
|
| 161 |
-
"result": "C1 zero202.45t/s; prefill7692/7940/8006t/s at8200/16227/32316tokens",
|
| 162 |
-
"conclusion": "Exploratory absolute observations; no advantage or independent qualification claim; workload and MTP confounders detailed.",
|
| 163 |
-
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/README.md",
|
| 164 |
-
"evidence_sha256": "sha256:16e1e1fedf7b49dcdc1d79c156438ee016d47817861c8abdfe32964d9fe2277b",
|
| 165 |
-
"category": "performance"
|
| 166 |
-
},
|
| 167 |
-
{
|
| 168 |
-
"name": "Paired current r27 NVFP4 versus FP8 MLA KV development KLD",
|
| 169 |
-
"status": "passed",
|
| 170 |
-
"conditions": "Same32 opened conditional-fit windows; TP4/DCP4, MTP off, maxseq1; fixed NVFP4 then FP8 order, one server start per arm.",
|
| 171 |
-
"measurement": "CPU FP64 teacher-to-student KL over2046 true-decode rows/window;20,000-resample paired window BCa95.",
|
| 172 |
-
"result": "NVFP4=0.0350078183; FP8=0.0310574767; FP8-minus-NVFP4=-0.0039503416, paired BCa95[-0.0092510457,-0.0018107907].",
|
| 173 |
-
"conclusion": "Audited development comparison only; no final holdout, independent replication, MTP or long-context qualification.",
|
| 174 |
-
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-kv-cf32-20260908/comparison.json",
|
| 175 |
-
"evidence_sha256": "sha256:9376acdce0d66738f5af7bc54188ad98d298dbb47dfd3cd3faece459becb77d6",
|
| 176 |
-
"category": "correctness"
|
| 177 |
-
}
|
| 178 |
-
],
|
| 179 |
-
"performance_claims": [
|
| 180 |
-
{
|
| 181 |
-
"name": "N/A"
|
| 182 |
-
}
|
| 183 |
-
]
|
| 184 |
},
|
| 185 |
"limitations": {
|
| 186 |
"known": [
|
| 187 |
-
"
|
| 188 |
-
"Historical
|
| 189 |
-
"
|
| 190 |
-
"
|
| 191 |
],
|
| 192 |
"untested": [
|
| 193 |
-
"
|
| 194 |
"Independent repeated speed qualification"
|
| 195 |
],
|
| 196 |
"unsupported": [
|
|
|
|
| 1 |
{
|
| 2 |
"schema_version": 4,
|
| 3 |
+
"record_id": "trellismx-reference-20260909",
|
| 4 |
+
"title": "TrellisMX GLM-5.3 Flash selected TP4 reference",
|
| 5 |
+
"summary": "Selected September9 FP8-math reference image with measured speed, exact source overlays and audited matched FP8/NVFP4 MLA KLD.",
|
| 6 |
"model_family": "GLM-5.3-Flash",
|
| 7 |
"release_class": "experimental",
|
| 8 |
"distribution_role": "custom",
|
|
|
|
| 10 |
"maintenance_status": "ephemeral",
|
| 11 |
"image": {
|
| 12 |
"repository": "verdictai/trellismx",
|
| 13 |
+
"tag": "glm53-flash-p8-r27-reference-20260909",
|
| 14 |
+
"digest": "sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf",
|
| 15 |
+
"reference": "verdictai/trellismx:glm53-flash-p8-r27-reference-20260909@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf"
|
| 16 |
},
|
| 17 |
"base_image": {
|
| 18 |
"repository": "voipmonitor/vllm",
|
|
|
|
| 33 |
"relationship": "Community source contract; does not qualify this custom P8 runtime."
|
| 34 |
},
|
| 35 |
"build": {
|
| 36 |
+
"recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/main/runtime-reference-20260909",
|
| 37 |
+
"recipe_commit": "See containing HF commit",
|
| 38 |
+
"build_command": "docker build --pull=false -t trellismx-reference-rebuild runtime-reference-20260909",
|
| 39 |
+
"source_manifest": "runtime-reference-20260909/source-manifest.json",
|
| 40 |
+
"source_manifest_sha256": "2629fdb6094b54ce1cc1cc59b7dc0fe2617e0fe2ed39ff49a4e92ed72a28b252",
|
| 41 |
+
"base": "verdictai/trellismx@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f",
|
| 42 |
+
"qualification": "Published image is the retained measured image, retagged without a rebuild. Recipe reproduces Python overlay; fresh rebuild not GPU tested.",
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 43 |
"environment_defaults": [
|
| 44 |
+
"TP4 DCP4 MTP3 maxseq24 batch4096 maxlen1000000 GMU0.97 KVnvfp4_ds_mla",
|
| 45 |
+
"NCCL_MIN_NCHANNELS=8 NCCL_MAX_NCHANNELS=8",
|
| 46 |
+
"VLLM_PCIE_ONESHOT_ALLREDUCE_MAX_SIZE=131072 VLLM_PCIE_ONESHOT_FUSED_ADD_RMS_NORM_MAX_SIZE=86016 VLLM_SHARED_EXPERTS_STREAM_TOKEN_THRESHOLD=4096"
|
| 47 |
],
|
| 48 |
"entrypoint_changes": [
|
| 49 |
+
"compose explicitly launches /bin/bash /release/serve-r27-production.sh; baked launcher defaults16, public compose/wrapper override24"
|
| 50 |
+
]
|
|
|
|
|
|
|
| 51 |
},
|
| 52 |
"changes": {
|
| 53 |
"inherited": [
|
| 54 |
"r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
|
| 55 |
],
|
| 56 |
"introduced": [
|
| 57 |
+
"Route-hoisted FP8 native expert dispatch, tile policy and grouped FC2 paths; exact changed Python files in source manifest.",
|
| 58 |
+
"Selected collective settings and24slot serving wrapper."
|
| 59 |
],
|
| 60 |
"compatibility_impact": [
|
| 61 |
"TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
|
|
|
|
| 63 |
},
|
| 64 |
"tested_configurations": [
|
| 65 |
{
|
| 66 |
+
"name": "Retained September9 selected reference image",
|
| 67 |
"hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
|
| 68 |
"topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
|
| 69 |
"power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
|
|
|
|
| 71 |
"cuda_runtime": "13.3",
|
| 72 |
"pytorch": "2.13.0",
|
| 73 |
"nccl": "2.31.2",
|
| 74 |
+
"engine_source": "Exact selected image Python source and manifest in runtime-reference-20260909",
|
| 75 |
"model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
|
| 76 |
"quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
|
| 77 |
"parallelism": "TP4/DCP4, EP off",
|
| 78 |
+
"kv_cache": "nvfp4_ds_mla; expanded benchmark engine capacity23562091 effective aggregate tokens",
|
| 79 |
"speculative_mode": "MTP3 probabilistic draft plus standard rejection",
|
| 80 |
"graph_mode": "FULL_AND_PIECEWISE",
|
| 81 |
+
"scheduler_limits": "24 sequences;4096 batched tokens;1000000 model length",
|
| 82 |
"cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
|
| 83 |
"launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
|
| 84 |
}
|
| 85 |
],
|
| 86 |
"validation": {
|
| 87 |
+
"speed_evidence": "results/speed-20260909/README.md",
|
| 88 |
+
"all_speed_index": "results/speed-20260909/benchmark-index.json",
|
| 89 |
+
"kld_evidence": "results/kld-reference-20260909/comparison.json",
|
| 90 |
+
"kld_audit": "results/kld-reference-20260909/audit.json",
|
| 91 |
+
"kld_level": "Already-opened conditional-fit development comparison; single server per arm; MTPoff; not independent replication or final qualification."
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 92 |
},
|
| 93 |
"limitations": {
|
| 94 |
"known": [
|
| 95 |
+
"Recipe Python-source reconstruction is separate from the measured immutable Docker image.",
|
| 96 |
+
"Historical comparisons differ in image/topology; new matched KLD holds reference checkpoint/TP4/DCP4 fixed and changes KV specialization.",
|
| 97 |
+
"TR3/EXL3 matched KLD requested, not measured yet.",
|
| 98 |
+
"KV aggregate capacity is not1M-context accuracy or full-capacity stress."
|
| 99 |
],
|
| 100 |
"untested": [
|
| 101 |
+
"Fresh downloaded deployment GPU test",
|
| 102 |
"Independent repeated speed qualification"
|
| 103 |
],
|
| 104 |
"unsupported": [
|
results/image-record-historical-20260908.json
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema_version": 4,
|
| 3 |
+
"record_id": "trellismx-r27-dcp4-20260908",
|
| 4 |
+
"title": "TrellisMX P8 GLM-5.3-Flash r27 DCP4",
|
| 5 |
+
"summary": "Reproducible r27 inference overlay with source parity, baked production launcher, measured speed receipts and audited paired NVFP4/FP8 KV development KLD.",
|
| 6 |
+
"model_family": "GLM-5.3-Flash",
|
| 7 |
+
"release_class": "experimental",
|
| 8 |
+
"distribution_role": "custom",
|
| 9 |
+
"qualification_status": "implemented",
|
| 10 |
+
"maintenance_status": "ephemeral",
|
| 11 |
+
"image": {
|
| 12 |
+
"repository": "verdictai/trellismx",
|
| 13 |
+
"tag": "glm53-flash-p8-r27-dcp4-20260908",
|
| 14 |
+
"digest": "sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f",
|
| 15 |
+
"reference": "verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f"
|
| 16 |
+
},
|
| 17 |
+
"base_image": {
|
| 18 |
+
"repository": "voipmonitor/vllm",
|
| 19 |
+
"tag": "jovian-judgement-community-20260906-r27",
|
| 20 |
+
"digest": "sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
|
| 21 |
+
"reference": "voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
|
| 22 |
+
"credit": "Local Inference Lab Jovian Judgement GLM r27; vLLM, B12X and component license obligations remain applicable."
|
| 23 |
+
},
|
| 24 |
+
"recommended_image": {
|
| 25 |
+
"reference": "voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512",
|
| 26 |
+
"relationship": "Pinned community runtime inherited by this custom overlay; not a matched numerical baseline."
|
| 27 |
+
},
|
| 28 |
+
"community_wiki": {
|
| 29 |
+
"repository": "https://github.com/local-inference-lab/rtx6kpro",
|
| 30 |
+
"commit": "94b71ac2a5f9c75f6b60dd1b6e6dffda492a4942",
|
| 31 |
+
"runbook_path": "models/glm-5.3-flash.md",
|
| 32 |
+
"runbook_url": "https://github.com/local-inference-lab/rtx6kpro/blob/94b71ac2a5f9c75f6b60dd1b6e6dffda492a4942/models/glm-5.3-flash.md",
|
| 33 |
+
"relationship": "Community source contract; does not qualify this custom P8 runtime."
|
| 34 |
+
},
|
| 35 |
+
"build": {
|
| 36 |
+
"recipe_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/tree/53d93dfbc9002df7b73178327dfe99773efb680e/runtime",
|
| 37 |
+
"recipe_commit": "53d93dfbc9002df7b73178327dfe99773efb680e",
|
| 38 |
+
"build_command": "docker build --pull=false -t trellismx-r27-rebuild runtime",
|
| 39 |
+
"components": [
|
| 40 |
+
{
|
| 41 |
+
"name": "vllm",
|
| 42 |
+
"repository": "https://github.com/local-inference-lab/vllm",
|
| 43 |
+
"commit": "c88fb8847dc8bc18bca56640759f137198750085",
|
| 44 |
+
"release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
|
| 45 |
+
"pull_requests": [],
|
| 46 |
+
"patches": [
|
| 47 |
+
{
|
| 48 |
+
"path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
|
| 49 |
+
"sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
|
| 50 |
+
"purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
|
| 51 |
+
"authors": [
|
| 52 |
+
"Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
|
| 53 |
+
]
|
| 54 |
+
}
|
| 55 |
+
],
|
| 56 |
+
"overlays": []
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"name": "b12x",
|
| 60 |
+
"repository": "https://github.com/local-inference-lab/b12x",
|
| 61 |
+
"commit": "7093ad77849cf181bcf0c30b897c54fd32dac40e",
|
| 62 |
+
"release_reference": "Local integration commit; exact complete overlay files are published in the pinned HF recipe. Inherited binaries remain r27.",
|
| 63 |
+
"pull_requests": [],
|
| 64 |
+
"patches": [
|
| 65 |
+
{
|
| 66 |
+
"path_or_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/53d93dfbc9002df7b73178327dfe99773efb680e/runtime-files.json",
|
| 67 |
+
"sha256": "sha256:e1dfd232f438f17fbfceb2c534e60666e7b9f5c2104808826c12dd18b5f92de1",
|
| 68 |
+
"purpose": "Complete SHA256 inventory for inference overlay files; pins local integration output beyond upstream source labels.",
|
| 69 |
+
"authors": [
|
| 70 |
+
"Brandon M. Music; Local Inference Lab, ExLlamaV3, KQuant, QSRT and w4a8 lineage credited in CITATION.cff"
|
| 71 |
+
]
|
| 72 |
+
}
|
| 73 |
+
],
|
| 74 |
+
"overlays": []
|
| 75 |
+
}
|
| 76 |
+
],
|
| 77 |
+
"package_changes": [
|
| 78 |
+
"No dependency installs; r27 native extension binaries and packages inherited."
|
| 79 |
+
],
|
| 80 |
+
"build_arguments": [
|
| 81 |
+
"JOVIAN_IMAGE=voipmonitor/vllm:jovian-judgement-community-20260906-r27@sha256:a298fe1cd207eaf97bd2ff2686716ed25b7009c09b36650eba732a4a7dc51512"
|
| 82 |
+
],
|
| 83 |
+
"environment_defaults": [
|
| 84 |
+
"TP=4 DCP=4 KV_CACHE_DTYPE=nvfp4_ds_mla NUM_SPECULATIVE_TOKENS=3 GPU_MEMORY_UTILIZATION=0.97 MAX_NUM_SEQS=16 MAX_NUM_BATCHED_TOKENS=4096 MAX_MODEL_LEN=1000000"
|
| 85 |
+
],
|
| 86 |
+
"entrypoint_changes": [
|
| 87 |
+
"Bakes deployed launcher at /opt/trellismx/serve.sh; executes inherited r27 serve-glm53-flash launcher. PORT=8000."
|
| 88 |
+
],
|
| 89 |
+
"result_tree": "N/A",
|
| 90 |
+
"integration_patch_sha256": "N/A"
|
| 91 |
+
},
|
| 92 |
+
"changes": {
|
| 93 |
+
"inherited": [
|
| 94 |
+
"r27 split-cache rebalancing, B12X attention, FlashKDA, probabilistic MTP and fairness scheduler; binaries and dependencies unchanged."
|
| 95 |
+
],
|
| 96 |
+
"introduced": [
|
| 97 |
+
"Native TrellisMX P8 checkpoint loader and B12X expert path; DCP4 persistent workspace integration; complete inference source in recipe.",
|
| 98 |
+
"Production launcher defaults baked into image; licenses included."
|
| 99 |
+
],
|
| 100 |
+
"compatibility_impact": [
|
| 101 |
+
"TP4 checkpoint with 168 sidecars, pinned carrier and four SM120 GPUs; no arbitrary model conversion claim."
|
| 102 |
+
]
|
| 103 |
+
},
|
| 104 |
+
"tested_configurations": [
|
| 105 |
+
{
|
| 106 |
+
"name": "Measured source-equivalent r27 DCP4 local deployment; public image CPU checked",
|
| 107 |
+
"hardware": "4x RTX PRO 6000 Blackwell 96GB, 2 Max-Q and 2 standard",
|
| 108 |
+
"topology": "TP4/DCP4, PCIe gen5 x16; no expert parallelism",
|
| 109 |
+
"power_and_clocks": "300 W cap per GPU; clocks and telemetry retained in benchmark JSON; no clock change by release",
|
| 110 |
+
"driver": "610.57.04",
|
| 111 |
+
"cuda_runtime": "13.3",
|
| 112 |
+
"pytorch": "2.13.0",
|
| 113 |
+
"nccl": "2.31.2",
|
| 114 |
+
"engine_source": "c88fb8847dc8bc18bca56640759f137198750085 overlay on r27 binaries",
|
| 115 |
+
"model_revision": "168 sidecar identities in trellismx-manifest.json; carrier 520de24eabf507659eaef7c70f14fd584527facc",
|
| 116 |
+
"quantization": "TrellisMX coupled K4/K5, 4.6587417643 routed bpw; native E4M3 UE8M0/32 MMA",
|
| 117 |
+
"parallelism": "TP4/DCP4, EP off",
|
| 118 |
+
"kv_cache": "nvfp4_ds_mla; engine reports 23424836 effective aggregate tokens; hybrid split layout",
|
| 119 |
+
"speculative_mode": "MTP3 probabilistic draft plus standard rejection",
|
| 120 |
+
"graph_mode": "FULL_AND_PIECEWISE",
|
| 121 |
+
"scheduler_limits": "16 sequences; 4096 batched tokens; 1000000 model length; prefill share 0.4",
|
| 122 |
+
"cache_policy": "chunked prefill and prefix caching; vram; GPU memory utilization0.97",
|
| 123 |
+
"launch_command": "MODEL_ROOT=/absolute/path/to/carrier docker compose -f compose.yaml up -d"
|
| 124 |
+
}
|
| 125 |
+
],
|
| 126 |
+
"validation": {
|
| 127 |
+
"commands": [
|
| 128 |
+
"docker build --pull=false -t trellismx-r27-rebuild runtime",
|
| 129 |
+
"MODEL_ROOT=/model docker compose -f compose.yaml config --quiet",
|
| 130 |
+
"docker run --rm --entrypoint python -v /absolute/path/to/checkpoint:/checkpoint:ro verdictai/trellismx:glm53-flash-p8-r27-dcp4-20260908@sha256:1c8a10d2b21bd6ed5a7ca4a29bcc3900d29acc3ce42e1d722b9ebaa74357de3f -c 'from vllm.utils.trellismx import load_overlay; print(len(load_overlay(\"/checkpoint\").records))'",
|
| 131 |
+
"Pinned llm_decode_bench.py commands and effective arguments in results/r27-20260908/README.md and JSON."
|
| 132 |
+
],
|
| 133 |
+
"results": [
|
| 134 |
+
{
|
| 135 |
+
"name": "Deployed inference source parity",
|
| 136 |
+
"status": "passed",
|
| 137 |
+
"conditions": "Current local image57a9967b75e6 vs public build context",
|
| 138 |
+
"measurement": "SHA256 every included deployed inference file",
|
| 139 |
+
"result": "413 files matched",
|
| 140 |
+
"conclusion": "Source parity only; rebuilt image is separately identified and is not a new GPU benchmark.",
|
| 141 |
+
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/runtime-parity.json",
|
| 142 |
+
"evidence_sha256": "sha256:033767e89483ccf6290f427decbbef96911511d704358da970cee40b512207fa",
|
| 143 |
+
"category": "correctness"
|
| 144 |
+
},
|
| 145 |
+
{
|
| 146 |
+
"name": "Public image CPU overlay import and manifest inventory",
|
| 147 |
+
"status": "passed",
|
| 148 |
+
"conditions": "Public image, no GPUs passed, local HF-layout checkpoint read-only",
|
| 149 |
+
"measurement": "Import native overlay validator; verify168 file inventory, sizes, and design identities and launcher SHA",
|
| 150 |
+
"result": "168 records loaded",
|
| 151 |
+
"conclusion": "CPU manifest/inventory gate only, no complete sidecar rehash or fresh GPU launch claim.",
|
| 152 |
+
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/public-image-cpu-check.txt",
|
| 153 |
+
"evidence_sha256": "sha256:ea5d2725e24abafef7edfd7266efbde829f61153a62426e93df67274eb454d62",
|
| 154 |
+
"category": "smoke"
|
| 155 |
+
},
|
| 156 |
+
{
|
| 157 |
+
"name": "Recorded current-server speed observations",
|
| 158 |
+
"status": "passed",
|
| 159 |
+
"conditions": "Four SM120 GPUs at300W; TP4/DCP4 MTP3 NVFP4 MLA KV; one run each, no matched baseline",
|
| 160 |
+
"measurement": "C1/C2 20s streaming cells; separate10s cold-prefill targets; full JSON/logs retained",
|
| 161 |
+
"result": "C1 zero202.45t/s; prefill7692/7940/8006t/s at8200/16227/32316tokens",
|
| 162 |
+
"conclusion": "Exploratory absolute observations; no advantage or independent qualification claim; workload and MTP confounders detailed.",
|
| 163 |
+
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-20260908/README.md",
|
| 164 |
+
"evidence_sha256": "sha256:16e1e1fedf7b49dcdc1d79c156438ee016d47817861c8abdfe32964d9fe2277b",
|
| 165 |
+
"category": "performance"
|
| 166 |
+
},
|
| 167 |
+
{
|
| 168 |
+
"name": "Paired current r27 NVFP4 versus FP8 MLA KV development KLD",
|
| 169 |
+
"status": "passed",
|
| 170 |
+
"conditions": "Same32 opened conditional-fit windows; TP4/DCP4, MTP off, maxseq1; fixed NVFP4 then FP8 order, one server start per arm.",
|
| 171 |
+
"measurement": "CPU FP64 teacher-to-student KL over2046 true-decode rows/window;20,000-resample paired window BCa95.",
|
| 172 |
+
"result": "NVFP4=0.0350078183; FP8=0.0310574767; FP8-minus-NVFP4=-0.0039503416, paired BCa95[-0.0092510457,-0.0018107907].",
|
| 173 |
+
"conclusion": "Audited development comparison only; no final holdout, independent replication, MTP or long-context qualification.",
|
| 174 |
+
"evidence_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/results/r27-kv-cf32-20260908/comparison.json",
|
| 175 |
+
"evidence_sha256": "sha256:9376acdce0d66738f5af7bc54188ad98d298dbb47dfd3cd3faece459becb77d6",
|
| 176 |
+
"category": "correctness"
|
| 177 |
+
}
|
| 178 |
+
],
|
| 179 |
+
"performance_claims": [
|
| 180 |
+
{
|
| 181 |
+
"name": "N/A"
|
| 182 |
+
}
|
| 183 |
+
]
|
| 184 |
+
},
|
| 185 |
+
"limitations": {
|
| 186 |
+
"known": [
|
| 187 |
+
"Existing absolute speed observations belong to source-equivalent local server image57a9967b75e6, not a newly benchmarked registry rebuild.",
|
| 188 |
+
"Historical KLD0.0341811459 uses DCP1; current paired DCP4 results use the same opened windows, MTP off and2046 true-decode rows per window. Different historical runtime/topology prevents KV-only attribution.",
|
| 189 |
+
"Weight upload is still in progress; release-status.json governs remote completeness.",
|
| 190 |
+
"Engine aggregate cache capacity does not establish1M-context accuracy or full-capacity stress."
|
| 191 |
+
],
|
| 192 |
+
"untested": [
|
| 193 |
+
"Clean download-to-GPU serving of public image",
|
| 194 |
+
"Independent repeated speed qualification"
|
| 195 |
+
],
|
| 196 |
+
"unsupported": [
|
| 197 |
+
"Other GPU architectures, TP sizes, arbitrary model encoders; inference-only release."
|
| 198 |
+
]
|
| 199 |
+
},
|
| 200 |
+
"support": {
|
| 201 |
+
"owner": "Brandon Music",
|
| 202 |
+
"contact": "https://github.com/brandonmmusic-max",
|
| 203 |
+
"issue_tracker": "https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/issues",
|
| 204 |
+
"support_thread": "https://github.com/brandonmmusic-max/glm53-hadamard-shapleymcg-kld/issues/5",
|
| 205 |
+
"thread_status": "active",
|
| 206 |
+
"support_commitment": "ephemeral",
|
| 207 |
+
"triage_policy": "Keep reports in this thread. Escalate upstream only after reproduction on the recommended image or a minimal reproducer identifies the responsible source change.",
|
| 208 |
+
"superseded_by": "N/A"
|
| 209 |
+
},
|
| 210 |
+
"publication": {
|
| 211 |
+
"record_url": "https://huggingface.co/brandonmusic/GLM-5.3-Flash-TrellisMX-MXFP8/blob/main/image-record.json",
|
| 212 |
+
"main_channel_link": "N/A",
|
| 213 |
+
"main_channel_link_count": 0,
|
| 214 |
+
"bot_listing": "not-applicable",
|
| 215 |
+
"maintainer_approval_url": "N/A"
|
| 216 |
+
}
|
| 217 |
+
}
|
results/kld-reference-20260909/README.md
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Reference-stack KV-cache KLD — September 9, 2026
|
| 2 |
+
|
| 3 |
+
| Cache | Mean KL(teacher || student) | Window BCa 95% interval |
|
| 4 |
+
| --- | ---: | --- |
|
| 5 |
+
| Current r27 DCP4 / NVFP4 MLA KV | 0.0354562238 | [0.0295558848, 0.0434620368] |
|
| 6 |
+
| Current r27 DCP4 / FP8 MLA KV | 0.0319451732 | [0.0268267482, 0.0387948613] |
|
| 7 |
+
|
| 8 |
+
FP8 minus NVFP4: -0.0035110506; paired-window BCa 95% interval
|
| 9 |
+
[-0.0081718772, -0.0013675941].
|
| 10 |
+
FP8 has lower observed KLD in 22/32 windows.
|
| 11 |
+
|
| 12 |
+
Exact same 32 previously opened conditional-fit windows and BF16 teacher as the
|
| 13 |
+
September 8 measurement: 2048 input tokens, 2047 captured predictions, exclude
|
| 14 |
+
row zero, leaving 2046 true-decode rows/window. CPU FP64 KL over vocabulary154880;
|
| 15 |
+
equal mean of window means; BCa20000 resamples with seed20260902. Teacher stored F32.
|
| 16 |
+
TP4/DCP4, MTP off, maxseq1, batch4096, GMU0.97, maxlen1M; prefix caching enabled,
|
| 17 |
+
all prefix-hit counters zero. FP8 first then NVFP4, one server preparation each.
|
| 18 |
+
Window uncertainty does not estimate server-run variability. These are development
|
| 19 |
+
measurements, not untouched-final qualification or independent reproduction.
|
| 20 |
+
|
| 21 |
+
Selected image: verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf.
|
| 22 |
+
Capture-only derivative image, seal, source identities and all64 window scores are
|
| 23 |
+
in comparison.json. Capture seam preserves pre-mask logits and forces exact histories.
|
| 24 |
+
NCCL8, plain one-shot cutoff131072, fused cutoff86016, shared expert threshold4096.
|
| 25 |
+
Checkpoint and FP8 weight/activation math held fixed; KV dtype and native cache
|
| 26 |
+
specialization vary. Capture timing is not a throughput measurement.
|
| 27 |
+
|
| 28 |
+
Raw logits retired only after hashes, durable numerical scores and receipts.
|
| 29 |
+
Audit recomputes aggregates from retained FP64 scores and verifies metadata,
|
| 30 |
+
zero prefix hits, row masks and receipt hashes; it is not independent recapture.
|
| 31 |
+
No measured windows excluded or rerun. Prior image/cache values remain historical.
|
| 32 |
+
The separate production recipe uses MTP3 and24slots; its KV capacity must not be
|
| 33 |
+
confused with capacity from this MTP-disabled correctness profile.
|
results/kld-reference-20260909/audit.json
ADDED
|
@@ -0,0 +1,30 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "passed",
|
| 3 |
+
"arms": [
|
| 4 |
+
{
|
| 5 |
+
"arm": "fp8",
|
| 6 |
+
"windows": 32,
|
| 7 |
+
"prediction_rows": 65504,
|
| 8 |
+
"true_decode_rows": 65472,
|
| 9 |
+
"all_receipt_hashes_and_masks_pass": true,
|
| 10 |
+
"all_prefix_hit_counters_zero": true,
|
| 11 |
+
"raw_logits_retired_after_scoring": true,
|
| 12 |
+
"source_runtime_audit_sha256": "6a3393053b83b33b41a7d49de2fab1803b1efc4843ec5862c1acd8467b993fa4",
|
| 13 |
+
"source_startup_sha256": "00ce74cfce281b627289f3322829d357467f5bc1f026677ef27cda93ed43fce3"
|
| 14 |
+
},
|
| 15 |
+
{
|
| 16 |
+
"arm": "nvfp4",
|
| 17 |
+
"windows": 32,
|
| 18 |
+
"prediction_rows": 65504,
|
| 19 |
+
"true_decode_rows": 65472,
|
| 20 |
+
"all_receipt_hashes_and_masks_pass": true,
|
| 21 |
+
"all_prefix_hit_counters_zero": true,
|
| 22 |
+
"raw_logits_retired_after_scoring": true,
|
| 23 |
+
"source_runtime_audit_sha256": "74d7d1914a55a73b5e9de196f557c658603a9501acd2fafb1e540d496b1c7185",
|
| 24 |
+
"source_startup_sha256": "cae3cb749c61cc18fd23286f41f79266e32bfdc551e0575b3515d21eb993036a"
|
| 25 |
+
}
|
| 26 |
+
],
|
| 27 |
+
"producing_analysis_sha256": "0b9a6cb0c0f1d1b4b7784190d38e6f8fdf3dbefe5cb619be6f0dc065a742d8a3",
|
| 28 |
+
"verifier_sha256": "124494f28c018e0945349fb4479d3be9f9164cdf7ce298989de25367f4106ade",
|
| 29 |
+
"claim": "Receipt and score audit by the same operator; not independent replication."
|
| 30 |
+
}
|
results/kld-reference-20260909/comparison.json
ADDED
|
@@ -0,0 +1,1053 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema": "trellismx.r27-kv-cf32.v1",
|
| 3 |
+
"evidence": "controlled comparison on already-opened conditional-fit; one prepared run per KV mode, fixed order; not final qualification",
|
| 4 |
+
"arms": {
|
| 5 |
+
"fp8": {
|
| 6 |
+
"mean_true_decode_kld": 0.03194517316654619,
|
| 7 |
+
"window_bca95": [
|
| 8 |
+
0.026826748240418287,
|
| 9 |
+
0.03879486133125964
|
| 10 |
+
],
|
| 11 |
+
"per_window": [
|
| 12 |
+
{
|
| 13 |
+
"window_id": "conditional-fit-0056",
|
| 14 |
+
"domain": "axis1_general",
|
| 15 |
+
"prediction_rows": 2047,
|
| 16 |
+
"true_decode_rows": 2046,
|
| 17 |
+
"true_decode_mean_kld": 0.02440979598998064,
|
| 18 |
+
"including_prefill_mean_kld": 0.02501134181859523,
|
| 19 |
+
"prefill_row_kld": 1.2557741071640407,
|
| 20 |
+
"raw_sha256": "72bc5007903d7885f04ff28c03128901bfd15d7a94385772af929125edb650ff",
|
| 21 |
+
"score_sha256": "3d00fe1e7f7505cb4186e5d67c6e7c5df53b904b6c041230b9e227ad34afcd93",
|
| 22 |
+
"teacher_sha256": "55a70a330f39c5f23294c23d270a1ddc53031def99a42206e2706d4d0bb4466b",
|
| 23 |
+
"token_sha256": "219d8cd8873c6cf41f52782297cc5cfa9c4ed9666f2bc30d84cee9caa68ea744",
|
| 24 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 25 |
+
"raw_bytes": 1268157440
|
| 26 |
+
},
|
| 27 |
+
{
|
| 28 |
+
"window_id": "conditional-fit-0008",
|
| 29 |
+
"domain": "axis1_general",
|
| 30 |
+
"prediction_rows": 2047,
|
| 31 |
+
"true_decode_rows": 2046,
|
| 32 |
+
"true_decode_mean_kld": 0.06994665102705569,
|
| 33 |
+
"including_prefill_mean_kld": 0.07040344776705193,
|
| 34 |
+
"prefill_row_kld": 1.0050095777993453,
|
| 35 |
+
"raw_sha256": "f9c863fed0667b09b18edeba791c9ab4018c68e010ff488c0b73c1b0a0a67721",
|
| 36 |
+
"score_sha256": "a1bfbe0a93478e2939a51800dea0ad26d56c6aca01ad3633f18a3709e1601bad",
|
| 37 |
+
"teacher_sha256": "040641247b2e060035d89c2e7345e029d7c142983585e7c5d63dc818486f650d",
|
| 38 |
+
"token_sha256": "581d72e3e78b15c02d9e73f6bc742260cb57cdcd74337f4419c8f90cbc7c9c11",
|
| 39 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 40 |
+
"raw_bytes": 1268157440
|
| 41 |
+
},
|
| 42 |
+
{
|
| 43 |
+
"window_id": "conditional-fit-0100",
|
| 44 |
+
"domain": "axis1_general",
|
| 45 |
+
"prediction_rows": 2047,
|
| 46 |
+
"true_decode_rows": 2046,
|
| 47 |
+
"true_decode_mean_kld": 0.022149540339761863,
|
| 48 |
+
"including_prefill_mean_kld": 0.023738251318587308,
|
| 49 |
+
"prefill_row_kld": 3.2742409139954454,
|
| 50 |
+
"raw_sha256": "7b082ebc4250bb8b032c82bdb02a15019b947dfe54131e776a0af539fab8b2d8",
|
| 51 |
+
"score_sha256": "cef09f41fd3bc3e395b4721344d0b546796c981917ac64c997c78cbcedebe6b9",
|
| 52 |
+
"teacher_sha256": "f62991383b3ef5e8db0ed71442b17c428972e58da1c868c6aeb0f3fa143c865b",
|
| 53 |
+
"token_sha256": "7f008ec99068f373c6e15dd059ee977fb70e6dc832f46e009f0f299d48d19892",
|
| 54 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 55 |
+
"raw_bytes": 1268157440
|
| 56 |
+
},
|
| 57 |
+
{
|
| 58 |
+
"window_id": "conditional-fit-0094",
|
| 59 |
+
"domain": "axis1_general",
|
| 60 |
+
"prediction_rows": 2047,
|
| 61 |
+
"true_decode_rows": 2046,
|
| 62 |
+
"true_decode_mean_kld": 0.016832794344000746,
|
| 63 |
+
"including_prefill_mean_kld": 0.016934019957995376,
|
| 64 |
+
"prefill_row_kld": 0.22404162619101314,
|
| 65 |
+
"raw_sha256": "29d51a000848b3f9a158070703dc4b25be593e4890deb83f572291379b54668d",
|
| 66 |
+
"score_sha256": "7ade9eefb18e3bdcffc63d64abcc88dd7260bf9b97b4be788da8e8c492710e35",
|
| 67 |
+
"teacher_sha256": "656911b8fb34d7f42e984dd21311caf9ef56db0b7618b414cad4089562d93cba",
|
| 68 |
+
"token_sha256": "6b6a9b21c49a9984cc8c5ea90e0bda6db5761d039c7cdd335dc65e30444e017b",
|
| 69 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 70 |
+
"raw_bytes": 1268157440
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
"window_id": "conditional-fit-0083",
|
| 74 |
+
"domain": "axis2_legal",
|
| 75 |
+
"prediction_rows": 2047,
|
| 76 |
+
"true_decode_rows": 2046,
|
| 77 |
+
"true_decode_mean_kld": 0.03015300663995215,
|
| 78 |
+
"including_prefill_mean_kld": 0.03048617506449642,
|
| 79 |
+
"prefill_row_kld": 0.7121487716820719,
|
| 80 |
+
"raw_sha256": "d71da9253ed6946f0fb5632c845d2ffb2f7c03b7b6f7cc25b959b0e14bfe8484",
|
| 81 |
+
"score_sha256": "0f1cc2f89d5738573eea4d2b3dc1a6fa7fc02180cd2e7cd8a2a5ba68892007a6",
|
| 82 |
+
"teacher_sha256": "66a7870a1f13c8b5d647c1681fca005bb0f55a6e502e94f282da673a6bed3f1b",
|
| 83 |
+
"token_sha256": "3df9d2872815e061ff8d32a9a19da0d11d80ac67070cccd28579b24ec3079552",
|
| 84 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 85 |
+
"raw_bytes": 1268157440
|
| 86 |
+
},
|
| 87 |
+
{
|
| 88 |
+
"window_id": "conditional-fit-0021",
|
| 89 |
+
"domain": "axis2_legal",
|
| 90 |
+
"prediction_rows": 2047,
|
| 91 |
+
"true_decode_rows": 2046,
|
| 92 |
+
"true_decode_mean_kld": 0.0390320697408021,
|
| 93 |
+
"including_prefill_mean_kld": 0.03997988073234378,
|
| 94 |
+
"prefill_row_kld": 1.9792011694266036,
|
| 95 |
+
"raw_sha256": "63ef61c53db88b1eaf3bd30a833593d48ea2bd25313a7c500c8d754e8142f04c",
|
| 96 |
+
"score_sha256": "6e41161d9b1edb5b19299d74682f682d5fd8643794befb9c111ff1557bcadc25",
|
| 97 |
+
"teacher_sha256": "ebf76910c4db8f6955ad03f2ec7352e2f8aaf15e0c939d28f9a7d78f6bd7451b",
|
| 98 |
+
"token_sha256": "dbd9f2539cb3e97127ae760a04ca5b3456b0b13d6ce8698f88b215cc68b8f1a9",
|
| 99 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 100 |
+
"raw_bytes": 1268157440
|
| 101 |
+
},
|
| 102 |
+
{
|
| 103 |
+
"window_id": "conditional-fit-0098",
|
| 104 |
+
"domain": "axis2_legal",
|
| 105 |
+
"prediction_rows": 2047,
|
| 106 |
+
"true_decode_rows": 2046,
|
| 107 |
+
"true_decode_mean_kld": 0.04434523531376396,
|
| 108 |
+
"including_prefill_mean_kld": 0.04552245091649955,
|
| 109 |
+
"prefill_row_kld": 2.4541055741135085,
|
| 110 |
+
"raw_sha256": "f2b99e96d0776ba966989383185fc18a776722a9de83a1d7d7d22f833cb0fe15",
|
| 111 |
+
"score_sha256": "33363d670f8d1e088ebadd3edbfc00e8aac3ebe3a47aac8c64edc9698f5bc44e",
|
| 112 |
+
"teacher_sha256": "16e3d9524100836c020535635f7722a4c93ce0fdddc2ad0781f25bda77e8bcc9",
|
| 113 |
+
"token_sha256": "749bda3291a0323d286b0018b7dc3e0f469de8afd5d194370cde357b7b315c7d",
|
| 114 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 115 |
+
"raw_bytes": 1268157440
|
| 116 |
+
},
|
| 117 |
+
{
|
| 118 |
+
"window_id": "conditional-fit-0080",
|
| 119 |
+
"domain": "axis2_legal",
|
| 120 |
+
"prediction_rows": 2047,
|
| 121 |
+
"true_decode_rows": 2046,
|
| 122 |
+
"true_decode_mean_kld": 0.04201227846375734,
|
| 123 |
+
"including_prefill_mean_kld": 0.04314755010220913,
|
| 124 |
+
"prefill_row_kld": 2.365913322374568,
|
| 125 |
+
"raw_sha256": "1646b7c5c5387e320517272052d9b5f01c168e44dbc8497ab66f3b55fd473db1",
|
| 126 |
+
"score_sha256": "8a1f8466c13bb37d2e1c8f04463d90905c9acddf054d1271678a2cea58b35bfe",
|
| 127 |
+
"teacher_sha256": "e9c1a77490a5db63d410297bd8febe06adf4df2a6f4e16552db187738c9289d2",
|
| 128 |
+
"token_sha256": "a31d43f38710eb305c684501a2b02909a4e82c43101c07e76e4a3ca4edb4d0dd",
|
| 129 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 130 |
+
"raw_bytes": 1268157440
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"window_id": "conditional-fit-0081",
|
| 134 |
+
"domain": "axis3_code_agentic",
|
| 135 |
+
"prediction_rows": 2047,
|
| 136 |
+
"true_decode_rows": 2046,
|
| 137 |
+
"true_decode_mean_kld": 0.0255516960804676,
|
| 138 |
+
"including_prefill_mean_kld": 0.025679992777695365,
|
| 139 |
+
"prefill_row_kld": 0.28817503530570604,
|
| 140 |
+
"raw_sha256": "dfbb9d556272241c4e45c7a48302895559e7567d5cf85ddfdc788a6dd616a117",
|
| 141 |
+
"score_sha256": "601136d9bf9208532067fb0fe807c5200ae263e4dd85d01c21b1df89fbed8095",
|
| 142 |
+
"teacher_sha256": "5ae07760d91030d1eba925b76ea463309baa49c2e7d5f5e14409a23ced7fd20d",
|
| 143 |
+
"token_sha256": "350628488b94d59f3e4f3e99a418913544ea1fe7eacee14a642b996d9c614d16",
|
| 144 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 145 |
+
"raw_bytes": 1268157440
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"window_id": "conditional-fit-0123",
|
| 149 |
+
"domain": "axis3_code_agentic",
|
| 150 |
+
"prediction_rows": 2047,
|
| 151 |
+
"true_decode_rows": 2046,
|
| 152 |
+
"true_decode_mean_kld": 0.033191169179572766,
|
| 153 |
+
"including_prefill_mean_kld": 0.03317634067093455,
|
| 154 |
+
"prefill_row_kld": 0.0028372119971554416,
|
| 155 |
+
"raw_sha256": "0cfa11435295489e24da83d30b42710f530f65a99e473c64755317a2391f8940",
|
| 156 |
+
"score_sha256": "0c5504a8e70139d48221451c5569e222ac3f55d13a17de9e65adfe9ae645bfa3",
|
| 157 |
+
"teacher_sha256": "4a24b5b750ba98c06d9bc69e19334aff89e7bd4f9429592598af5d62de7ecb42",
|
| 158 |
+
"token_sha256": "aab84ee426f91621d123aa68bd53f75e220fbaf8cb270db1feab1529a62ddbf0",
|
| 159 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 160 |
+
"raw_bytes": 1268157440
|
| 161 |
+
},
|
| 162 |
+
{
|
| 163 |
+
"window_id": "conditional-fit-0090",
|
| 164 |
+
"domain": "axis3_code_agentic",
|
| 165 |
+
"prediction_rows": 2047,
|
| 166 |
+
"true_decode_rows": 2046,
|
| 167 |
+
"true_decode_mean_kld": 0.03953329012985432,
|
| 168 |
+
"including_prefill_mean_kld": 0.04086400060183276,
|
| 169 |
+
"prefill_row_kld": 2.7634976262697193,
|
| 170 |
+
"raw_sha256": "8067287b0733598d6f9d01ad790e350dab5bc4bc32eba7430a05d38adbfdf455",
|
| 171 |
+
"score_sha256": "4ac0add16333368e2109970786f1ab2b8a6c1f0532218bdb3dbe94fb3045551e",
|
| 172 |
+
"teacher_sha256": "db2a5fe33c9e457a22ebeb31a7073c1dac666b877503e9b4e90a8d6abf7b8a32",
|
| 173 |
+
"token_sha256": "a06e76993db717afee847fcc4d86f10b52dd23cdce5dae916f32d238d08d886c",
|
| 174 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 175 |
+
"raw_bytes": 1268157440
|
| 176 |
+
},
|
| 177 |
+
{
|
| 178 |
+
"window_id": "conditional-fit-0062",
|
| 179 |
+
"domain": "axis3_code_agentic",
|
| 180 |
+
"prediction_rows": 2047,
|
| 181 |
+
"true_decode_rows": 2046,
|
| 182 |
+
"true_decode_mean_kld": 0.02626102483081332,
|
| 183 |
+
"including_prefill_mean_kld": 0.02663675046149981,
|
| 184 |
+
"prefill_row_kld": 0.7953713908460541,
|
| 185 |
+
"raw_sha256": "9f2f72b2e7ea4ffb425c094bd8d3df7452743a16ec6bcc22065dd25f613670f3",
|
| 186 |
+
"score_sha256": "9ecc5d9141126ec6f81b86eb28e9c785088f25d994ee8558b5bc7218ca325940",
|
| 187 |
+
"teacher_sha256": "d6f7883a533625c7c75119fe7403b48bf81885b1de0b311a1a0571949f7f231c",
|
| 188 |
+
"token_sha256": "7c6d2ef2f4b7c7091a517f6ccd778009f9ee144165c9aecf2f8f28aea7b3753f",
|
| 189 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 190 |
+
"raw_bytes": 1268157440
|
| 191 |
+
},
|
| 192 |
+
{
|
| 193 |
+
"window_id": "conditional-fit-0047",
|
| 194 |
+
"domain": "axis4_reasoning_termination",
|
| 195 |
+
"prediction_rows": 2047,
|
| 196 |
+
"true_decode_rows": 2046,
|
| 197 |
+
"true_decode_mean_kld": 0.02201209190929644,
|
| 198 |
+
"including_prefill_mean_kld": 0.026359335852221596,
|
| 199 |
+
"prefill_row_kld": 8.920820443077092,
|
| 200 |
+
"raw_sha256": "6594c75d2eb3fef5ac2049c9924e2be8119f2e0e9948b5c114f1e9b5ea9233b3",
|
| 201 |
+
"score_sha256": "92f15873c7c6837cebdb8a1d16767da58ec6a2a13091394ad3028b2e2c0fe716",
|
| 202 |
+
"teacher_sha256": "8513cdcb9940786c7c42bf14d2f5f2cd452791e4ec6036c06a1f973eb969018d",
|
| 203 |
+
"token_sha256": "89730ee6826d4302c352156d5df766c51622a8b6681b8052d4ca6e7d481177fe",
|
| 204 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 205 |
+
"raw_bytes": 1268157440
|
| 206 |
+
},
|
| 207 |
+
{
|
| 208 |
+
"window_id": "conditional-fit-0051",
|
| 209 |
+
"domain": "axis4_reasoning_termination",
|
| 210 |
+
"prediction_rows": 2047,
|
| 211 |
+
"true_decode_rows": 2046,
|
| 212 |
+
"true_decode_mean_kld": 0.013267071463714407,
|
| 213 |
+
"including_prefill_mean_kld": 0.013608489001681363,
|
| 214 |
+
"prefill_row_kld": 0.7121487716820719,
|
| 215 |
+
"raw_sha256": "43626b428d36c588448d0d2756b020f63e23237c04d852b958f79ca7f601298c",
|
| 216 |
+
"score_sha256": "30293b5eabd64e4bdfce7177844412ab5834b7f6c199aba692592cc6206665fa",
|
| 217 |
+
"teacher_sha256": "177477c19b54dc0b0230e19202d48ab71d52f2b1a37fe383511bb64f0aae2fcf",
|
| 218 |
+
"token_sha256": "c79269b16560e7301549593443c9394007e7a957ad149c46224e7ad271dc6a18",
|
| 219 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 220 |
+
"raw_bytes": 1268157440
|
| 221 |
+
},
|
| 222 |
+
{
|
| 223 |
+
"window_id": "conditional-fit-0055",
|
| 224 |
+
"domain": "axis4_reasoning_termination",
|
| 225 |
+
"prediction_rows": 2047,
|
| 226 |
+
"true_decode_rows": 2046,
|
| 227 |
+
"true_decode_mean_kld": 0.020627011525195565,
|
| 228 |
+
"including_prefill_mean_kld": 0.021473521460828236,
|
| 229 |
+
"prefill_row_kld": 1.7534328497652714,
|
| 230 |
+
"raw_sha256": "e470e9ded871cbe5d724b0416e973bb98b60bbe4cf76f6348feb871494464fbd",
|
| 231 |
+
"score_sha256": "2c690f7a591f3e932abfd73138a1185cdb253a37ce48cb85292b712ebd412ff8",
|
| 232 |
+
"teacher_sha256": "7679c828eae4bf08f17598d044904693fc90c05b9d3f189dddeeb846100e43a7",
|
| 233 |
+
"token_sha256": "feec9ac508172dd275ec6452a03493acab5ffdff9bb57161317c77f24daf90f2",
|
| 234 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 235 |
+
"raw_bytes": 1268157440
|
| 236 |
+
},
|
| 237 |
+
{
|
| 238 |
+
"window_id": "conditional-fit-0031",
|
| 239 |
+
"domain": "axis4_reasoning_termination",
|
| 240 |
+
"prediction_rows": 2047,
|
| 241 |
+
"true_decode_rows": 2046,
|
| 242 |
+
"true_decode_mean_kld": 0.01655011698258689,
|
| 243 |
+
"including_prefill_mean_kld": 0.020900029208329205,
|
| 244 |
+
"prefill_row_kld": 8.920820443077092,
|
| 245 |
+
"raw_sha256": "e94b496a3cec9e30b6bcbd429f1d4757f38bcd031ce303872442bf994404bec8",
|
| 246 |
+
"score_sha256": "4486c4b80fe284775cb7e57a99199749eb5b8892e397e56d1904c3d27459e1a7",
|
| 247 |
+
"teacher_sha256": "e04a6c04a951efb5a178927e91457535cc67745ce5c81e785cb6d963aac6ce83",
|
| 248 |
+
"token_sha256": "2cff3482b4212537d0c35b88b8a0f479502c8905952430c973d25b1ab0bb5818",
|
| 249 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 250 |
+
"raw_bytes": 1268157440
|
| 251 |
+
},
|
| 252 |
+
{
|
| 253 |
+
"window_id": "conditional-fit-0003",
|
| 254 |
+
"domain": "axis4_reasoning_termination",
|
| 255 |
+
"prediction_rows": 2047,
|
| 256 |
+
"true_decode_rows": 2046,
|
| 257 |
+
"true_decode_mean_kld": 0.019912232579636858,
|
| 258 |
+
"including_prefill_mean_kld": 0.022605333926454176,
|
| 259 |
+
"prefill_row_kld": 5.5326906895146735,
|
| 260 |
+
"raw_sha256": "2f0ef47bdc989be608cabfb611ac42a5710712f17f219bec543b0594ab7dadd6",
|
| 261 |
+
"score_sha256": "b72e81ff056ef12b3edc2b59ed5425223aef7120e8f8f144409329512b0cdb6c",
|
| 262 |
+
"teacher_sha256": "7677c89cc3dee8e2a64e3c7947add39475f828921a3a8fe2e92b379b3ed8320b",
|
| 263 |
+
"token_sha256": "d416537c6e8273747bf025e97287a0bb6e3ff047bfee65cf54e4dfe287569b93",
|
| 264 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 265 |
+
"raw_bytes": 1268157440
|
| 266 |
+
},
|
| 267 |
+
{
|
| 268 |
+
"window_id": "conditional-fit-0009",
|
| 269 |
+
"domain": "axis2_legal",
|
| 270 |
+
"prediction_rows": 2047,
|
| 271 |
+
"true_decode_rows": 2046,
|
| 272 |
+
"true_decode_mean_kld": 0.04340137042823817,
|
| 273 |
+
"including_prefill_mean_kld": 0.04350757595049225,
|
| 274 |
+
"prefill_row_kld": 0.2608040744823556,
|
| 275 |
+
"raw_sha256": "6e8cc5a72d7df8f166b66eb2e84d1fad1947b65002b6112bddf3e824f4adeb39",
|
| 276 |
+
"score_sha256": "5274f9fb536c8add3acf2f38dcb460002193f9eb13e9425607330a29a9389876",
|
| 277 |
+
"teacher_sha256": "00701ad8bf4a4a5eedcedd4eafd956830c9ff1c790b15074ca721a3462abf930",
|
| 278 |
+
"token_sha256": "fc1166707f5938a914faf364dab717b0de3ea0e8451ed1a3fb399469e659b105",
|
| 279 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 280 |
+
"raw_bytes": 1268157440
|
| 281 |
+
},
|
| 282 |
+
{
|
| 283 |
+
"window_id": "conditional-fit-0010",
|
| 284 |
+
"domain": "axis3_code_agentic",
|
| 285 |
+
"prediction_rows": 2047,
|
| 286 |
+
"true_decode_rows": 2046,
|
| 287 |
+
"true_decode_mean_kld": 0.016893295516952813,
|
| 288 |
+
"including_prefill_mean_kld": 0.016892189177210668,
|
| 289 |
+
"prefill_row_kld": 0.014628618064772496,
|
| 290 |
+
"raw_sha256": "30eddaed5438035e494759f7a8e5def1346ff8cc3d1144ff135459781ee8c36d",
|
| 291 |
+
"score_sha256": "a982b303a18638a696fd13660a23cfedfea7373a61663b08e0438d7834345726",
|
| 292 |
+
"teacher_sha256": "f9bf4519890b70d972dfdf9760181418eb7c55b6237c0108f01d44df89943278",
|
| 293 |
+
"token_sha256": "ddd9d769a76eba74bc8e680ff25736a3b9146371801d1d2b29e4dc48f6d256a2",
|
| 294 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 295 |
+
"raw_bytes": 1268157440
|
| 296 |
+
},
|
| 297 |
+
{
|
| 298 |
+
"window_id": "conditional-fit-0011",
|
| 299 |
+
"domain": "axis4_reasoning_termination",
|
| 300 |
+
"prediction_rows": 2047,
|
| 301 |
+
"true_decode_rows": 2046,
|
| 302 |
+
"true_decode_mean_kld": 0.016308449522884096,
|
| 303 |
+
"including_prefill_mean_kld": 0.02065847980796188,
|
| 304 |
+
"prefill_row_kld": 8.920820443077092,
|
| 305 |
+
"raw_sha256": "b6098a0ba73fd75f000bbfd23b395eac80f5c3529f5cc55b592cd44c49d3f272",
|
| 306 |
+
"score_sha256": "4c9d123aaaaf9eb62e494447f1cd08cb61b905d70df3e4b461ce06c48fd4b15b",
|
| 307 |
+
"teacher_sha256": "9f37c4ead38bb80a5d85f825b156e5b493b6d3f1ca9dabfad06b0416f37db286",
|
| 308 |
+
"token_sha256": "c4b96acfa47944be3485b20eab76eb16d59913c9fedb5001a537335a3ee13614",
|
| 309 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 310 |
+
"raw_bytes": 1268157440
|
| 311 |
+
},
|
| 312 |
+
{
|
| 313 |
+
"window_id": "conditional-fit-0030",
|
| 314 |
+
"domain": "axis3_code_agentic",
|
| 315 |
+
"prediction_rows": 2047,
|
| 316 |
+
"true_decode_rows": 2046,
|
| 317 |
+
"true_decode_mean_kld": 0.02726396346416777,
|
| 318 |
+
"including_prefill_mean_kld": 0.027262084282667205,
|
| 319 |
+
"prefill_row_kld": 0.023417278932510124,
|
| 320 |
+
"raw_sha256": "bc8f5f049ba9fea541f21abd4109e523067d52c9b5ef8cd3df6be759c2d1d954",
|
| 321 |
+
"score_sha256": "5f7dc8e0eafcb5371f8f2a34cd406787b5c35d15cede4c2b167f685629c2f84f",
|
| 322 |
+
"teacher_sha256": "90c08d120cfac926855e15c106dba1e242099bcd0ae6da7655bfaa54d5ce2420",
|
| 323 |
+
"token_sha256": "d644a79da2ba01cefc5f2020eab9e8662d45ba3025377844e4bf4d3f1d36802d",
|
| 324 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 325 |
+
"raw_bytes": 1268157440
|
| 326 |
+
},
|
| 327 |
+
{
|
| 328 |
+
"window_id": "conditional-fit-0032",
|
| 329 |
+
"domain": "axis1_general",
|
| 330 |
+
"prediction_rows": 2047,
|
| 331 |
+
"true_decode_rows": 2046,
|
| 332 |
+
"true_decode_mean_kld": 0.033507635709301424,
|
| 333 |
+
"including_prefill_mean_kld": 0.03461520895529943,
|
| 334 |
+
"prefill_row_kld": 2.3007100702672156,
|
| 335 |
+
"raw_sha256": "06846c021ae34fc753d82f7163b2c4fecb1ea4d093d1a3c4e494afbff734a9a0",
|
| 336 |
+
"score_sha256": "e561f17f20aa32b915f1df05ee4af640cb8d053ad9c8df19c6fffdf27d9f01b0",
|
| 337 |
+
"teacher_sha256": "1aa70ccf8e224c3a58834e2232a3a2477768db29ffeca39f5b27c17ad8c40c2b",
|
| 338 |
+
"token_sha256": "88745c333ead140a6fb05929031613429cbd42e25881d92add968a94a5d2d930",
|
| 339 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 340 |
+
"raw_bytes": 1268157440
|
| 341 |
+
},
|
| 342 |
+
{
|
| 343 |
+
"window_id": "conditional-fit-0035",
|
| 344 |
+
"domain": "axis4_reasoning_termination",
|
| 345 |
+
"prediction_rows": 2047,
|
| 346 |
+
"true_decode_rows": 2046,
|
| 347 |
+
"true_decode_mean_kld": 0.015753355570421665,
|
| 348 |
+
"including_prefill_mean_kld": 0.015747717272730583,
|
| 349 |
+
"prefill_row_kld": 0.004211760196777308,
|
| 350 |
+
"raw_sha256": "c6735072659f7f4c6f1c9dc5aa12bd496cd81be2c55f5198023d60e5b2bebcc8",
|
| 351 |
+
"score_sha256": "0d8da276e156afdb165e71e5982d7980741407a093267641016b7880cd2c0a9d",
|
| 352 |
+
"teacher_sha256": "4e958dce74e093eb62ba5ac210fe0a4e50f12f1875de678769700cde206b429d",
|
| 353 |
+
"token_sha256": "49f702136da39db9e762b6815a1952c0eba0a1fb04af22fb28a27bb63e2a41dc",
|
| 354 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 355 |
+
"raw_bytes": 1268157440
|
| 356 |
+
},
|
| 357 |
+
{
|
| 358 |
+
"window_id": "conditional-fit-0040",
|
| 359 |
+
"domain": "axis1_general",
|
| 360 |
+
"prediction_rows": 2047,
|
| 361 |
+
"true_decode_rows": 2046,
|
| 362 |
+
"true_decode_mean_kld": 0.04597109826675457,
|
| 363 |
+
"including_prefill_mean_kld": 0.04706420401414631,
|
| 364 |
+
"prefill_row_kld": 2.2835585631776425,
|
| 365 |
+
"raw_sha256": "615796ddf57e73c358a672505908dee823ae3da813ad0e43d69d2e5b15c07d80",
|
| 366 |
+
"score_sha256": "53206f5c9c253f5406ea7b054cfa816db5112665f5c2be6561e2016d1fa402ed",
|
| 367 |
+
"teacher_sha256": "f4c96798fc9eb5564d9aec52779cc0ce88bc55312dcd7663240938d711c8a90d",
|
| 368 |
+
"token_sha256": "50f3bc08ab37c551a05364eb48f0d726742f19331e79114dd5a9fb3c42ce6d4f",
|
| 369 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 370 |
+
"raw_bytes": 1268157440
|
| 371 |
+
},
|
| 372 |
+
{
|
| 373 |
+
"window_id": "conditional-fit-0041",
|
| 374 |
+
"domain": "axis2_legal",
|
| 375 |
+
"prediction_rows": 2047,
|
| 376 |
+
"true_decode_rows": 2046,
|
| 377 |
+
"true_decode_mean_kld": 0.07671995932113433,
|
| 378 |
+
"including_prefill_mean_kld": 0.077568246897156,
|
| 379 |
+
"prefill_row_kld": 1.8131646274374966,
|
| 380 |
+
"raw_sha256": "f3486e51a876933be29f0ff6656920a2b778351cbc722f24c0829d6e5735901e",
|
| 381 |
+
"score_sha256": "f94f09523579b471bd22d57bee5657add70718f81bf13271592ae266f0c401a3",
|
| 382 |
+
"teacher_sha256": "c2bbe4ea31532baceff966423c6ce6e042aebb5a2eb1de45e382a7d3df6f6827",
|
| 383 |
+
"token_sha256": "8acf2614f0f787c185af112f21bcdf247c9b6ae76d292ada28f476e2a7b0241d",
|
| 384 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 385 |
+
"raw_bytes": 1268157440
|
| 386 |
+
},
|
| 387 |
+
{
|
| 388 |
+
"window_id": "conditional-fit-0046",
|
| 389 |
+
"domain": "axis3_code_agentic",
|
| 390 |
+
"prediction_rows": 2047,
|
| 391 |
+
"true_decode_rows": 2046,
|
| 392 |
+
"true_decode_mean_kld": 0.07234383010849382,
|
| 393 |
+
"including_prefill_mean_kld": 0.07241689437092738,
|
| 394 |
+
"prefill_row_kld": 0.22190637530999638,
|
| 395 |
+
"raw_sha256": "04c26659f459b862e0584309731b22e5d21d01a9e4449b726053a16b3796d642",
|
| 396 |
+
"score_sha256": "fcd1f44a5795aa1c23ff4c21f90db35b93a5419ece26a0553a4518b3e566970b",
|
| 397 |
+
"teacher_sha256": "3caa5c58fb0b0e80129d551b39867fd52fefd93830060bb5655862fdd0e3b8ce",
|
| 398 |
+
"token_sha256": "3f2df1c090abf4473163a078c168fec33ab98ed6415985b92449caa861c04444",
|
| 399 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 400 |
+
"raw_bytes": 1268157440
|
| 401 |
+
},
|
| 402 |
+
{
|
| 403 |
+
"window_id": "conditional-fit-0060",
|
| 404 |
+
"domain": "axis1_general",
|
| 405 |
+
"prediction_rows": 2047,
|
| 406 |
+
"true_decode_rows": 2046,
|
| 407 |
+
"true_decode_mean_kld": 0.019628373069938383,
|
| 408 |
+
"including_prefill_mean_kld": 0.020660688213730436,
|
| 409 |
+
"prefill_row_kld": 2.1327774724122643,
|
| 410 |
+
"raw_sha256": "688a90f82f45f3dad38f0a72a3b663e0a71ffc5e9ca80528e53e43879bba4610",
|
| 411 |
+
"score_sha256": "d96ce7e3e1f13a8b49afb666fc3051cb23760b27f7ede74d4c5405d18de22f44",
|
| 412 |
+
"teacher_sha256": "d9e6344664b92527ee58d3bca038702676b357e7052ade03ac4bc9c70908cc34",
|
| 413 |
+
"token_sha256": "04ffe38aafcd9407b2846d59e8d797bee04b97f00683f8bc74d475a2d6b8592b",
|
| 414 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 415 |
+
"raw_bytes": 1268157440
|
| 416 |
+
},
|
| 417 |
+
{
|
| 418 |
+
"window_id": "conditional-fit-0063",
|
| 419 |
+
"domain": "axis4_reasoning_termination",
|
| 420 |
+
"prediction_rows": 2047,
|
| 421 |
+
"true_decode_rows": 2046,
|
| 422 |
+
"true_decode_mean_kld": 0.026129365452294863,
|
| 423 |
+
"including_prefill_mean_kld": 0.026535091377105725,
|
| 424 |
+
"prefill_row_kld": 0.8566503335401285,
|
| 425 |
+
"raw_sha256": "01445f5b0df02c5d68900360da39e4d7c52ffc099dc8f89d58ceecce3d06b305",
|
| 426 |
+
"score_sha256": "b34a6f8bb4a69bfc147bc76df6b8ca1fae7425c53a02ba4cd570b92df5cefc77",
|
| 427 |
+
"teacher_sha256": "a9bb59659437c1df5b3f3ef0b54eda61402842a2597b76b0124f58b59ecf3ab4",
|
| 428 |
+
"token_sha256": "38003e2d26db850dd8ff39f2e95fa3d1be9a34838c70d9f7e87056b7ef50b67d",
|
| 429 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 430 |
+
"raw_bytes": 1268157440
|
| 431 |
+
},
|
| 432 |
+
{
|
| 433 |
+
"window_id": "conditional-fit-0074",
|
| 434 |
+
"domain": "axis2_legal",
|
| 435 |
+
"prediction_rows": 2047,
|
| 436 |
+
"true_decode_rows": 2046,
|
| 437 |
+
"true_decode_mean_kld": 0.04218677374740498,
|
| 438 |
+
"including_prefill_mean_kld": 0.04276227996673919,
|
| 439 |
+
"prefill_row_kld": 1.2202480047245254,
|
| 440 |
+
"raw_sha256": "6b1b3fadfdc741d4dae747cc9bba50570a49e4e33a23221c4693775904ec7b7b",
|
| 441 |
+
"score_sha256": "67e46c4c9f47e3dcdace4de241760311be0a3d8e50d34e2efa8f801ad0c51ee1",
|
| 442 |
+
"teacher_sha256": "440341730d063ae27bbc15bcbfc879d9d16b3b25691d97b8cfd60be8f32f2313",
|
| 443 |
+
"token_sha256": "6f6618094ffce3f44b50f94cc533d338a3bfbbda7432890cd5ad2c13efd50129",
|
| 444 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 445 |
+
"raw_bytes": 1268157440
|
| 446 |
+
},
|
| 447 |
+
{
|
| 448 |
+
"window_id": "conditional-fit-0099",
|
| 449 |
+
"domain": "axis3_code_agentic",
|
| 450 |
+
"prediction_rows": 2047,
|
| 451 |
+
"true_decode_rows": 2046,
|
| 452 |
+
"true_decode_mean_kld": 0.014557495693748427,
|
| 453 |
+
"including_prefill_mean_kld": 0.014584803788056863,
|
| 454 |
+
"prefill_row_kld": 0.07045716474311164,
|
| 455 |
+
"raw_sha256": "d17316f3263c27d2276a4eaf0e0c9a0db0ff6442d7414be32326aab7fbeb7152",
|
| 456 |
+
"score_sha256": "9af87d56dc24e6b616ffd759b55c7c64c3698effc4a5432e1a93ff9ca68725a1",
|
| 457 |
+
"teacher_sha256": "da8dea840dd57564f0b3675872eaf297631b38a6d8dabcde7f8b54db06623e9a",
|
| 458 |
+
"token_sha256": "07e37956eb606026bdd7479ac5516275e92c0ee834c3c0439a3f300f9099608d",
|
| 459 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 460 |
+
"raw_bytes": 1268157440
|
| 461 |
+
},
|
| 462 |
+
{
|
| 463 |
+
"window_id": "conditional-fit-0113",
|
| 464 |
+
"domain": "axis2_legal",
|
| 465 |
+
"prediction_rows": 2047,
|
| 466 |
+
"true_decode_rows": 2046,
|
| 467 |
+
"true_decode_mean_kld": 0.053326012340420864,
|
| 468 |
+
"including_prefill_mean_kld": 0.05418572832239306,
|
| 469 |
+
"prefill_row_kld": 1.8131646274374966,
|
| 470 |
+
"raw_sha256": "c4e382d7ab1b99a2bc226cabd882f2c7c4e0763163aaff6a1e1374372378ee32",
|
| 471 |
+
"score_sha256": "ca2567079ef41e93fe9ba19a536fa45619090dcf218a0dd1c1b8b40924fee362",
|
| 472 |
+
"teacher_sha256": "6dc5f38f3436d6833b0ab68ce34e0f7ee5cde80359c26602526d8de512fd3207",
|
| 473 |
+
"token_sha256": "742ab50951388e712d20dab49b12ad7eb31490cfe5091a2fbc76f31ed8fa1b09",
|
| 474 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 475 |
+
"raw_bytes": 1268157440
|
| 476 |
+
},
|
| 477 |
+
{
|
| 478 |
+
"window_id": "conditional-fit-0118",
|
| 479 |
+
"domain": "axis1_general",
|
| 480 |
+
"prediction_rows": 2047,
|
| 481 |
+
"true_decode_rows": 2046,
|
| 482 |
+
"true_decode_mean_kld": 0.012467486577109195,
|
| 483 |
+
"including_prefill_mean_kld": 0.012881541710171803,
|
| 484 |
+
"prefill_row_kld": 0.8600383439562671,
|
| 485 |
+
"raw_sha256": "6051661893d329024bb0b34ae027b8ec6075f01634cea83d4814916c1440f807",
|
| 486 |
+
"score_sha256": "a3e717d8dbdf162e89757147d2143be0ca13d6aee8396556d949646361133efc",
|
| 487 |
+
"teacher_sha256": "e296287593aa0dc89b695a6a082a224a85904757c4a3cefa69729394d1469f89",
|
| 488 |
+
"token_sha256": "2e03fdfcf4d54fc3174ee4620d5efd9f2e5c10e2e6e0b1217577f08392a09749",
|
| 489 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 490 |
+
"raw_bytes": 1268157440
|
| 491 |
+
}
|
| 492 |
+
],
|
| 493 |
+
"domains": {
|
| 494 |
+
"axis1_general": 0.030614171915487813,
|
| 495 |
+
"axis2_legal": 0.04639708824943424,
|
| 496 |
+
"axis3_code_agentic": 0.031949470625508854,
|
| 497 |
+
"axis4_reasoning_termination": 0.018819961875753848
|
| 498 |
+
},
|
| 499 |
+
"correctness_profile_engine_capacity_tokens": [
|
| 500 |
+
19355555
|
| 501 |
+
]
|
| 502 |
+
},
|
| 503 |
+
"nvfp4": {
|
| 504 |
+
"mean_true_decode_kld": 0.03545622377215776,
|
| 505 |
+
"window_bca95": [
|
| 506 |
+
0.029555884771305184,
|
| 507 |
+
0.04346203678097485
|
| 508 |
+
],
|
| 509 |
+
"per_window": [
|
| 510 |
+
{
|
| 511 |
+
"window_id": "conditional-fit-0056",
|
| 512 |
+
"domain": "axis1_general",
|
| 513 |
+
"prediction_rows": 2047,
|
| 514 |
+
"true_decode_rows": 2046,
|
| 515 |
+
"true_decode_mean_kld": 0.023563083317839253,
|
| 516 |
+
"including_prefill_mean_kld": 0.025769605740888088,
|
| 517 |
+
"prefill_row_kld": 4.540314483298789,
|
| 518 |
+
"raw_sha256": "435e6c97b6f953bed64d5bc5fe1a66a22a35b5c8a0a5e181b0054577b69c89cd",
|
| 519 |
+
"score_sha256": "76b63f8ba02de8d5c913bcc6870fcf73890f72f8eeecae981741c148a0886c71",
|
| 520 |
+
"teacher_sha256": "55a70a330f39c5f23294c23d270a1ddc53031def99a42206e2706d4d0bb4466b",
|
| 521 |
+
"token_sha256": "219d8cd8873c6cf41f52782297cc5cfa9c4ed9666f2bc30d84cee9caa68ea744",
|
| 522 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 523 |
+
"raw_bytes": 1268157440
|
| 524 |
+
},
|
| 525 |
+
{
|
| 526 |
+
"window_id": "conditional-fit-0008",
|
| 527 |
+
"domain": "axis1_general",
|
| 528 |
+
"prediction_rows": 2047,
|
| 529 |
+
"true_decode_rows": 2046,
|
| 530 |
+
"true_decode_mean_kld": 0.06856101759452098,
|
| 531 |
+
"including_prefill_mean_kld": 0.06958029764551615,
|
| 532 |
+
"prefill_row_kld": 2.1550272819816234,
|
| 533 |
+
"raw_sha256": "0bcd5339efda6b717b15343f5de53e83d0757e1fd96aa9b8e08840b11fd32be2",
|
| 534 |
+
"score_sha256": "592c290d69ded521dfbceb1bfda56d0a23550bf6bc90c1b33692cecfe34aa2ee",
|
| 535 |
+
"teacher_sha256": "040641247b2e060035d89c2e7345e029d7c142983585e7c5d63dc818486f650d",
|
| 536 |
+
"token_sha256": "581d72e3e78b15c02d9e73f6bc742260cb57cdcd74337f4419c8f90cbc7c9c11",
|
| 537 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 538 |
+
"raw_bytes": 1268157440
|
| 539 |
+
},
|
| 540 |
+
{
|
| 541 |
+
"window_id": "conditional-fit-0100",
|
| 542 |
+
"domain": "axis1_general",
|
| 543 |
+
"prediction_rows": 2047,
|
| 544 |
+
"true_decode_rows": 2046,
|
| 545 |
+
"true_decode_mean_kld": 0.02667201133477343,
|
| 546 |
+
"including_prefill_mean_kld": 0.027072764083237236,
|
| 547 |
+
"prefill_row_kld": 0.8470128874401833,
|
| 548 |
+
"raw_sha256": "b75fba8ee764641d60da5e9aa16fc0d8331c1f6fc90bf9da54c8fd9d3583dd6b",
|
| 549 |
+
"score_sha256": "83558033705994a6c4563da73feca28252fde4e18e6fdb0e1a1597fbaf785ebe",
|
| 550 |
+
"teacher_sha256": "f62991383b3ef5e8db0ed71442b17c428972e58da1c868c6aeb0f3fa143c865b",
|
| 551 |
+
"token_sha256": "7f008ec99068f373c6e15dd059ee977fb70e6dc832f46e009f0f299d48d19892",
|
| 552 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 553 |
+
"raw_bytes": 1268157440
|
| 554 |
+
},
|
| 555 |
+
{
|
| 556 |
+
"window_id": "conditional-fit-0094",
|
| 557 |
+
"domain": "axis1_general",
|
| 558 |
+
"prediction_rows": 2047,
|
| 559 |
+
"true_decode_rows": 2046,
|
| 560 |
+
"true_decode_mean_kld": 0.01980333069056471,
|
| 561 |
+
"including_prefill_mean_kld": 0.019910660629571305,
|
| 562 |
+
"prefill_row_kld": 0.2395077158370706,
|
| 563 |
+
"raw_sha256": "387038afed9bd884c7430caed10c37159f96b475f782f74c42f39d161289c416",
|
| 564 |
+
"score_sha256": "4a29ea890ee1c5fb5ef4b38af4faf3a2d0ccbf4b9e039cd4e2fae795ab9adad0",
|
| 565 |
+
"teacher_sha256": "656911b8fb34d7f42e984dd21311caf9ef56db0b7618b414cad4089562d93cba",
|
| 566 |
+
"token_sha256": "6b6a9b21c49a9984cc8c5ea90e0bda6db5761d039c7cdd335dc65e30444e017b",
|
| 567 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 568 |
+
"raw_bytes": 1268157440
|
| 569 |
+
},
|
| 570 |
+
{
|
| 571 |
+
"window_id": "conditional-fit-0083",
|
| 572 |
+
"domain": "axis2_legal",
|
| 573 |
+
"prediction_rows": 2047,
|
| 574 |
+
"true_decode_rows": 2046,
|
| 575 |
+
"true_decode_mean_kld": 0.029895885238883084,
|
| 576 |
+
"including_prefill_mean_kld": 0.03010492019686818,
|
| 577 |
+
"prefill_row_kld": 0.4577904442343803,
|
| 578 |
+
"raw_sha256": "caeca2ce6c7e3883a51eb1fe7af4f3f945fac4122c763b7ff6770fa3e8fc3532",
|
| 579 |
+
"score_sha256": "cf6e4f767052c8393f8f7f422237981e13589db63db732ef56478b955b0cfec2",
|
| 580 |
+
"teacher_sha256": "66a7870a1f13c8b5d647c1681fca005bb0f55a6e502e94f282da673a6bed3f1b",
|
| 581 |
+
"token_sha256": "3df9d2872815e061ff8d32a9a19da0d11d80ac67070cccd28579b24ec3079552",
|
| 582 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 583 |
+
"raw_bytes": 1268157440
|
| 584 |
+
},
|
| 585 |
+
{
|
| 586 |
+
"window_id": "conditional-fit-0021",
|
| 587 |
+
"domain": "axis2_legal",
|
| 588 |
+
"prediction_rows": 2047,
|
| 589 |
+
"true_decode_rows": 2046,
|
| 590 |
+
"true_decode_mean_kld": 0.04397075315325536,
|
| 591 |
+
"including_prefill_mean_kld": 0.044495762997015596,
|
| 592 |
+
"prefill_row_kld": 1.1186659033304427,
|
| 593 |
+
"raw_sha256": "ff68e13dedc4ed15d16f1e3b7651f3b030cd406b4e0840480b33dc2fa2dbdfdb",
|
| 594 |
+
"score_sha256": "677465d5f820174bf8485fb26bc023cc12882f11a493e62ca2ce0b114ac7bcef",
|
| 595 |
+
"teacher_sha256": "ebf76910c4db8f6955ad03f2ec7352e2f8aaf15e0c939d28f9a7d78f6bd7451b",
|
| 596 |
+
"token_sha256": "dbd9f2539cb3e97127ae760a04ca5b3456b0b13d6ce8698f88b215cc68b8f1a9",
|
| 597 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 598 |
+
"raw_bytes": 1268157440
|
| 599 |
+
},
|
| 600 |
+
{
|
| 601 |
+
"window_id": "conditional-fit-0098",
|
| 602 |
+
"domain": "axis2_legal",
|
| 603 |
+
"prediction_rows": 2047,
|
| 604 |
+
"true_decode_rows": 2046,
|
| 605 |
+
"true_decode_mean_kld": 0.05041230633987613,
|
| 606 |
+
"including_prefill_mean_kld": 0.051236972335366136,
|
| 607 |
+
"prefill_row_kld": 1.738503599107944,
|
| 608 |
+
"raw_sha256": "4b290f9027a497d7af5bc5c06174dae87e2723fb95b93f3d2e02ffcf355d5234",
|
| 609 |
+
"score_sha256": "1e3fcd31518d465b473c11fbed05d857b572785311b32c88e7fa6ba9a7e75a18",
|
| 610 |
+
"teacher_sha256": "16e3d9524100836c020535635f7722a4c93ce0fdddc2ad0781f25bda77e8bcc9",
|
| 611 |
+
"token_sha256": "749bda3291a0323d286b0018b7dc3e0f469de8afd5d194370cde357b7b315c7d",
|
| 612 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 613 |
+
"raw_bytes": 1268157440
|
| 614 |
+
},
|
| 615 |
+
{
|
| 616 |
+
"window_id": "conditional-fit-0080",
|
| 617 |
+
"domain": "axis2_legal",
|
| 618 |
+
"prediction_rows": 2047,
|
| 619 |
+
"true_decode_rows": 2046,
|
| 620 |
+
"true_decode_mean_kld": 0.047528968125352955,
|
| 621 |
+
"including_prefill_mean_kld": 0.04910846403209107,
|
| 622 |
+
"prefill_row_kld": 3.2807570892182745,
|
| 623 |
+
"raw_sha256": "68f5c356fbf444f5d0022938812e368f1ba97c6cc2f7541072c1609923160030",
|
| 624 |
+
"score_sha256": "87dda561523fd7657fe08d596e9f14bc21a47d1cba5f5f425e392b4ea4d436dc",
|
| 625 |
+
"teacher_sha256": "e9c1a77490a5db63d410297bd8febe06adf4df2a6f4e16552db187738c9289d2",
|
| 626 |
+
"token_sha256": "a31d43f38710eb305c684501a2b02909a4e82c43101c07e76e4a3ca4edb4d0dd",
|
| 627 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 628 |
+
"raw_bytes": 1268157440
|
| 629 |
+
},
|
| 630 |
+
{
|
| 631 |
+
"window_id": "conditional-fit-0081",
|
| 632 |
+
"domain": "axis3_code_agentic",
|
| 633 |
+
"prediction_rows": 2047,
|
| 634 |
+
"true_decode_rows": 2046,
|
| 635 |
+
"true_decode_mean_kld": 0.025434826960204975,
|
| 636 |
+
"including_prefill_mean_kld": 0.026087226927511527,
|
| 637 |
+
"prefill_row_kld": 1.360897560036721,
|
| 638 |
+
"raw_sha256": "55a8ed57d62691e0e1252f3f244d14fe16a7382025f276fd851612b4b5d71ffc",
|
| 639 |
+
"score_sha256": "b3a51927b33af5a34e3a67c19f25e8cb70e335a25e09b270ea8916213a430f55",
|
| 640 |
+
"teacher_sha256": "5ae07760d91030d1eba925b76ea463309baa49c2e7d5f5e14409a23ced7fd20d",
|
| 641 |
+
"token_sha256": "350628488b94d59f3e4f3e99a418913544ea1fe7eacee14a642b996d9c614d16",
|
| 642 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 643 |
+
"raw_bytes": 1268157440
|
| 644 |
+
},
|
| 645 |
+
{
|
| 646 |
+
"window_id": "conditional-fit-0123",
|
| 647 |
+
"domain": "axis3_code_agentic",
|
| 648 |
+
"prediction_rows": 2047,
|
| 649 |
+
"true_decode_rows": 2046,
|
| 650 |
+
"true_decode_mean_kld": 0.03625410701112817,
|
| 651 |
+
"including_prefill_mean_kld": 0.03623887373163693,
|
| 652 |
+
"prefill_row_kld": 0.0050715838925684655,
|
| 653 |
+
"raw_sha256": "9b464ec3a977d1ee3cebc601ffb2a2169d0cf3a9ce70673ce1ea96d807947b12",
|
| 654 |
+
"score_sha256": "47f8a43f8c66147cabaedafeea53145f78935ed971f3cc45d5ec7f8870970954",
|
| 655 |
+
"teacher_sha256": "4a24b5b750ba98c06d9bc69e19334aff89e7bd4f9429592598af5d62de7ecb42",
|
| 656 |
+
"token_sha256": "aab84ee426f91621d123aa68bd53f75e220fbaf8cb270db1feab1529a62ddbf0",
|
| 657 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 658 |
+
"raw_bytes": 1268157440
|
| 659 |
+
},
|
| 660 |
+
{
|
| 661 |
+
"window_id": "conditional-fit-0090",
|
| 662 |
+
"domain": "axis3_code_agentic",
|
| 663 |
+
"prediction_rows": 2047,
|
| 664 |
+
"true_decode_rows": 2046,
|
| 665 |
+
"true_decode_mean_kld": 0.04104338344204917,
|
| 666 |
+
"including_prefill_mean_kld": 0.04195193367340904,
|
| 667 |
+
"prefill_row_kld": 1.9008457070357228,
|
| 668 |
+
"raw_sha256": "890278e84ba9ed1f86276a9702da857639ec6542257c80a8c84ccfb5c9485f63",
|
| 669 |
+
"score_sha256": "5e6e6d5b823b9f005abb50e738519564f2fc407837530abdf906b7cad8348e39",
|
| 670 |
+
"teacher_sha256": "db2a5fe33c9e457a22ebeb31a7073c1dac666b877503e9b4e90a8d6abf7b8a32",
|
| 671 |
+
"token_sha256": "a06e76993db717afee847fcc4d86f10b52dd23cdce5dae916f32d238d08d886c",
|
| 672 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 673 |
+
"raw_bytes": 1268157440
|
| 674 |
+
},
|
| 675 |
+
{
|
| 676 |
+
"window_id": "conditional-fit-0062",
|
| 677 |
+
"domain": "axis3_code_agentic",
|
| 678 |
+
"prediction_rows": 2047,
|
| 679 |
+
"true_decode_rows": 2046,
|
| 680 |
+
"true_decode_mean_kld": 0.02918700579461782,
|
| 681 |
+
"including_prefill_mean_kld": 0.030030454221016484,
|
| 682 |
+
"prefill_row_kld": 1.7557259346326815,
|
| 683 |
+
"raw_sha256": "4ee3b4a4a812c85274c3c836dcd86d0b123e38e4e33bbab437385e1955636035",
|
| 684 |
+
"score_sha256": "6a54c1c1ebae3e5b81823df2342e1d41c4b77ee7c9bbf26d205163b937fcca75",
|
| 685 |
+
"teacher_sha256": "d6f7883a533625c7c75119fe7403b48bf81885b1de0b311a1a0571949f7f231c",
|
| 686 |
+
"token_sha256": "7c6d2ef2f4b7c7091a517f6ccd778009f9ee144165c9aecf2f8f28aea7b3753f",
|
| 687 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 688 |
+
"raw_bytes": 1268157440
|
| 689 |
+
},
|
| 690 |
+
{
|
| 691 |
+
"window_id": "conditional-fit-0047",
|
| 692 |
+
"domain": "axis4_reasoning_termination",
|
| 693 |
+
"prediction_rows": 2047,
|
| 694 |
+
"true_decode_rows": 2046,
|
| 695 |
+
"true_decode_mean_kld": 0.025365434304478303,
|
| 696 |
+
"including_prefill_mean_kld": 0.03031257573187498,
|
| 697 |
+
"prefill_row_kld": 10.152163936185477,
|
| 698 |
+
"raw_sha256": "dd5c4505db35935c58c9f74bb6e74a6edd3c19374f835664d81a6e8de5b33eb9",
|
| 699 |
+
"score_sha256": "d6e3feb817873e71e0dd76cdfa4a8f638ca2b3b8ee42264f726be54b4d8f0845",
|
| 700 |
+
"teacher_sha256": "8513cdcb9940786c7c42bf14d2f5f2cd452791e4ec6036c06a1f973eb969018d",
|
| 701 |
+
"token_sha256": "89730ee6826d4302c352156d5df766c51622a8b6681b8052d4ca6e7d481177fe",
|
| 702 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 703 |
+
"raw_bytes": 1268157440
|
| 704 |
+
},
|
| 705 |
+
{
|
| 706 |
+
"window_id": "conditional-fit-0051",
|
| 707 |
+
"domain": "axis4_reasoning_termination",
|
| 708 |
+
"prediction_rows": 2047,
|
| 709 |
+
"true_decode_rows": 2046,
|
| 710 |
+
"true_decode_mean_kld": 0.012924581711263027,
|
| 711 |
+
"including_prefill_mean_kld": 0.013141907486799479,
|
| 712 |
+
"prefill_row_kld": 0.4577904442343803,
|
| 713 |
+
"raw_sha256": "d3d0a1b6cbfdcecf0842b87193c339608d8d51fb39007b7db0ec03f073db51a7",
|
| 714 |
+
"score_sha256": "c590acb2bb149f97ecf9be58ea5884de08396d26e18935a548f2d11c56fd8ce8",
|
| 715 |
+
"teacher_sha256": "177477c19b54dc0b0230e19202d48ab71d52f2b1a37fe383511bb64f0aae2fcf",
|
| 716 |
+
"token_sha256": "c79269b16560e7301549593443c9394007e7a957ad149c46224e7ad271dc6a18",
|
| 717 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 718 |
+
"raw_bytes": 1268157440
|
| 719 |
+
},
|
| 720 |
+
{
|
| 721 |
+
"window_id": "conditional-fit-0055",
|
| 722 |
+
"domain": "axis4_reasoning_termination",
|
| 723 |
+
"prediction_rows": 2047,
|
| 724 |
+
"true_decode_rows": 2046,
|
| 725 |
+
"true_decode_mean_kld": 0.023650142194727084,
|
| 726 |
+
"including_prefill_mean_kld": 0.02412364237713099,
|
| 727 |
+
"prefill_row_kld": 0.9929050155755266,
|
| 728 |
+
"raw_sha256": "0ab63005f5783e2d377aa953c64abe217cc6d1957ba9d164a3daf62b4a4f3693",
|
| 729 |
+
"score_sha256": "c7ca6c112700a4f07e16ad2806d6f2b18483a929964476311de92a1594a43746",
|
| 730 |
+
"teacher_sha256": "7679c828eae4bf08f17598d044904693fc90c05b9d3f189dddeeb846100e43a7",
|
| 731 |
+
"token_sha256": "feec9ac508172dd275ec6452a03493acab5ffdff9bb57161317c77f24daf90f2",
|
| 732 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 733 |
+
"raw_bytes": 1268157440
|
| 734 |
+
},
|
| 735 |
+
{
|
| 736 |
+
"window_id": "conditional-fit-0031",
|
| 737 |
+
"domain": "axis4_reasoning_termination",
|
| 738 |
+
"prediction_rows": 2047,
|
| 739 |
+
"true_decode_rows": 2046,
|
| 740 |
+
"true_decode_mean_kld": 0.02037624180208121,
|
| 741 |
+
"including_prefill_mean_kld": 0.02532582054872674,
|
| 742 |
+
"prefill_row_kld": 10.152163936185477,
|
| 743 |
+
"raw_sha256": "0e31a3893a84a0ec7e1fdb6fb71c2a14d4fb2175229b0019bc57e59728d0380a",
|
| 744 |
+
"score_sha256": "74ac2939aafb8dbe1f1654aacbb4abd0b3b295da5ce29ee0788188acb38dc105",
|
| 745 |
+
"teacher_sha256": "e04a6c04a951efb5a178927e91457535cc67745ce5c81e785cb6d963aac6ce83",
|
| 746 |
+
"token_sha256": "2cff3482b4212537d0c35b88b8a0f479502c8905952430c973d25b1ab0bb5818",
|
| 747 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 748 |
+
"raw_bytes": 1268157440
|
| 749 |
+
},
|
| 750 |
+
{
|
| 751 |
+
"window_id": "conditional-fit-0003",
|
| 752 |
+
"domain": "axis4_reasoning_termination",
|
| 753 |
+
"prediction_rows": 2047,
|
| 754 |
+
"true_decode_rows": 2046,
|
| 755 |
+
"true_decode_mean_kld": 0.02690073030649202,
|
| 756 |
+
"including_prefill_mean_kld": 0.029946398068044746,
|
| 757 |
+
"prefill_row_kld": 6.2613826382049185,
|
| 758 |
+
"raw_sha256": "bc400c83ee2139f177dabf51511a8683e1a73e03f57fc40554c77f563c7da8e0",
|
| 759 |
+
"score_sha256": "dbfa39c66bc06fa999aaac571e8377131c13782d16e47b119caf725f1c7d28f3",
|
| 760 |
+
"teacher_sha256": "7677c89cc3dee8e2a64e3c7947add39475f828921a3a8fe2e92b379b3ed8320b",
|
| 761 |
+
"token_sha256": "d416537c6e8273747bf025e97287a0bb6e3ff047bfee65cf54e4dfe287569b93",
|
| 762 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 763 |
+
"raw_bytes": 1268157440
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"window_id": "conditional-fit-0009",
|
| 767 |
+
"domain": "axis2_legal",
|
| 768 |
+
"prediction_rows": 2047,
|
| 769 |
+
"true_decode_rows": 2046,
|
| 770 |
+
"true_decode_mean_kld": 0.04604780270271153,
|
| 771 |
+
"including_prefill_mean_kld": 0.046153134975366834,
|
| 772 |
+
"prefill_row_kld": 0.26166296482811324,
|
| 773 |
+
"raw_sha256": "b425869ce07538ae045c4d809021ae9564be0171c2776a9e7ab6465ffe11c5d7",
|
| 774 |
+
"score_sha256": "7ce0191e8d42f8aa8b9a45d584ac950a90b92023e131bb2dc60ffc35ee081583",
|
| 775 |
+
"teacher_sha256": "00701ad8bf4a4a5eedcedd4eafd956830c9ff1c790b15074ca721a3462abf930",
|
| 776 |
+
"token_sha256": "fc1166707f5938a914faf364dab717b0de3ea0e8451ed1a3fb399469e659b105",
|
| 777 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 778 |
+
"raw_bytes": 1268157440
|
| 779 |
+
},
|
| 780 |
+
{
|
| 781 |
+
"window_id": "conditional-fit-0010",
|
| 782 |
+
"domain": "axis3_code_agentic",
|
| 783 |
+
"prediction_rows": 2047,
|
| 784 |
+
"true_decode_rows": 2046,
|
| 785 |
+
"true_decode_mean_kld": 0.01563325095084946,
|
| 786 |
+
"including_prefill_mean_kld": 0.015634848716246288,
|
| 787 |
+
"prefill_row_kld": 0.018903876718152742,
|
| 788 |
+
"raw_sha256": "0020ca7d06c49315301acd377253c6fa721a5e8db71c88efaa33eb4ce96ee31b",
|
| 789 |
+
"score_sha256": "53fa079c5a28e23de66c8f16e72d56125b2721d5943a1ea387b4d2ac9ffaa5cf",
|
| 790 |
+
"teacher_sha256": "f9bf4519890b70d972dfdf9760181418eb7c55b6237c0108f01d44df89943278",
|
| 791 |
+
"token_sha256": "ddd9d769a76eba74bc8e680ff25736a3b9146371801d1d2b29e4dc48f6d256a2",
|
| 792 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 793 |
+
"raw_bytes": 1268157440
|
| 794 |
+
},
|
| 795 |
+
{
|
| 796 |
+
"window_id": "conditional-fit-0011",
|
| 797 |
+
"domain": "axis4_reasoning_termination",
|
| 798 |
+
"prediction_rows": 2047,
|
| 799 |
+
"true_decode_rows": 2046,
|
| 800 |
+
"true_decode_mean_kld": 0.018165048102336066,
|
| 801 |
+
"including_prefill_mean_kld": 0.023115707060852503,
|
| 802 |
+
"prefill_row_kld": 10.152163936185477,
|
| 803 |
+
"raw_sha256": "e5460f80bb681f0001beea1baead1660b76ba8565ec4aa06cbbb13e7f59b256f",
|
| 804 |
+
"score_sha256": "1870f3a7be94e65bf76df49d9fdf3690cd4c8563146acbc0186efc9e7c6c7cce",
|
| 805 |
+
"teacher_sha256": "9f37c4ead38bb80a5d85f825b156e5b493b6d3f1ca9dabfad06b0416f37db286",
|
| 806 |
+
"token_sha256": "c4b96acfa47944be3485b20eab76eb16d59913c9fedb5001a537335a3ee13614",
|
| 807 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 808 |
+
"raw_bytes": 1268157440
|
| 809 |
+
},
|
| 810 |
+
{
|
| 811 |
+
"window_id": "conditional-fit-0030",
|
| 812 |
+
"domain": "axis3_code_agentic",
|
| 813 |
+
"prediction_rows": 2047,
|
| 814 |
+
"true_decode_rows": 2046,
|
| 815 |
+
"true_decode_mean_kld": 0.024984801688836707,
|
| 816 |
+
"including_prefill_mean_kld": 0.024978325612706537,
|
| 817 |
+
"prefill_row_kld": 0.011728273850378172,
|
| 818 |
+
"raw_sha256": "9b7a816b646bd6c974b15105d1f7f92d162d594ca262d7c79fa1bcf1afdb980a",
|
| 819 |
+
"score_sha256": "fd35a10287015d380956b9985ce62b7df231d250fc9dbf2f923d3e974a7d29f6",
|
| 820 |
+
"teacher_sha256": "90c08d120cfac926855e15c106dba1e242099bcd0ae6da7655bfaa54d5ce2420",
|
| 821 |
+
"token_sha256": "d644a79da2ba01cefc5f2020eab9e8662d45ba3025377844e4bf4d3f1d36802d",
|
| 822 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 823 |
+
"raw_bytes": 1268157440
|
| 824 |
+
},
|
| 825 |
+
{
|
| 826 |
+
"window_id": "conditional-fit-0032",
|
| 827 |
+
"domain": "axis1_general",
|
| 828 |
+
"prediction_rows": 2047,
|
| 829 |
+
"true_decode_rows": 2046,
|
| 830 |
+
"true_decode_mean_kld": 0.021718153455878218,
|
| 831 |
+
"including_prefill_mean_kld": 0.022319411629929006,
|
| 832 |
+
"prefill_row_kld": 1.2524936357378378,
|
| 833 |
+
"raw_sha256": "b364f38f47ad85d8efe4e8426747acb301a9e63117501450139127fc75039721",
|
| 834 |
+
"score_sha256": "fb18bbe03c546a46151b364d26cf895f7991b8f30fe53489b9447d7cc7e0fa7c",
|
| 835 |
+
"teacher_sha256": "1aa70ccf8e224c3a58834e2232a3a2477768db29ffeca39f5b27c17ad8c40c2b",
|
| 836 |
+
"token_sha256": "88745c333ead140a6fb05929031613429cbd42e25881d92add968a94a5d2d930",
|
| 837 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 838 |
+
"raw_bytes": 1268157440
|
| 839 |
+
},
|
| 840 |
+
{
|
| 841 |
+
"window_id": "conditional-fit-0035",
|
| 842 |
+
"domain": "axis4_reasoning_termination",
|
| 843 |
+
"prediction_rows": 2047,
|
| 844 |
+
"true_decode_rows": 2046,
|
| 845 |
+
"true_decode_mean_kld": 0.032422367247282986,
|
| 846 |
+
"including_prefill_mean_kld": 0.03463729855715423,
|
| 847 |
+
"prefill_row_kld": 4.566386758553708,
|
| 848 |
+
"raw_sha256": "a4c51cbc1ac997dd9c3bb9215dadeca429e88747a33aa61d37445974baac480c",
|
| 849 |
+
"score_sha256": "37eefbf56891e3d8af693c6f8685bbd172337a6f8bf081849e805e78b3148866",
|
| 850 |
+
"teacher_sha256": "4e958dce74e093eb62ba5ac210fe0a4e50f12f1875de678769700cde206b429d",
|
| 851 |
+
"token_sha256": "49f702136da39db9e762b6815a1952c0eba0a1fb04af22fb28a27bb63e2a41dc",
|
| 852 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 853 |
+
"raw_bytes": 1268157440
|
| 854 |
+
},
|
| 855 |
+
{
|
| 856 |
+
"window_id": "conditional-fit-0040",
|
| 857 |
+
"domain": "axis1_general",
|
| 858 |
+
"prediction_rows": 2047,
|
| 859 |
+
"true_decode_rows": 2046,
|
| 860 |
+
"true_decode_mean_kld": 0.05400247271460698,
|
| 861 |
+
"including_prefill_mean_kld": 0.0549407898731226,
|
| 862 |
+
"prefill_row_kld": 1.9747376961960739,
|
| 863 |
+
"raw_sha256": "fd04d2487f3904f117be851809799cc8c320f950d589b3c328db83f9cda9dfcd",
|
| 864 |
+
"score_sha256": "c79be837a60e6ba89adb4b80483b1b10d49aeac719841479ca4eb8980618e9b7",
|
| 865 |
+
"teacher_sha256": "f4c96798fc9eb5564d9aec52779cc0ce88bc55312dcd7663240938d711c8a90d",
|
| 866 |
+
"token_sha256": "50f3bc08ab37c551a05364eb48f0d726742f19331e79114dd5a9fb3c42ce6d4f",
|
| 867 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 868 |
+
"raw_bytes": 1268157440
|
| 869 |
+
},
|
| 870 |
+
{
|
| 871 |
+
"window_id": "conditional-fit-0041",
|
| 872 |
+
"domain": "axis2_legal",
|
| 873 |
+
"prediction_rows": 2047,
|
| 874 |
+
"true_decode_rows": 2046,
|
| 875 |
+
"true_decode_mean_kld": 0.08657045244168773,
|
| 876 |
+
"including_prefill_mean_kld": 0.08720775297518737,
|
| 877 |
+
"prefill_row_kld": 1.3911246445154333,
|
| 878 |
+
"raw_sha256": "8bbe8f464764b145ae367b5ffdbafd78c30bacf85f8bc90f22bbe2db6e7da597",
|
| 879 |
+
"score_sha256": "b571a88deab449256a1c881ef8550aad88b9ea274c98f6182267c11da48e79bf",
|
| 880 |
+
"teacher_sha256": "c2bbe4ea31532baceff966423c6ce6e042aebb5a2eb1de45e382a7d3df6f6827",
|
| 881 |
+
"token_sha256": "8acf2614f0f787c185af112f21bcdf247c9b6ae76d292ada28f476e2a7b0241d",
|
| 882 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 883 |
+
"raw_bytes": 1268157440
|
| 884 |
+
},
|
| 885 |
+
{
|
| 886 |
+
"window_id": "conditional-fit-0046",
|
| 887 |
+
"domain": "axis3_code_agentic",
|
| 888 |
+
"prediction_rows": 2047,
|
| 889 |
+
"true_decode_rows": 2046,
|
| 890 |
+
"true_decode_mean_kld": 0.06463482692444868,
|
| 891 |
+
"including_prefill_mean_kld": 0.0648417166461593,
|
| 892 |
+
"prefill_row_kld": 0.48813808726609126,
|
| 893 |
+
"raw_sha256": "daa52d49b4550a1fd0bdf38c70a737b613369f485bd5051fafae07cb8ce6a5eb",
|
| 894 |
+
"score_sha256": "b25aa33f209bc0e72888e76efe9feef250e2c106ca34142f4c0dbd353a355532",
|
| 895 |
+
"teacher_sha256": "3caa5c58fb0b0e80129d551b39867fd52fefd93830060bb5655862fdd0e3b8ce",
|
| 896 |
+
"token_sha256": "3f2df1c090abf4473163a078c168fec33ab98ed6415985b92449caa861c04444",
|
| 897 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 898 |
+
"raw_bytes": 1268157440
|
| 899 |
+
},
|
| 900 |
+
{
|
| 901 |
+
"window_id": "conditional-fit-0060",
|
| 902 |
+
"domain": "axis1_general",
|
| 903 |
+
"prediction_rows": 2047,
|
| 904 |
+
"true_decode_rows": 2046,
|
| 905 |
+
"true_decode_mean_kld": 0.015704907348830933,
|
| 906 |
+
"including_prefill_mean_kld": 0.01629535873627198,
|
| 907 |
+
"prefill_row_kld": 1.2243588974406492,
|
| 908 |
+
"raw_sha256": "e4067368b626a98b028811ad6772aeaeadd85f04dcdeb9f91dab6658078b3a74",
|
| 909 |
+
"score_sha256": "bab86bd14f05d787511ea71f9bdf76096edf975451528afd499e35b3701a5454",
|
| 910 |
+
"teacher_sha256": "d9e6344664b92527ee58d3bca038702676b357e7052ade03ac4bc9c70908cc34",
|
| 911 |
+
"token_sha256": "04ffe38aafcd9407b2846d59e8d797bee04b97f00683f8bc74d475a2d6b8592b",
|
| 912 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 913 |
+
"raw_bytes": 1268157440
|
| 914 |
+
},
|
| 915 |
+
{
|
| 916 |
+
"window_id": "conditional-fit-0063",
|
| 917 |
+
"domain": "axis4_reasoning_termination",
|
| 918 |
+
"prediction_rows": 2047,
|
| 919 |
+
"true_decode_rows": 2046,
|
| 920 |
+
"true_decode_mean_kld": 0.03132559147375623,
|
| 921 |
+
"including_prefill_mean_kld": 0.031814505567999064,
|
| 922 |
+
"prefill_row_kld": 1.0321327423888638,
|
| 923 |
+
"raw_sha256": "614a33a2276349c6d3ef14734996f4579e50cc22ef755b409471361e7b24699d",
|
| 924 |
+
"score_sha256": "17cfb59297e5f31cfa88eae36930a85afbfd41d4edfdbd51e9c612bfe75af43f",
|
| 925 |
+
"teacher_sha256": "a9bb59659437c1df5b3f3ef0b54eda61402842a2597b76b0124f58b59ecf3ab4",
|
| 926 |
+
"token_sha256": "38003e2d26db850dd8ff39f2e95fa3d1be9a34838c70d9f7e87056b7ef50b67d",
|
| 927 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 928 |
+
"raw_bytes": 1268157440
|
| 929 |
+
},
|
| 930 |
+
{
|
| 931 |
+
"window_id": "conditional-fit-0074",
|
| 932 |
+
"domain": "axis2_legal",
|
| 933 |
+
"prediction_rows": 2047,
|
| 934 |
+
"true_decode_rows": 2046,
|
| 935 |
+
"true_decode_mean_kld": 0.08556837423676537,
|
| 936 |
+
"including_prefill_mean_kld": 0.0867002508399446,
|
| 937 |
+
"prefill_row_kld": 2.4025197809446555,
|
| 938 |
+
"raw_sha256": "9abd2d8047eef0aa95de284928fde34187426078a2759b214f339b67fafea057",
|
| 939 |
+
"score_sha256": "02b7030207fb7736d57c060fbc8208a57254ef4f66b81b6954e3b8b1cb2ca972",
|
| 940 |
+
"teacher_sha256": "440341730d063ae27bbc15bcbfc879d9d16b3b25691d97b8cfd60be8f32f2313",
|
| 941 |
+
"token_sha256": "6f6618094ffce3f44b50f94cc533d338a3bfbbda7432890cd5ad2c13efd50129",
|
| 942 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 943 |
+
"raw_bytes": 1268157440
|
| 944 |
+
},
|
| 945 |
+
{
|
| 946 |
+
"window_id": "conditional-fit-0099",
|
| 947 |
+
"domain": "axis3_code_agentic",
|
| 948 |
+
"prediction_rows": 2047,
|
| 949 |
+
"true_decode_rows": 2046,
|
| 950 |
+
"true_decode_mean_kld": 0.01659116406314401,
|
| 951 |
+
"including_prefill_mean_kld": 0.016611742343204117,
|
| 952 |
+
"prefill_row_kld": 0.058714903346186786,
|
| 953 |
+
"raw_sha256": "5cb40de2fe89feb4655bd5ab724266f6376e2d8559eb6f14838fff92e0f57eda",
|
| 954 |
+
"score_sha256": "c52bf6a766f6db58d17972e8cdb67aaec0d255c27eee59affc117e2bf110f423",
|
| 955 |
+
"teacher_sha256": "da8dea840dd57564f0b3675872eaf297631b38a6d8dabcde7f8b54db06623e9a",
|
| 956 |
+
"token_sha256": "07e37956eb606026bdd7479ac5516275e92c0ee834c3c0439a3f300f9099608d",
|
| 957 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 958 |
+
"raw_bytes": 1268157440
|
| 959 |
+
},
|
| 960 |
+
{
|
| 961 |
+
"window_id": "conditional-fit-0113",
|
| 962 |
+
"domain": "axis2_legal",
|
| 963 |
+
"prediction_rows": 2047,
|
| 964 |
+
"true_decode_rows": 2046,
|
| 965 |
+
"true_decode_mean_kld": 0.0561128193667057,
|
| 966 |
+
"including_prefill_mean_kld": 0.056764999056568295,
|
| 967 |
+
"prefill_row_kld": 1.3911246445154333,
|
| 968 |
+
"raw_sha256": "c232c29068dada31c94b8aba188f91fd007fd4bc2170cec51b51ada2659f0593",
|
| 969 |
+
"score_sha256": "19dcdfe7a722a3f784e204d6b7126be29a6fd82803522814a9f13efba2d99c8c",
|
| 970 |
+
"teacher_sha256": "6dc5f38f3436d6833b0ab68ce34e0f7ee5cde80359c26602526d8de512fd3207",
|
| 971 |
+
"token_sha256": "742ab50951388e712d20dab49b12ad7eb31490cfe5091a2fbc76f31ed8fa1b09",
|
| 972 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 973 |
+
"raw_bytes": 1268157440
|
| 974 |
+
},
|
| 975 |
+
{
|
| 976 |
+
"window_id": "conditional-fit-0118",
|
| 977 |
+
"domain": "axis1_general",
|
| 978 |
+
"prediction_rows": 2047,
|
| 979 |
+
"true_decode_rows": 2046,
|
| 980 |
+
"true_decode_mean_kld": 0.013573318669100111,
|
| 981 |
+
"including_prefill_mean_kld": 0.01402729782126711,
|
| 982 |
+
"prefill_row_kld": 0.9428686431549456,
|
| 983 |
+
"raw_sha256": "1dd59d9376e947ed299c24653b27953232374ae9f1b124692f8dfb9534afdb85",
|
| 984 |
+
"score_sha256": "4f2f5d0ac0658964a5bc7efa3ddb34089a34adc96e31a3e17a428179686992d0",
|
| 985 |
+
"teacher_sha256": "e296287593aa0dc89b695a6a082a224a85904757c4a3cefa69729394d1469f89",
|
| 986 |
+
"token_sha256": "2e03fdfcf4d54fc3174ee4620d5efd9f2e5c10e2e6e0b1217577f08392a09749",
|
| 987 |
+
"raw_retirement": "after durable score and verified raw hash",
|
| 988 |
+
"raw_bytes": 1268157440
|
| 989 |
+
}
|
| 990 |
+
],
|
| 991 |
+
"domains": {
|
| 992 |
+
"axis1_general": 0.030449786890764326,
|
| 993 |
+
"axis2_legal": 0.05576342020065474,
|
| 994 |
+
"axis3_code_agentic": 0.031720420854409875,
|
| 995 |
+
"axis4_reasoning_termination": 0.023891267142802115
|
| 996 |
+
},
|
| 997 |
+
"correctness_profile_engine_capacity_tokens": [
|
| 998 |
+
31557971
|
| 999 |
+
]
|
| 1000 |
+
}
|
| 1001 |
+
},
|
| 1002 |
+
"fp8_minus_nvfp4": -0.003511050605611574,
|
| 1003 |
+
"paired_window_bca95": [
|
| 1004 |
+
-0.008171877159734635,
|
| 1005 |
+
-0.001367594090752266
|
| 1006 |
+
],
|
| 1007 |
+
"fp8_lower_windows": 22,
|
| 1008 |
+
"true_decode_rows_per_window": 2046,
|
| 1009 |
+
"windows": 32,
|
| 1010 |
+
"historical_dcp1_context_only": 0.03418114591027796,
|
| 1011 |
+
"limits": [
|
| 1012 |
+
"DCP4 new versus DCP1 historical; different runtime image.",
|
| 1013 |
+
"MTP off and maxseq1 forced decode; not MTP quality or throughput evidence.",
|
| 1014 |
+
"Fixed FP8 then NVFP4 order; one server preparation per arm, window intervals do not capture server-run variability."
|
| 1015 |
+
],
|
| 1016 |
+
"receipt_audit": [
|
| 1017 |
+
{
|
| 1018 |
+
"arm": "fp8",
|
| 1019 |
+
"windows": 32,
|
| 1020 |
+
"prediction_rows": 65504,
|
| 1021 |
+
"true_decode_rows": 65472,
|
| 1022 |
+
"all_receipt_hashes_and_masks_pass": true,
|
| 1023 |
+
"all_prefix_hit_counters_zero": true,
|
| 1024 |
+
"raw_logits_retired_after_scoring": true,
|
| 1025 |
+
"source_runtime_audit_sha256": "6a3393053b83b33b41a7d49de2fab1803b1efc4843ec5862c1acd8467b993fa4",
|
| 1026 |
+
"source_startup_sha256": "00ce74cfce281b627289f3322829d357467f5bc1f026677ef27cda93ed43fce3"
|
| 1027 |
+
},
|
| 1028 |
+
{
|
| 1029 |
+
"arm": "nvfp4",
|
| 1030 |
+
"windows": 32,
|
| 1031 |
+
"prediction_rows": 65504,
|
| 1032 |
+
"true_decode_rows": 65472,
|
| 1033 |
+
"all_receipt_hashes_and_masks_pass": true,
|
| 1034 |
+
"all_prefix_hit_counters_zero": true,
|
| 1035 |
+
"raw_logits_retired_after_scoring": true,
|
| 1036 |
+
"source_runtime_audit_sha256": "74d7d1914a55a73b5e9de196f557c658603a9501acd2fafb1e540d496b1c7185",
|
| 1037 |
+
"source_startup_sha256": "cae3cb749c61cc18fd23286f41f79266e32bfdc551e0575b3515d21eb993036a"
|
| 1038 |
+
}
|
| 1039 |
+
],
|
| 1040 |
+
"public_base_image": "verdictai/trellismx@sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf",
|
| 1041 |
+
"capture_image_local_id": "sha256:0405a1c0dc128b51069798a5d00b346257bbd006c0deb7e6530a3d005d75de71",
|
| 1042 |
+
"plan_seal": {
|
| 1043 |
+
"algorithm": "sha256",
|
| 1044 |
+
"created_at": "2026-09-09T20:32:50.484730+00:00",
|
| 1045 |
+
"plan_sha256": "8bd19d67a09ddc6ef6bea1afe67306f1a6859e927cb31969730b51e5d13e24fe"
|
| 1046 |
+
},
|
| 1047 |
+
"metric_sha256": "752af6740595791f300cf47f15ef81f8ca85245d5282fc600ef26ee57eae41e6",
|
| 1048 |
+
"role_manifest_sha256": "b5d7e4524eb98ddfbd230a5d9a44de0dc5dbeb796c03e859898b5838e4463d14",
|
| 1049 |
+
"teacher_revision": "7c378d5f17dba158c4c803eff27c346dd0615660",
|
| 1050 |
+
"teacher_model_revision": "a6c167b62691b2bac901344b65cb651a70f53e43",
|
| 1051 |
+
"capture_cpu_test_sha256": "d470ec0385cb9164d789a3a04c5aa638464b656f1c6a0ac12fdc9c2f8a5f37bc",
|
| 1052 |
+
"prior_attempts": []
|
| 1053 |
+
}
|
results/speed-20260909/README.md
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# September 9 TrellisMX FP8 speed experiments
|
| 2 |
+
|
| 3 |
+
Selected runtime: **token-map-hoist reference**, local image identity `sha256:ca6b80188dce154b91f49108b7d87792d2ba6328935afc71b44d1c0e6f6a1adf`. TP4/DCP4, MTP3 probabilistic, CUDA graphs, NVFP4 MLA KV, 24 sequences, batch4096, four RTX PRO6000 GPUs at300W. Expert math remains E4M3 FP8 with UE8M0 scales. Checkpoint unchanged.
|
| 4 |
+
|
| 5 |
+
All rates are tokens/sec; C4 is aggregate. Single exploratory server runs; MTP acceptance/output trajectories and sustained clocks differ. Later cooled screens require90seconds minimum idle and all GPUs<=55C continuously30seconds before each cell. Earlier runs retain their original protocols; do not silently treat them as cooled replications. No independent replication or superiority claim. No measurement reached300t/s C1 or20000t/s prefill.
|
| 6 |
+
|
| 7 |
+
| Measurement | Selected reference | Task-count | Selective grid-floor |
|
| 8 |
+
|---|---:|---:|---:|
|
| 9 |
+
| 32K prefill | 8407.000 | 8439.000 | 8423.000 |
|
| 10 |
+
| 64K prefill | 8407.000 | 8451.000 | 8448.000 |
|
| 11 |
+
| 8K C1 decode | 222.100 | 212.098 | 214.588 |
|
| 12 |
+
| 8K C4 decode aggregate | 318.560 | 325.083 | 326.620 |
|
| 13 |
+
| 16K C1 decode | 216.636 | 208.708 | 217.293 |
|
| 14 |
+
| 16K C4 decode aggregate | 319.157 | 353.532 | 343.859 |
|
| 15 |
+
| 0K C1 decode | 204.611 | 195.372 | 185.422 |
|
| 16 |
+
| 0K C4 decode aggregate | 327.670 | 333.729 | 325.106 |
|
| 17 |
+
|
| 18 |
+
## Selected reference extended context
|
| 19 |
+
|
| 20 |
+
| Measurement | Tokens/sec |
|
| 21 |
+
|---|---:|
|
| 22 |
+
| 32K prefill | 8457.000 |
|
| 23 |
+
| 64K prefill | 8443.000 |
|
| 24 |
+
| 128K prefill | 8323.000 |
|
| 25 |
+
| 32K C1 decode | 204.930 |
|
| 26 |
+
| 32K C4 decode aggregate | 329.343 |
|
| 27 |
+
| 64K C1 decode | 199.446 |
|
| 28 |
+
| 64K C4 decode aggregate | 330.096 |
|
| 29 |
+
| 128K C1 decode | 204.445 |
|
| 30 |
+
| 128K C4 decode aggregate | 334.086 |
|
| 31 |
+
| 8K C8 decode aggregate | 594.639 |
|
| 32 |
+
| 16K C8 decode aggregate | 602.965 |
|
| 33 |
+
| 0K C1 decode | 204.611 |
|
| 34 |
+
| 0K C4 decode aggregate | 327.670 |
|
| 35 |
+
|
| 36 |
+
Task-count and grid-floor were not measured at C1 contexts32K/64K/128K. Their shorter-context results cannot rank long-context C1. Reference KV capacity is23,562,091 tokens in the expanded run; separate zero-context reference sessions reported23,568,627. Use the explicit engine KV metric, not the benchmark's generic block-times-DCP estimate.
|
| 37 |
+
|
| 38 |
+
Reference is retained for balanced use: best observed C1 at0K/8K and effectively tied at16K. Task-count is a concurrent-throughput alternative, especially16K C4. Neither alternative has a meaningful prefill improvement established. Both alternatives passed172 exact representative K4/K5 component comparisons and FP8 compiled-opcode checks; these are not full-model quality evaluations.
|
| 39 |
+
|
| 40 |
+
`benchmark-index.json` indexes every discovered llm_decode_bench result in today's campaign, including earlier/negative/profile diagnostics. Results under profiling paths are diagnostic, not serving throughput qualification. `evidence/` includes raw benchmark JSON and associated logs/receipts plus candidate summaries/failures. Local host/path prefixes are redacted; `public-file-inventory.json` records both original and public hashes. Historical raw receipt hashes refer to the originals, not the redacted copies. Raw local archives are retained separately. KLD is published separately with cache dtype and capture-image provenance.
|
results/speed-20260909/benchmark-index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-command.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0,8k,32k",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--skip-prefill",
|
| 25 |
+
"--cell-warmup-timeout-seconds",
|
| 26 |
+
"180"
|
| 27 |
+
]
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "cd13e14d3d3f0b8b357a7d0c2f80a963c62ba2aa169ff9bbde9ad08d9d3cd255"
|
| 5 |
+
}
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.json
ADDED
|
@@ -0,0 +1,1394 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T07:18:01.203111",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 32768,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 17031168,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0,
|
| 42 |
+
8192,
|
| 43 |
+
32768
|
| 44 |
+
],
|
| 45 |
+
"startup_diagnostics_available": true,
|
| 46 |
+
"nvidia_p2p_override_effective": true,
|
| 47 |
+
"p2pmark_status": "not_run",
|
| 48 |
+
"amd_fabric_status": "not_run"
|
| 49 |
+
},
|
| 50 |
+
"startup_diagnostics": {
|
| 51 |
+
"version": "0.4.29",
|
| 52 |
+
"server_url": "http://127.0.0.1:8001",
|
| 53 |
+
"hostname": "<host>",
|
| 54 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 55 |
+
"env": {},
|
| 56 |
+
"args": {
|
| 57 |
+
"concurrency": "1,2,4",
|
| 58 |
+
"contexts": "0,8k,32k",
|
| 59 |
+
"max_tokens": 8192,
|
| 60 |
+
"duration": 20.0,
|
| 61 |
+
"request_count": 0,
|
| 62 |
+
"run_burst": false,
|
| 63 |
+
"standalone_prefill": false,
|
| 64 |
+
"prefill_only": false,
|
| 65 |
+
"skip_prefill": true,
|
| 66 |
+
"prefill_contexts": "8k,64k,128k",
|
| 67 |
+
"prefill_metric": "client",
|
| 68 |
+
"dcp_size": 0,
|
| 69 |
+
"kv_budget": 0
|
| 70 |
+
},
|
| 71 |
+
"nvidia_p2p_override": {
|
| 72 |
+
"effective": true,
|
| 73 |
+
"configured": true,
|
| 74 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 75 |
+
"params_available": true,
|
| 76 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 77 |
+
"modprobe_available": true,
|
| 78 |
+
"runtime": {
|
| 79 |
+
"ForceP2P": "0x11",
|
| 80 |
+
"RMForceP2PType": "1",
|
| 81 |
+
"RMPcieP2PType": "2",
|
| 82 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 83 |
+
"EnableResizableBar": "1",
|
| 84 |
+
"DmaRemapPeerMmio": "1"
|
| 85 |
+
},
|
| 86 |
+
"expected": {
|
| 87 |
+
"ForceP2P": "0x11",
|
| 88 |
+
"RMForceP2PType": "1",
|
| 89 |
+
"RMPcieP2PType": "2",
|
| 90 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 91 |
+
"EnableResizableBar": "1"
|
| 92 |
+
},
|
| 93 |
+
"missing": [],
|
| 94 |
+
"mismatched": {},
|
| 95 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 96 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 97 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 98 |
+
},
|
| 99 |
+
"p2pmark": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"amd_fabric": {
|
| 103 |
+
"status": "not_run"
|
| 104 |
+
},
|
| 105 |
+
"nvidia_smi_query": {
|
| 106 |
+
"cmd": [
|
| 107 |
+
"nvidia-smi",
|
| 108 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 109 |
+
"--format=csv,noheader,nounits"
|
| 110 |
+
],
|
| 111 |
+
"returncode": 0,
|
| 112 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 113 |
+
"stderr": ""
|
| 114 |
+
},
|
| 115 |
+
"nvidia_smi_topo": {
|
| 116 |
+
"cmd": [
|
| 117 |
+
"nvidia-smi",
|
| 118 |
+
"topo",
|
| 119 |
+
"-m"
|
| 120 |
+
],
|
| 121 |
+
"returncode": 0,
|
| 122 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 123 |
+
"stderr": ""
|
| 124 |
+
}
|
| 125 |
+
},
|
| 126 |
+
"nvidia_p2p_override": {
|
| 127 |
+
"effective": true,
|
| 128 |
+
"configured": true,
|
| 129 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 130 |
+
"params_available": true,
|
| 131 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 132 |
+
"modprobe_available": true,
|
| 133 |
+
"runtime": {
|
| 134 |
+
"ForceP2P": "0x11",
|
| 135 |
+
"RMForceP2PType": "1",
|
| 136 |
+
"RMPcieP2PType": "2",
|
| 137 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 138 |
+
"EnableResizableBar": "1",
|
| 139 |
+
"DmaRemapPeerMmio": "1"
|
| 140 |
+
},
|
| 141 |
+
"expected": {
|
| 142 |
+
"ForceP2P": "0x11",
|
| 143 |
+
"RMForceP2PType": "1",
|
| 144 |
+
"RMPcieP2PType": "2",
|
| 145 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 146 |
+
"EnableResizableBar": "1"
|
| 147 |
+
},
|
| 148 |
+
"missing": [],
|
| 149 |
+
"mismatched": {},
|
| 150 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 151 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 152 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 153 |
+
},
|
| 154 |
+
"p2pmark": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"amd_fabric": {
|
| 158 |
+
"status": "not_run"
|
| 159 |
+
},
|
| 160 |
+
"hardware_run_summary": {
|
| 161 |
+
"samples": 112,
|
| 162 |
+
"duration_seconds": 267.276,
|
| 163 |
+
"gpu_count": 4,
|
| 164 |
+
"cpu_util_avg_pct": 10.31,
|
| 165 |
+
"cpu_temp_max_c": 77.25,
|
| 166 |
+
"gpu_util_avg_pct": 91.56,
|
| 167 |
+
"gpu_util_max_pct": 100.0,
|
| 168 |
+
"mem_util_avg_pct": 39.56,
|
| 169 |
+
"mem_util_max_pct": 55.0,
|
| 170 |
+
"temp_avg_c": 67.18,
|
| 171 |
+
"temp_max_c": 84.0,
|
| 172 |
+
"power_total_avg_w": 1101.05,
|
| 173 |
+
"power_total_max_w": 1178.9,
|
| 174 |
+
"power_limit_total_w": 1200.0,
|
| 175 |
+
"vram_used_avg_mb": 380174.0,
|
| 176 |
+
"vram_used_max_mb": 380174.0,
|
| 177 |
+
"vram_total_mb": 391548.0,
|
| 178 |
+
"vram_used_avg_pct": 97.1,
|
| 179 |
+
"vram_used_max_pct": 97.1,
|
| 180 |
+
"pcie_rx_avg_mb_s": 11487.79,
|
| 181 |
+
"pcie_rx_max_mb_s": 71064.0,
|
| 182 |
+
"pcie_tx_avg_mb_s": 11408.48,
|
| 183 |
+
"pcie_tx_max_mb_s": 66265.0
|
| 184 |
+
},
|
| 185 |
+
"event_log": [
|
| 186 |
+
"07:13:31 benchmark start engine=vllm",
|
| 187 |
+
"07:13:31 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 188 |
+
"07:13:31 startup decode concurrency=1,2,4 contexts=0,8k,32k",
|
| 189 |
+
"07:13:31 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 190 |
+
"07:13:31 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 191 |
+
"07:13:31 startup KV cache budget from vLLM metrics: 17,031,168 tokens (2079 blocks x 2048; local 4,257,792 \u00d7 CP 4; CP source: local process)",
|
| 192 |
+
"07:13:31 startup model context length: 1,000,000 tokens",
|
| 193 |
+
"07:13:31 startup prefill tests: skipped",
|
| 194 |
+
"07:13:31 startup calibrating padding text run=wwbkdgkvesfu up_to=32k",
|
| 195 |
+
"07:13:31 startup context 8k: 50,544 chars (8,192 prompt tokens via /tokenize)",
|
| 196 |
+
"07:13:31 startup context 32k: 205,140 chars (32,768 prompt tokens via /tokenize)",
|
| 197 |
+
"07:13:31 startup token targeting: /tokenize exact",
|
| 198 |
+
"07:13:31 startup startup preparation done",
|
| 199 |
+
"07:13:31 hardware monitor interval=2s",
|
| 200 |
+
"07:13:31 decode warmup start",
|
| 201 |
+
"07:13:31 decode warmup start C=1 ctx=32k 3s",
|
| 202 |
+
"07:13:31 cell start C=1 ctx=32k",
|
| 203 |
+
"07:13:39 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 204 |
+
"07:13:42 cell done C=1 ctx=32k 175.7 tok/s",
|
| 205 |
+
"07:13:42 decode warmup done C=1 ctx=32k",
|
| 206 |
+
"07:13:44 cell start C=1 ctx=0",
|
| 207 |
+
"07:13:50 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 208 |
+
"07:14:10 cell done C=1 ctx=0 167.2 tok/s",
|
| 209 |
+
"07:14:12 cell start C=1 ctx=8k",
|
| 210 |
+
"07:14:18 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 211 |
+
"07:14:38 cell done C=1 ctx=8k 173.0 tok/s",
|
| 212 |
+
"07:14:40 cell start C=1 ctx=32k",
|
| 213 |
+
"07:14:49 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 214 |
+
"07:15:09 cell done C=1 ctx=32k 186.4 tok/s",
|
| 215 |
+
"07:15:11 cell start C=2 ctx=0",
|
| 216 |
+
"07:15:17 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 217 |
+
"07:15:37 cell done C=2 ctx=0 244.6 tok/s",
|
| 218 |
+
"07:15:39 cell start C=4 ctx=0",
|
| 219 |
+
"07:15:44 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 220 |
+
"07:16:04 cell done C=4 ctx=0 292.9 tok/s",
|
| 221 |
+
"07:16:06 cell start C=2 ctx=8k",
|
| 222 |
+
"07:16:12 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 223 |
+
"07:16:32 cell done C=2 ctx=8k 246.0 tok/s",
|
| 224 |
+
"07:16:34 cell start C=4 ctx=8k",
|
| 225 |
+
"07:16:41 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 226 |
+
"07:17:01 cell done C=4 ctx=8k 298.4 tok/s",
|
| 227 |
+
"07:17:04 cell start C=2 ctx=32k",
|
| 228 |
+
"07:17:09 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 229 |
+
"07:17:29 cell done C=2 ctx=32k 252.6 tok/s",
|
| 230 |
+
"07:17:31 cell start C=4 ctx=32k",
|
| 231 |
+
"07:17:39 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 232 |
+
"07:17:59 cell done C=4 ctx=32k 297.7 tok/s"
|
| 233 |
+
],
|
| 234 |
+
"prefill": {},
|
| 235 |
+
"results": [
|
| 236 |
+
{
|
| 237 |
+
"concurrency": 1,
|
| 238 |
+
"context_tokens": 0,
|
| 239 |
+
"benchmark_mode": "duration",
|
| 240 |
+
"request_count_target": 0,
|
| 241 |
+
"warmup_request_count": 0,
|
| 242 |
+
"measurement_seconds": 19.988585,
|
| 243 |
+
"measurement_wall_seconds": 20.00075,
|
| 244 |
+
"client_output_tokens": 3342,
|
| 245 |
+
"server_output_tokens": 3342,
|
| 246 |
+
"aggregate_source": "openai_continuous_usage",
|
| 247 |
+
"aggregate_tps": 167.19542416324822,
|
| 248 |
+
"per_request_avg_tps": 167.19542416324822,
|
| 249 |
+
"ttft_avg": 0.07083372515626252,
|
| 250 |
+
"ttft_p50": 0.07083372515626252,
|
| 251 |
+
"ttft_p90": 0.07083372515626252,
|
| 252 |
+
"ttft_p99": 0.07083372515626252,
|
| 253 |
+
"time_to_second_token_avg": 0.012936464976519346,
|
| 254 |
+
"time_to_second_token_p50": 0.012936464976519346,
|
| 255 |
+
"time_to_second_token_p90": 0.012936464976519346,
|
| 256 |
+
"time_to_second_token_p99": 0.012936464976519346,
|
| 257 |
+
"request_latency_avg": 0.0,
|
| 258 |
+
"request_latency_p50": 0.0,
|
| 259 |
+
"request_latency_p90": 0.0,
|
| 260 |
+
"request_latency_p99": 0.0,
|
| 261 |
+
"inter_token_latency_avg": 0.005911081440074571,
|
| 262 |
+
"inter_token_latency_p50": 0.005911081440074571,
|
| 263 |
+
"inter_token_latency_p90": 0.005911081440074571,
|
| 264 |
+
"inter_token_latency_p99": 0.005911081440074571,
|
| 265 |
+
"output_tps_per_user_avg": 169.17378150475705,
|
| 266 |
+
"output_tps_per_user_p50": 169.17378150475705,
|
| 267 |
+
"output_tps_per_user_p90": 169.17378150475705,
|
| 268 |
+
"output_tps_per_user_p99": 169.17378150475705,
|
| 269 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 270 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 271 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 272 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.01407805126159353,
|
| 274 |
+
"chunk_inter_token_latency_p50": 0.01407805126159353,
|
| 275 |
+
"chunk_inter_token_latency_p90": 0.01407805126159353,
|
| 276 |
+
"chunk_inter_token_latency_p99": 0.01407805126159353,
|
| 277 |
+
"input_seq_len_avg": 78.0,
|
| 278 |
+
"output_seq_len_avg": 4307.0,
|
| 279 |
+
"output_seq_len_p50": 4307.0,
|
| 280 |
+
"output_seq_len_p90": 4307.0,
|
| 281 |
+
"output_seq_len_p99": 4307.0,
|
| 282 |
+
"request_count": 1,
|
| 283 |
+
"completed_request_count": 0,
|
| 284 |
+
"request_samples": [
|
| 285 |
+
{
|
| 286 |
+
"ttft": 0.07083372515626252,
|
| 287 |
+
"time_to_second_token": 0.012936464976519346,
|
| 288 |
+
"latency": 0.0,
|
| 289 |
+
"inter_token_latency_avg": 0.005911081440074571,
|
| 290 |
+
"chunk_inter_token_latency_avg": 0.01407805126159353,
|
| 291 |
+
"input_tokens": 78,
|
| 292 |
+
"output_tokens": 4307,
|
| 293 |
+
"output_tps_per_user": 169.17378150475705,
|
| 294 |
+
"e2e_output_tps_per_user": 0.0,
|
| 295 |
+
"completed": false
|
| 296 |
+
}
|
| 297 |
+
],
|
| 298 |
+
"total_tokens": 3342,
|
| 299 |
+
"wall_time": 25.542422840138897,
|
| 300 |
+
"num_completed": 1,
|
| 301 |
+
"num_errors": 0,
|
| 302 |
+
"server_gen_throughput": 167.0490340328882,
|
| 303 |
+
"server_utilization": 0.0101058710298364,
|
| 304 |
+
"server_spec_accept_rate": 0.49765258215962443,
|
| 305 |
+
"server_spec_accept_length": 0.0,
|
| 306 |
+
"avg_running_reqs": 1,
|
| 307 |
+
"max_running_reqs": 1,
|
| 308 |
+
"effective_concurrency": 1,
|
| 309 |
+
"avg_queue_reqs": 0,
|
| 310 |
+
"max_queue_reqs": 0,
|
| 311 |
+
"queue_fraction": 0.0,
|
| 312 |
+
"underfilled": false,
|
| 313 |
+
"warmup_timed_out": false,
|
| 314 |
+
"warmup_duration": 5.535,
|
| 315 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 316 |
+
"timeout_reason": "",
|
| 317 |
+
"capacity_limited": false,
|
| 318 |
+
"hardware_summary": {
|
| 319 |
+
"samples": 9,
|
| 320 |
+
"duration_seconds": 19.305,
|
| 321 |
+
"gpu_count": 4,
|
| 322 |
+
"cpu_util_avg_pct": 10.88,
|
| 323 |
+
"cpu_temp_max_c": 76.12,
|
| 324 |
+
"gpu_util_avg_pct": 99.0,
|
| 325 |
+
"gpu_util_max_pct": 99.0,
|
| 326 |
+
"mem_util_avg_pct": 48.47,
|
| 327 |
+
"mem_util_max_pct": 53.0,
|
| 328 |
+
"temp_avg_c": 65.39,
|
| 329 |
+
"temp_max_c": 80.0,
|
| 330 |
+
"power_total_avg_w": 1150.89,
|
| 331 |
+
"power_total_max_w": 1153.46,
|
| 332 |
+
"power_limit_total_w": 1200.0,
|
| 333 |
+
"vram_used_avg_mb": 380174.0,
|
| 334 |
+
"vram_used_max_mb": 380174.0,
|
| 335 |
+
"vram_total_mb": 391548.0,
|
| 336 |
+
"vram_used_avg_pct": 97.1,
|
| 337 |
+
"vram_used_max_pct": 97.1,
|
| 338 |
+
"pcie_rx_avg_mb_s": 8406.89,
|
| 339 |
+
"pcie_rx_max_mb_s": 8711.0,
|
| 340 |
+
"pcie_tx_avg_mb_s": 8352.33,
|
| 341 |
+
"pcie_tx_max_mb_s": 8558.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
{
|
| 345 |
+
"concurrency": 1,
|
| 346 |
+
"context_tokens": 8192,
|
| 347 |
+
"benchmark_mode": "duration",
|
| 348 |
+
"request_count_target": 0,
|
| 349 |
+
"warmup_request_count": 0,
|
| 350 |
+
"measurement_seconds": 19.99916,
|
| 351 |
+
"measurement_wall_seconds": 20.000275,
|
| 352 |
+
"client_output_tokens": 3459,
|
| 353 |
+
"server_output_tokens": 3459,
|
| 354 |
+
"aggregate_source": "openai_continuous_usage",
|
| 355 |
+
"aggregate_tps": 172.95726809626717,
|
| 356 |
+
"per_request_avg_tps": 172.95726809626717,
|
| 357 |
+
"ttft_avg": 0.5792566961608827,
|
| 358 |
+
"ttft_p50": 0.5792566961608827,
|
| 359 |
+
"ttft_p90": 0.5792566961608827,
|
| 360 |
+
"ttft_p99": 0.5792566961608827,
|
| 361 |
+
"time_to_second_token_avg": 0.0174833619967103,
|
| 362 |
+
"time_to_second_token_p50": 0.0174833619967103,
|
| 363 |
+
"time_to_second_token_p90": 0.0174833619967103,
|
| 364 |
+
"time_to_second_token_p99": 0.0174833619967103,
|
| 365 |
+
"request_latency_avg": 0.0,
|
| 366 |
+
"request_latency_p50": 0.0,
|
| 367 |
+
"request_latency_p90": 0.0,
|
| 368 |
+
"request_latency_p99": 0.0,
|
| 369 |
+
"inter_token_latency_avg": 0.005649759039614473,
|
| 370 |
+
"inter_token_latency_p50": 0.005649759039614473,
|
| 371 |
+
"inter_token_latency_p90": 0.005649759039614473,
|
| 372 |
+
"inter_token_latency_p99": 0.005649759039614473,
|
| 373 |
+
"output_tps_per_user_avg": 176.9986990574801,
|
| 374 |
+
"output_tps_per_user_p50": 176.9986990574801,
|
| 375 |
+
"output_tps_per_user_p90": 176.9986990574801,
|
| 376 |
+
"output_tps_per_user_p99": 176.9986990574801,
|
| 377 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 378 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 379 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 380 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 381 |
+
"chunk_inter_token_latency_avg": 0.014258915671407954,
|
| 382 |
+
"chunk_inter_token_latency_p50": 0.014258915671407954,
|
| 383 |
+
"chunk_inter_token_latency_p90": 0.014258915671407954,
|
| 384 |
+
"chunk_inter_token_latency_p99": 0.014258915671407954,
|
| 385 |
+
"input_seq_len_avg": 8192.0,
|
| 386 |
+
"output_seq_len_avg": 4241.0,
|
| 387 |
+
"output_seq_len_p50": 4241.0,
|
| 388 |
+
"output_seq_len_p90": 4241.0,
|
| 389 |
+
"output_seq_len_p99": 4241.0,
|
| 390 |
+
"request_count": 1,
|
| 391 |
+
"completed_request_count": 0,
|
| 392 |
+
"request_samples": [
|
| 393 |
+
{
|
| 394 |
+
"ttft": 0.5792566961608827,
|
| 395 |
+
"time_to_second_token": 0.0174833619967103,
|
| 396 |
+
"latency": 0.0,
|
| 397 |
+
"inter_token_latency_avg": 0.005649759039614473,
|
| 398 |
+
"chunk_inter_token_latency_avg": 0.014258915671407954,
|
| 399 |
+
"input_tokens": 8192,
|
| 400 |
+
"output_tokens": 4241,
|
| 401 |
+
"output_tps_per_user": 176.9986990574801,
|
| 402 |
+
"e2e_output_tps_per_user": 0.0,
|
| 403 |
+
"completed": false
|
| 404 |
+
}
|
| 405 |
+
],
|
| 406 |
+
"total_tokens": 3459,
|
| 407 |
+
"wall_time": 26.074295089114457,
|
| 408 |
+
"num_completed": 1,
|
| 409 |
+
"num_errors": 0,
|
| 410 |
+
"server_gen_throughput": 172.904696376223,
|
| 411 |
+
"server_utilization": 0.010587102983638075,
|
| 412 |
+
"server_spec_accept_rate": 0.539906103286385,
|
| 413 |
+
"server_spec_accept_length": 0.0,
|
| 414 |
+
"avg_running_reqs": 1,
|
| 415 |
+
"max_running_reqs": 1,
|
| 416 |
+
"effective_concurrency": 1,
|
| 417 |
+
"avg_queue_reqs": 0,
|
| 418 |
+
"max_queue_reqs": 0,
|
| 419 |
+
"queue_fraction": 0.0,
|
| 420 |
+
"underfilled": false,
|
| 421 |
+
"warmup_timed_out": false,
|
| 422 |
+
"warmup_duration": 6.06,
|
| 423 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 424 |
+
"timeout_reason": "",
|
| 425 |
+
"capacity_limited": false,
|
| 426 |
+
"hardware_summary": {
|
| 427 |
+
"samples": 8,
|
| 428 |
+
"duration_seconds": 16.905,
|
| 429 |
+
"gpu_count": 4,
|
| 430 |
+
"cpu_util_avg_pct": 10.94,
|
| 431 |
+
"cpu_temp_max_c": 75.25,
|
| 432 |
+
"gpu_util_avg_pct": 99.0,
|
| 433 |
+
"gpu_util_max_pct": 99.0,
|
| 434 |
+
"mem_util_avg_pct": 49.0,
|
| 435 |
+
"mem_util_max_pct": 55.0,
|
| 436 |
+
"temp_avg_c": 66.22,
|
| 437 |
+
"temp_max_c": 81.0,
|
| 438 |
+
"power_total_avg_w": 1153.53,
|
| 439 |
+
"power_total_max_w": 1154.66,
|
| 440 |
+
"power_limit_total_w": 1200.0,
|
| 441 |
+
"vram_used_avg_mb": 380174.0,
|
| 442 |
+
"vram_used_max_mb": 380174.0,
|
| 443 |
+
"vram_total_mb": 391548.0,
|
| 444 |
+
"vram_used_avg_pct": 97.1,
|
| 445 |
+
"vram_used_max_pct": 97.1,
|
| 446 |
+
"pcie_rx_avg_mb_s": 8274.25,
|
| 447 |
+
"pcie_rx_max_mb_s": 8601.0,
|
| 448 |
+
"pcie_tx_avg_mb_s": 8292.38,
|
| 449 |
+
"pcie_tx_max_mb_s": 8476.0
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"concurrency": 1,
|
| 454 |
+
"context_tokens": 32768,
|
| 455 |
+
"benchmark_mode": "duration",
|
| 456 |
+
"request_count_target": 0,
|
| 457 |
+
"warmup_request_count": 0,
|
| 458 |
+
"measurement_seconds": 19.999218,
|
| 459 |
+
"measurement_wall_seconds": 20.000322,
|
| 460 |
+
"client_output_tokens": 3727,
|
| 461 |
+
"server_output_tokens": 3727,
|
| 462 |
+
"aggregate_source": "openai_continuous_usage",
|
| 463 |
+
"aggregate_tps": 186.35728993398365,
|
| 464 |
+
"per_request_avg_tps": 186.35728993398365,
|
| 465 |
+
"ttft_avg": 0.5975865018554032,
|
| 466 |
+
"ttft_p50": 0.5975865018554032,
|
| 467 |
+
"ttft_p90": 0.5975865018554032,
|
| 468 |
+
"ttft_p99": 0.5975865018554032,
|
| 469 |
+
"time_to_second_token_avg": 0.011153812054544687,
|
| 470 |
+
"time_to_second_token_p50": 0.011153812054544687,
|
| 471 |
+
"time_to_second_token_p90": 0.011153812054544687,
|
| 472 |
+
"time_to_second_token_p99": 0.011153812054544687,
|
| 473 |
+
"request_latency_avg": 0.0,
|
| 474 |
+
"request_latency_p50": 0.0,
|
| 475 |
+
"request_latency_p90": 0.0,
|
| 476 |
+
"request_latency_p99": 0.0,
|
| 477 |
+
"inter_token_latency_avg": 0.005334254001541856,
|
| 478 |
+
"inter_token_latency_p50": 0.005334254001541856,
|
| 479 |
+
"inter_token_latency_p90": 0.005334254001541856,
|
| 480 |
+
"inter_token_latency_p99": 0.005334254001541856,
|
| 481 |
+
"output_tps_per_user_avg": 187.4676383447342,
|
| 482 |
+
"output_tps_per_user_p50": 187.4676383447342,
|
| 483 |
+
"output_tps_per_user_p90": 187.4676383447342,
|
| 484 |
+
"output_tps_per_user_p99": 187.4676383447342,
|
| 485 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 486 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 487 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 488 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 489 |
+
"chunk_inter_token_latency_avg": 0.014272389806152839,
|
| 490 |
+
"chunk_inter_token_latency_p50": 0.014272389806152839,
|
| 491 |
+
"chunk_inter_token_latency_p90": 0.014272389806152839,
|
| 492 |
+
"chunk_inter_token_latency_p99": 0.014272389806152839,
|
| 493 |
+
"input_seq_len_avg": 32768.0,
|
| 494 |
+
"output_seq_len_avg": 4488.0,
|
| 495 |
+
"output_seq_len_p50": 4488.0,
|
| 496 |
+
"output_seq_len_p90": 4488.0,
|
| 497 |
+
"output_seq_len_p99": 4488.0,
|
| 498 |
+
"request_count": 1,
|
| 499 |
+
"completed_request_count": 0,
|
| 500 |
+
"request_samples": [
|
| 501 |
+
{
|
| 502 |
+
"ttft": 0.5975865018554032,
|
| 503 |
+
"time_to_second_token": 0.011153812054544687,
|
| 504 |
+
"latency": 0.0,
|
| 505 |
+
"inter_token_latency_avg": 0.005334254001541856,
|
| 506 |
+
"chunk_inter_token_latency_avg": 0.014272389806152839,
|
| 507 |
+
"input_tokens": 32768,
|
| 508 |
+
"output_tokens": 4488,
|
| 509 |
+
"output_tps_per_user": 187.4676383447342,
|
| 510 |
+
"e2e_output_tps_per_user": 0.0,
|
| 511 |
+
"completed": false
|
| 512 |
+
}
|
| 513 |
+
],
|
| 514 |
+
"total_tokens": 3727,
|
| 515 |
+
"wall_time": 29.12444451614283,
|
| 516 |
+
"num_completed": 1,
|
| 517 |
+
"num_errors": 0,
|
| 518 |
+
"server_gen_throughput": 186.30161122127362,
|
| 519 |
+
"server_utilization": 0.012030798845043322,
|
| 520 |
+
"server_spec_accept_rate": 0.4225352112676056,
|
| 521 |
+
"server_spec_accept_length": 0.0,
|
| 522 |
+
"avg_running_reqs": 1,
|
| 523 |
+
"max_running_reqs": 1,
|
| 524 |
+
"effective_concurrency": 1,
|
| 525 |
+
"avg_queue_reqs": 0,
|
| 526 |
+
"max_queue_reqs": 0,
|
| 527 |
+
"queue_fraction": 0.0,
|
| 528 |
+
"underfilled": false,
|
| 529 |
+
"warmup_timed_out": false,
|
| 530 |
+
"warmup_duration": 9.11,
|
| 531 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 532 |
+
"timeout_reason": "",
|
| 533 |
+
"capacity_limited": false,
|
| 534 |
+
"hardware_summary": {
|
| 535 |
+
"samples": 8,
|
| 536 |
+
"duration_seconds": 16.884,
|
| 537 |
+
"gpu_count": 4,
|
| 538 |
+
"cpu_util_avg_pct": 10.89,
|
| 539 |
+
"cpu_temp_max_c": 76.12,
|
| 540 |
+
"gpu_util_avg_pct": 99.03,
|
| 541 |
+
"gpu_util_max_pct": 100.0,
|
| 542 |
+
"mem_util_avg_pct": 48.38,
|
| 543 |
+
"mem_util_max_pct": 54.0,
|
| 544 |
+
"temp_avg_c": 67.03,
|
| 545 |
+
"temp_max_c": 82.0,
|
| 546 |
+
"power_total_avg_w": 1154.29,
|
| 547 |
+
"power_total_max_w": 1155.42,
|
| 548 |
+
"power_limit_total_w": 1200.0,
|
| 549 |
+
"vram_used_avg_mb": 380174.0,
|
| 550 |
+
"vram_used_max_mb": 380174.0,
|
| 551 |
+
"vram_total_mb": 391548.0,
|
| 552 |
+
"vram_used_avg_pct": 97.1,
|
| 553 |
+
"vram_used_max_pct": 97.1,
|
| 554 |
+
"pcie_rx_avg_mb_s": 8367.62,
|
| 555 |
+
"pcie_rx_max_mb_s": 8719.0,
|
| 556 |
+
"pcie_tx_avg_mb_s": 8179.25,
|
| 557 |
+
"pcie_tx_max_mb_s": 8438.0
|
| 558 |
+
}
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"concurrency": 2,
|
| 562 |
+
"context_tokens": 0,
|
| 563 |
+
"benchmark_mode": "duration",
|
| 564 |
+
"request_count_target": 0,
|
| 565 |
+
"warmup_request_count": 0,
|
| 566 |
+
"measurement_seconds": 19.996612,
|
| 567 |
+
"measurement_wall_seconds": 20.000729,
|
| 568 |
+
"client_output_tokens": 4892,
|
| 569 |
+
"server_output_tokens": 4892,
|
| 570 |
+
"aggregate_source": "openai_continuous_usage",
|
| 571 |
+
"aggregate_tps": 244.64143701284553,
|
| 572 |
+
"per_request_avg_tps": 122.32071850642276,
|
| 573 |
+
"ttft_avg": 0.11364343203604221,
|
| 574 |
+
"ttft_p50": 0.11364343203604221,
|
| 575 |
+
"ttft_p90": 0.14825461693108083,
|
| 576 |
+
"ttft_p99": 0.1560421335324645,
|
| 577 |
+
"time_to_second_token_avg": 0.015344579587690532,
|
| 578 |
+
"time_to_second_token_p50": 0.015344579587690532,
|
| 579 |
+
"time_to_second_token_p90": 0.016974025568924845,
|
| 580 |
+
"time_to_second_token_p99": 0.017340650914702563,
|
| 581 |
+
"request_latency_avg": 0.0,
|
| 582 |
+
"request_latency_p50": 0.0,
|
| 583 |
+
"request_latency_p90": 0.0,
|
| 584 |
+
"request_latency_p99": 0.0,
|
| 585 |
+
"inter_token_latency_avg": 0.008021675154484334,
|
| 586 |
+
"inter_token_latency_p50": 0.008021675154484334,
|
| 587 |
+
"inter_token_latency_p90": 0.008142962393239039,
|
| 588 |
+
"inter_token_latency_p99": 0.008170252021958846,
|
| 589 |
+
"output_tps_per_user_avg": 124.70678698569867,
|
| 590 |
+
"output_tps_per_user_p50": 124.70678698569867,
|
| 591 |
+
"output_tps_per_user_p90": 126.59234599378327,
|
| 592 |
+
"output_tps_per_user_p99": 127.0165967706023,
|
| 593 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 594 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 595 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 596 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 597 |
+
"chunk_inter_token_latency_avg": 0.020790530252109002,
|
| 598 |
+
"chunk_inter_token_latency_p50": 0.020790530252109002,
|
| 599 |
+
"chunk_inter_token_latency_p90": 0.02081209152359912,
|
| 600 |
+
"chunk_inter_token_latency_p99": 0.020816942809684397,
|
| 601 |
+
"input_seq_len_avg": 78.0,
|
| 602 |
+
"output_seq_len_avg": 3170.5,
|
| 603 |
+
"output_seq_len_p50": 3170.5,
|
| 604 |
+
"output_seq_len_p90": 3214.1,
|
| 605 |
+
"output_seq_len_p99": 3223.91,
|
| 606 |
+
"request_count": 2,
|
| 607 |
+
"completed_request_count": 0,
|
| 608 |
+
"request_samples": [
|
| 609 |
+
{
|
| 610 |
+
"ttft": 0.07037945091724396,
|
| 611 |
+
"time_to_second_token": 0.013307772111147642,
|
| 612 |
+
"latency": 0.0,
|
| 613 |
+
"inter_token_latency_avg": 0.008173284202927714,
|
| 614 |
+
"chunk_inter_token_latency_avg": 0.02081748184147165,
|
| 615 |
+
"input_tokens": 78,
|
| 616 |
+
"output_tokens": 3116,
|
| 617 |
+
"output_tps_per_user": 122.34983822559292,
|
| 618 |
+
"e2e_output_tps_per_user": 0.0,
|
| 619 |
+
"completed": false
|
| 620 |
+
},
|
| 621 |
+
{
|
| 622 |
+
"ttft": 0.15690741315484047,
|
| 623 |
+
"time_to_second_token": 0.017381387064233422,
|
| 624 |
+
"latency": 0.0,
|
| 625 |
+
"inter_token_latency_avg": 0.007870066106040956,
|
| 626 |
+
"chunk_inter_token_latency_avg": 0.02076357866274635,
|
| 627 |
+
"input_tokens": 78,
|
| 628 |
+
"output_tokens": 3225,
|
| 629 |
+
"output_tps_per_user": 127.06373574580442,
|
| 630 |
+
"e2e_output_tps_per_user": 0.0,
|
| 631 |
+
"completed": false
|
| 632 |
+
}
|
| 633 |
+
],
|
| 634 |
+
"total_tokens": 4892,
|
| 635 |
+
"wall_time": 25.551810663193464,
|
| 636 |
+
"num_completed": 2,
|
| 637 |
+
"num_errors": 0,
|
| 638 |
+
"server_gen_throughput": 244.53218903662054,
|
| 639 |
+
"server_utilization": 0.0202117420596728,
|
| 640 |
+
"server_spec_accept_rate": 0.5034013605442177,
|
| 641 |
+
"server_spec_accept_length": 0.0,
|
| 642 |
+
"avg_running_reqs": 2,
|
| 643 |
+
"max_running_reqs": 2,
|
| 644 |
+
"effective_concurrency": 2,
|
| 645 |
+
"avg_queue_reqs": 0,
|
| 646 |
+
"max_queue_reqs": 0,
|
| 647 |
+
"queue_fraction": 0.0,
|
| 648 |
+
"underfilled": false,
|
| 649 |
+
"warmup_timed_out": false,
|
| 650 |
+
"warmup_duration": 5.534,
|
| 651 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 652 |
+
"timeout_reason": "",
|
| 653 |
+
"capacity_limited": false,
|
| 654 |
+
"hardware_summary": {
|
| 655 |
+
"samples": 9,
|
| 656 |
+
"duration_seconds": 19.285,
|
| 657 |
+
"gpu_count": 4,
|
| 658 |
+
"cpu_util_avg_pct": 10.84,
|
| 659 |
+
"cpu_temp_max_c": 76.12,
|
| 660 |
+
"gpu_util_avg_pct": 100.0,
|
| 661 |
+
"gpu_util_max_pct": 100.0,
|
| 662 |
+
"mem_util_avg_pct": 45.14,
|
| 663 |
+
"mem_util_max_pct": 49.0,
|
| 664 |
+
"temp_avg_c": 67.78,
|
| 665 |
+
"temp_max_c": 83.0,
|
| 666 |
+
"power_total_avg_w": 1176.88,
|
| 667 |
+
"power_total_max_w": 1178.9,
|
| 668 |
+
"power_limit_total_w": 1200.0,
|
| 669 |
+
"vram_used_avg_mb": 380174.0,
|
| 670 |
+
"vram_used_max_mb": 380174.0,
|
| 671 |
+
"vram_total_mb": 391548.0,
|
| 672 |
+
"vram_used_avg_pct": 97.1,
|
| 673 |
+
"vram_used_max_pct": 97.1,
|
| 674 |
+
"pcie_rx_avg_mb_s": 11281.22,
|
| 675 |
+
"pcie_rx_max_mb_s": 11385.0,
|
| 676 |
+
"pcie_tx_avg_mb_s": 11144.0,
|
| 677 |
+
"pcie_tx_max_mb_s": 11241.0
|
| 678 |
+
}
|
| 679 |
+
},
|
| 680 |
+
{
|
| 681 |
+
"concurrency": 4,
|
| 682 |
+
"context_tokens": 0,
|
| 683 |
+
"benchmark_mode": "duration",
|
| 684 |
+
"request_count_target": 0,
|
| 685 |
+
"warmup_request_count": 0,
|
| 686 |
+
"measurement_seconds": 19.968882,
|
| 687 |
+
"measurement_wall_seconds": 20.001101,
|
| 688 |
+
"client_output_tokens": 5848,
|
| 689 |
+
"server_output_tokens": 5848,
|
| 690 |
+
"aggregate_source": "openai_continuous_usage",
|
| 691 |
+
"aggregate_tps": 292.85565306969767,
|
| 692 |
+
"per_request_avg_tps": 73.21391326742442,
|
| 693 |
+
"ttft_avg": 0.1554050333215855,
|
| 694 |
+
"ttft_p50": 0.18278046406339854,
|
| 695 |
+
"ttft_p90": 0.18279754216782748,
|
| 696 |
+
"ttft_p99": 0.18280153354629874,
|
| 697 |
+
"time_to_second_token_avg": 0.026175830571446568,
|
| 698 |
+
"time_to_second_token_p50": 0.030380802578292787,
|
| 699 |
+
"time_to_second_token_p90": 0.030411765072494747,
|
| 700 |
+
"time_to_second_token_p99": 0.030420526592060924,
|
| 701 |
+
"request_latency_avg": 0.0,
|
| 702 |
+
"request_latency_p50": 0.0,
|
| 703 |
+
"request_latency_p90": 0.0,
|
| 704 |
+
"request_latency_p99": 0.0,
|
| 705 |
+
"inter_token_latency_avg": 0.013267129396052025,
|
| 706 |
+
"inter_token_latency_p50": 0.013429554256049785,
|
| 707 |
+
"inter_token_latency_p90": 0.013534146595532138,
|
| 708 |
+
"inter_token_latency_p99": 0.013559169974355398,
|
| 709 |
+
"output_tps_per_user_avg": 75.43238689615268,
|
| 710 |
+
"output_tps_per_user_p50": 74.46328652304663,
|
| 711 |
+
"output_tps_per_user_p90": 77.75213899442375,
|
| 712 |
+
"output_tps_per_user_p99": 78.93575450758087,
|
| 713 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 714 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 715 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 716 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 717 |
+
"chunk_inter_token_latency_avg": 0.034184321112551014,
|
| 718 |
+
"chunk_inter_token_latency_p50": 0.03419339492657416,
|
| 719 |
+
"chunk_inter_token_latency_p90": 0.034255501869857555,
|
| 720 |
+
"chunk_inter_token_latency_p99": 0.03427054421702008,
|
| 721 |
+
"input_seq_len_avg": 78.0,
|
| 722 |
+
"output_seq_len_avg": 1913.0,
|
| 723 |
+
"output_seq_len_p50": 1890.5,
|
| 724 |
+
"output_seq_len_p90": 1969.7,
|
| 725 |
+
"output_seq_len_p99": 1999.67,
|
| 726 |
+
"request_count": 4,
|
| 727 |
+
"completed_request_count": 0,
|
| 728 |
+
"request_samples": [
|
| 729 |
+
{
|
| 730 |
+
"ttft": 0.0732572281267494,
|
| 731 |
+
"time_to_second_token": 0.013520217034965754,
|
| 732 |
+
"latency": 0.0,
|
| 733 |
+
"inter_token_latency_avg": 0.013469271168953313,
|
| 734 |
+
"chunk_inter_token_latency_avg": 0.03427221558892703,
|
| 735 |
+
"input_tokens": 78,
|
| 736 |
+
"output_tokens": 1889,
|
| 737 |
+
"output_tps_per_user": 74.24306686355838,
|
| 738 |
+
"e2e_output_tps_per_user": 0.0,
|
| 739 |
+
"completed": false
|
| 740 |
+
},
|
| 741 |
+
{
|
| 742 |
+
"ttft": 0.18280197703279555,
|
| 743 |
+
"time_to_second_token": 0.030421500094234943,
|
| 744 |
+
"latency": 0.0,
|
| 745 |
+
"inter_token_latency_avg": 0.012647458722328322,
|
| 746 |
+
"chunk_inter_token_latency_avg": 0.034216503192028784,
|
| 747 |
+
"input_tokens": 78,
|
| 748 |
+
"output_tokens": 2003,
|
| 749 |
+
"output_tps_per_user": 79.06726734237611,
|
| 750 |
+
"e2e_output_tps_per_user": 0.0,
|
| 751 |
+
"completed": false
|
| 752 |
+
},
|
| 753 |
+
{
|
| 754 |
+
"ttft": 0.18278719414956868,
|
| 755 |
+
"time_to_second_token": 0.030389050021767616,
|
| 756 |
+
"latency": 0.0,
|
| 757 |
+
"inter_token_latency_avg": 0.01338983734314626,
|
| 758 |
+
"chunk_inter_token_latency_avg": 0.034170286661119535,
|
| 759 |
+
"input_tokens": 78,
|
| 760 |
+
"output_tokens": 1892,
|
| 761 |
+
"output_tps_per_user": 74.68350618253487,
|
| 762 |
+
"e2e_output_tps_per_user": 0.0,
|
| 763 |
+
"completed": false
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"ttft": 0.1827737339772284,
|
| 767 |
+
"time_to_second_token": 0.030372555134817958,
|
| 768 |
+
"latency": 0.0,
|
| 769 |
+
"inter_token_latency_avg": 0.013561950349780205,
|
| 770 |
+
"chunk_inter_token_latency_avg": 0.03407827900812872,
|
| 771 |
+
"input_tokens": 78,
|
| 772 |
+
"output_tokens": 1868,
|
| 773 |
+
"output_tps_per_user": 73.73570719614136,
|
| 774 |
+
"e2e_output_tps_per_user": 0.0,
|
| 775 |
+
"completed": false
|
| 776 |
+
}
|
| 777 |
+
],
|
| 778 |
+
"total_tokens": 5848,
|
| 779 |
+
"wall_time": 25.541023290948942,
|
| 780 |
+
"num_completed": 4,
|
| 781 |
+
"num_errors": 0,
|
| 782 |
+
"server_gen_throughput": 292.3212847040937,
|
| 783 |
+
"server_utilization": 0.04042348411934549,
|
| 784 |
+
"server_spec_accept_rate": 0.49712643678160917,
|
| 785 |
+
"server_spec_accept_length": 0.0,
|
| 786 |
+
"avg_running_reqs": 4,
|
| 787 |
+
"max_running_reqs": 4,
|
| 788 |
+
"effective_concurrency": 4,
|
| 789 |
+
"avg_queue_reqs": 0,
|
| 790 |
+
"max_queue_reqs": 0,
|
| 791 |
+
"queue_fraction": 0.0,
|
| 792 |
+
"underfilled": false,
|
| 793 |
+
"warmup_timed_out": false,
|
| 794 |
+
"warmup_duration": 5.535,
|
| 795 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 796 |
+
"timeout_reason": "",
|
| 797 |
+
"capacity_limited": false,
|
| 798 |
+
"hardware_summary": {
|
| 799 |
+
"samples": 8,
|
| 800 |
+
"duration_seconds": 16.833,
|
| 801 |
+
"gpu_count": 4,
|
| 802 |
+
"cpu_util_avg_pct": 10.81,
|
| 803 |
+
"cpu_temp_max_c": 77.0,
|
| 804 |
+
"gpu_util_avg_pct": 100.0,
|
| 805 |
+
"gpu_util_max_pct": 100.0,
|
| 806 |
+
"mem_util_avg_pct": 39.75,
|
| 807 |
+
"mem_util_max_pct": 43.0,
|
| 808 |
+
"temp_avg_c": 68.5,
|
| 809 |
+
"temp_max_c": 84.0,
|
| 810 |
+
"power_total_avg_w": 1172.95,
|
| 811 |
+
"power_total_max_w": 1173.93,
|
| 812 |
+
"power_limit_total_w": 1200.0,
|
| 813 |
+
"vram_used_avg_mb": 380174.0,
|
| 814 |
+
"vram_used_max_mb": 380174.0,
|
| 815 |
+
"vram_total_mb": 391548.0,
|
| 816 |
+
"vram_used_avg_pct": 97.1,
|
| 817 |
+
"vram_used_max_pct": 97.1,
|
| 818 |
+
"pcie_rx_avg_mb_s": 7881.25,
|
| 819 |
+
"pcie_rx_max_mb_s": 8056.0,
|
| 820 |
+
"pcie_tx_avg_mb_s": 7872.12,
|
| 821 |
+
"pcie_tx_max_mb_s": 8191.0
|
| 822 |
+
}
|
| 823 |
+
},
|
| 824 |
+
{
|
| 825 |
+
"concurrency": 2,
|
| 826 |
+
"context_tokens": 8192,
|
| 827 |
+
"benchmark_mode": "duration",
|
| 828 |
+
"request_count_target": 0,
|
| 829 |
+
"warmup_request_count": 0,
|
| 830 |
+
"measurement_seconds": 19.997371,
|
| 831 |
+
"measurement_wall_seconds": 20.000506,
|
| 832 |
+
"client_output_tokens": 4919,
|
| 833 |
+
"server_output_tokens": 4919,
|
| 834 |
+
"aggregate_source": "openai_continuous_usage",
|
| 835 |
+
"aggregate_tps": 245.98233342981337,
|
| 836 |
+
"per_request_avg_tps": 122.99116671490668,
|
| 837 |
+
"ttft_avg": 0.9427531745750457,
|
| 838 |
+
"ttft_p50": 0.9427531745750457,
|
| 839 |
+
"ttft_p90": 1.225604160549119,
|
| 840 |
+
"ttft_p99": 1.2892456323932855,
|
| 841 |
+
"time_to_second_token_avg": 0.022013463429175317,
|
| 842 |
+
"time_to_second_token_p50": 0.022013463429175317,
|
| 843 |
+
"time_to_second_token_p90": 0.025811466271989048,
|
| 844 |
+
"time_to_second_token_p99": 0.026666016911622136,
|
| 845 |
+
"request_latency_avg": 0.0,
|
| 846 |
+
"request_latency_p50": 0.0,
|
| 847 |
+
"request_latency_p90": 0.0,
|
| 848 |
+
"request_latency_p99": 0.0,
|
| 849 |
+
"inter_token_latency_avg": 0.008114433301139572,
|
| 850 |
+
"inter_token_latency_p50": 0.008114433301139572,
|
| 851 |
+
"inter_token_latency_p90": 0.00823075733708024,
|
| 852 |
+
"inter_token_latency_p99": 0.008256930245166891,
|
| 853 |
+
"output_tps_per_user_avg": 123.27677949684329,
|
| 854 |
+
"output_tps_per_user_p50": 123.27677949684329,
|
| 855 |
+
"output_tps_per_user_p90": 125.04400734833438,
|
| 856 |
+
"output_tps_per_user_p99": 125.44163361491988,
|
| 857 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 858 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 859 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 860 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 861 |
+
"chunk_inter_token_latency_avg": 0.021151300976615273,
|
| 862 |
+
"chunk_inter_token_latency_p50": 0.021151300976615273,
|
| 863 |
+
"chunk_inter_token_latency_p90": 0.021420214613202804,
|
| 864 |
+
"chunk_inter_token_latency_p99": 0.021480720181434997,
|
| 865 |
+
"input_seq_len_avg": 8192.0,
|
| 866 |
+
"output_seq_len_avg": 2907.5,
|
| 867 |
+
"output_seq_len_p50": 2907.5,
|
| 868 |
+
"output_seq_len_p90": 2914.3,
|
| 869 |
+
"output_seq_len_p99": 2915.83,
|
| 870 |
+
"request_count": 2,
|
| 871 |
+
"completed_request_count": 0,
|
| 872 |
+
"request_samples": [
|
| 873 |
+
{
|
| 874 |
+
"ttft": 0.5891894421074539,
|
| 875 |
+
"time_to_second_token": 0.017265959875658154,
|
| 876 |
+
"latency": 0.0,
|
| 877 |
+
"inter_token_latency_avg": 0.008259838346065407,
|
| 878 |
+
"chunk_inter_token_latency_avg": 0.021487443022349687,
|
| 879 |
+
"input_tokens": 8192,
|
| 880 |
+
"output_tokens": 2899,
|
| 881 |
+
"output_tps_per_user": 121.06774468247944,
|
| 882 |
+
"e2e_output_tps_per_user": 0.0,
|
| 883 |
+
"completed": false
|
| 884 |
+
},
|
| 885 |
+
{
|
| 886 |
+
"ttft": 1.2963169070426375,
|
| 887 |
+
"time_to_second_token": 0.02676096698269248,
|
| 888 |
+
"latency": 0.0,
|
| 889 |
+
"inter_token_latency_avg": 0.007969028256213737,
|
| 890 |
+
"chunk_inter_token_latency_avg": 0.020815158930880862,
|
| 891 |
+
"input_tokens": 8192,
|
| 892 |
+
"output_tokens": 2916,
|
| 893 |
+
"output_tps_per_user": 125.48581431120715,
|
| 894 |
+
"e2e_output_tps_per_user": 0.0,
|
| 895 |
+
"completed": false
|
| 896 |
+
}
|
| 897 |
+
],
|
| 898 |
+
"total_tokens": 4919,
|
| 899 |
+
"wall_time": 25.566069599008188,
|
| 900 |
+
"num_completed": 2,
|
| 901 |
+
"num_errors": 0,
|
| 902 |
+
"server_gen_throughput": 245.8762565579269,
|
| 903 |
+
"server_utilization": 0.02117420596727626,
|
| 904 |
+
"server_spec_accept_rate": 0.5510204081632653,
|
| 905 |
+
"server_spec_accept_length": 0.0,
|
| 906 |
+
"avg_running_reqs": 2,
|
| 907 |
+
"max_running_reqs": 2,
|
| 908 |
+
"effective_concurrency": 2,
|
| 909 |
+
"avg_queue_reqs": 0,
|
| 910 |
+
"max_queue_reqs": 0,
|
| 911 |
+
"queue_fraction": 0.0,
|
| 912 |
+
"underfilled": false,
|
| 913 |
+
"warmup_timed_out": false,
|
| 914 |
+
"warmup_duration": 5.547,
|
| 915 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 916 |
+
"timeout_reason": "",
|
| 917 |
+
"capacity_limited": false,
|
| 918 |
+
"hardware_summary": {
|
| 919 |
+
"samples": 8,
|
| 920 |
+
"duration_seconds": 16.882,
|
| 921 |
+
"gpu_count": 4,
|
| 922 |
+
"cpu_util_avg_pct": 10.89,
|
| 923 |
+
"cpu_temp_max_c": 75.62,
|
| 924 |
+
"gpu_util_avg_pct": 100.0,
|
| 925 |
+
"gpu_util_max_pct": 100.0,
|
| 926 |
+
"mem_util_avg_pct": 45.12,
|
| 927 |
+
"mem_util_max_pct": 49.0,
|
| 928 |
+
"temp_avg_c": 68.28,
|
| 929 |
+
"temp_max_c": 84.0,
|
| 930 |
+
"power_total_avg_w": 1178.0,
|
| 931 |
+
"power_total_max_w": 1178.89,
|
| 932 |
+
"power_limit_total_w": 1200.0,
|
| 933 |
+
"vram_used_avg_mb": 380174.0,
|
| 934 |
+
"vram_used_max_mb": 380174.0,
|
| 935 |
+
"vram_total_mb": 391548.0,
|
| 936 |
+
"vram_used_avg_pct": 97.1,
|
| 937 |
+
"vram_used_max_pct": 97.1,
|
| 938 |
+
"pcie_rx_avg_mb_s": 11233.75,
|
| 939 |
+
"pcie_rx_max_mb_s": 11359.0,
|
| 940 |
+
"pcie_tx_avg_mb_s": 11151.88,
|
| 941 |
+
"pcie_tx_max_mb_s": 11299.0
|
| 942 |
+
}
|
| 943 |
+
},
|
| 944 |
+
{
|
| 945 |
+
"concurrency": 4,
|
| 946 |
+
"context_tokens": 8192,
|
| 947 |
+
"benchmark_mode": "duration",
|
| 948 |
+
"request_count_target": 0,
|
| 949 |
+
"warmup_request_count": 0,
|
| 950 |
+
"measurement_seconds": 19.996605,
|
| 951 |
+
"measurement_wall_seconds": 20.000706,
|
| 952 |
+
"client_output_tokens": 5967,
|
| 953 |
+
"server_output_tokens": 5967,
|
| 954 |
+
"aggregate_source": "openai_continuous_usage",
|
| 955 |
+
"aggregate_tps": 298.40065698894307,
|
| 956 |
+
"per_request_avg_tps": 74.60016424723577,
|
| 957 |
+
"ttft_avg": 2.7296009025303647,
|
| 958 |
+
"ttft_p50": 3.442606173455715,
|
| 959 |
+
"ttft_p90": 3.4429665624164043,
|
| 960 |
+
"ttft_p99": 3.443046331773512,
|
| 961 |
+
"time_to_second_token_avg": 0.04401593271177262,
|
| 962 |
+
"time_to_second_token_p50": 0.05272976343985647,
|
| 963 |
+
"time_to_second_token_p90": 0.05277097581420094,
|
| 964 |
+
"time_to_second_token_p99": 0.052783893730957064,
|
| 965 |
+
"request_latency_avg": 0.0,
|
| 966 |
+
"request_latency_p50": 0.0,
|
| 967 |
+
"request_latency_p90": 0.0,
|
| 968 |
+
"request_latency_p99": 0.0,
|
| 969 |
+
"inter_token_latency_avg": 0.013244349253168272,
|
| 970 |
+
"inter_token_latency_p50": 0.01323177277428084,
|
| 971 |
+
"inter_token_latency_p90": 0.013676198067583552,
|
| 972 |
+
"inter_token_latency_p99": 0.013812555348605936,
|
| 973 |
+
"output_tps_per_user_avg": 75.57579651538197,
|
| 974 |
+
"output_tps_per_user_p50": 75.57923042394843,
|
| 975 |
+
"output_tps_per_user_p90": 78.00785040565367,
|
| 976 |
+
"output_tps_per_user_p99": 78.74432054765556,
|
| 977 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 978 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 979 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 980 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 981 |
+
"chunk_inter_token_latency_avg": 0.03463163790764121,
|
| 982 |
+
"chunk_inter_token_latency_p50": 0.03453151627338626,
|
| 983 |
+
"chunk_inter_token_latency_p90": 0.03481200221493731,
|
| 984 |
+
"chunk_inter_token_latency_p99": 0.03492015582317328,
|
| 985 |
+
"input_seq_len_avg": 8192.0,
|
| 986 |
+
"output_seq_len_avg": 1798.5,
|
| 987 |
+
"output_seq_len_p50": 1790.5,
|
| 988 |
+
"output_seq_len_p90": 1861.2,
|
| 989 |
+
"output_seq_len_p99": 1876.32,
|
| 990 |
+
"request_count": 4,
|
| 991 |
+
"completed_request_count": 0,
|
| 992 |
+
"request_samples": [
|
| 993 |
+
{
|
| 994 |
+
"ttft": 0.5901360681746155,
|
| 995 |
+
"time_to_second_token": 0.01781887491233647,
|
| 996 |
+
"latency": 0.0,
|
| 997 |
+
"inter_token_latency_avg": 0.013827706157608423,
|
| 998 |
+
"chunk_inter_token_latency_avg": 0.03493217289075506,
|
| 999 |
+
"input_tokens": 8192,
|
| 1000 |
+
"output_tokens": 1878,
|
| 1001 |
+
"output_tps_per_user": 72.31857465019748,
|
| 1002 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1003 |
+
"completed": false
|
| 1004 |
+
},
|
| 1005 |
+
{
|
| 1006 |
+
"ttft": 3.443055195035413,
|
| 1007 |
+
"time_to_second_token": 0.052722041960805655,
|
| 1008 |
+
"latency": 0.0,
|
| 1009 |
+
"inter_token_latency_avg": 0.012686145306502984,
|
| 1010 |
+
"chunk_inter_token_latency_avg": 0.03453134619303727,
|
| 1011 |
+
"input_tokens": 8192,
|
| 1012 |
+
"output_tokens": 1822,
|
| 1013 |
+
"output_tps_per_user": 78.82615056343354,
|
| 1014 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1015 |
+
"completed": false
|
| 1016 |
+
},
|
| 1017 |
+
{
|
| 1018 |
+
"ttft": 3.4427597529720515,
|
| 1019 |
+
"time_to_second_token": 0.052737484918907285,
|
| 1020 |
+
"latency": 0.0,
|
| 1021 |
+
"inter_token_latency_avg": 0.013322679190858855,
|
| 1022 |
+
"chunk_inter_token_latency_avg": 0.034531428575409945,
|
| 1023 |
+
"input_tokens": 8192,
|
| 1024 |
+
"output_tokens": 1735,
|
| 1025 |
+
"output_tps_per_user": 75.05997747706289,
|
| 1026 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1027 |
+
"completed": false
|
| 1028 |
+
},
|
| 1029 |
+
{
|
| 1030 |
+
"ttft": 3.442452593939379,
|
| 1031 |
+
"time_to_second_token": 0.052785329055041075,
|
| 1032 |
+
"latency": 0.0,
|
| 1033 |
+
"inter_token_latency_avg": 0.013140866357702825,
|
| 1034 |
+
"chunk_inter_token_latency_avg": 0.03453160397136258,
|
| 1035 |
+
"input_tokens": 8192,
|
| 1036 |
+
"output_tokens": 1759,
|
| 1037 |
+
"output_tps_per_user": 76.09848337083397,
|
| 1038 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1039 |
+
"completed": false
|
| 1040 |
+
}
|
| 1041 |
+
],
|
| 1042 |
+
"total_tokens": 5967,
|
| 1043 |
+
"wall_time": 27.59907579794526,
|
| 1044 |
+
"num_completed": 4,
|
| 1045 |
+
"num_errors": 0,
|
| 1046 |
+
"server_gen_throughput": 298.2659723183749,
|
| 1047 |
+
"server_utilization": 0.04234841193455241,
|
| 1048 |
+
"server_spec_accept_rate": 0.4511494252873563,
|
| 1049 |
+
"server_spec_accept_length": 0.0,
|
| 1050 |
+
"avg_running_reqs": 4,
|
| 1051 |
+
"max_running_reqs": 4,
|
| 1052 |
+
"effective_concurrency": 4,
|
| 1053 |
+
"avg_queue_reqs": 0,
|
| 1054 |
+
"max_queue_reqs": 0,
|
| 1055 |
+
"queue_fraction": 0.0,
|
| 1056 |
+
"underfilled": false,
|
| 1057 |
+
"warmup_timed_out": false,
|
| 1058 |
+
"warmup_duration": 7.567,
|
| 1059 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1060 |
+
"timeout_reason": "",
|
| 1061 |
+
"capacity_limited": false,
|
| 1062 |
+
"hardware_summary": {
|
| 1063 |
+
"samples": 9,
|
| 1064 |
+
"duration_seconds": 19.211,
|
| 1065 |
+
"gpu_count": 4,
|
| 1066 |
+
"cpu_util_avg_pct": 10.83,
|
| 1067 |
+
"cpu_temp_max_c": 77.0,
|
| 1068 |
+
"gpu_util_avg_pct": 100.0,
|
| 1069 |
+
"gpu_util_max_pct": 100.0,
|
| 1070 |
+
"mem_util_avg_pct": 39.47,
|
| 1071 |
+
"mem_util_max_pct": 43.0,
|
| 1072 |
+
"temp_avg_c": 68.94,
|
| 1073 |
+
"temp_max_c": 84.0,
|
| 1074 |
+
"power_total_avg_w": 1173.61,
|
| 1075 |
+
"power_total_max_w": 1174.17,
|
| 1076 |
+
"power_limit_total_w": 1200.0,
|
| 1077 |
+
"vram_used_avg_mb": 380174.0,
|
| 1078 |
+
"vram_used_max_mb": 380174.0,
|
| 1079 |
+
"vram_total_mb": 391548.0,
|
| 1080 |
+
"vram_used_avg_pct": 97.1,
|
| 1081 |
+
"vram_used_max_pct": 97.1,
|
| 1082 |
+
"pcie_rx_avg_mb_s": 7879.33,
|
| 1083 |
+
"pcie_rx_max_mb_s": 8354.0,
|
| 1084 |
+
"pcie_tx_avg_mb_s": 7779.89,
|
| 1085 |
+
"pcie_tx_max_mb_s": 8190.0
|
| 1086 |
+
}
|
| 1087 |
+
},
|
| 1088 |
+
{
|
| 1089 |
+
"concurrency": 2,
|
| 1090 |
+
"context_tokens": 32768,
|
| 1091 |
+
"benchmark_mode": "duration",
|
| 1092 |
+
"request_count_target": 0,
|
| 1093 |
+
"warmup_request_count": 0,
|
| 1094 |
+
"measurement_seconds": 19.988436,
|
| 1095 |
+
"measurement_wall_seconds": 20.000537,
|
| 1096 |
+
"client_output_tokens": 5050,
|
| 1097 |
+
"server_output_tokens": 5050,
|
| 1098 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1099 |
+
"aggregate_tps": 252.6460844518971,
|
| 1100 |
+
"per_request_avg_tps": 126.32304222594856,
|
| 1101 |
+
"ttft_avg": 0.951587092014961,
|
| 1102 |
+
"ttft_p50": 0.951587092014961,
|
| 1103 |
+
"ttft_p90": 1.232194907194935,
|
| 1104 |
+
"ttft_p99": 1.295331665610429,
|
| 1105 |
+
"time_to_second_token_avg": 0.0161319924518466,
|
| 1106 |
+
"time_to_second_token_p50": 0.0161319924518466,
|
| 1107 |
+
"time_to_second_token_p90": 0.019481185637414456,
|
| 1108 |
+
"time_to_second_token_p99": 0.020234754104167224,
|
| 1109 |
+
"request_latency_avg": 0.0,
|
| 1110 |
+
"request_latency_p50": 0.0,
|
| 1111 |
+
"request_latency_p90": 0.0,
|
| 1112 |
+
"request_latency_p99": 0.0,
|
| 1113 |
+
"inter_token_latency_avg": 0.007939165504351434,
|
| 1114 |
+
"inter_token_latency_p50": 0.007939165504351434,
|
| 1115 |
+
"inter_token_latency_p90": 0.008045622734254517,
|
| 1116 |
+
"inter_token_latency_p99": 0.008069575610982711,
|
| 1117 |
+
"output_tps_per_user_avg": 125.99321968668501,
|
| 1118 |
+
"output_tps_per_user_p50": 125.99321968668501,
|
| 1119 |
+
"output_tps_per_user_p90": 127.68267799903076,
|
| 1120 |
+
"output_tps_per_user_p99": 128.06280611930853,
|
| 1121 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1122 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1123 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1124 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1125 |
+
"chunk_inter_token_latency_avg": 0.02123036904289779,
|
| 1126 |
+
"chunk_inter_token_latency_p50": 0.02123036904289779,
|
| 1127 |
+
"chunk_inter_token_latency_p90": 0.021406878769873895,
|
| 1128 |
+
"chunk_inter_token_latency_p99": 0.02144659345844352,
|
| 1129 |
+
"input_seq_len_avg": 32768.0,
|
| 1130 |
+
"output_seq_len_avg": 2971.0,
|
| 1131 |
+
"output_seq_len_p50": 2971.0,
|
| 1132 |
+
"output_seq_len_p90": 3046.2,
|
| 1133 |
+
"output_seq_len_p99": 3063.12,
|
| 1134 |
+
"request_count": 2,
|
| 1135 |
+
"completed_request_count": 0,
|
| 1136 |
+
"request_samples": [
|
| 1137 |
+
{
|
| 1138 |
+
"ttft": 0.6008273230399936,
|
| 1139 |
+
"time_to_second_token": 0.01194550096988678,
|
| 1140 |
+
"latency": 0.0,
|
| 1141 |
+
"inter_token_latency_avg": 0.007806093966972579,
|
| 1142 |
+
"chunk_inter_token_latency_avg": 0.021451006201617922,
|
| 1143 |
+
"input_tokens": 32768,
|
| 1144 |
+
"output_tokens": 3065,
|
| 1145 |
+
"output_tps_per_user": 128.1050425771172,
|
| 1146 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1147 |
+
"completed": false
|
| 1148 |
+
},
|
| 1149 |
+
{
|
| 1150 |
+
"ttft": 1.3023468609899282,
|
| 1151 |
+
"time_to_second_token": 0.02031848393380642,
|
| 1152 |
+
"latency": 0.0,
|
| 1153 |
+
"inter_token_latency_avg": 0.008072237041730289,
|
| 1154 |
+
"chunk_inter_token_latency_avg": 0.021009731884177655,
|
| 1155 |
+
"input_tokens": 32768,
|
| 1156 |
+
"output_tokens": 2877,
|
| 1157 |
+
"output_tps_per_user": 123.88139679625283,
|
| 1158 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1159 |
+
"completed": false
|
| 1160 |
+
}
|
| 1161 |
+
],
|
| 1162 |
+
"total_tokens": 5050,
|
| 1163 |
+
"wall_time": 25.579677499830723,
|
| 1164 |
+
"num_completed": 2,
|
| 1165 |
+
"num_errors": 0,
|
| 1166 |
+
"server_gen_throughput": 252.43712151026958,
|
| 1167 |
+
"server_utilization": 0.012030798845043322,
|
| 1168 |
+
"server_spec_accept_rate": 0.5173611111111112,
|
| 1169 |
+
"server_spec_accept_length": 0.0,
|
| 1170 |
+
"avg_running_reqs": 2,
|
| 1171 |
+
"max_running_reqs": 2,
|
| 1172 |
+
"effective_concurrency": 2,
|
| 1173 |
+
"avg_queue_reqs": 0,
|
| 1174 |
+
"max_queue_reqs": 0,
|
| 1175 |
+
"queue_fraction": 0.0,
|
| 1176 |
+
"underfilled": false,
|
| 1177 |
+
"warmup_timed_out": false,
|
| 1178 |
+
"warmup_duration": 5.549,
|
| 1179 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1180 |
+
"timeout_reason": "",
|
| 1181 |
+
"capacity_limited": false,
|
| 1182 |
+
"hardware_summary": {
|
| 1183 |
+
"samples": 8,
|
| 1184 |
+
"duration_seconds": 16.873,
|
| 1185 |
+
"gpu_count": 4,
|
| 1186 |
+
"cpu_util_avg_pct": 10.84,
|
| 1187 |
+
"cpu_temp_max_c": 77.25,
|
| 1188 |
+
"gpu_util_avg_pct": 100.0,
|
| 1189 |
+
"gpu_util_max_pct": 100.0,
|
| 1190 |
+
"mem_util_avg_pct": 45.06,
|
| 1191 |
+
"mem_util_max_pct": 49.0,
|
| 1192 |
+
"temp_avg_c": 68.59,
|
| 1193 |
+
"temp_max_c": 84.0,
|
| 1194 |
+
"power_total_avg_w": 1178.44,
|
| 1195 |
+
"power_total_max_w": 1178.81,
|
| 1196 |
+
"power_limit_total_w": 1200.0,
|
| 1197 |
+
"vram_used_avg_mb": 380174.0,
|
| 1198 |
+
"vram_used_max_mb": 380174.0,
|
| 1199 |
+
"vram_total_mb": 391548.0,
|
| 1200 |
+
"vram_used_avg_pct": 97.1,
|
| 1201 |
+
"vram_used_max_pct": 97.1,
|
| 1202 |
+
"pcie_rx_avg_mb_s": 11056.12,
|
| 1203 |
+
"pcie_rx_max_mb_s": 11191.0,
|
| 1204 |
+
"pcie_tx_avg_mb_s": 11029.38,
|
| 1205 |
+
"pcie_tx_max_mb_s": 11153.0
|
| 1206 |
+
}
|
| 1207 |
+
},
|
| 1208 |
+
{
|
| 1209 |
+
"concurrency": 4,
|
| 1210 |
+
"context_tokens": 32768,
|
| 1211 |
+
"benchmark_mode": "duration",
|
| 1212 |
+
"request_count_target": 0,
|
| 1213 |
+
"warmup_request_count": 0,
|
| 1214 |
+
"measurement_seconds": 19.966272,
|
| 1215 |
+
"measurement_wall_seconds": 20.000472,
|
| 1216 |
+
"client_output_tokens": 5944,
|
| 1217 |
+
"server_output_tokens": 5956,
|
| 1218 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1219 |
+
"aggregate_tps": 297.7020479777527,
|
| 1220 |
+
"per_request_avg_tps": 74.42551199443818,
|
| 1221 |
+
"ttft_avg": 2.7296407557441853,
|
| 1222 |
+
"ttft_p50": 3.4367186579620466,
|
| 1223 |
+
"ttft_p90": 3.437307026120834,
|
| 1224 |
+
"ttft_p99": 3.4374052726081574,
|
| 1225 |
+
"time_to_second_token_avg": 0.02786868193652481,
|
| 1226 |
+
"time_to_second_token_p50": 0.03361363697331399,
|
| 1227 |
+
"time_to_second_token_p90": 0.033683925354853275,
|
| 1228 |
+
"time_to_second_token_p99": 0.03370359269436449,
|
| 1229 |
+
"request_latency_avg": 0.0,
|
| 1230 |
+
"request_latency_p50": 0.0,
|
| 1231 |
+
"request_latency_p90": 0.0,
|
| 1232 |
+
"request_latency_p99": 0.0,
|
| 1233 |
+
"inter_token_latency_avg": 0.013227158501728438,
|
| 1234 |
+
"inter_token_latency_p50": 0.013280406418363698,
|
| 1235 |
+
"inter_token_latency_p90": 0.01328424798952288,
|
| 1236 |
+
"inter_token_latency_p99": 0.013284256693709504,
|
| 1237 |
+
"output_tps_per_user_avg": 75.60591881108353,
|
| 1238 |
+
"output_tps_per_user_p50": 75.29890661417849,
|
| 1239 |
+
"output_tps_per_user_p90": 76.18032211148439,
|
| 1240 |
+
"output_tps_per_user_p99": 76.51194460773044,
|
| 1241 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1242 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1243 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1244 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1245 |
+
"chunk_inter_token_latency_avg": 0.03458492849846504,
|
| 1246 |
+
"chunk_inter_token_latency_p50": 0.0345171884634303,
|
| 1247 |
+
"chunk_inter_token_latency_p90": 0.03476874527228148,
|
| 1248 |
+
"chunk_inter_token_latency_p99": 0.034855800227265865,
|
| 1249 |
+
"input_seq_len_avg": 32768.0,
|
| 1250 |
+
"output_seq_len_avg": 1799.75,
|
| 1251 |
+
"output_seq_len_p50": 1738.5,
|
| 1252 |
+
"output_seq_len_p90": 1910.5,
|
| 1253 |
+
"output_seq_len_p99": 1976.6499999999999,
|
| 1254 |
+
"request_count": 4,
|
| 1255 |
+
"completed_request_count": 0,
|
| 1256 |
+
"request_samples": [
|
| 1257 |
+
{
|
| 1258 |
+
"ttft": 0.6077095181681216,
|
| 1259 |
+
"time_to_second_token": 0.01054167584516108,
|
| 1260 |
+
"latency": 0.0,
|
| 1261 |
+
"inter_token_latency_avg": 0.013063563509345002,
|
| 1262 |
+
"chunk_inter_token_latency_avg": 0.03486547300004191,
|
| 1263 |
+
"input_tokens": 32768,
|
| 1264 |
+
"output_tokens": 1984,
|
| 1265 |
+
"output_tps_per_user": 76.54879155175779,
|
| 1266 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1267 |
+
"completed": false
|
| 1268 |
+
},
|
| 1269 |
+
{
|
| 1270 |
+
"ttft": 3.4374161888845265,
|
| 1271 |
+
"time_to_second_token": 0.03363293595612049,
|
| 1272 |
+
"latency": 0.0,
|
| 1273 |
+
"inter_token_latency_avg": 0.013284225423113112,
|
| 1274 |
+
"chunk_inter_token_latency_avg": 0.034491329686020145,
|
| 1275 |
+
"input_tokens": 32768,
|
| 1276 |
+
"output_tokens": 1738,
|
| 1277 |
+
"output_tps_per_user": 75.27725314417718,
|
| 1278 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1279 |
+
"completed": false
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"ttft": 3.4370523130055517,
|
| 1283 |
+
"time_to_second_token": 0.033594337990507483,
|
| 1284 |
+
"latency": 0.0,
|
| 1285 |
+
"inter_token_latency_avg": 0.013276587413614283,
|
| 1286 |
+
"chunk_inter_token_latency_avg": 0.03443986406695765,
|
| 1287 |
+
"input_tokens": 32768,
|
| 1288 |
+
"output_tokens": 1739,
|
| 1289 |
+
"output_tps_per_user": 75.3205600841798,
|
| 1290 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1291 |
+
"completed": false
|
| 1292 |
+
},
|
| 1293 |
+
{
|
| 1294 |
+
"ttft": 3.4363850029185414,
|
| 1295 |
+
"time_to_second_token": 0.03370577795431018,
|
| 1296 |
+
"latency": 0.0,
|
| 1297 |
+
"inter_token_latency_avg": 0.013284257660841351,
|
| 1298 |
+
"chunk_inter_token_latency_avg": 0.03454304724084046,
|
| 1299 |
+
"input_tokens": 32768,
|
| 1300 |
+
"output_tokens": 1738,
|
| 1301 |
+
"output_tps_per_user": 75.27707046421935,
|
| 1302 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1303 |
+
"completed": false
|
| 1304 |
+
}
|
| 1305 |
+
],
|
| 1306 |
+
"total_tokens": 5944,
|
| 1307 |
+
"wall_time": 27.572524111019447,
|
| 1308 |
+
"num_completed": 4,
|
| 1309 |
+
"num_errors": 0,
|
| 1310 |
+
"server_gen_throughput": 297.6989316734227,
|
| 1311 |
+
"server_utilization": 0.04379210779595766,
|
| 1312 |
+
"server_spec_accept_rate": 0.5028735632183908,
|
| 1313 |
+
"server_spec_accept_length": 0.0,
|
| 1314 |
+
"avg_running_reqs": 4,
|
| 1315 |
+
"max_running_reqs": 4,
|
| 1316 |
+
"effective_concurrency": 4,
|
| 1317 |
+
"avg_queue_reqs": 0,
|
| 1318 |
+
"max_queue_reqs": 0,
|
| 1319 |
+
"queue_fraction": 0.0,
|
| 1320 |
+
"underfilled": false,
|
| 1321 |
+
"warmup_timed_out": false,
|
| 1322 |
+
"warmup_duration": 7.566,
|
| 1323 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1324 |
+
"timeout_reason": "",
|
| 1325 |
+
"capacity_limited": false,
|
| 1326 |
+
"hardware_summary": {
|
| 1327 |
+
"samples": 9,
|
| 1328 |
+
"duration_seconds": 19.263,
|
| 1329 |
+
"gpu_count": 4,
|
| 1330 |
+
"cpu_util_avg_pct": 11.06,
|
| 1331 |
+
"cpu_temp_max_c": 76.75,
|
| 1332 |
+
"gpu_util_avg_pct": 100.0,
|
| 1333 |
+
"gpu_util_max_pct": 100.0,
|
| 1334 |
+
"mem_util_avg_pct": 38.81,
|
| 1335 |
+
"mem_util_max_pct": 43.0,
|
| 1336 |
+
"temp_avg_c": 69.0,
|
| 1337 |
+
"temp_max_c": 84.0,
|
| 1338 |
+
"power_total_avg_w": 1173.28,
|
| 1339 |
+
"power_total_max_w": 1174.27,
|
| 1340 |
+
"power_limit_total_w": 1200.0,
|
| 1341 |
+
"vram_used_avg_mb": 380174.0,
|
| 1342 |
+
"vram_used_max_mb": 380174.0,
|
| 1343 |
+
"vram_total_mb": 391548.0,
|
| 1344 |
+
"vram_used_avg_pct": 97.1,
|
| 1345 |
+
"vram_used_max_pct": 97.1,
|
| 1346 |
+
"pcie_rx_avg_mb_s": 7830.56,
|
| 1347 |
+
"pcie_rx_max_mb_s": 8017.0,
|
| 1348 |
+
"pcie_tx_avg_mb_s": 7803.78,
|
| 1349 |
+
"pcie_tx_max_mb_s": 8098.0
|
| 1350 |
+
}
|
| 1351 |
+
}
|
| 1352 |
+
],
|
| 1353 |
+
"summary_table": {
|
| 1354 |
+
"0": {
|
| 1355 |
+
"1": 167.19542416324822,
|
| 1356 |
+
"2": 244.64143701284553,
|
| 1357 |
+
"4": 292.85565306969767
|
| 1358 |
+
},
|
| 1359 |
+
"8192": {
|
| 1360 |
+
"1": 172.95726809626717,
|
| 1361 |
+
"2": 245.98233342981337,
|
| 1362 |
+
"4": 298.40065698894307
|
| 1363 |
+
},
|
| 1364 |
+
"32768": {
|
| 1365 |
+
"1": 186.35728993398365,
|
| 1366 |
+
"2": 252.6460844518971,
|
| 1367 |
+
"4": 297.7020479777527
|
| 1368 |
+
}
|
| 1369 |
+
},
|
| 1370 |
+
"burst_results": [],
|
| 1371 |
+
"burst_summary_table": {},
|
| 1372 |
+
"methodology": {
|
| 1373 |
+
"prefill": {
|
| 1374 |
+
"name": "Prefill",
|
| 1375 |
+
"present": false,
|
| 1376 |
+
"mode": "skipped",
|
| 1377 |
+
"formula": "prompt_tokens / TTFT",
|
| 1378 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 1379 |
+
},
|
| 1380 |
+
"sustained_decode": {
|
| 1381 |
+
"name": "Sustained Decode",
|
| 1382 |
+
"present": true,
|
| 1383 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 1384 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 1385 |
+
},
|
| 1386 |
+
"burst_e2e_decode": {
|
| 1387 |
+
"name": "Burst / E2E Decode",
|
| 1388 |
+
"present": false,
|
| 1389 |
+
"status": "not run; use --run-burst",
|
| 1390 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 1391 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 1392 |
+
}
|
| 1393 |
+
}
|
| 1394 |
+
}
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/decode-cap8192.log
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0', '8k', '32k'] │
|
| 16 |
+
│ Duration: 20.0s per decode test | Max tokens: 8192 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill: skipped | Sustained decode: 9 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 17,031,168 tokens (2079 blocks × 2048; local
|
| 23 |
+
4,257,792 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: skipped
|
| 26 |
+
Calibrating padding text (run=wwbkdgkvesfu, up to 32k)...
|
| 27 |
+
8k: 50,544 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,140 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
Token targeting: /tokenize exact
|
| 30 |
+
Done.
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
llm-decode-bench v0.4.29
|
| 35 |
+
╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
|
| 36 |
+
│ Sustained Decode │
|
| 37 |
+
│ Steady-state decode throughput after the engine has admitted the requested │
|
| 38 |
+
│ concurrency and passed warmup. Use this as the main tuning/regression signal │
|
| 39 |
+
│ for kernels, NCCL, DCP, MTP, and scheduler changes. │
|
| 40 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 41 |
+
Aggregate tok/s + TTFT/ITL
|
| 42 |
+
╭────────────┬─────────────┬─────────────┬──────────────╮
|
| 43 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 44 |
+
├────────────┼─────────────┼─────────────┼──────────────┤
|
| 45 |
+
│ 0 │ 167.2 71/6 │ 244.6 114/8 │ 292.9 183/13 │
|
| 46 |
+
│ 8k │ 173.0 579/6 │ 246.0 943/8 │ 298.4 3k/13 │
|
| 47 |
+
│ 32k │ 186.4 598/5 │ 252.6 952/8 │ 297.7 3k/13 │
|
| 48 |
+
╰────────────┴─────────────┴─────────────┴──────────────╯
|
| 49 |
+
Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
|
| 50 |
+
(continuous completion_tokens when the server supports it). Prometheus is kept
|
| 51 |
+
as validation/scheduler data.
|
| 52 |
+
Aggregate source(s): openai_continuous_usage
|
| 53 |
+
Per-Request tok/s
|
| 54 |
+
╭────────────┬───────┬───────┬──────╮
|
| 55 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 56 |
+
├────────────┼───────┼───────┼──────┤
|
| 57 |
+
│ 0 │ 167.2 │ 122.3 │ 73.2 │
|
| 58 |
+
│ 8k │ 173.0 │ 123.0 │ 74.6 │
|
| 59 |
+
│ 32k │ 186.4 │ 126.3 │ 74.4 │
|
| 60 |
+
╰────────────┴───────┴───────┴──────╯
|
| 61 |
+
Client request latency: p50 /
|
| 62 |
+
p90 ms
|
| 63 |
+
╭────────────┬─────┬─────┬─────╮
|
| 64 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 65 |
+
├────────────┼─────┼─────┼─────┤
|
| 66 |
+
│ 0 │ —/— │ —/— │ —/— │
|
| 67 |
+
│ 8k │ —/— │ —/— │ —/— │
|
| 68 |
+
│ 32k │ —/— │ —/— │ —/— │
|
| 69 |
+
╰────────────┴─────┴─────┴─────╯
|
| 70 |
+
Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
|
| 71 |
+
coordinate. ITL is computed from observed generated tokens, including streams
|
| 72 |
+
stopped at the measurement boundary; a missing ITL means no stream produced at
|
| 73 |
+
least two measured output tokens. Per-request tok/s and request latency are
|
| 74 |
+
shown in separate per-cell matrices. Completion/sample counts and full
|
| 75 |
+
request-level distributions remain in JSON under request_samples.
|
| 76 |
+
Sustained mode: client latency metrics explain request UX variance; aggregate
|
| 77 |
+
tok/s remains the primary throughput signal.
|
| 78 |
+
ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
|
| 79 |
+
Hardware Summary
|
| 80 |
+
╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
|
| 81 |
+
│ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
|
| 82 |
+
├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
|
| 83 |
+
│ 0 │ │ sust… │ 99/99% │ 48% │ 1151/1… │ 80C │ 76C │ 97… │ 8407/8352 │
|
| 84 |
+
│ … │ │ sust… │ 99/99% │ 49% │ 1154/1… │ 81C │ 75C │ 97… │ 8274/8292 │
|
| 85 |
+
│ … │ │ sust… │ 99/100% │ 48% │ 1154/1… │ 82C │ 76C │ 97… │ 8368/8179 │
|
| 86 |
+
│ 0 │ │ sust… │ 100/100% │ 45% │ 1177/1… │ 83C │ 76C │ 97… │ 11281/11144 │
|
| 87 |
+
│ 0 │ │ sust… │ 100/100% │ 40% │ 1173/1… │ 84C │ 77C │ 97… │ 7881/7872 │
|
| 88 |
+
│ … │ │ sust… │ 100/100% │ 45% │ 1178/1… │ 84C │ 76C │ 97… │ 11234/11152 │
|
| 89 |
+
│ … │ │ sust… │ 100/100% │ 39% │ 1174/1… │ 84C │ 77C │ 97… │ 7879/7780 │
|
| 90 |
+
│ … │ │ sust… │ 100/100% │ 45% │ 1178/1… │ 84C │ 77C │ 97… │ 11056/11029 │
|
| 91 |
+
│ … │ │ sust… │ 100/100% │ 39% │ 1173/1… │ 84C │ 77C │ 97… │ 7831/7804 │
|
| 92 |
+
╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
|
| 93 |
+
╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
|
| 94 |
+
│ avg 1,101 W | max 1,179 W | limit 1,200 W | over 4m 27s | 112 samples │
|
| 95 |
+
╰───────────────────────────────────────────────────────────────────────╯
|
| 96 |
+
Hardware summary is sampled from nvidia-smi during the measured part of each
|
| 97 |
+
cell. Whole-run GPU power is the sampled sum of GPU power draw across the
|
| 98 |
+
complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
|
| 99 |
+
a coarse live diagnostic, not a per-kernel NCCL profiler.
|
| 100 |
+
|
| 101 |
+
╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
|
| 102 |
+
│ Burst / E2E Decode │
|
| 103 |
+
│ Not run. Re-run with --run-burst to append a finite client-facing request │
|
| 104 |
+
│ burst after Sustained Decode. This is intentionally disabled by default │
|
| 105 |
+
│ because it adds another full decode matrix. │
|
| 106 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 107 |
+
|
| 108 |
+
╭────────────────────────────── Primary Summary ───────────────────────────────╮
|
| 109 |
+
│ Primary matrices repeated last so the important numbers are visible without │
|
| 110 |
+
│ scrolling back through diagnostics. │
|
| 111 |
+
╰─────────────────────────────────────��────────────────────────────────────────╯
|
| 112 |
+
Aggregate decode tok/s
|
| 113 |
+
╭────────────┬───────┬───────┬───────╮
|
| 114 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 115 |
+
├────────────┼───────┼───────┼───────┤
|
| 116 |
+
│ 0 │ 167.2 │ 244.6 │ 292.9 │
|
| 117 |
+
│ 8k │ 173.0 │ 246.0 │ 298.4 │
|
| 118 |
+
│ 32k │ 186.4 │ 252.6 │ 297.7 │
|
| 119 |
+
╰────────────┴───────┴───────┴───────╯
|
| 120 |
+
|
| 121 |
+
Results saved to
|
| 122 |
+
<campaign>/batch16384-speed-w
|
| 123 |
+
indow-01/results-01/batch16384/rep-1/decode-cap8192.json
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-command.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--prefill-only",
|
| 25 |
+
"--prefill-contexts",
|
| 26 |
+
"8k,32k,64k,128k",
|
| 27 |
+
"--prefill-duration",
|
| 28 |
+
"20",
|
| 29 |
+
"--cell-warmup-timeout-seconds",
|
| 30 |
+
"180"
|
| 31 |
+
]
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "0068157e646ee939ef2c7a0c5c860e6ec4582eecfe6284929bdeaa7272046797"
|
| 5 |
+
}
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.json
ADDED
|
@@ -0,0 +1,396 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T07:13:30.053751",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "standalone_cold",
|
| 15 |
+
"standalone_prefill": true,
|
| 16 |
+
"prefill_only": true,
|
| 17 |
+
"skip_prefill": false,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 0,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 17031168,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0
|
| 42 |
+
],
|
| 43 |
+
"startup_diagnostics_available": true,
|
| 44 |
+
"nvidia_p2p_override_effective": true,
|
| 45 |
+
"p2pmark_status": "not_run",
|
| 46 |
+
"amd_fabric_status": "not_run"
|
| 47 |
+
},
|
| 48 |
+
"startup_diagnostics": {
|
| 49 |
+
"version": "0.4.29",
|
| 50 |
+
"server_url": "http://127.0.0.1:8001",
|
| 51 |
+
"hostname": "<host>",
|
| 52 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 53 |
+
"env": {},
|
| 54 |
+
"args": {
|
| 55 |
+
"concurrency": "1,2,4",
|
| 56 |
+
"contexts": "0",
|
| 57 |
+
"max_tokens": 8192,
|
| 58 |
+
"duration": 20.0,
|
| 59 |
+
"request_count": 0,
|
| 60 |
+
"run_burst": false,
|
| 61 |
+
"standalone_prefill": true,
|
| 62 |
+
"prefill_only": true,
|
| 63 |
+
"skip_prefill": false,
|
| 64 |
+
"prefill_contexts": "8k,32k,64k,128k",
|
| 65 |
+
"prefill_metric": "client",
|
| 66 |
+
"dcp_size": 0,
|
| 67 |
+
"kv_budget": 0
|
| 68 |
+
},
|
| 69 |
+
"nvidia_p2p_override": {
|
| 70 |
+
"effective": true,
|
| 71 |
+
"configured": true,
|
| 72 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 73 |
+
"params_available": true,
|
| 74 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 75 |
+
"modprobe_available": true,
|
| 76 |
+
"runtime": {
|
| 77 |
+
"ForceP2P": "0x11",
|
| 78 |
+
"RMForceP2PType": "1",
|
| 79 |
+
"RMPcieP2PType": "2",
|
| 80 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 81 |
+
"EnableResizableBar": "1",
|
| 82 |
+
"DmaRemapPeerMmio": "1"
|
| 83 |
+
},
|
| 84 |
+
"expected": {
|
| 85 |
+
"ForceP2P": "0x11",
|
| 86 |
+
"RMForceP2PType": "1",
|
| 87 |
+
"RMPcieP2PType": "2",
|
| 88 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 89 |
+
"EnableResizableBar": "1"
|
| 90 |
+
},
|
| 91 |
+
"missing": [],
|
| 92 |
+
"mismatched": {},
|
| 93 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 94 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 95 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 96 |
+
},
|
| 97 |
+
"p2pmark": {
|
| 98 |
+
"status": "not_run"
|
| 99 |
+
},
|
| 100 |
+
"amd_fabric": {
|
| 101 |
+
"status": "not_run"
|
| 102 |
+
},
|
| 103 |
+
"nvidia_smi_query": {
|
| 104 |
+
"cmd": [
|
| 105 |
+
"nvidia-smi",
|
| 106 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 107 |
+
"--format=csv,noheader,nounits"
|
| 108 |
+
],
|
| 109 |
+
"returncode": 0,
|
| 110 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 111 |
+
"stderr": ""
|
| 112 |
+
},
|
| 113 |
+
"nvidia_smi_topo": {
|
| 114 |
+
"cmd": [
|
| 115 |
+
"nvidia-smi",
|
| 116 |
+
"topo",
|
| 117 |
+
"-m"
|
| 118 |
+
],
|
| 119 |
+
"returncode": 0,
|
| 120 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 121 |
+
"stderr": ""
|
| 122 |
+
}
|
| 123 |
+
},
|
| 124 |
+
"nvidia_p2p_override": {
|
| 125 |
+
"effective": true,
|
| 126 |
+
"configured": true,
|
| 127 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 128 |
+
"params_available": true,
|
| 129 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 130 |
+
"modprobe_available": true,
|
| 131 |
+
"runtime": {
|
| 132 |
+
"ForceP2P": "0x11",
|
| 133 |
+
"RMForceP2PType": "1",
|
| 134 |
+
"RMPcieP2PType": "2",
|
| 135 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 136 |
+
"EnableResizableBar": "1",
|
| 137 |
+
"DmaRemapPeerMmio": "1"
|
| 138 |
+
},
|
| 139 |
+
"expected": {
|
| 140 |
+
"ForceP2P": "0x11",
|
| 141 |
+
"RMForceP2PType": "1",
|
| 142 |
+
"RMPcieP2PType": "2",
|
| 143 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 144 |
+
"EnableResizableBar": "1"
|
| 145 |
+
},
|
| 146 |
+
"missing": [],
|
| 147 |
+
"mismatched": {},
|
| 148 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 149 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 150 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 151 |
+
},
|
| 152 |
+
"p2pmark": {
|
| 153 |
+
"status": "not_run"
|
| 154 |
+
},
|
| 155 |
+
"amd_fabric": {
|
| 156 |
+
"status": "not_run"
|
| 157 |
+
},
|
| 158 |
+
"hardware_run_summary": {
|
| 159 |
+
"samples": 46,
|
| 160 |
+
"duration_seconds": 108.093,
|
| 161 |
+
"gpu_count": 4,
|
| 162 |
+
"cpu_util_avg_pct": 10.0,
|
| 163 |
+
"cpu_temp_max_c": 76.25,
|
| 164 |
+
"gpu_util_avg_pct": 84.77,
|
| 165 |
+
"gpu_util_max_pct": 100.0,
|
| 166 |
+
"mem_util_avg_pct": 22.67,
|
| 167 |
+
"mem_util_max_pct": 35.0,
|
| 168 |
+
"temp_avg_c": 59.3,
|
| 169 |
+
"temp_max_c": 80.0,
|
| 170 |
+
"power_total_avg_w": 1039.43,
|
| 171 |
+
"power_total_max_w": 1180.51,
|
| 172 |
+
"power_limit_total_w": 1200.0,
|
| 173 |
+
"vram_used_avg_mb": 376719.74,
|
| 174 |
+
"vram_used_max_mb": 380174.0,
|
| 175 |
+
"vram_total_mb": 391548.0,
|
| 176 |
+
"vram_used_avg_pct": 96.22,
|
| 177 |
+
"vram_used_max_pct": 97.1,
|
| 178 |
+
"pcie_rx_avg_mb_s": 42900.63,
|
| 179 |
+
"pcie_rx_max_mb_s": 68701.0,
|
| 180 |
+
"pcie_tx_avg_mb_s": 40511.74,
|
| 181 |
+
"pcie_tx_max_mb_s": 63363.0
|
| 182 |
+
},
|
| 183 |
+
"event_log": [],
|
| 184 |
+
"prefill": {
|
| 185 |
+
"8192": {
|
| 186 |
+
"ttft_seconds": 1.041,
|
| 187 |
+
"prefill_seconds": 1.041,
|
| 188 |
+
"tok_per_sec": 7874.0,
|
| 189 |
+
"client_ttft_seconds": 1.041,
|
| 190 |
+
"client_tok_per_sec": 7874.0,
|
| 191 |
+
"prompt_tokens": 8194,
|
| 192 |
+
"samples": 14,
|
| 193 |
+
"method": "client",
|
| 194 |
+
"server_validation": {
|
| 195 |
+
"method": "",
|
| 196 |
+
"tok_per_sec": 0.0,
|
| 197 |
+
"prefill_seconds": 0.0,
|
| 198 |
+
"prompt_tokens": 0,
|
| 199 |
+
"request_prompt_tokens": 0,
|
| 200 |
+
"cached_tokens": 0,
|
| 201 |
+
"token_source": "",
|
| 202 |
+
"samples": 0,
|
| 203 |
+
"invalid_reason": ""
|
| 204 |
+
},
|
| 205 |
+
"hardware_summary": {
|
| 206 |
+
"samples": 9,
|
| 207 |
+
"duration_seconds": 19.127,
|
| 208 |
+
"gpu_count": 4,
|
| 209 |
+
"cpu_util_avg_pct": 10.29,
|
| 210 |
+
"cpu_temp_max_c": 75.25,
|
| 211 |
+
"gpu_util_avg_pct": 65.11,
|
| 212 |
+
"gpu_util_max_pct": 100.0,
|
| 213 |
+
"mem_util_avg_pct": 18.42,
|
| 214 |
+
"mem_util_max_pct": 33.0,
|
| 215 |
+
"temp_avg_c": 52.0,
|
| 216 |
+
"temp_max_c": 66.0,
|
| 217 |
+
"power_total_avg_w": 963.89,
|
| 218 |
+
"power_total_max_w": 1159.22,
|
| 219 |
+
"power_limit_total_w": 1200.0,
|
| 220 |
+
"vram_used_avg_mb": 366586.0,
|
| 221 |
+
"vram_used_max_mb": 366586.0,
|
| 222 |
+
"vram_total_mb": 391548.0,
|
| 223 |
+
"vram_used_avg_pct": 93.62,
|
| 224 |
+
"vram_used_max_pct": 93.62,
|
| 225 |
+
"pcie_rx_avg_mb_s": 33986.56,
|
| 226 |
+
"pcie_rx_max_mb_s": 58738.0,
|
| 227 |
+
"pcie_tx_avg_mb_s": 32431.22,
|
| 228 |
+
"pcie_tx_max_mb_s": 56269.0
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"32768": {
|
| 232 |
+
"ttft_seconds": 4.08,
|
| 233 |
+
"prefill_seconds": 4.08,
|
| 234 |
+
"tok_per_sec": 8033.0,
|
| 235 |
+
"client_ttft_seconds": 4.08,
|
| 236 |
+
"client_tok_per_sec": 8033.0,
|
| 237 |
+
"prompt_tokens": 32770,
|
| 238 |
+
"samples": 5,
|
| 239 |
+
"method": "client",
|
| 240 |
+
"server_validation": {
|
| 241 |
+
"method": "",
|
| 242 |
+
"tok_per_sec": 0.0,
|
| 243 |
+
"prefill_seconds": 0.0,
|
| 244 |
+
"prompt_tokens": 0,
|
| 245 |
+
"request_prompt_tokens": 0,
|
| 246 |
+
"cached_tokens": 0,
|
| 247 |
+
"token_source": "",
|
| 248 |
+
"samples": 0,
|
| 249 |
+
"invalid_reason": ""
|
| 250 |
+
},
|
| 251 |
+
"hardware_summary": {
|
| 252 |
+
"samples": 9,
|
| 253 |
+
"duration_seconds": 19.245,
|
| 254 |
+
"gpu_count": 4,
|
| 255 |
+
"cpu_util_avg_pct": 10.29,
|
| 256 |
+
"cpu_temp_max_c": 75.75,
|
| 257 |
+
"gpu_util_avg_pct": 97.5,
|
| 258 |
+
"gpu_util_max_pct": 100.0,
|
| 259 |
+
"mem_util_avg_pct": 25.89,
|
| 260 |
+
"mem_util_max_pct": 33.0,
|
| 261 |
+
"temp_avg_c": 58.39,
|
| 262 |
+
"temp_max_c": 73.0,
|
| 263 |
+
"power_total_avg_w": 1058.45,
|
| 264 |
+
"power_total_max_w": 1148.19,
|
| 265 |
+
"power_limit_total_w": 1200.0,
|
| 266 |
+
"vram_used_avg_mb": 380045.11,
|
| 267 |
+
"vram_used_max_mb": 380174.0,
|
| 268 |
+
"vram_total_mb": 391548.0,
|
| 269 |
+
"vram_used_avg_pct": 97.07,
|
| 270 |
+
"vram_used_max_pct": 97.1,
|
| 271 |
+
"pcie_rx_avg_mb_s": 50337.22,
|
| 272 |
+
"pcie_rx_max_mb_s": 61317.0,
|
| 273 |
+
"pcie_tx_avg_mb_s": 47808.0,
|
| 274 |
+
"pcie_tx_max_mb_s": 54706.0
|
| 275 |
+
}
|
| 276 |
+
},
|
| 277 |
+
"65536": {
|
| 278 |
+
"ttft_seconds": 8.239,
|
| 279 |
+
"prefill_seconds": 8.239,
|
| 280 |
+
"tok_per_sec": 7955.0,
|
| 281 |
+
"client_ttft_seconds": 8.239,
|
| 282 |
+
"client_tok_per_sec": 7955.0,
|
| 283 |
+
"prompt_tokens": 65538,
|
| 284 |
+
"samples": 3,
|
| 285 |
+
"method": "client",
|
| 286 |
+
"server_validation": {
|
| 287 |
+
"method": "",
|
| 288 |
+
"tok_per_sec": 0.0,
|
| 289 |
+
"prefill_seconds": 0.0,
|
| 290 |
+
"prompt_tokens": 0,
|
| 291 |
+
"request_prompt_tokens": 0,
|
| 292 |
+
"cached_tokens": 0,
|
| 293 |
+
"token_source": "",
|
| 294 |
+
"samples": 0,
|
| 295 |
+
"invalid_reason": ""
|
| 296 |
+
},
|
| 297 |
+
"hardware_summary": {
|
| 298 |
+
"samples": 11,
|
| 299 |
+
"duration_seconds": 24.061,
|
| 300 |
+
"gpu_count": 4,
|
| 301 |
+
"cpu_util_avg_pct": 10.33,
|
| 302 |
+
"cpu_temp_max_c": 76.0,
|
| 303 |
+
"gpu_util_avg_pct": 94.18,
|
| 304 |
+
"gpu_util_max_pct": 100.0,
|
| 305 |
+
"mem_util_avg_pct": 25.11,
|
| 306 |
+
"mem_util_max_pct": 35.0,
|
| 307 |
+
"temp_avg_c": 61.48,
|
| 308 |
+
"temp_max_c": 77.0,
|
| 309 |
+
"power_total_avg_w": 1093.56,
|
| 310 |
+
"power_total_max_w": 1180.51,
|
| 311 |
+
"power_limit_total_w": 1200.0,
|
| 312 |
+
"vram_used_avg_mb": 380174.0,
|
| 313 |
+
"vram_used_max_mb": 380174.0,
|
| 314 |
+
"vram_total_mb": 391548.0,
|
| 315 |
+
"vram_used_avg_pct": 97.1,
|
| 316 |
+
"vram_used_max_pct": 97.1,
|
| 317 |
+
"pcie_rx_avg_mb_s": 50410.73,
|
| 318 |
+
"pcie_rx_max_mb_s": 59338.0,
|
| 319 |
+
"pcie_tx_avg_mb_s": 47142.09,
|
| 320 |
+
"pcie_tx_max_mb_s": 60407.0
|
| 321 |
+
}
|
| 322 |
+
},
|
| 323 |
+
"131072": {
|
| 324 |
+
"ttft_seconds": 16.814,
|
| 325 |
+
"prefill_seconds": 16.814,
|
| 326 |
+
"tok_per_sec": 7796.0,
|
| 327 |
+
"client_ttft_seconds": 16.814,
|
| 328 |
+
"client_tok_per_sec": 7796.0,
|
| 329 |
+
"prompt_tokens": 131074,
|
| 330 |
+
"samples": 2,
|
| 331 |
+
"method": "client",
|
| 332 |
+
"server_validation": {
|
| 333 |
+
"method": "",
|
| 334 |
+
"tok_per_sec": 0.0,
|
| 335 |
+
"prefill_seconds": 0.0,
|
| 336 |
+
"prompt_tokens": 0,
|
| 337 |
+
"request_prompt_tokens": 0,
|
| 338 |
+
"cached_tokens": 0,
|
| 339 |
+
"token_source": "",
|
| 340 |
+
"samples": 0,
|
| 341 |
+
"invalid_reason": ""
|
| 342 |
+
},
|
| 343 |
+
"hardware_summary": {
|
| 344 |
+
"samples": 15,
|
| 345 |
+
"duration_seconds": 33.649,
|
| 346 |
+
"gpu_count": 4,
|
| 347 |
+
"cpu_util_avg_pct": 10.22,
|
| 348 |
+
"cpu_temp_max_c": 76.25,
|
| 349 |
+
"gpu_util_avg_pct": 93.33,
|
| 350 |
+
"gpu_util_max_pct": 100.0,
|
| 351 |
+
"mem_util_avg_pct": 24.52,
|
| 352 |
+
"mem_util_max_pct": 31.0,
|
| 353 |
+
"temp_avg_c": 64.23,
|
| 354 |
+
"temp_max_c": 80.0,
|
| 355 |
+
"power_total_avg_w": 1110.04,
|
| 356 |
+
"power_total_max_w": 1154.26,
|
| 357 |
+
"power_limit_total_w": 1200.0,
|
| 358 |
+
"vram_used_avg_mb": 380174.0,
|
| 359 |
+
"vram_used_max_mb": 380174.0,
|
| 360 |
+
"vram_total_mb": 391548.0,
|
| 361 |
+
"vram_used_avg_pct": 97.1,
|
| 362 |
+
"vram_used_max_pct": 97.1,
|
| 363 |
+
"pcie_rx_avg_mb_s": 43409.47,
|
| 364 |
+
"pcie_rx_max_mb_s": 68701.0,
|
| 365 |
+
"pcie_tx_avg_mb_s": 40960.33,
|
| 366 |
+
"pcie_tx_max_mb_s": 63363.0
|
| 367 |
+
}
|
| 368 |
+
}
|
| 369 |
+
},
|
| 370 |
+
"results": [],
|
| 371 |
+
"summary_table": {},
|
| 372 |
+
"burst_results": [],
|
| 373 |
+
"burst_summary_table": {},
|
| 374 |
+
"methodology": {
|
| 375 |
+
"prefill": {
|
| 376 |
+
"name": "Prefill",
|
| 377 |
+
"present": true,
|
| 378 |
+
"mode": "standalone_cold",
|
| 379 |
+
"formula": "prompt_tokens / TTFT",
|
| 380 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 381 |
+
},
|
| 382 |
+
"sustained_decode": {
|
| 383 |
+
"name": "Sustained Decode",
|
| 384 |
+
"present": false,
|
| 385 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 386 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 387 |
+
},
|
| 388 |
+
"burst_e2e_decode": {
|
| 389 |
+
"name": "Burst / E2E Decode",
|
| 390 |
+
"present": false,
|
| 391 |
+
"status": "not run; use --run-burst",
|
| 392 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 393 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 394 |
+
}
|
| 395 |
+
}
|
| 396 |
+
}
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/batch16384/rep-1/prefill.log
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0'] │
|
| 16 |
+
│ Decode: skipped (--prefill-only) | Max tokens: 8192 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill-only: standalone cold profile (client) | Sustained decode: 0 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 17,031,168 tokens (2079 blocks × 2048; local
|
| 23 |
+
4,257,792 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: standalone cold profile ['8k', '32k', '64k', '128k']
|
| 26 |
+
Calibrating padding text (run=jowlqgbaqdiz, up to 128k)...
|
| 27 |
+
8k: 50,558 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,152 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
64k: 411,264 chars (65,536 prompt tokens via /tokenize)
|
| 30 |
+
128k: 823,408 chars (131,072 prompt tokens via /tokenize)
|
| 31 |
+
Token targeting: /tokenize exact
|
| 32 |
+
Done.
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
llm-decode-bench v0.4.29
|
| 37 |
+
Prefill Speed (C=1, client ISL / TTFT)
|
| 38 |
+
|
| 39 |
+
Client PCIe rx/tx
|
| 40 |
+
Context Tokens TTFT (s) tok/s Server tok/s avg N
|
| 41 |
+
──────────────────────────────────────────────────────────────────────────────
|
| 42 |
+
8k 8,194 1.04 7,874 — 33987/32431 14
|
| 43 |
+
32k 32,770 4.08 8,033 — 50337/47808 5
|
| 44 |
+
64k 65,538 8.24 7,955 — 50411/47142 3
|
| 45 |
+
128k 131,074 16.81 7,796 — 43409/40960 2
|
| 46 |
+
|
| 47 |
+
Client tok/s = prompt_tokens / TTFT. Integrated scout rows come from the
|
| 48 |
+
prefix-cache scout request that decode needs anyway. Server tok/s is optional
|
| 49 |
+
Prometheus validation when the engine exports prefill counters and the exact
|
| 50 |
+
counter delta is uncontaminated; for vLLM this uses newly computed KV tokens,
|
| 51 |
+
not request prompt tokens.
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
Results saved to
|
| 55 |
+
<campaign>/batch16384-speed-w
|
| 56 |
+
indow-01/results-01/batch16384/rep-1/prefill.json
|
results/speed-20260909/evidence/batch16384-speed-window-01/results-01/failure.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"error": "KeyboardInterrupt()"
|
| 3 |
+
}
|
results/speed-20260909/evidence/candidate-graph-analysis-02/REPORT.md
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Candidate FP8 serving graph profile
|
| 2 |
+
|
| 3 |
+
All four planned cells completed on retained image sha256:9bb99e0b47f00c4ccf77f2d4be77bb7c3ce8d6b5636169aa1f1fddda040ab45e. Runtime/source receipts match TP4/DCP4, MTP3 probabilistic, NVFP4 KV, graphs, maxseq24, batch4096, four300W GPUs. Original production and clients stayed stopped. These are instrumented diagnostic traces, not new throughput claims.
|
| 4 |
+
|
| 5 |
+
| Profile | Graph kernel events mapped | Non-graph kernel events |
|
| 6 |
+
|---|---:|---:|
|
| 7 |
+
| 8K C1 decode | 1020936 | 31624 |
|
| 8 |
+
| 8K C4 decode | 429400 | 16400 |
|
| 9 |
+
| 32K cold prefill plus setup | 214348 | 238940 |
|
| 10 |
+
| 64K cold prefill plus setup | 219208 | 350852 |
|
| 11 |
+
|
| 12 |
+
All1,883,892graph-associated events map via explicit process-scoped clone/original node IDs to captured topology. All6,368graph receipts succeeded. Both decode clients report zero errors and no underfill/capacity/warmup-timeout flags. Raw capture timings include start/stop RPC latency; nominal1.5seconds is the controller sleep between profiling RPCs, not an exact measured trace duration.
|
| 13 |
+
|
| 14 |
+
Target P8 direct FC1/FC2 kernels are present in both decode traces. P8 grouped FC1/FC2 kernels are additionally present in both prefill traces. The serving adapter directly dispatches P8NativeTPMoE with use_a16=False; missing target runtime raises an error. Marlin kernels are recorded separately and must retain their layer-attribution boundary. Four representative compiled target regimes were independently checked for QMMA.SF.16832.F32.E4M3.E4M3.E8; no W4A16 target replacement was made.
|
| 15 |
+
|
| 16 |
+
Do not remove waits from this evidence. Many graph edges have non-default dependency/port types, which the analyzer preserves without interpreting them as full-completion barriers. Some default-edge kernel timestamps overlap by approximately1.0-1.44microseconds in C1; keep these anomalies instead of treating them as dependency violations or deleting edges. Cross-rank, eager-prefill and external-stream dependencies remain incomplete. Summed durations are not wall time or attainable speedup fractions.
|
| 17 |
+
|
| 18 |
+
Next bounded candidate comes from actual grouped FC2 launch geometry:256CTAs,128threads, registers199(K4)/216(K5), reported dynamic shared34816/38912bytes and executed partition102400bytes where available. These resources admit2blocks per188-SM device, or376CTAs, but current grid exposes only256. Test376CTAs with unchanged task stride/ownership and unchanged FP8/Hadamard/orderedK512 math. More CTAs might reduce imbalance or hurt locality; CPU task-coverage/source-scope proof passed and incremental component test is running. This does not assume the300C1/20000prefill targets are reachable from this change.
|
| 19 |
+
|
| 20 |
+
All raw Nsight reports and graph records are indexed by graph-profile-preservation-02/INDEX.json. Exported SQLite and these analyses are additional artifacts, hashed separately here. Earlier failed profiling attempts remain preserved.
|
results/speed-20260909/evidence/candidate-graph-analysis-02/SUMMARY.json
ADDED
|
@@ -0,0 +1,242 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "all-four-profiles-completed-and-graph-associated-kernels-mapped",
|
| 3 |
+
"profiles": [
|
| 4 |
+
{
|
| 5 |
+
"profile": "8K C1 decode",
|
| 6 |
+
"coverage": {
|
| 7 |
+
"total_kernel_events": 1052560,
|
| 8 |
+
"matched_graph_kernel_events": 1020936,
|
| 9 |
+
"non_graph_kernel_events": 31624
|
| 10 |
+
},
|
| 11 |
+
"kernel_categories": {
|
| 12 |
+
"other": {
|
| 13 |
+
"calls": 1010080,
|
| 14 |
+
"summed_duration_ns": 5759231895
|
| 15 |
+
},
|
| 16 |
+
"Marlin-needs-layer-attribution": {
|
| 17 |
+
"calls": 2832,
|
| 18 |
+
"summed_duration_ns": 64929376
|
| 19 |
+
},
|
| 20 |
+
"target-P8-direct-FC1": {
|
| 21 |
+
"calls": 19824,
|
| 22 |
+
"summed_duration_ns": 1332961683
|
| 23 |
+
},
|
| 24 |
+
"target-P8-direct-FC2": {
|
| 25 |
+
"calls": 19824,
|
| 26 |
+
"summed_duration_ns": 476408029
|
| 27 |
+
}
|
| 28 |
+
},
|
| 29 |
+
"edge_checks": {
|
| 30 |
+
"both_endpoints_timed": 1122416,
|
| 31 |
+
"nondefault_edges_not_interpreted_as_completion": 433768,
|
| 32 |
+
"default_completion_edges": 688648
|
| 33 |
+
},
|
| 34 |
+
"timing_anomaly_count": 299,
|
| 35 |
+
"grouped_fc2_resources": [],
|
| 36 |
+
"analysis_sha256": "8f6e2864cbd68a184a6f8b33b74adaa31da50612ce33e99e230b1205a54e0a27"
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"profile": "8K C4 decode",
|
| 40 |
+
"coverage": {
|
| 41 |
+
"total_kernel_events": 445800,
|
| 42 |
+
"matched_graph_kernel_events": 429400,
|
| 43 |
+
"non_graph_kernel_events": 16400
|
| 44 |
+
},
|
| 45 |
+
"kernel_categories": {
|
| 46 |
+
"other": {
|
| 47 |
+
"calls": 427800,
|
| 48 |
+
"summed_duration_ns": 3763933505
|
| 49 |
+
},
|
| 50 |
+
"Marlin-needs-layer-attribution": {
|
| 51 |
+
"calls": 1200,
|
| 52 |
+
"summed_duration_ns": 66829405
|
| 53 |
+
},
|
| 54 |
+
"target-P8-direct-FC1": {
|
| 55 |
+
"calls": 8400,
|
| 56 |
+
"summed_duration_ns": 2249130432
|
| 57 |
+
},
|
| 58 |
+
"target-P8-direct-FC2": {
|
| 59 |
+
"calls": 8400,
|
| 60 |
+
"summed_duration_ns": 846519484
|
| 61 |
+
}
|
| 62 |
+
},
|
| 63 |
+
"edge_checks": {
|
| 64 |
+
"both_endpoints_timed": 473200,
|
| 65 |
+
"default_completion_edges": 279800,
|
| 66 |
+
"nondefault_edges_not_interpreted_as_completion": 193400
|
| 67 |
+
},
|
| 68 |
+
"timing_anomaly_count": 0,
|
| 69 |
+
"grouped_fc2_resources": [],
|
| 70 |
+
"analysis_sha256": "ecdbc033af4eae4984b62014ccc2e7f586a2b4e98992b3edba945b887f56fa57"
|
| 71 |
+
},
|
| 72 |
+
{
|
| 73 |
+
"profile": "32K cold prefill plus setup",
|
| 74 |
+
"coverage": {
|
| 75 |
+
"total_kernel_events": 453288,
|
| 76 |
+
"matched_graph_kernel_events": 214348,
|
| 77 |
+
"non_graph_kernel_events": 238940
|
| 78 |
+
},
|
| 79 |
+
"kernel_categories": {
|
| 80 |
+
"other": {
|
| 81 |
+
"calls": 438168,
|
| 82 |
+
"summed_duration_ns": 25529631362
|
| 83 |
+
},
|
| 84 |
+
"Marlin-needs-layer-attribution": {
|
| 85 |
+
"calls": 1008,
|
| 86 |
+
"summed_duration_ns": 21740194
|
| 87 |
+
},
|
| 88 |
+
"target-P8-direct-FC1": {
|
| 89 |
+
"calls": 3864,
|
| 90 |
+
"summed_duration_ns": 246439082
|
| 91 |
+
},
|
| 92 |
+
"target-P8-direct-FC2": {
|
| 93 |
+
"calls": 3864,
|
| 94 |
+
"summed_duration_ns": 87860594
|
| 95 |
+
},
|
| 96 |
+
"target-P8-grouped-FC1": {
|
| 97 |
+
"calls": 3192,
|
| 98 |
+
"summed_duration_ns": 4867068677
|
| 99 |
+
},
|
| 100 |
+
"target-P8-grouped-FC2": {
|
| 101 |
+
"calls": 3192,
|
| 102 |
+
"summed_duration_ns": 3764819288
|
| 103 |
+
}
|
| 104 |
+
},
|
| 105 |
+
"edge_checks": {
|
| 106 |
+
"both_endpoints_timed": 232380,
|
| 107 |
+
"nondefault_edges_not_interpreted_as_completion": 87884,
|
| 108 |
+
"default_completion_edges": 144496
|
| 109 |
+
},
|
| 110 |
+
"timing_anomaly_count": 69,
|
| 111 |
+
"grouped_fc2_resources": [
|
| 112 |
+
{
|
| 113 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
|
| 114 |
+
"grid_z": 256,
|
| 115 |
+
"block_x": 128,
|
| 116 |
+
"registers": 199,
|
| 117 |
+
"dynamic_shared": 34816,
|
| 118 |
+
"shared_partition": 0,
|
| 119 |
+
"requested_percent": null,
|
| 120 |
+
"calls": 1800
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
|
| 124 |
+
"grid_z": 256,
|
| 125 |
+
"block_x": 128,
|
| 126 |
+
"registers": 199,
|
| 127 |
+
"dynamic_shared": 34816,
|
| 128 |
+
"shared_partition": 102400,
|
| 129 |
+
"requested_percent": 68,
|
| 130 |
+
"calls": 100
|
| 131 |
+
},
|
| 132 |
+
{
|
| 133 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
|
| 134 |
+
"grid_z": 256,
|
| 135 |
+
"block_x": 128,
|
| 136 |
+
"registers": 216,
|
| 137 |
+
"dynamic_shared": 38912,
|
| 138 |
+
"shared_partition": 0,
|
| 139 |
+
"requested_percent": null,
|
| 140 |
+
"calls": 1224
|
| 141 |
+
},
|
| 142 |
+
{
|
| 143 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
|
| 144 |
+
"grid_z": 256,
|
| 145 |
+
"block_x": 128,
|
| 146 |
+
"registers": 216,
|
| 147 |
+
"dynamic_shared": 38912,
|
| 148 |
+
"shared_partition": 102400,
|
| 149 |
+
"requested_percent": 76,
|
| 150 |
+
"calls": 68
|
| 151 |
+
}
|
| 152 |
+
],
|
| 153 |
+
"analysis_sha256": "dbc87a37bea8fdc6ce706e723c40a34d0596c40a075e8db8d9c506ead76f76cd"
|
| 154 |
+
},
|
| 155 |
+
{
|
| 156 |
+
"profile": "64K cold prefill plus setup",
|
| 157 |
+
"coverage": {
|
| 158 |
+
"total_kernel_events": 570060,
|
| 159 |
+
"matched_graph_kernel_events": 219208,
|
| 160 |
+
"non_graph_kernel_events": 350852
|
| 161 |
+
},
|
| 162 |
+
"kernel_categories": {
|
| 163 |
+
"other": {
|
| 164 |
+
"calls": 551700,
|
| 165 |
+
"summed_duration_ns": 39640395478
|
| 166 |
+
},
|
| 167 |
+
"target-P8-grouped-FC1": {
|
| 168 |
+
"calls": 4704,
|
| 169 |
+
"summed_duration_ns": 7145088363
|
| 170 |
+
},
|
| 171 |
+
"target-P8-grouped-FC2": {
|
| 172 |
+
"calls": 4704,
|
| 173 |
+
"summed_duration_ns": 5633648267
|
| 174 |
+
},
|
| 175 |
+
"Marlin-needs-layer-attribution": {
|
| 176 |
+
"calls": 1224,
|
| 177 |
+
"summed_duration_ns": 25492544
|
| 178 |
+
},
|
| 179 |
+
"target-P8-direct-FC1": {
|
| 180 |
+
"calls": 3864,
|
| 181 |
+
"summed_duration_ns": 247245351
|
| 182 |
+
},
|
| 183 |
+
"target-P8-direct-FC2": {
|
| 184 |
+
"calls": 3864,
|
| 185 |
+
"summed_duration_ns": 87240073
|
| 186 |
+
}
|
| 187 |
+
},
|
| 188 |
+
"edge_checks": {
|
| 189 |
+
"both_endpoints_timed": 237024,
|
| 190 |
+
"default_completion_edges": 148096,
|
| 191 |
+
"nondefault_edges_not_interpreted_as_completion": 88928
|
| 192 |
+
},
|
| 193 |
+
"timing_anomaly_count": 61,
|
| 194 |
+
"grouped_fc2_resources": [
|
| 195 |
+
{
|
| 196 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
|
| 197 |
+
"grid_z": 256,
|
| 198 |
+
"block_x": 128,
|
| 199 |
+
"registers": 199,
|
| 200 |
+
"dynamic_shared": 34816,
|
| 201 |
+
"shared_partition": 0,
|
| 202 |
+
"requested_percent": null,
|
| 203 |
+
"calls": 2700
|
| 204 |
+
},
|
| 205 |
+
{
|
| 206 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o754974721_tensorptri32gmemalign16o1_2",
|
| 207 |
+
"grid_z": 256,
|
| 208 |
+
"block_x": 128,
|
| 209 |
+
"registers": 199,
|
| 210 |
+
"dynamic_shared": 34816,
|
| 211 |
+
"shared_partition": 102400,
|
| 212 |
+
"requested_percent": 68,
|
| 213 |
+
"calls": 100
|
| 214 |
+
},
|
| 215 |
+
{
|
| 216 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
|
| 217 |
+
"grid_z": 256,
|
| 218 |
+
"block_x": 128,
|
| 219 |
+
"registers": 216,
|
| 220 |
+
"dynamic_shared": 38912,
|
| 221 |
+
"shared_partition": 0,
|
| 222 |
+
"requested_percent": null,
|
| 223 |
+
"calls": 1836
|
| 224 |
+
},
|
| 225 |
+
{
|
| 226 |
+
"kernel": "kernel_cutlass_kernel_b12xmoe_sharedkernelsp8_coupled_prefill_fc2P8CoupledPrefillFC2Kernel_object_at__tensorptri32gmemalign16o1_tensorptri32gmemalign16o943718401_tensorptri32gmemalign16o1_2",
|
| 227 |
+
"grid_z": 256,
|
| 228 |
+
"block_x": 128,
|
| 229 |
+
"registers": 216,
|
| 230 |
+
"dynamic_shared": 38912,
|
| 231 |
+
"shared_partition": 102400,
|
| 232 |
+
"requested_percent": 76,
|
| 233 |
+
"calls": 68
|
| 234 |
+
}
|
| 235 |
+
],
|
| 236 |
+
"analysis_sha256": "a5bb8b2ec42a9930a512254c254e23c163d31b3bfd8afc9edcaed2aa1c33c55d"
|
| 237 |
+
}
|
| 238 |
+
],
|
| 239 |
+
"throughput_claim": false,
|
| 240 |
+
"next_candidate": "grouped-fc2-grid376-component-01",
|
| 241 |
+
"scope": "All graph-associated kernel events matched; eager/non-graph prefill kernels remain outside graph-topology coverage."
|
| 242 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-01/results-01/failure.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"error": "RuntimeError(\"trellismx-candidate-graph-profile-01 exited: {'Status': 'exited', 'Running': False, 'Paused': False, 'Restarting': False, 'OOMKilled': False, 'Dead': False, 'Pid': 0, 'ExitCode': 1, 'Error': '', 'StartedAt': '2026-09-09T14:00:23.959294351Z', 'FinishedAt': '2026-09-09T14:03:17.551677807Z'}\")"
|
| 3 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c1-8k/result.json
ADDED
|
@@ -0,0 +1,345 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T10:10:50.723841",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 8192,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": 0.0,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29188096,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1
|
| 37 |
+
],
|
| 38 |
+
"context_lengths": [
|
| 39 |
+
8192
|
| 40 |
+
],
|
| 41 |
+
"startup_diagnostics_available": true,
|
| 42 |
+
"nvidia_p2p_override_effective": true,
|
| 43 |
+
"p2pmark_status": "not_run",
|
| 44 |
+
"amd_fabric_status": "not_run"
|
| 45 |
+
},
|
| 46 |
+
"startup_diagnostics": {
|
| 47 |
+
"version": "0.4.29",
|
| 48 |
+
"server_url": "http://127.0.0.1:8001",
|
| 49 |
+
"hostname": "<host>",
|
| 50 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 51 |
+
"env": {},
|
| 52 |
+
"args": {
|
| 53 |
+
"concurrency": "1",
|
| 54 |
+
"contexts": "8k",
|
| 55 |
+
"max_tokens": 8192,
|
| 56 |
+
"duration": 20.0,
|
| 57 |
+
"request_count": 0,
|
| 58 |
+
"run_burst": false,
|
| 59 |
+
"standalone_prefill": false,
|
| 60 |
+
"prefill_only": false,
|
| 61 |
+
"skip_prefill": true,
|
| 62 |
+
"prefill_contexts": "8k,64k,128k",
|
| 63 |
+
"prefill_metric": "client",
|
| 64 |
+
"dcp_size": 0,
|
| 65 |
+
"kv_budget": 0
|
| 66 |
+
},
|
| 67 |
+
"nvidia_p2p_override": {
|
| 68 |
+
"effective": true,
|
| 69 |
+
"configured": true,
|
| 70 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 71 |
+
"params_available": true,
|
| 72 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 73 |
+
"modprobe_available": true,
|
| 74 |
+
"runtime": {
|
| 75 |
+
"ForceP2P": "0x11",
|
| 76 |
+
"RMForceP2PType": "1",
|
| 77 |
+
"RMPcieP2PType": "2",
|
| 78 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 79 |
+
"EnableResizableBar": "1",
|
| 80 |
+
"DmaRemapPeerMmio": "1"
|
| 81 |
+
},
|
| 82 |
+
"expected": {
|
| 83 |
+
"ForceP2P": "0x11",
|
| 84 |
+
"RMForceP2PType": "1",
|
| 85 |
+
"RMPcieP2PType": "2",
|
| 86 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 87 |
+
"EnableResizableBar": "1"
|
| 88 |
+
},
|
| 89 |
+
"missing": [],
|
| 90 |
+
"mismatched": {},
|
| 91 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 92 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 93 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 94 |
+
},
|
| 95 |
+
"p2pmark": {
|
| 96 |
+
"status": "not_run"
|
| 97 |
+
},
|
| 98 |
+
"amd_fabric": {
|
| 99 |
+
"status": "not_run"
|
| 100 |
+
},
|
| 101 |
+
"nvidia_smi_query": {
|
| 102 |
+
"cmd": [
|
| 103 |
+
"nvidia-smi",
|
| 104 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 105 |
+
"--format=csv,noheader,nounits"
|
| 106 |
+
],
|
| 107 |
+
"returncode": 0,
|
| 108 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 109 |
+
"stderr": ""
|
| 110 |
+
},
|
| 111 |
+
"nvidia_smi_topo": {
|
| 112 |
+
"cmd": [
|
| 113 |
+
"nvidia-smi",
|
| 114 |
+
"topo",
|
| 115 |
+
"-m"
|
| 116 |
+
],
|
| 117 |
+
"returncode": 0,
|
| 118 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 119 |
+
"stderr": ""
|
| 120 |
+
}
|
| 121 |
+
},
|
| 122 |
+
"nvidia_p2p_override": {
|
| 123 |
+
"effective": true,
|
| 124 |
+
"configured": true,
|
| 125 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 126 |
+
"params_available": true,
|
| 127 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 128 |
+
"modprobe_available": true,
|
| 129 |
+
"runtime": {
|
| 130 |
+
"ForceP2P": "0x11",
|
| 131 |
+
"RMForceP2PType": "1",
|
| 132 |
+
"RMPcieP2PType": "2",
|
| 133 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 134 |
+
"EnableResizableBar": "1",
|
| 135 |
+
"DmaRemapPeerMmio": "1"
|
| 136 |
+
},
|
| 137 |
+
"expected": {
|
| 138 |
+
"ForceP2P": "0x11",
|
| 139 |
+
"RMForceP2PType": "1",
|
| 140 |
+
"RMPcieP2PType": "2",
|
| 141 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 142 |
+
"EnableResizableBar": "1"
|
| 143 |
+
},
|
| 144 |
+
"missing": [],
|
| 145 |
+
"mismatched": {},
|
| 146 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 147 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 148 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 149 |
+
},
|
| 150 |
+
"p2pmark": {
|
| 151 |
+
"status": "not_run"
|
| 152 |
+
},
|
| 153 |
+
"amd_fabric": {
|
| 154 |
+
"status": "not_run"
|
| 155 |
+
},
|
| 156 |
+
"hardware_run_summary": {
|
| 157 |
+
"samples": 19,
|
| 158 |
+
"duration_seconds": 43.426,
|
| 159 |
+
"gpu_count": 4,
|
| 160 |
+
"cpu_util_avg_pct": 8.54,
|
| 161 |
+
"cpu_temp_max_c": 75.12,
|
| 162 |
+
"gpu_util_avg_pct": 64.61,
|
| 163 |
+
"gpu_util_max_pct": 100.0,
|
| 164 |
+
"mem_util_avg_pct": 35.28,
|
| 165 |
+
"mem_util_max_pct": 60.0,
|
| 166 |
+
"temp_avg_c": 48.41,
|
| 167 |
+
"temp_max_c": 67.0,
|
| 168 |
+
"power_total_avg_w": 874.72,
|
| 169 |
+
"power_total_max_w": 1154.84,
|
| 170 |
+
"power_limit_total_w": 1200.0,
|
| 171 |
+
"vram_used_avg_mb": 386175.47,
|
| 172 |
+
"vram_used_max_mb": 386578.0,
|
| 173 |
+
"vram_total_mb": 391548.0,
|
| 174 |
+
"vram_used_avg_pct": 98.63,
|
| 175 |
+
"vram_used_max_pct": 98.73,
|
| 176 |
+
"pcie_rx_avg_mb_s": 8453.16,
|
| 177 |
+
"pcie_rx_max_mb_s": 48004.0,
|
| 178 |
+
"pcie_tx_avg_mb_s": 8921.84,
|
| 179 |
+
"pcie_tx_max_mb_s": 47296.0
|
| 180 |
+
},
|
| 181 |
+
"event_log": [
|
| 182 |
+
"10:10:04 benchmark start engine=vllm",
|
| 183 |
+
"10:10:04 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 184 |
+
"10:10:04 startup decode concurrency=1 contexts=8k",
|
| 185 |
+
"10:10:04 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 186 |
+
"10:10:04 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 187 |
+
"10:10:04 startup KV cache budget from vLLM metrics: 29,188,096 tokens (3563 blocks x 2048; local 7,297,024 \u00d7 CP 4; CP source: local process)",
|
| 188 |
+
"10:10:04 startup model context length: 1,000,000 tokens",
|
| 189 |
+
"10:10:04 startup prefill tests: skipped",
|
| 190 |
+
"10:10:04 startup calibrating padding text run=tmxrepeatabc up_to=8k",
|
| 191 |
+
"10:10:04 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
|
| 192 |
+
"10:10:04 startup token targeting: /tokenize exact",
|
| 193 |
+
"10:10:04 startup startup preparation done",
|
| 194 |
+
"10:10:04 hardware monitor interval=2s",
|
| 195 |
+
"10:10:04 decode warmup start",
|
| 196 |
+
"10:10:05 decode warmup start C=1 ctx=8k 3s",
|
| 197 |
+
"10:10:05 cell start C=1 ctx=8k",
|
| 198 |
+
"10:10:10 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 199 |
+
"10:10:21 cell done C=1 ctx=8k 212.1 tok/s",
|
| 200 |
+
"10:10:21 decode warmup done C=1 ctx=8k",
|
| 201 |
+
"10:10:23 cell start C=1 ctx=8k",
|
| 202 |
+
"10:10:28 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 203 |
+
"10:10:48 cell done C=1 ctx=8k 215.3 tok/s"
|
| 204 |
+
],
|
| 205 |
+
"prefill": {},
|
| 206 |
+
"results": [
|
| 207 |
+
{
|
| 208 |
+
"concurrency": 1,
|
| 209 |
+
"context_tokens": 8192,
|
| 210 |
+
"benchmark_mode": "duration",
|
| 211 |
+
"request_count_target": 0,
|
| 212 |
+
"warmup_request_count": 0,
|
| 213 |
+
"measurement_seconds": 19.989377,
|
| 214 |
+
"measurement_wall_seconds": 20.000509,
|
| 215 |
+
"client_output_tokens": 4303,
|
| 216 |
+
"server_output_tokens": 4306,
|
| 217 |
+
"aggregate_source": "openai_continuous_usage",
|
| 218 |
+
"aggregate_tps": 215.26434223990609,
|
| 219 |
+
"per_request_avg_tps": 215.26434223990609,
|
| 220 |
+
"ttft_avg": 0.5413568730000407,
|
| 221 |
+
"ttft_p50": 0.5413568730000407,
|
| 222 |
+
"ttft_p90": 0.5413568730000407,
|
| 223 |
+
"ttft_p99": 0.5413568730000407,
|
| 224 |
+
"time_to_second_token_avg": 0.014796818839386106,
|
| 225 |
+
"time_to_second_token_p50": 0.014796818839386106,
|
| 226 |
+
"time_to_second_token_p90": 0.014796818839386106,
|
| 227 |
+
"time_to_second_token_p99": 0.014796818839386106,
|
| 228 |
+
"request_latency_avg": 0.0,
|
| 229 |
+
"request_latency_p50": 0.0,
|
| 230 |
+
"request_latency_p90": 0.0,
|
| 231 |
+
"request_latency_p99": 0.0,
|
| 232 |
+
"inter_token_latency_avg": 0.004611418287442928,
|
| 233 |
+
"inter_token_latency_p50": 0.004611418287442928,
|
| 234 |
+
"inter_token_latency_p90": 0.004611418287442928,
|
| 235 |
+
"inter_token_latency_p99": 0.004611418287442928,
|
| 236 |
+
"output_tps_per_user_avg": 216.8530238783671,
|
| 237 |
+
"output_tps_per_user_p50": 216.8530238783671,
|
| 238 |
+
"output_tps_per_user_p90": 216.8530238783671,
|
| 239 |
+
"output_tps_per_user_p99": 216.8530238783671,
|
| 240 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 241 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 242 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 243 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 244 |
+
"chunk_inter_token_latency_avg": 0.012825661236893406,
|
| 245 |
+
"chunk_inter_token_latency_p50": 0.012825661236893406,
|
| 246 |
+
"chunk_inter_token_latency_p90": 0.012825661236893406,
|
| 247 |
+
"chunk_inter_token_latency_p99": 0.012825661236893406,
|
| 248 |
+
"input_seq_len_avg": 8192.0,
|
| 249 |
+
"output_seq_len_avg": 5202.0,
|
| 250 |
+
"output_seq_len_p50": 5202.0,
|
| 251 |
+
"output_seq_len_p90": 5202.0,
|
| 252 |
+
"output_seq_len_p99": 5202.0,
|
| 253 |
+
"request_count": 1,
|
| 254 |
+
"completed_request_count": 0,
|
| 255 |
+
"request_samples": [
|
| 256 |
+
{
|
| 257 |
+
"ttft": 0.5413568730000407,
|
| 258 |
+
"time_to_second_token": 0.014796818839386106,
|
| 259 |
+
"latency": 0.0,
|
| 260 |
+
"inter_token_latency_avg": 0.004611418287442928,
|
| 261 |
+
"chunk_inter_token_latency_avg": 0.012825661236893406,
|
| 262 |
+
"input_tokens": 8192,
|
| 263 |
+
"output_tokens": 5202,
|
| 264 |
+
"output_tps_per_user": 216.8530238783671,
|
| 265 |
+
"e2e_output_tps_per_user": 0.0,
|
| 266 |
+
"completed": false
|
| 267 |
+
}
|
| 268 |
+
],
|
| 269 |
+
"total_tokens": 4303,
|
| 270 |
+
"wall_time": 25.557856339029968,
|
| 271 |
+
"num_completed": 1,
|
| 272 |
+
"num_errors": 0,
|
| 273 |
+
"server_gen_throughput": 215.2383033924078,
|
| 274 |
+
"server_utilization": 0.006176305446378483,
|
| 275 |
+
"server_spec_accept_rate": 0.3924050632911392,
|
| 276 |
+
"server_spec_accept_length": 0.0,
|
| 277 |
+
"avg_running_reqs": 1,
|
| 278 |
+
"max_running_reqs": 1,
|
| 279 |
+
"effective_concurrency": 1,
|
| 280 |
+
"avg_queue_reqs": 0,
|
| 281 |
+
"max_queue_reqs": 0,
|
| 282 |
+
"queue_fraction": 0.0,
|
| 283 |
+
"underfilled": false,
|
| 284 |
+
"warmup_timed_out": false,
|
| 285 |
+
"warmup_duration": 5.552,
|
| 286 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 287 |
+
"timeout_reason": "",
|
| 288 |
+
"capacity_limited": false,
|
| 289 |
+
"hardware_summary": {
|
| 290 |
+
"samples": 9,
|
| 291 |
+
"duration_seconds": 19.391,
|
| 292 |
+
"gpu_count": 4,
|
| 293 |
+
"cpu_util_avg_pct": 10.9,
|
| 294 |
+
"cpu_temp_max_c": 75.12,
|
| 295 |
+
"gpu_util_avg_pct": 99.0,
|
| 296 |
+
"gpu_util_max_pct": 99.0,
|
| 297 |
+
"mem_util_avg_pct": 56.72,
|
| 298 |
+
"mem_util_max_pct": 60.0,
|
| 299 |
+
"temp_avg_c": 53.08,
|
| 300 |
+
"temp_max_c": 67.0,
|
| 301 |
+
"power_total_avg_w": 1148.08,
|
| 302 |
+
"power_total_max_w": 1154.84,
|
| 303 |
+
"power_limit_total_w": 1200.0,
|
| 304 |
+
"vram_used_avg_mb": 386578.0,
|
| 305 |
+
"vram_used_max_mb": 386578.0,
|
| 306 |
+
"vram_total_mb": 391548.0,
|
| 307 |
+
"vram_used_avg_pct": 98.73,
|
| 308 |
+
"vram_used_max_pct": 98.73,
|
| 309 |
+
"pcie_rx_avg_mb_s": 9520.33,
|
| 310 |
+
"pcie_rx_max_mb_s": 9579.0,
|
| 311 |
+
"pcie_tx_avg_mb_s": 9139.0,
|
| 312 |
+
"pcie_tx_max_mb_s": 9237.0
|
| 313 |
+
}
|
| 314 |
+
}
|
| 315 |
+
],
|
| 316 |
+
"summary_table": {
|
| 317 |
+
"8192": {
|
| 318 |
+
"1": 215.26434223990609
|
| 319 |
+
}
|
| 320 |
+
},
|
| 321 |
+
"burst_results": [],
|
| 322 |
+
"burst_summary_table": {},
|
| 323 |
+
"methodology": {
|
| 324 |
+
"prefill": {
|
| 325 |
+
"name": "Prefill",
|
| 326 |
+
"present": false,
|
| 327 |
+
"mode": "skipped",
|
| 328 |
+
"formula": "prompt_tokens / TTFT",
|
| 329 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 330 |
+
},
|
| 331 |
+
"sustained_decode": {
|
| 332 |
+
"name": "Sustained Decode",
|
| 333 |
+
"present": true,
|
| 334 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 335 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 336 |
+
},
|
| 337 |
+
"burst_e2e_decode": {
|
| 338 |
+
"name": "Burst / E2E Decode",
|
| 339 |
+
"present": false,
|
| 340 |
+
"status": "not run; use --run-burst",
|
| 341 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 342 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 343 |
+
}
|
| 344 |
+
}
|
| 345 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/decode-c4-8k/result.json
ADDED
|
@@ -0,0 +1,381 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T10:11:44.827451",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 8192,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": 0.0,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29188096,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
4
|
| 37 |
+
],
|
| 38 |
+
"context_lengths": [
|
| 39 |
+
8192
|
| 40 |
+
],
|
| 41 |
+
"startup_diagnostics_available": true,
|
| 42 |
+
"nvidia_p2p_override_effective": true,
|
| 43 |
+
"p2pmark_status": "not_run",
|
| 44 |
+
"amd_fabric_status": "not_run"
|
| 45 |
+
},
|
| 46 |
+
"startup_diagnostics": {
|
| 47 |
+
"version": "0.4.29",
|
| 48 |
+
"server_url": "http://127.0.0.1:8001",
|
| 49 |
+
"hostname": "<host>",
|
| 50 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 51 |
+
"env": {},
|
| 52 |
+
"args": {
|
| 53 |
+
"concurrency": "4",
|
| 54 |
+
"contexts": "8k",
|
| 55 |
+
"max_tokens": 8192,
|
| 56 |
+
"duration": 20.0,
|
| 57 |
+
"request_count": 0,
|
| 58 |
+
"run_burst": false,
|
| 59 |
+
"standalone_prefill": false,
|
| 60 |
+
"prefill_only": false,
|
| 61 |
+
"skip_prefill": true,
|
| 62 |
+
"prefill_contexts": "8k,64k,128k",
|
| 63 |
+
"prefill_metric": "client",
|
| 64 |
+
"dcp_size": 0,
|
| 65 |
+
"kv_budget": 0
|
| 66 |
+
},
|
| 67 |
+
"nvidia_p2p_override": {
|
| 68 |
+
"effective": true,
|
| 69 |
+
"configured": true,
|
| 70 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 71 |
+
"params_available": true,
|
| 72 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 73 |
+
"modprobe_available": true,
|
| 74 |
+
"runtime": {
|
| 75 |
+
"ForceP2P": "0x11",
|
| 76 |
+
"RMForceP2PType": "1",
|
| 77 |
+
"RMPcieP2PType": "2",
|
| 78 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 79 |
+
"EnableResizableBar": "1",
|
| 80 |
+
"DmaRemapPeerMmio": "1"
|
| 81 |
+
},
|
| 82 |
+
"expected": {
|
| 83 |
+
"ForceP2P": "0x11",
|
| 84 |
+
"RMForceP2PType": "1",
|
| 85 |
+
"RMPcieP2PType": "2",
|
| 86 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 87 |
+
"EnableResizableBar": "1"
|
| 88 |
+
},
|
| 89 |
+
"missing": [],
|
| 90 |
+
"mismatched": {},
|
| 91 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 92 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 93 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 94 |
+
},
|
| 95 |
+
"p2pmark": {
|
| 96 |
+
"status": "not_run"
|
| 97 |
+
},
|
| 98 |
+
"amd_fabric": {
|
| 99 |
+
"status": "not_run"
|
| 100 |
+
},
|
| 101 |
+
"nvidia_smi_query": {
|
| 102 |
+
"cmd": [
|
| 103 |
+
"nvidia-smi",
|
| 104 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 105 |
+
"--format=csv,noheader,nounits"
|
| 106 |
+
],
|
| 107 |
+
"returncode": 0,
|
| 108 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 109 |
+
"stderr": ""
|
| 110 |
+
},
|
| 111 |
+
"nvidia_smi_topo": {
|
| 112 |
+
"cmd": [
|
| 113 |
+
"nvidia-smi",
|
| 114 |
+
"topo",
|
| 115 |
+
"-m"
|
| 116 |
+
],
|
| 117 |
+
"returncode": 0,
|
| 118 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 119 |
+
"stderr": ""
|
| 120 |
+
}
|
| 121 |
+
},
|
| 122 |
+
"nvidia_p2p_override": {
|
| 123 |
+
"effective": true,
|
| 124 |
+
"configured": true,
|
| 125 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 126 |
+
"params_available": true,
|
| 127 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 128 |
+
"modprobe_available": true,
|
| 129 |
+
"runtime": {
|
| 130 |
+
"ForceP2P": "0x11",
|
| 131 |
+
"RMForceP2PType": "1",
|
| 132 |
+
"RMPcieP2PType": "2",
|
| 133 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 134 |
+
"EnableResizableBar": "1",
|
| 135 |
+
"DmaRemapPeerMmio": "1"
|
| 136 |
+
},
|
| 137 |
+
"expected": {
|
| 138 |
+
"ForceP2P": "0x11",
|
| 139 |
+
"RMForceP2PType": "1",
|
| 140 |
+
"RMPcieP2PType": "2",
|
| 141 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 142 |
+
"EnableResizableBar": "1"
|
| 143 |
+
},
|
| 144 |
+
"missing": [],
|
| 145 |
+
"mismatched": {},
|
| 146 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 147 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 148 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 149 |
+
},
|
| 150 |
+
"p2pmark": {
|
| 151 |
+
"status": "not_run"
|
| 152 |
+
},
|
| 153 |
+
"amd_fabric": {
|
| 154 |
+
"status": "not_run"
|
| 155 |
+
},
|
| 156 |
+
"hardware_run_summary": {
|
| 157 |
+
"samples": 21,
|
| 158 |
+
"duration_seconds": 48.105,
|
| 159 |
+
"gpu_count": 4,
|
| 160 |
+
"cpu_util_avg_pct": 9.28,
|
| 161 |
+
"cpu_temp_max_c": 76.0,
|
| 162 |
+
"gpu_util_avg_pct": 76.04,
|
| 163 |
+
"gpu_util_max_pct": 100.0,
|
| 164 |
+
"mem_util_avg_pct": 35.81,
|
| 165 |
+
"mem_util_max_pct": 59.0,
|
| 166 |
+
"temp_avg_c": 57.21,
|
| 167 |
+
"temp_max_c": 75.0,
|
| 168 |
+
"power_total_avg_w": 970.98,
|
| 169 |
+
"power_total_max_w": 1182.0,
|
| 170 |
+
"power_limit_total_w": 1200.0,
|
| 171 |
+
"vram_used_avg_mb": 386578.0,
|
| 172 |
+
"vram_used_max_mb": 386578.0,
|
| 173 |
+
"vram_total_mb": 391548.0,
|
| 174 |
+
"vram_used_avg_pct": 98.73,
|
| 175 |
+
"vram_used_max_pct": 98.73,
|
| 176 |
+
"pcie_rx_avg_mb_s": 9912.24,
|
| 177 |
+
"pcie_rx_max_mb_s": 43706.0,
|
| 178 |
+
"pcie_tx_avg_mb_s": 10658.48,
|
| 179 |
+
"pcie_tx_max_mb_s": 52315.0
|
| 180 |
+
},
|
| 181 |
+
"event_log": [
|
| 182 |
+
"10:10:54 benchmark start engine=vllm",
|
| 183 |
+
"10:10:54 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 184 |
+
"10:10:54 startup decode concurrency=4 contexts=8k",
|
| 185 |
+
"10:10:54 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 186 |
+
"10:10:54 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 187 |
+
"10:10:54 startup KV cache budget from vLLM metrics: 29,188,096 tokens (3563 blocks x 2048; local 7,297,024 \u00d7 CP 4; CP source: local process)",
|
| 188 |
+
"10:10:54 startup model context length: 1,000,000 tokens",
|
| 189 |
+
"10:10:54 startup prefill tests: skipped",
|
| 190 |
+
"10:10:54 startup calibrating padding text run=tmxrepeatabc up_to=8k",
|
| 191 |
+
"10:10:54 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
|
| 192 |
+
"10:10:54 startup token targeting: /tokenize exact",
|
| 193 |
+
"10:10:54 startup startup preparation done",
|
| 194 |
+
"10:10:54 hardware monitor interval=2s",
|
| 195 |
+
"10:10:54 decode warmup start",
|
| 196 |
+
"10:10:55 decode warmup start C=1 ctx=8k 3s",
|
| 197 |
+
"10:10:55 cell start C=1 ctx=8k",
|
| 198 |
+
"10:11:00 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 199 |
+
"10:11:03 cell done C=1 ctx=8k 219.4 tok/s",
|
| 200 |
+
"10:11:03 decode warmup done C=1 ctx=8k",
|
| 201 |
+
"10:11:05 cell start C=4 ctx=8k",
|
| 202 |
+
"10:11:22 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 203 |
+
"10:11:42 cell done C=4 ctx=8k 335.7 tok/s"
|
| 204 |
+
],
|
| 205 |
+
"prefill": {},
|
| 206 |
+
"results": [
|
| 207 |
+
{
|
| 208 |
+
"concurrency": 4,
|
| 209 |
+
"context_tokens": 8192,
|
| 210 |
+
"benchmark_mode": "duration",
|
| 211 |
+
"request_count_target": 0,
|
| 212 |
+
"warmup_request_count": 0,
|
| 213 |
+
"measurement_seconds": 19.991854,
|
| 214 |
+
"measurement_wall_seconds": 20.00195,
|
| 215 |
+
"client_output_tokens": 6712,
|
| 216 |
+
"server_output_tokens": 6712,
|
| 217 |
+
"aggregate_source": "openai_continuous_usage",
|
| 218 |
+
"aggregate_tps": 335.7367410580083,
|
| 219 |
+
"per_request_avg_tps": 83.93418526450208,
|
| 220 |
+
"ttft_avg": 4.347738385258708,
|
| 221 |
+
"ttft_p50": 1.9870852205203846,
|
| 222 |
+
"ttft_p90": 9.619254515157083,
|
| 223 |
+
"ttft_p99": 12.563032215915152,
|
| 224 |
+
"time_to_second_token_avg": 0.031753852672409266,
|
| 225 |
+
"time_to_second_token_p50": 0.03671444847714156,
|
| 226 |
+
"time_to_second_token_p90": 0.036814691172912715,
|
| 227 |
+
"time_to_second_token_p99": 0.036851499965414404,
|
| 228 |
+
"request_latency_avg": 0.0,
|
| 229 |
+
"request_latency_p50": 0.0,
|
| 230 |
+
"request_latency_p90": 0.0,
|
| 231 |
+
"request_latency_p99": 0.0,
|
| 232 |
+
"inter_token_latency_avg": 0.01416472446590232,
|
| 233 |
+
"inter_token_latency_p50": 0.014673835583566086,
|
| 234 |
+
"inter_token_latency_p90": 0.015258161326212282,
|
| 235 |
+
"inter_token_latency_p99": 0.015378115281519295,
|
| 236 |
+
"output_tps_per_user_avg": 71.30240784736792,
|
| 237 |
+
"output_tps_per_user_p50": 68.1721620239172,
|
| 238 |
+
"output_tps_per_user_p90": 79.55850802561741,
|
| 239 |
+
"output_tps_per_user_p99": 83.46057979574924,
|
| 240 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 241 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 242 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 243 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 244 |
+
"chunk_inter_token_latency_avg": 0.03829464632355565,
|
| 245 |
+
"chunk_inter_token_latency_p50": 0.04061762081634814,
|
| 246 |
+
"chunk_inter_token_latency_p90": 0.04061850702094191,
|
| 247 |
+
"chunk_inter_token_latency_p99": 0.04061852844490954,
|
| 248 |
+
"input_seq_len_avg": 8192.0,
|
| 249 |
+
"output_seq_len_avg": 2266.5,
|
| 250 |
+
"output_seq_len_p50": 2333.0,
|
| 251 |
+
"output_seq_len_p90": 2389.0,
|
| 252 |
+
"output_seq_len_p99": 2405.2,
|
| 253 |
+
"request_count": 4,
|
| 254 |
+
"completed_request_count": 0,
|
| 255 |
+
"request_samples": [
|
| 256 |
+
{
|
| 257 |
+
"ttft": 0.5266644728835672,
|
| 258 |
+
"time_to_second_token": 0.01673092390410602,
|
| 259 |
+
"latency": 0.0,
|
| 260 |
+
"inter_token_latency_avg": 0.01539144349877563,
|
| 261 |
+
"chunk_inter_token_latency_avg": 0.04061679015537416,
|
| 262 |
+
"input_tokens": 8192,
|
| 263 |
+
"output_tokens": 2347,
|
| 264 |
+
"output_tps_per_user": 64.97116401587341,
|
| 265 |
+
"e2e_output_tps_per_user": 0.0,
|
| 266 |
+
"completed": false
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"ttft": 1.9872382539324462,
|
| 270 |
+
"time_to_second_token": 0.036709635984152555,
|
| 271 |
+
"latency": 0.0,
|
| 272 |
+
"inter_token_latency_avg": 0.014947169590231138,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.040618451477322126,
|
| 274 |
+
"input_tokens": 8192,
|
| 275 |
+
"output_tokens": 2319,
|
| 276 |
+
"output_tps_per_user": 66.90229838922544,
|
| 277 |
+
"e2e_output_tps_per_user": 0.0,
|
| 278 |
+
"completed": false
|
| 279 |
+
},
|
| 280 |
+
{
|
| 281 |
+
"ttft": 1.986932187108323,
|
| 282 |
+
"time_to_second_token": 0.03671926097013056,
|
| 283 |
+
"latency": 0.0,
|
| 284 |
+
"inter_token_latency_avg": 0.014400501576901033,
|
| 285 |
+
"chunk_inter_token_latency_avg": 0.04061853082535039,
|
| 286 |
+
"input_tokens": 8192,
|
| 287 |
+
"output_tokens": 2407,
|
| 288 |
+
"output_tps_per_user": 69.44202565860894,
|
| 289 |
+
"e2e_output_tps_per_user": 0.0,
|
| 290 |
+
"completed": false
|
| 291 |
+
},
|
| 292 |
+
{
|
| 293 |
+
"ttft": 12.890118627110496,
|
| 294 |
+
"time_to_second_token": 0.036855589831247926,
|
| 295 |
+
"latency": 0.0,
|
| 296 |
+
"inter_token_latency_avg": 0.011919783197701478,
|
| 297 |
+
"chunk_inter_token_latency_avg": 0.031324812836175914,
|
| 298 |
+
"input_tokens": 8192,
|
| 299 |
+
"output_tokens": 1993,
|
| 300 |
+
"output_tps_per_user": 83.89414332576389,
|
| 301 |
+
"e2e_output_tps_per_user": 0.0,
|
| 302 |
+
"completed": false
|
| 303 |
+
}
|
| 304 |
+
],
|
| 305 |
+
"total_tokens": 6712,
|
| 306 |
+
"wall_time": 37.68367512291297,
|
| 307 |
+
"num_completed": 4,
|
| 308 |
+
"num_errors": 0,
|
| 309 |
+
"server_gen_throughput": 335.4828005774738,
|
| 310 |
+
"server_utilization": 0.02470522178551371,
|
| 311 |
+
"server_spec_accept_rate": 0.5364583333333334,
|
| 312 |
+
"server_spec_accept_length": 0.0,
|
| 313 |
+
"avg_running_reqs": 4,
|
| 314 |
+
"max_running_reqs": 4,
|
| 315 |
+
"effective_concurrency": 4,
|
| 316 |
+
"avg_queue_reqs": 0,
|
| 317 |
+
"max_queue_reqs": 0,
|
| 318 |
+
"queue_fraction": 0.0,
|
| 319 |
+
"underfilled": false,
|
| 320 |
+
"warmup_timed_out": false,
|
| 321 |
+
"warmup_duration": 17.659,
|
| 322 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 323 |
+
"timeout_reason": "",
|
| 324 |
+
"capacity_limited": false,
|
| 325 |
+
"hardware_summary": {
|
| 326 |
+
"samples": 8,
|
| 327 |
+
"duration_seconds": 16.884,
|
| 328 |
+
"gpu_count": 4,
|
| 329 |
+
"cpu_util_avg_pct": 10.94,
|
| 330 |
+
"cpu_temp_max_c": 75.75,
|
| 331 |
+
"gpu_util_avg_pct": 100.0,
|
| 332 |
+
"gpu_util_max_pct": 100.0,
|
| 333 |
+
"mem_util_avg_pct": 45.19,
|
| 334 |
+
"mem_util_max_pct": 48.0,
|
| 335 |
+
"temp_avg_c": 60.38,
|
| 336 |
+
"temp_max_c": 75.0,
|
| 337 |
+
"power_total_avg_w": 1179.07,
|
| 338 |
+
"power_total_max_w": 1180.51,
|
| 339 |
+
"power_limit_total_w": 1200.0,
|
| 340 |
+
"vram_used_avg_mb": 386578.0,
|
| 341 |
+
"vram_used_max_mb": 386578.0,
|
| 342 |
+
"vram_total_mb": 391548.0,
|
| 343 |
+
"vram_used_avg_pct": 98.73,
|
| 344 |
+
"vram_used_max_pct": 98.73,
|
| 345 |
+
"pcie_rx_avg_mb_s": 8712.12,
|
| 346 |
+
"pcie_rx_max_mb_s": 9059.0,
|
| 347 |
+
"pcie_tx_avg_mb_s": 8716.25,
|
| 348 |
+
"pcie_tx_max_mb_s": 9186.0
|
| 349 |
+
}
|
| 350 |
+
}
|
| 351 |
+
],
|
| 352 |
+
"summary_table": {
|
| 353 |
+
"8192": {
|
| 354 |
+
"4": 335.7367410580083
|
| 355 |
+
}
|
| 356 |
+
},
|
| 357 |
+
"burst_results": [],
|
| 358 |
+
"burst_summary_table": {},
|
| 359 |
+
"methodology": {
|
| 360 |
+
"prefill": {
|
| 361 |
+
"name": "Prefill",
|
| 362 |
+
"present": false,
|
| 363 |
+
"mode": "skipped",
|
| 364 |
+
"formula": "prompt_tokens / TTFT",
|
| 365 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 366 |
+
},
|
| 367 |
+
"sustained_decode": {
|
| 368 |
+
"name": "Sustained Decode",
|
| 369 |
+
"present": true,
|
| 370 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 371 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 372 |
+
},
|
| 373 |
+
"burst_e2e_decode": {
|
| 374 |
+
"name": "Burst / E2E Decode",
|
| 375 |
+
"present": false,
|
| 376 |
+
"status": "not run; use --run-burst",
|
| 377 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 378 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 379 |
+
}
|
| 380 |
+
}
|
| 381 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-32k/result.json
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T10:11:59.320922",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "standalone_cold",
|
| 15 |
+
"standalone_prefill": true,
|
| 16 |
+
"prefill_only": true,
|
| 17 |
+
"skip_prefill": false,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 0,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": 0.0,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29188096,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
4
|
| 38 |
+
],
|
| 39 |
+
"context_lengths": [
|
| 40 |
+
8192
|
| 41 |
+
],
|
| 42 |
+
"startup_diagnostics_available": true,
|
| 43 |
+
"nvidia_p2p_override_effective": true,
|
| 44 |
+
"p2pmark_status": "not_run",
|
| 45 |
+
"amd_fabric_status": "not_run"
|
| 46 |
+
},
|
| 47 |
+
"startup_diagnostics": {
|
| 48 |
+
"version": "0.4.29",
|
| 49 |
+
"server_url": "http://127.0.0.1:8001",
|
| 50 |
+
"hostname": "<host>",
|
| 51 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 52 |
+
"env": {},
|
| 53 |
+
"args": {
|
| 54 |
+
"concurrency": "1,4",
|
| 55 |
+
"contexts": "8k",
|
| 56 |
+
"max_tokens": 8192,
|
| 57 |
+
"duration": 20.0,
|
| 58 |
+
"request_count": 0,
|
| 59 |
+
"run_burst": false,
|
| 60 |
+
"standalone_prefill": true,
|
| 61 |
+
"prefill_only": true,
|
| 62 |
+
"skip_prefill": false,
|
| 63 |
+
"prefill_contexts": "32k",
|
| 64 |
+
"prefill_metric": "client",
|
| 65 |
+
"dcp_size": 0,
|
| 66 |
+
"kv_budget": 0
|
| 67 |
+
},
|
| 68 |
+
"nvidia_p2p_override": {
|
| 69 |
+
"effective": true,
|
| 70 |
+
"configured": true,
|
| 71 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 72 |
+
"params_available": true,
|
| 73 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 74 |
+
"modprobe_available": true,
|
| 75 |
+
"runtime": {
|
| 76 |
+
"ForceP2P": "0x11",
|
| 77 |
+
"RMForceP2PType": "1",
|
| 78 |
+
"RMPcieP2PType": "2",
|
| 79 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 80 |
+
"EnableResizableBar": "1",
|
| 81 |
+
"DmaRemapPeerMmio": "1"
|
| 82 |
+
},
|
| 83 |
+
"expected": {
|
| 84 |
+
"ForceP2P": "0x11",
|
| 85 |
+
"RMForceP2PType": "1",
|
| 86 |
+
"RMPcieP2PType": "2",
|
| 87 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 88 |
+
"EnableResizableBar": "1"
|
| 89 |
+
},
|
| 90 |
+
"missing": [],
|
| 91 |
+
"mismatched": {},
|
| 92 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 93 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 94 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 95 |
+
},
|
| 96 |
+
"p2pmark": {
|
| 97 |
+
"status": "not_run"
|
| 98 |
+
},
|
| 99 |
+
"amd_fabric": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"nvidia_smi_query": {
|
| 103 |
+
"cmd": [
|
| 104 |
+
"nvidia-smi",
|
| 105 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 106 |
+
"--format=csv,noheader,nounits"
|
| 107 |
+
],
|
| 108 |
+
"returncode": 0,
|
| 109 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 110 |
+
"stderr": ""
|
| 111 |
+
},
|
| 112 |
+
"nvidia_smi_topo": {
|
| 113 |
+
"cmd": [
|
| 114 |
+
"nvidia-smi",
|
| 115 |
+
"topo",
|
| 116 |
+
"-m"
|
| 117 |
+
],
|
| 118 |
+
"returncode": 0,
|
| 119 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 120 |
+
"stderr": ""
|
| 121 |
+
}
|
| 122 |
+
},
|
| 123 |
+
"nvidia_p2p_override": {
|
| 124 |
+
"effective": true,
|
| 125 |
+
"configured": true,
|
| 126 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 127 |
+
"params_available": true,
|
| 128 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 129 |
+
"modprobe_available": true,
|
| 130 |
+
"runtime": {
|
| 131 |
+
"ForceP2P": "0x11",
|
| 132 |
+
"RMForceP2PType": "1",
|
| 133 |
+
"RMPcieP2PType": "2",
|
| 134 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 135 |
+
"EnableResizableBar": "1",
|
| 136 |
+
"DmaRemapPeerMmio": "1"
|
| 137 |
+
},
|
| 138 |
+
"expected": {
|
| 139 |
+
"ForceP2P": "0x11",
|
| 140 |
+
"RMForceP2PType": "1",
|
| 141 |
+
"RMPcieP2PType": "2",
|
| 142 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 143 |
+
"EnableResizableBar": "1"
|
| 144 |
+
},
|
| 145 |
+
"missing": [],
|
| 146 |
+
"mismatched": {},
|
| 147 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 148 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 149 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 150 |
+
},
|
| 151 |
+
"p2pmark": {
|
| 152 |
+
"status": "not_run"
|
| 153 |
+
},
|
| 154 |
+
"amd_fabric": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"hardware_run_summary": {
|
| 158 |
+
"samples": 4,
|
| 159 |
+
"duration_seconds": 7.229,
|
| 160 |
+
"gpu_count": 4,
|
| 161 |
+
"cpu_util_avg_pct": 8.88,
|
| 162 |
+
"cpu_temp_max_c": 73.88,
|
| 163 |
+
"gpu_util_avg_pct": 72.19,
|
| 164 |
+
"gpu_util_max_pct": 100.0,
|
| 165 |
+
"mem_util_avg_pct": 19.81,
|
| 166 |
+
"mem_util_max_pct": 32.0,
|
| 167 |
+
"temp_avg_c": 59.81,
|
| 168 |
+
"temp_max_c": 74.0,
|
| 169 |
+
"power_total_avg_w": 936.62,
|
| 170 |
+
"power_total_max_w": 1134.55,
|
| 171 |
+
"power_limit_total_w": 1200.0,
|
| 172 |
+
"vram_used_avg_mb": 386578.0,
|
| 173 |
+
"vram_used_max_mb": 386578.0,
|
| 174 |
+
"vram_total_mb": 391548.0,
|
| 175 |
+
"vram_used_avg_pct": 98.73,
|
| 176 |
+
"vram_used_max_pct": 98.73,
|
| 177 |
+
"pcie_rx_avg_mb_s": 43008.75,
|
| 178 |
+
"pcie_rx_max_mb_s": 58512.0,
|
| 179 |
+
"pcie_tx_avg_mb_s": 40414.0,
|
| 180 |
+
"pcie_tx_max_mb_s": 46717.0
|
| 181 |
+
},
|
| 182 |
+
"event_log": [],
|
| 183 |
+
"prefill": {
|
| 184 |
+
"32768": {
|
| 185 |
+
"ttft_seconds": 4.13,
|
| 186 |
+
"prefill_seconds": 4.13,
|
| 187 |
+
"tok_per_sec": 7935.0,
|
| 188 |
+
"client_ttft_seconds": 4.13,
|
| 189 |
+
"client_tok_per_sec": 7935.0,
|
| 190 |
+
"prompt_tokens": 32770,
|
| 191 |
+
"samples": 1,
|
| 192 |
+
"method": "client",
|
| 193 |
+
"server_validation": {
|
| 194 |
+
"method": "",
|
| 195 |
+
"tok_per_sec": 0.0,
|
| 196 |
+
"prefill_seconds": 0.0,
|
| 197 |
+
"prompt_tokens": 0,
|
| 198 |
+
"request_prompt_tokens": 0,
|
| 199 |
+
"cached_tokens": 0,
|
| 200 |
+
"token_source": "",
|
| 201 |
+
"samples": 0,
|
| 202 |
+
"invalid_reason": ""
|
| 203 |
+
},
|
| 204 |
+
"hardware_summary": {
|
| 205 |
+
"samples": 2,
|
| 206 |
+
"duration_seconds": 2.407,
|
| 207 |
+
"gpu_count": 4,
|
| 208 |
+
"cpu_util_avg_pct": 12.25,
|
| 209 |
+
"cpu_temp_max_c": 73.88,
|
| 210 |
+
"gpu_util_avg_pct": 94.5,
|
| 211 |
+
"gpu_util_max_pct": 100.0,
|
| 212 |
+
"mem_util_avg_pct": 26.25,
|
| 213 |
+
"mem_util_max_pct": 32.0,
|
| 214 |
+
"temp_avg_c": 61.75,
|
| 215 |
+
"temp_max_c": 74.0,
|
| 216 |
+
"power_total_avg_w": 1127.2,
|
| 217 |
+
"power_total_max_w": 1133.91,
|
| 218 |
+
"power_limit_total_w": 1200.0,
|
| 219 |
+
"vram_used_avg_mb": 386578.0,
|
| 220 |
+
"vram_used_max_mb": 386578.0,
|
| 221 |
+
"vram_total_mb": 391548.0,
|
| 222 |
+
"vram_used_avg_pct": 98.73,
|
| 223 |
+
"vram_used_max_pct": 98.73,
|
| 224 |
+
"pcie_rx_avg_mb_s": 54350.5,
|
| 225 |
+
"pcie_rx_max_mb_s": 58512.0,
|
| 226 |
+
"pcie_tx_avg_mb_s": 45629.0,
|
| 227 |
+
"pcie_tx_max_mb_s": 46717.0
|
| 228 |
+
}
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"results": [],
|
| 232 |
+
"summary_table": {},
|
| 233 |
+
"burst_results": [],
|
| 234 |
+
"burst_summary_table": {},
|
| 235 |
+
"methodology": {
|
| 236 |
+
"prefill": {
|
| 237 |
+
"name": "Prefill",
|
| 238 |
+
"present": true,
|
| 239 |
+
"mode": "standalone_cold",
|
| 240 |
+
"formula": "prompt_tokens / TTFT",
|
| 241 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 242 |
+
},
|
| 243 |
+
"sustained_decode": {
|
| 244 |
+
"name": "Sustained Decode",
|
| 245 |
+
"present": false,
|
| 246 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 247 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 248 |
+
},
|
| 249 |
+
"burst_e2e_decode": {
|
| 250 |
+
"name": "Burst / E2E Decode",
|
| 251 |
+
"present": false,
|
| 252 |
+
"status": "not run; use --run-burst",
|
| 253 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 254 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 255 |
+
}
|
| 256 |
+
}
|
| 257 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/prefill-64k/result.json
ADDED
|
@@ -0,0 +1,257 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T10:12:28.058934",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "standalone_cold",
|
| 15 |
+
"standalone_prefill": true,
|
| 16 |
+
"prefill_only": true,
|
| 17 |
+
"skip_prefill": false,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 0,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": 0.0,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29188096,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
4
|
| 38 |
+
],
|
| 39 |
+
"context_lengths": [
|
| 40 |
+
8192
|
| 41 |
+
],
|
| 42 |
+
"startup_diagnostics_available": true,
|
| 43 |
+
"nvidia_p2p_override_effective": true,
|
| 44 |
+
"p2pmark_status": "not_run",
|
| 45 |
+
"amd_fabric_status": "not_run"
|
| 46 |
+
},
|
| 47 |
+
"startup_diagnostics": {
|
| 48 |
+
"version": "0.4.29",
|
| 49 |
+
"server_url": "http://127.0.0.1:8001",
|
| 50 |
+
"hostname": "<host>",
|
| 51 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 52 |
+
"env": {},
|
| 53 |
+
"args": {
|
| 54 |
+
"concurrency": "1,4",
|
| 55 |
+
"contexts": "8k",
|
| 56 |
+
"max_tokens": 8192,
|
| 57 |
+
"duration": 20.0,
|
| 58 |
+
"request_count": 0,
|
| 59 |
+
"run_burst": false,
|
| 60 |
+
"standalone_prefill": true,
|
| 61 |
+
"prefill_only": true,
|
| 62 |
+
"skip_prefill": false,
|
| 63 |
+
"prefill_contexts": "64k",
|
| 64 |
+
"prefill_metric": "client",
|
| 65 |
+
"dcp_size": 0,
|
| 66 |
+
"kv_budget": 0
|
| 67 |
+
},
|
| 68 |
+
"nvidia_p2p_override": {
|
| 69 |
+
"effective": true,
|
| 70 |
+
"configured": true,
|
| 71 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 72 |
+
"params_available": true,
|
| 73 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 74 |
+
"modprobe_available": true,
|
| 75 |
+
"runtime": {
|
| 76 |
+
"ForceP2P": "0x11",
|
| 77 |
+
"RMForceP2PType": "1",
|
| 78 |
+
"RMPcieP2PType": "2",
|
| 79 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 80 |
+
"EnableResizableBar": "1",
|
| 81 |
+
"DmaRemapPeerMmio": "1"
|
| 82 |
+
},
|
| 83 |
+
"expected": {
|
| 84 |
+
"ForceP2P": "0x11",
|
| 85 |
+
"RMForceP2PType": "1",
|
| 86 |
+
"RMPcieP2PType": "2",
|
| 87 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 88 |
+
"EnableResizableBar": "1"
|
| 89 |
+
},
|
| 90 |
+
"missing": [],
|
| 91 |
+
"mismatched": {},
|
| 92 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 93 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 94 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 95 |
+
},
|
| 96 |
+
"p2pmark": {
|
| 97 |
+
"status": "not_run"
|
| 98 |
+
},
|
| 99 |
+
"amd_fabric": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"nvidia_smi_query": {
|
| 103 |
+
"cmd": [
|
| 104 |
+
"nvidia-smi",
|
| 105 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 106 |
+
"--format=csv,noheader,nounits"
|
| 107 |
+
],
|
| 108 |
+
"returncode": 0,
|
| 109 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 110 |
+
"stderr": ""
|
| 111 |
+
},
|
| 112 |
+
"nvidia_smi_topo": {
|
| 113 |
+
"cmd": [
|
| 114 |
+
"nvidia-smi",
|
| 115 |
+
"topo",
|
| 116 |
+
"-m"
|
| 117 |
+
],
|
| 118 |
+
"returncode": 0,
|
| 119 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 120 |
+
"stderr": ""
|
| 121 |
+
}
|
| 122 |
+
},
|
| 123 |
+
"nvidia_p2p_override": {
|
| 124 |
+
"effective": true,
|
| 125 |
+
"configured": true,
|
| 126 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 127 |
+
"params_available": true,
|
| 128 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 129 |
+
"modprobe_available": true,
|
| 130 |
+
"runtime": {
|
| 131 |
+
"ForceP2P": "0x11",
|
| 132 |
+
"RMForceP2PType": "1",
|
| 133 |
+
"RMPcieP2PType": "2",
|
| 134 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 135 |
+
"EnableResizableBar": "1",
|
| 136 |
+
"DmaRemapPeerMmio": "1"
|
| 137 |
+
},
|
| 138 |
+
"expected": {
|
| 139 |
+
"ForceP2P": "0x11",
|
| 140 |
+
"RMForceP2PType": "1",
|
| 141 |
+
"RMPcieP2PType": "2",
|
| 142 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 143 |
+
"EnableResizableBar": "1"
|
| 144 |
+
},
|
| 145 |
+
"missing": [],
|
| 146 |
+
"mismatched": {},
|
| 147 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 148 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 149 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 150 |
+
},
|
| 151 |
+
"p2pmark": {
|
| 152 |
+
"status": "not_run"
|
| 153 |
+
},
|
| 154 |
+
"amd_fabric": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"hardware_run_summary": {
|
| 158 |
+
"samples": 6,
|
| 159 |
+
"duration_seconds": 12.079,
|
| 160 |
+
"gpu_count": 4,
|
| 161 |
+
"cpu_util_avg_pct": 9.23,
|
| 162 |
+
"cpu_temp_max_c": 74.12,
|
| 163 |
+
"gpu_util_avg_pct": 86.92,
|
| 164 |
+
"gpu_util_max_pct": 100.0,
|
| 165 |
+
"mem_util_avg_pct": 23.62,
|
| 166 |
+
"mem_util_max_pct": 37.0,
|
| 167 |
+
"temp_avg_c": 59.46,
|
| 168 |
+
"temp_max_c": 74.0,
|
| 169 |
+
"power_total_avg_w": 1002.06,
|
| 170 |
+
"power_total_max_w": 1132.26,
|
| 171 |
+
"power_limit_total_w": 1200.0,
|
| 172 |
+
"vram_used_avg_mb": 386844.67,
|
| 173 |
+
"vram_used_max_mb": 386978.0,
|
| 174 |
+
"vram_total_mb": 391548.0,
|
| 175 |
+
"vram_used_avg_pct": 98.8,
|
| 176 |
+
"vram_used_max_pct": 98.83,
|
| 177 |
+
"pcie_rx_avg_mb_s": 43749.5,
|
| 178 |
+
"pcie_rx_max_mb_s": 56738.0,
|
| 179 |
+
"pcie_tx_avg_mb_s": 46172.67,
|
| 180 |
+
"pcie_tx_max_mb_s": 58712.0
|
| 181 |
+
},
|
| 182 |
+
"event_log": [],
|
| 183 |
+
"prefill": {
|
| 184 |
+
"65536": {
|
| 185 |
+
"ttft_seconds": 8.202,
|
| 186 |
+
"prefill_seconds": 8.202,
|
| 187 |
+
"tok_per_sec": 7991.0,
|
| 188 |
+
"client_ttft_seconds": 8.202,
|
| 189 |
+
"client_tok_per_sec": 7991.0,
|
| 190 |
+
"prompt_tokens": 65538,
|
| 191 |
+
"samples": 1,
|
| 192 |
+
"method": "client",
|
| 193 |
+
"server_validation": {
|
| 194 |
+
"method": "",
|
| 195 |
+
"tok_per_sec": 0.0,
|
| 196 |
+
"prefill_seconds": 0.0,
|
| 197 |
+
"prompt_tokens": 0,
|
| 198 |
+
"request_prompt_tokens": 0,
|
| 199 |
+
"cached_tokens": 0,
|
| 200 |
+
"token_source": "",
|
| 201 |
+
"samples": 0,
|
| 202 |
+
"invalid_reason": ""
|
| 203 |
+
},
|
| 204 |
+
"hardware_summary": {
|
| 205 |
+
"samples": 4,
|
| 206 |
+
"duration_seconds": 7.26,
|
| 207 |
+
"gpu_count": 4,
|
| 208 |
+
"cpu_util_avg_pct": 11.1,
|
| 209 |
+
"cpu_temp_max_c": 74.12,
|
| 210 |
+
"gpu_util_avg_pct": 99.75,
|
| 211 |
+
"gpu_util_max_pct": 100.0,
|
| 212 |
+
"mem_util_avg_pct": 26.44,
|
| 213 |
+
"mem_util_max_pct": 29.0,
|
| 214 |
+
"temp_avg_c": 61.12,
|
| 215 |
+
"temp_max_c": 74.0,
|
| 216 |
+
"power_total_avg_w": 1131.9,
|
| 217 |
+
"power_total_max_w": 1132.26,
|
| 218 |
+
"power_limit_total_w": 1200.0,
|
| 219 |
+
"vram_used_avg_mb": 386928.0,
|
| 220 |
+
"vram_used_max_mb": 386978.0,
|
| 221 |
+
"vram_total_mb": 391548.0,
|
| 222 |
+
"vram_used_avg_pct": 98.82,
|
| 223 |
+
"vram_used_max_pct": 98.83,
|
| 224 |
+
"pcie_rx_avg_mb_s": 53121.5,
|
| 225 |
+
"pcie_rx_max_mb_s": 56738.0,
|
| 226 |
+
"pcie_tx_avg_mb_s": 52848.5,
|
| 227 |
+
"pcie_tx_max_mb_s": 57089.0
|
| 228 |
+
}
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"results": [],
|
| 232 |
+
"summary_table": {},
|
| 233 |
+
"burst_results": [],
|
| 234 |
+
"burst_summary_table": {},
|
| 235 |
+
"methodology": {
|
| 236 |
+
"prefill": {
|
| 237 |
+
"name": "Prefill",
|
| 238 |
+
"present": true,
|
| 239 |
+
"mode": "standalone_cold",
|
| 240 |
+
"formula": "prompt_tokens / TTFT",
|
| 241 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 242 |
+
},
|
| 243 |
+
"sustained_decode": {
|
| 244 |
+
"name": "Sustained Decode",
|
| 245 |
+
"present": false,
|
| 246 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 247 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 248 |
+
},
|
| 249 |
+
"burst_e2e_decode": {
|
| 250 |
+
"name": "Burst / E2E Decode",
|
| 251 |
+
"present": false,
|
| 252 |
+
"status": "not run; use --run-burst",
|
| 253 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 254 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 255 |
+
}
|
| 256 |
+
}
|
| 257 |
+
}
|
results/speed-20260909/evidence/candidate-graph-profile-02/results-01/result.json
ADDED
|
@@ -0,0 +1,7 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"status": "four-profile-cells-completed",
|
| 3 |
+
"graph_join": "pending",
|
| 4 |
+
"target_p8_dispatch": "pending Nsight classification",
|
| 5 |
+
"throughput_claim": false,
|
| 6 |
+
"production_restarted": false
|
| 7 |
+
}
|
results/speed-20260909/evidence/candidate-kernel-window-02/results-01/result.json
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"scope": "representative component equality",
|
| 3 |
+
"results": [
|
| 4 |
+
{
|
| 5 |
+
"rank": 0,
|
| 6 |
+
"status": "passed"
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
"rank": 1,
|
| 10 |
+
"status": "passed"
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"rank": 2,
|
| 14 |
+
"status": "passed"
|
| 15 |
+
},
|
| 16 |
+
{
|
| 17 |
+
"rank": 3,
|
| 18 |
+
"status": "passed"
|
| 19 |
+
}
|
| 20 |
+
],
|
| 21 |
+
"full_model": "NOT TESTED",
|
| 22 |
+
"throughput": "NOT MEASURED"
|
| 23 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-command.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"512",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0,8k,32k",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--skip-prefill",
|
| 25 |
+
"--cell-warmup-timeout-seconds",
|
| 26 |
+
"180"
|
| 27 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "e084ee41fd496dbfcf36f6f66715a1e622256de26889bd9d64e4c6c0f7335b4d"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.json
ADDED
|
@@ -0,0 +1,2366 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:26:52.004095",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 32768,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 512,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0,
|
| 42 |
+
8192,
|
| 43 |
+
32768
|
| 44 |
+
],
|
| 45 |
+
"startup_diagnostics_available": true,
|
| 46 |
+
"nvidia_p2p_override_effective": true,
|
| 47 |
+
"p2pmark_status": "not_run",
|
| 48 |
+
"amd_fabric_status": "not_run"
|
| 49 |
+
},
|
| 50 |
+
"startup_diagnostics": {
|
| 51 |
+
"version": "0.4.29",
|
| 52 |
+
"server_url": "http://127.0.0.1:8001",
|
| 53 |
+
"hostname": "<host>",
|
| 54 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 55 |
+
"env": {},
|
| 56 |
+
"args": {
|
| 57 |
+
"concurrency": "1,2,4",
|
| 58 |
+
"contexts": "0,8k,32k",
|
| 59 |
+
"max_tokens": 512,
|
| 60 |
+
"duration": 20.0,
|
| 61 |
+
"request_count": 0,
|
| 62 |
+
"run_burst": false,
|
| 63 |
+
"standalone_prefill": false,
|
| 64 |
+
"prefill_only": false,
|
| 65 |
+
"skip_prefill": true,
|
| 66 |
+
"prefill_contexts": "8k,64k,128k",
|
| 67 |
+
"prefill_metric": "client",
|
| 68 |
+
"dcp_size": 0,
|
| 69 |
+
"kv_budget": 0
|
| 70 |
+
},
|
| 71 |
+
"nvidia_p2p_override": {
|
| 72 |
+
"effective": true,
|
| 73 |
+
"configured": true,
|
| 74 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 75 |
+
"params_available": true,
|
| 76 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 77 |
+
"modprobe_available": true,
|
| 78 |
+
"runtime": {
|
| 79 |
+
"ForceP2P": "0x11",
|
| 80 |
+
"RMForceP2PType": "1",
|
| 81 |
+
"RMPcieP2PType": "2",
|
| 82 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 83 |
+
"EnableResizableBar": "1",
|
| 84 |
+
"DmaRemapPeerMmio": "1"
|
| 85 |
+
},
|
| 86 |
+
"expected": {
|
| 87 |
+
"ForceP2P": "0x11",
|
| 88 |
+
"RMForceP2PType": "1",
|
| 89 |
+
"RMPcieP2PType": "2",
|
| 90 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 91 |
+
"EnableResizableBar": "1"
|
| 92 |
+
},
|
| 93 |
+
"missing": [],
|
| 94 |
+
"mismatched": {},
|
| 95 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 96 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 97 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 98 |
+
},
|
| 99 |
+
"p2pmark": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"amd_fabric": {
|
| 103 |
+
"status": "not_run"
|
| 104 |
+
},
|
| 105 |
+
"nvidia_smi_query": {
|
| 106 |
+
"cmd": [
|
| 107 |
+
"nvidia-smi",
|
| 108 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 109 |
+
"--format=csv,noheader,nounits"
|
| 110 |
+
],
|
| 111 |
+
"returncode": 0,
|
| 112 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 113 |
+
"stderr": ""
|
| 114 |
+
},
|
| 115 |
+
"nvidia_smi_topo": {
|
| 116 |
+
"cmd": [
|
| 117 |
+
"nvidia-smi",
|
| 118 |
+
"topo",
|
| 119 |
+
"-m"
|
| 120 |
+
],
|
| 121 |
+
"returncode": 0,
|
| 122 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 123 |
+
"stderr": ""
|
| 124 |
+
}
|
| 125 |
+
},
|
| 126 |
+
"nvidia_p2p_override": {
|
| 127 |
+
"effective": true,
|
| 128 |
+
"configured": true,
|
| 129 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 130 |
+
"params_available": true,
|
| 131 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 132 |
+
"modprobe_available": true,
|
| 133 |
+
"runtime": {
|
| 134 |
+
"ForceP2P": "0x11",
|
| 135 |
+
"RMForceP2PType": "1",
|
| 136 |
+
"RMPcieP2PType": "2",
|
| 137 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 138 |
+
"EnableResizableBar": "1",
|
| 139 |
+
"DmaRemapPeerMmio": "1"
|
| 140 |
+
},
|
| 141 |
+
"expected": {
|
| 142 |
+
"ForceP2P": "0x11",
|
| 143 |
+
"RMForceP2PType": "1",
|
| 144 |
+
"RMPcieP2PType": "2",
|
| 145 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 146 |
+
"EnableResizableBar": "1"
|
| 147 |
+
},
|
| 148 |
+
"missing": [],
|
| 149 |
+
"mismatched": {},
|
| 150 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 151 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 152 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 153 |
+
},
|
| 154 |
+
"p2pmark": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"amd_fabric": {
|
| 158 |
+
"status": "not_run"
|
| 159 |
+
},
|
| 160 |
+
"hardware_run_summary": {
|
| 161 |
+
"samples": 113,
|
| 162 |
+
"duration_seconds": 270.016,
|
| 163 |
+
"gpu_count": 4,
|
| 164 |
+
"cpu_util_avg_pct": 10.99,
|
| 165 |
+
"cpu_temp_max_c": 77.12,
|
| 166 |
+
"gpu_util_avg_pct": 91.62,
|
| 167 |
+
"gpu_util_max_pct": 100.0,
|
| 168 |
+
"mem_util_avg_pct": 33.39,
|
| 169 |
+
"mem_util_max_pct": 56.0,
|
| 170 |
+
"temp_avg_c": 67.84,
|
| 171 |
+
"temp_max_c": 84.0,
|
| 172 |
+
"power_total_avg_w": 1098.07,
|
| 173 |
+
"power_total_max_w": 1177.71,
|
| 174 |
+
"power_limit_total_w": 1200.0,
|
| 175 |
+
"vram_used_avg_mb": 384778.0,
|
| 176 |
+
"vram_used_max_mb": 384778.0,
|
| 177 |
+
"vram_total_mb": 391548.0,
|
| 178 |
+
"vram_used_avg_pct": 98.27,
|
| 179 |
+
"vram_used_max_pct": 98.27,
|
| 180 |
+
"pcie_rx_avg_mb_s": 16074.05,
|
| 181 |
+
"pcie_rx_max_mb_s": 74971.0,
|
| 182 |
+
"pcie_tx_avg_mb_s": 16658.8,
|
| 183 |
+
"pcie_tx_max_mb_s": 78099.0
|
| 184 |
+
},
|
| 185 |
+
"event_log": [
|
| 186 |
+
"02:22:19 benchmark start engine=vllm",
|
| 187 |
+
"02:22:19 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 188 |
+
"02:22:19 startup decode concurrency=1,2,4 contexts=0,8k,32k",
|
| 189 |
+
"02:22:19 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 190 |
+
"02:22:19 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 191 |
+
"02:22:19 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
|
| 192 |
+
"02:22:19 startup model context length: 1,000,000 tokens",
|
| 193 |
+
"02:22:19 startup prefill tests: skipped",
|
| 194 |
+
"02:22:19 startup calibrating padding text run=ftbfeppyynkm up_to=32k",
|
| 195 |
+
"02:22:19 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
|
| 196 |
+
"02:22:19 startup context 32k: 205,152 chars (32,768 prompt tokens via /tokenize)",
|
| 197 |
+
"02:22:19 startup token targeting: /tokenize exact",
|
| 198 |
+
"02:22:19 startup startup preparation done",
|
| 199 |
+
"02:22:19 hardware monitor interval=2s",
|
| 200 |
+
"02:22:19 decode warmup start",
|
| 201 |
+
"02:22:19 decode warmup start C=1 ctx=32k 3s",
|
| 202 |
+
"02:22:19 cell start C=1 ctx=32k",
|
| 203 |
+
"02:22:27 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 204 |
+
"02:22:30 cell done C=1 ctx=32k 172.8 tok/s",
|
| 205 |
+
"02:22:30 decode warmup done C=1 ctx=32k",
|
| 206 |
+
"02:22:32 cell start C=1 ctx=0",
|
| 207 |
+
"02:22:38 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 208 |
+
"02:22:58 cell done C=1 ctx=0 181.4 tok/s",
|
| 209 |
+
"02:23:00 cell start C=1 ctx=8k",
|
| 210 |
+
"02:23:06 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 211 |
+
"02:23:26 cell done C=1 ctx=8k 161.5 tok/s",
|
| 212 |
+
"02:23:28 cell start C=1 ctx=32k",
|
| 213 |
+
"02:23:37 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 214 |
+
"02:23:57 cell done C=1 ctx=32k 162.9 tok/s",
|
| 215 |
+
"02:23:59 cell start C=2 ctx=0",
|
| 216 |
+
"02:24:05 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 217 |
+
"02:24:25 cell done C=2 ctx=0 257.3 tok/s",
|
| 218 |
+
"02:24:27 cell start C=4 ctx=0",
|
| 219 |
+
"02:24:32 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 220 |
+
"02:24:52 cell done C=4 ctx=0 309.0 tok/s",
|
| 221 |
+
"02:24:54 cell start C=2 ctx=8k",
|
| 222 |
+
"02:25:00 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 223 |
+
"02:25:20 cell done C=2 ctx=8k 179.5 tok/s",
|
| 224 |
+
"02:25:22 cell start C=4 ctx=8k",
|
| 225 |
+
"02:25:31 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 226 |
+
"02:25:51 cell done C=4 ctx=8k 217.9 tok/s",
|
| 227 |
+
"02:25:53 cell start C=2 ctx=32k",
|
| 228 |
+
"02:25:58 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 229 |
+
"02:26:18 cell done C=2 ctx=32k 187.5 tok/s",
|
| 230 |
+
"02:26:20 cell start C=4 ctx=32k",
|
| 231 |
+
"02:26:29 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 232 |
+
"02:26:49 cell done C=4 ctx=32k 217.2 tok/s"
|
| 233 |
+
],
|
| 234 |
+
"prefill": {},
|
| 235 |
+
"results": [
|
| 236 |
+
{
|
| 237 |
+
"concurrency": 1,
|
| 238 |
+
"context_tokens": 0,
|
| 239 |
+
"benchmark_mode": "duration",
|
| 240 |
+
"request_count_target": 0,
|
| 241 |
+
"warmup_request_count": 0,
|
| 242 |
+
"measurement_seconds": 19.997167,
|
| 243 |
+
"measurement_wall_seconds": 20.00031,
|
| 244 |
+
"client_output_tokens": 3627,
|
| 245 |
+
"server_output_tokens": 3627,
|
| 246 |
+
"aggregate_source": "openai_continuous_usage",
|
| 247 |
+
"aggregate_tps": 181.37569616773123,
|
| 248 |
+
"per_request_avg_tps": 181.37569616773123,
|
| 249 |
+
"ttft_avg": 0.08000486893579364,
|
| 250 |
+
"ttft_p50": 0.08374336501583457,
|
| 251 |
+
"ttft_p90": 0.08476541170384735,
|
| 252 |
+
"ttft_p99": 0.08516110226279125,
|
| 253 |
+
"time_to_second_token_avg": 0.013620631862431764,
|
| 254 |
+
"time_to_second_token_p50": 0.013637091615237296,
|
| 255 |
+
"time_to_second_token_p90": 0.01392088970169425,
|
| 256 |
+
"time_to_second_token_p99": 0.014014782523736358,
|
| 257 |
+
"request_latency_avg": 2.806835822191917,
|
| 258 |
+
"request_latency_p50": 2.8223699629306793,
|
| 259 |
+
"request_latency_p90": 2.8927299182862045,
|
| 260 |
+
"request_latency_p99": 2.9908720326423643,
|
| 261 |
+
"inter_token_latency_avg": 0.0052722589844875455,
|
| 262 |
+
"inter_token_latency_p50": 0.00534923418599077,
|
| 263 |
+
"inter_token_latency_p90": 0.005470371673862338,
|
| 264 |
+
"inter_token_latency_p99": 0.005685154040414228,
|
| 265 |
+
"output_tps_per_user_avg": 190.17533606228403,
|
| 266 |
+
"output_tps_per_user_p50": 186.94466223296394,
|
| 267 |
+
"output_tps_per_user_p90": 203.41232058199725,
|
| 268 |
+
"output_tps_per_user_p99": 211.44483943260568,
|
| 269 |
+
"e2e_output_tps_per_user_avg": 182.67236866069015,
|
| 270 |
+
"e2e_output_tps_per_user_p50": 181.4078263036614,
|
| 271 |
+
"e2e_output_tps_per_user_p90": 189.76977705704618,
|
| 272 |
+
"e2e_output_tps_per_user_p99": 196.55013284253087,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.014190152055605577,
|
| 274 |
+
"chunk_inter_token_latency_p50": 0.01418408028440155,
|
| 275 |
+
"chunk_inter_token_latency_p90": 0.014293591843861054,
|
| 276 |
+
"chunk_inter_token_latency_p99": 0.01437531164027474,
|
| 277 |
+
"input_seq_len_avg": 78.0,
|
| 278 |
+
"output_seq_len_avg": 512.0,
|
| 279 |
+
"output_seq_len_p50": 512.0,
|
| 280 |
+
"output_seq_len_p90": 512.0,
|
| 281 |
+
"output_seq_len_p99": 512.0,
|
| 282 |
+
"request_count": 10,
|
| 283 |
+
"completed_request_count": 9,
|
| 284 |
+
"request_samples": [
|
| 285 |
+
{
|
| 286 |
+
"ttft": 0.07054085680283606,
|
| 287 |
+
"time_to_second_token": 0.013252795208245516,
|
| 288 |
+
"latency": 2.5949868359602988,
|
| 289 |
+
"inter_token_latency_avg": 0.004940207395611473,
|
| 290 |
+
"chunk_inter_token_latency_avg": 0.014024699884208127,
|
| 291 |
+
"input_tokens": 78,
|
| 292 |
+
"output_tokens": 512,
|
| 293 |
+
"output_tps_per_user": 202.42065158809498,
|
| 294 |
+
"e2e_output_tps_per_user": 197.30350570758472,
|
| 295 |
+
"completed": true
|
| 296 |
+
},
|
| 297 |
+
{
|
| 298 |
+
"ttft": 0.08365814504213631,
|
| 299 |
+
"time_to_second_token": 0.013909297995269299,
|
| 300 |
+
"latency": 2.8654682198539376,
|
| 301 |
+
"inter_token_latency_avg": 0.0054438553323127225,
|
| 302 |
+
"chunk_inter_token_latency_avg": 0.014120863323917774,
|
| 303 |
+
"input_tokens": 78,
|
| 304 |
+
"output_tokens": 512,
|
| 305 |
+
"output_tps_per_user": 183.69334579197354,
|
| 306 |
+
"e2e_output_tps_per_user": 178.67935035974622,
|
| 307 |
+
"completed": true
|
| 308 |
+
},
|
| 309 |
+
{
|
| 310 |
+
"ttft": 0.08446813188493252,
|
| 311 |
+
"time_to_second_token": 0.013657593168318272,
|
| 312 |
+
"latency": 3.001776712015271,
|
| 313 |
+
"inter_token_latency_avg": 0.005709018747808882,
|
| 314 |
+
"chunk_inter_token_latency_avg": 0.014230773561611409,
|
| 315 |
+
"input_tokens": 78,
|
| 316 |
+
"output_tokens": 512,
|
| 317 |
+
"output_tps_per_user": 175.16144965959333,
|
| 318 |
+
"e2e_output_tps_per_user": 170.56565131930282,
|
| 319 |
+
"completed": true
|
| 320 |
+
},
|
| 321 |
+
{
|
| 322 |
+
"ttft": 0.07495116395875812,
|
| 323 |
+
"time_to_second_token": 0.01324723707512021,
|
| 324 |
+
"latency": 2.8223699629306793,
|
| 325 |
+
"inter_token_latency_avg": 0.005376553422645638,
|
| 326 |
+
"chunk_inter_token_latency_avg": 0.014384391617654037,
|
| 327 |
+
"input_tokens": 78,
|
| 328 |
+
"output_tokens": 512,
|
| 329 |
+
"output_tps_per_user": 185.99275807212763,
|
| 330 |
+
"e2e_output_tps_per_user": 181.4078263036614,
|
| 331 |
+
"completed": true
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"ttft": 0.08520506788045168,
|
| 335 |
+
"time_to_second_token": 0.013890709029510617,
|
| 336 |
+
"latency": 2.7519469358958304,
|
| 337 |
+
"inter_token_latency_avg": 0.0052186729315369445,
|
| 338 |
+
"chunk_inter_token_latency_avg": 0.014184797170294567,
|
| 339 |
+
"input_tokens": 78,
|
| 340 |
+
"output_tokens": 512,
|
| 341 |
+
"output_tps_per_user": 191.61959623047144,
|
| 342 |
+
"e2e_output_tps_per_user": 186.05009904863252,
|
| 343 |
+
"completed": true
|
| 344 |
+
},
|
| 345 |
+
{
|
| 346 |
+
"ttft": 0.08382858498953283,
|
| 347 |
+
"time_to_second_token": 0.014025215059518814,
|
| 348 |
+
"latency": 2.826261157169938,
|
| 349 |
+
"inter_token_latency_avg": 0.005366795640274766,
|
| 350 |
+
"chunk_inter_token_latency_avg": 0.014283502980106277,
|
| 351 |
+
"input_tokens": 78,
|
| 352 |
+
"output_tokens": 512,
|
| 353 |
+
"output_tps_per_user": 186.3309257568083,
|
| 354 |
+
"e2e_output_tps_per_user": 181.158064144606,
|
| 355 |
+
"completed": true
|
| 356 |
+
},
|
| 357 |
+
{
|
| 358 |
+
"ttft": 0.07493811403401196,
|
| 359 |
+
"time_to_second_token": 0.013330142945051193,
|
| 360 |
+
"latency": 2.725051681045443,
|
| 361 |
+
"inter_token_latency_avg": 0.005186132225071294,
|
| 362 |
+
"chunk_inter_token_latency_avg": 0.0140963487606991,
|
| 363 |
+
"input_tokens": 78,
|
| 364 |
+
"output_tokens": 512,
|
| 365 |
+
"output_tps_per_user": 192.8219252038552,
|
| 366 |
+
"e2e_output_tps_per_user": 187.88634489441154,
|
| 367 |
+
"completed": true
|
| 368 |
+
},
|
| 369 |
+
{
|
| 370 |
+
"ttft": 0.08471656101755798,
|
| 371 |
+
"time_to_second_token": 0.013534185010939837,
|
| 372 |
+
"latency": 2.8092013269197196,
|
| 373 |
+
"inter_token_latency_avg": 0.005331672731706774,
|
| 374 |
+
"chunk_inter_token_latency_avg": 0.014264318146084616,
|
| 375 |
+
"input_tokens": 78,
|
| 376 |
+
"output_tokens": 512,
|
| 377 |
+
"output_tps_per_user": 187.5583987091196,
|
| 378 |
+
"e2e_output_tps_per_user": 182.25820808699618,
|
| 379 |
+
"completed": true
|
| 380 |
+
},
|
| 381 |
+
{
|
| 382 |
+
"ttft": 0.08452034182846546,
|
| 383 |
+
"time_to_second_token": 0.01361659006215632,
|
| 384 |
+
"latency": 2.8644595679361373,
|
| 385 |
+
"inter_token_latency_avg": 0.005440194180249847,
|
| 386 |
+
"chunk_inter_token_latency_avg": 0.01418336339850853,
|
| 387 |
+
"input_tokens": 78,
|
| 388 |
+
"output_tokens": 512,
|
| 389 |
+
"output_tps_per_user": 183.81696808367857,
|
| 390 |
+
"e2e_output_tps_per_user": 178.74226808127003,
|
| 391 |
+
"completed": true
|
| 392 |
+
},
|
| 393 |
+
{
|
| 394 |
+
"ttft": 0.07322172191925347,
|
| 395 |
+
"time_to_second_token": 0.013742553070187569,
|
| 396 |
+
"latency": 0.0,
|
| 397 |
+
"inter_token_latency_avg": 0.00470948723765711,
|
| 398 |
+
"chunk_inter_token_latency_avg": 0.01412846171297133,
|
| 399 |
+
"input_tokens": 78,
|
| 400 |
+
"output_tokens": 43,
|
| 401 |
+
"output_tps_per_user": 212.33734152711773,
|
| 402 |
+
"e2e_output_tps_per_user": 0.0,
|
| 403 |
+
"completed": false
|
| 404 |
+
}
|
| 405 |
+
],
|
| 406 |
+
"total_tokens": 3627,
|
| 407 |
+
"wall_time": 25.549715572968125,
|
| 408 |
+
"num_completed": 1,
|
| 409 |
+
"num_errors": 0,
|
| 410 |
+
"server_gen_throughput": 181.29852055793228,
|
| 411 |
+
"server_utilization": 0.005862646566164198,
|
| 412 |
+
"server_spec_accept_rate": 0.5046296296296297,
|
| 413 |
+
"server_spec_accept_length": 0.0,
|
| 414 |
+
"avg_running_reqs": 1,
|
| 415 |
+
"max_running_reqs": 1,
|
| 416 |
+
"effective_concurrency": 1,
|
| 417 |
+
"avg_queue_reqs": 0,
|
| 418 |
+
"max_queue_reqs": 0,
|
| 419 |
+
"queue_fraction": 0.0,
|
| 420 |
+
"underfilled": false,
|
| 421 |
+
"warmup_timed_out": false,
|
| 422 |
+
"warmup_duration": 5.538,
|
| 423 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 424 |
+
"timeout_reason": "",
|
| 425 |
+
"capacity_limited": false,
|
| 426 |
+
"hardware_summary": {
|
| 427 |
+
"samples": 8,
|
| 428 |
+
"duration_seconds": 16.928,
|
| 429 |
+
"gpu_count": 4,
|
| 430 |
+
"cpu_util_avg_pct": 11.61,
|
| 431 |
+
"cpu_temp_max_c": 76.0,
|
| 432 |
+
"gpu_util_avg_pct": 99.0,
|
| 433 |
+
"gpu_util_max_pct": 99.0,
|
| 434 |
+
"mem_util_avg_pct": 44.03,
|
| 435 |
+
"mem_util_max_pct": 56.0,
|
| 436 |
+
"temp_avg_c": 67.31,
|
| 437 |
+
"temp_max_c": 82.0,
|
| 438 |
+
"power_total_avg_w": 1151.08,
|
| 439 |
+
"power_total_max_w": 1152.57,
|
| 440 |
+
"power_limit_total_w": 1200.0,
|
| 441 |
+
"vram_used_avg_mb": 384778.0,
|
| 442 |
+
"vram_used_max_mb": 384778.0,
|
| 443 |
+
"vram_total_mb": 391548.0,
|
| 444 |
+
"vram_used_avg_pct": 98.27,
|
| 445 |
+
"vram_used_max_pct": 98.27,
|
| 446 |
+
"pcie_rx_avg_mb_s": 8368.88,
|
| 447 |
+
"pcie_rx_max_mb_s": 8659.0,
|
| 448 |
+
"pcie_tx_avg_mb_s": 8253.75,
|
| 449 |
+
"pcie_tx_max_mb_s": 8495.0
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"concurrency": 1,
|
| 454 |
+
"context_tokens": 8192,
|
| 455 |
+
"benchmark_mode": "duration",
|
| 456 |
+
"request_count_target": 0,
|
| 457 |
+
"warmup_request_count": 0,
|
| 458 |
+
"measurement_seconds": 19.997908,
|
| 459 |
+
"measurement_wall_seconds": 20.001023,
|
| 460 |
+
"client_output_tokens": 3229,
|
| 461 |
+
"server_output_tokens": 3229,
|
| 462 |
+
"aggregate_source": "openai_continuous_usage",
|
| 463 |
+
"aggregate_tps": 161.46688548018994,
|
| 464 |
+
"per_request_avg_tps": 161.46688548018994,
|
| 465 |
+
"ttft_avg": 0.6216339806560427,
|
| 466 |
+
"ttft_p50": 0.6253726300783455,
|
| 467 |
+
"ttft_p90": 0.6310847522690892,
|
| 468 |
+
"ttft_p99": 0.631135935941711,
|
| 469 |
+
"time_to_second_token_avg": 0.017570865049492568,
|
| 470 |
+
"time_to_second_token_p50": 0.01762134348973632,
|
| 471 |
+
"time_to_second_token_p90": 0.018190144654363395,
|
| 472 |
+
"time_to_second_token_p99": 0.0183273093868047,
|
| 473 |
+
"request_latency_avg": 3.1911156099023565,
|
| 474 |
+
"request_latency_p50": 3.198385320138186,
|
| 475 |
+
"request_latency_p90": 3.315028171055019,
|
| 476 |
+
"request_latency_p99": 3.317255131038837,
|
| 477 |
+
"inter_token_latency_avg": 0.00510038525208451,
|
| 478 |
+
"inter_token_latency_p50": 0.0050694173082997274,
|
| 479 |
+
"inter_token_latency_p90": 0.005415639657450936,
|
| 480 |
+
"inter_token_latency_p99": 0.005569194878906556,
|
| 481 |
+
"output_tps_per_user_avg": 196.5708670855281,
|
| 482 |
+
"output_tps_per_user_p50": 197.2764163282376,
|
| 483 |
+
"output_tps_per_user_p90": 207.7094766010595,
|
| 484 |
+
"output_tps_per_user_p99": 208.07124550938897,
|
| 485 |
+
"e2e_output_tps_per_user_avg": 160.58636690583256,
|
| 486 |
+
"e2e_output_tps_per_user_p50": 160.0807747510169,
|
| 487 |
+
"e2e_output_tps_per_user_p90": 165.8642163873898,
|
| 488 |
+
"e2e_output_tps_per_user_p99": 166.33389003015049,
|
| 489 |
+
"chunk_inter_token_latency_avg": 0.014235832994823009,
|
| 490 |
+
"chunk_inter_token_latency_p50": 0.014211497073928412,
|
| 491 |
+
"chunk_inter_token_latency_p90": 0.014303687557771127,
|
| 492 |
+
"chunk_inter_token_latency_p99": 0.014314021112869283,
|
| 493 |
+
"input_seq_len_avg": 8192.0,
|
| 494 |
+
"output_seq_len_avg": 512.0,
|
| 495 |
+
"output_seq_len_p50": 512.0,
|
| 496 |
+
"output_seq_len_p90": 512.0,
|
| 497 |
+
"output_seq_len_p99": 512.0,
|
| 498 |
+
"request_count": 8,
|
| 499 |
+
"completed_request_count": 7,
|
| 500 |
+
"request_samples": [
|
| 501 |
+
{
|
| 502 |
+
"ttft": 0.5874758099671453,
|
| 503 |
+
"time_to_second_token": 0.01834254991263151,
|
| 504 |
+
"latency": 3.317502571037039,
|
| 505 |
+
"inter_token_latency_avg": 0.00534251812342445,
|
| 506 |
+
"chunk_inter_token_latency_avg": 0.014218889380572364,
|
| 507 |
+
"input_tokens": 8192,
|
| 508 |
+
"output_tokens": 512,
|
| 509 |
+
"output_tps_per_user": 187.17765235375927,
|
| 510 |
+
"e2e_output_tps_per_user": 154.33296253330428,
|
| 511 |
+
"completed": true
|
| 512 |
+
},
|
| 513 |
+
{
|
| 514 |
+
"ttft": 0.6288027700502425,
|
| 515 |
+
"time_to_second_token": 0.017373620066791773,
|
| 516 |
+
"latency": 3.3133785710670054,
|
| 517 |
+
"inter_token_latency_avg": 0.005253572996118909,
|
| 518 |
+
"chunk_inter_token_latency_avg": 0.01420410476728446,
|
| 519 |
+
"input_tokens": 8192,
|
| 520 |
+
"output_tokens": 512,
|
| 521 |
+
"output_tps_per_user": 190.34664612802612,
|
| 522 |
+
"e2e_output_tps_per_user": 154.52505321030097,
|
| 523 |
+
"completed": true
|
| 524 |
+
},
|
| 525 |
+
{
|
| 526 |
+
"ttft": 0.6219424901064485,
|
| 527 |
+
"time_to_second_token": 0.017552112927660346,
|
| 528 |
+
"latency": 3.1045689159072936,
|
| 529 |
+
"inter_token_latency_avg": 0.004858368739336292,
|
| 530 |
+
"chunk_inter_token_latency_avg": 0.014267967964372673,
|
| 531 |
+
"input_tokens": 8192,
|
| 532 |
+
"output_tokens": 512,
|
| 533 |
+
"output_tps_per_user": 205.8304039179643,
|
| 534 |
+
"e2e_output_tps_per_user": 164.91822661001254,
|
| 535 |
+
"completed": true
|
| 536 |
+
},
|
| 537 |
+
{
|
| 538 |
+
"ttft": 0.6217653600033373,
|
| 539 |
+
"time_to_second_token": 0.01812482811510563,
|
| 540 |
+
"latency": 3.077180569060147,
|
| 541 |
+
"inter_token_latency_avg": 0.004805117825942876,
|
| 542 |
+
"chunk_inter_token_latency_avg": 0.014193151497438206,
|
| 543 |
+
"input_tokens": 8192,
|
| 544 |
+
"output_tokens": 512,
|
| 545 |
+
"output_tps_per_user": 208.11144205475892,
|
| 546 |
+
"e2e_output_tps_per_user": 166.38607599045721,
|
| 547 |
+
"completed": true
|
| 548 |
+
},
|
| 549 |
+
{
|
| 550 |
+
"ttft": 0.6305665450636297,
|
| 551 |
+
"time_to_second_token": 0.016708664130419493,
|
| 552 |
+
"latency": 3.198385320138186,
|
| 553 |
+
"inter_token_latency_avg": 0.005025085665507938,
|
| 554 |
+
"chunk_inter_token_latency_avg": 0.014186844061185394,
|
| 555 |
+
"input_tokens": 8192,
|
| 556 |
+
"output_tokens": 512,
|
| 557 |
+
"output_tps_per_user": 199.00158257280566,
|
| 558 |
+
"e2e_output_tps_per_user": 160.0807747510169,
|
| 559 |
+
"completed": true
|
| 560 |
+
},
|
| 561 |
+
{
|
| 562 |
+
"ttft": 0.6203168679494411,
|
| 563 |
+
"time_to_second_token": 0.01790391909889877,
|
| 564 |
+
"latency": 3.233442581957206,
|
| 565 |
+
"inter_token_latency_avg": 0.005113748951091517,
|
| 566 |
+
"chunk_inter_token_latency_avg": 0.01420177018482481,
|
| 567 |
+
"input_tokens": 8192,
|
| 568 |
+
"output_tokens": 512,
|
| 569 |
+
"output_tps_per_user": 195.55125008366954,
|
| 570 |
+
"e2e_output_tps_per_user": 158.34516526039127,
|
| 571 |
+
"completed": true
|
| 572 |
+
},
|
| 573 |
+
{
|
| 574 |
+
"ttft": 0.6311416230164468,
|
| 575 |
+
"time_to_second_token": 0.01687065209262073,
|
| 576 |
+
"latency": 3.093350740149617,
|
| 577 |
+
"inter_token_latency_avg": 0.004818413145074698,
|
| 578 |
+
"chunk_inter_token_latency_avg": 0.014315169285657967,
|
| 579 |
+
"input_tokens": 8192,
|
| 580 |
+
"output_tokens": 512,
|
| 581 |
+
"output_tps_per_user": 207.53720569233118,
|
| 582 |
+
"e2e_output_tps_per_user": 165.51630998534486,
|
| 583 |
+
"completed": true
|
| 584 |
+
},
|
| 585 |
+
{
|
| 586 |
+
"ttft": 0.6310603790916502,
|
| 587 |
+
"time_to_second_token": 0.01769057405181229,
|
| 588 |
+
"latency": 0.0,
|
| 589 |
+
"inter_token_latency_avg": 0.005586256570179402,
|
| 590 |
+
"chunk_inter_token_latency_avg": 0.014298766817248195,
|
| 591 |
+
"input_tokens": 8192,
|
| 592 |
+
"output_tokens": 280,
|
| 593 |
+
"output_tps_per_user": 179.01075388090973,
|
| 594 |
+
"e2e_output_tps_per_user": 0.0,
|
| 595 |
+
"completed": false
|
| 596 |
+
}
|
| 597 |
+
],
|
| 598 |
+
"total_tokens": 3229,
|
| 599 |
+
"wall_time": 26.068293957971036,
|
| 600 |
+
"num_completed": 1,
|
| 601 |
+
"num_errors": 0,
|
| 602 |
+
"server_gen_throughput": 161.40326963780146,
|
| 603 |
+
"server_utilization": 0.006141820212172022,
|
| 604 |
+
"server_spec_accept_rate": 0.5918367346938775,
|
| 605 |
+
"server_spec_accept_length": 0.0,
|
| 606 |
+
"avg_running_reqs": 1,
|
| 607 |
+
"max_running_reqs": 1,
|
| 608 |
+
"effective_concurrency": 1,
|
| 609 |
+
"avg_queue_reqs": 0,
|
| 610 |
+
"max_queue_reqs": 0,
|
| 611 |
+
"queue_fraction": 0.0,
|
| 612 |
+
"underfilled": false,
|
| 613 |
+
"warmup_timed_out": false,
|
| 614 |
+
"warmup_duration": 6.056,
|
| 615 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 616 |
+
"timeout_reason": "",
|
| 617 |
+
"capacity_limited": false,
|
| 618 |
+
"hardware_summary": {
|
| 619 |
+
"samples": 8,
|
| 620 |
+
"duration_seconds": 16.899,
|
| 621 |
+
"gpu_count": 4,
|
| 622 |
+
"cpu_util_avg_pct": 11.6,
|
| 623 |
+
"cpu_temp_max_c": 75.88,
|
| 624 |
+
"gpu_util_avg_pct": 99.12,
|
| 625 |
+
"gpu_util_max_pct": 100.0,
|
| 626 |
+
"mem_util_avg_pct": 40.22,
|
| 627 |
+
"mem_util_max_pct": 55.0,
|
| 628 |
+
"temp_avg_c": 67.81,
|
| 629 |
+
"temp_max_c": 83.0,
|
| 630 |
+
"power_total_avg_w": 1153.3,
|
| 631 |
+
"power_total_max_w": 1156.82,
|
| 632 |
+
"power_limit_total_w": 1200.0,
|
| 633 |
+
"vram_used_avg_mb": 384778.0,
|
| 634 |
+
"vram_used_max_mb": 384778.0,
|
| 635 |
+
"vram_total_mb": 391548.0,
|
| 636 |
+
"vram_used_avg_pct": 98.27,
|
| 637 |
+
"vram_used_max_pct": 98.27,
|
| 638 |
+
"pcie_rx_avg_mb_s": 12913.25,
|
| 639 |
+
"pcie_rx_max_mb_s": 39230.0,
|
| 640 |
+
"pcie_tx_avg_mb_s": 11280.75,
|
| 641 |
+
"pcie_tx_max_mb_s": 30681.0
|
| 642 |
+
}
|
| 643 |
+
},
|
| 644 |
+
{
|
| 645 |
+
"concurrency": 1,
|
| 646 |
+
"context_tokens": 32768,
|
| 647 |
+
"benchmark_mode": "duration",
|
| 648 |
+
"request_count_target": 0,
|
| 649 |
+
"warmup_request_count": 0,
|
| 650 |
+
"measurement_seconds": 19.99527,
|
| 651 |
+
"measurement_wall_seconds": 20.000365,
|
| 652 |
+
"client_output_tokens": 3258,
|
| 653 |
+
"server_output_tokens": 3258,
|
| 654 |
+
"aggregate_source": "openai_continuous_usage",
|
| 655 |
+
"aggregate_tps": 162.9385363574482,
|
| 656 |
+
"per_request_avg_tps": 162.9385363574482,
|
| 657 |
+
"ttft_avg": 0.6264620197180193,
|
| 658 |
+
"ttft_p50": 0.6282767069060355,
|
| 659 |
+
"ttft_p90": 0.6339533890830353,
|
| 660 |
+
"ttft_p99": 0.6355810678633861,
|
| 661 |
+
"time_to_second_token_avg": 0.011372056760592386,
|
| 662 |
+
"time_to_second_token_p50": 0.011673877947032452,
|
| 663 |
+
"time_to_second_token_p90": 0.013470914796926081,
|
| 664 |
+
"time_to_second_token_p99": 0.013671491749119014,
|
| 665 |
+
"request_latency_avg": 3.179016282981528,
|
| 666 |
+
"request_latency_p50": 3.1919217659160495,
|
| 667 |
+
"request_latency_p90": 3.2463170818053184,
|
| 668 |
+
"request_latency_p99": 3.250220402767882,
|
| 669 |
+
"inter_token_latency_avg": 0.004995384074006483,
|
| 670 |
+
"inter_token_latency_p50": 0.005010807180026302,
|
| 671 |
+
"inter_token_latency_p90": 0.005126464669228168,
|
| 672 |
+
"inter_token_latency_p99": 0.00513858911541697,
|
| 673 |
+
"output_tps_per_user_avg": 200.31161173007573,
|
| 674 |
+
"output_tps_per_user_p50": 199.57010444464703,
|
| 675 |
+
"output_tps_per_user_p90": 206.65625468620667,
|
| 676 |
+
"output_tps_per_user_p99": 210.60956431602483,
|
| 677 |
+
"e2e_output_tps_per_user_avg": 161.12331942575418,
|
| 678 |
+
"e2e_output_tps_per_user_p50": 160.40493393893104,
|
| 679 |
+
"e2e_output_tps_per_user_p90": 165.31501161627682,
|
| 680 |
+
"e2e_output_tps_per_user_p99": 167.56189180026823,
|
| 681 |
+
"chunk_inter_token_latency_avg": 0.014215011037926982,
|
| 682 |
+
"chunk_inter_token_latency_p50": 0.014209193651253612,
|
| 683 |
+
"chunk_inter_token_latency_p90": 0.014249588821529074,
|
| 684 |
+
"chunk_inter_token_latency_p99": 0.014264281110023109,
|
| 685 |
+
"input_seq_len_avg": 32768.0,
|
| 686 |
+
"output_seq_len_avg": 512.0,
|
| 687 |
+
"output_seq_len_p50": 512.0,
|
| 688 |
+
"output_seq_len_p90": 512.0,
|
| 689 |
+
"output_seq_len_p99": 512.0,
|
| 690 |
+
"request_count": 8,
|
| 691 |
+
"completed_request_count": 7,
|
| 692 |
+
"request_samples": [
|
| 693 |
+
{
|
| 694 |
+
"ttft": 0.6073537350166589,
|
| 695 |
+
"time_to_second_token": 0.01270562899298966,
|
| 696 |
+
"latency": 3.2036298899911344,
|
| 697 |
+
"inter_token_latency_avg": 0.005080775254353181,
|
| 698 |
+
"chunk_inter_token_latency_avg": 0.014187301393303145,
|
| 699 |
+
"input_tokens": 32768,
|
| 700 |
+
"output_tokens": 512,
|
| 701 |
+
"output_tps_per_user": 196.82035711837585,
|
| 702 |
+
"e2e_output_tps_per_user": 159.81871114375727,
|
| 703 |
+
"completed": true
|
| 704 |
+
},
|
| 705 |
+
{
|
| 706 |
+
"ttft": 0.624475359916687,
|
| 707 |
+
"time_to_second_token": 0.013375401962548494,
|
| 708 |
+
"latency": 3.1919217659160495,
|
| 709 |
+
"inter_token_latency_avg": 0.005024356958902862,
|
| 710 |
+
"chunk_inter_token_latency_avg": 0.014184786773477141,
|
| 711 |
+
"input_tokens": 32768,
|
| 712 |
+
"output_tokens": 512,
|
| 713 |
+
"output_tps_per_user": 199.03044472747092,
|
| 714 |
+
"e2e_output_tps_per_user": 160.40493393893104,
|
| 715 |
+
"completed": true
|
| 716 |
+
},
|
| 717 |
+
{
|
| 718 |
+
"ttft": 0.6302267559804022,
|
| 719 |
+
"time_to_second_token": 0.01065197098068893,
|
| 720 |
+
"latency": 3.1838252879679203,
|
| 721 |
+
"inter_token_latency_avg": 0.004997257401149742,
|
| 722 |
+
"chunk_inter_token_latency_avg": 0.014265913586522447,
|
| 723 |
+
"input_tokens": 32768,
|
| 724 |
+
"output_tokens": 512,
|
| 725 |
+
"output_tps_per_user": 200.10976416182314,
|
| 726 |
+
"e2e_output_tps_per_user": 160.81284420188285,
|
| 727 |
+
"completed": true
|
| 728 |
+
},
|
| 729 |
+
{
|
| 730 |
+
"ttft": 0.6267525688745081,
|
| 731 |
+
"time_to_second_token": 0.010681336978450418,
|
| 732 |
+
"latency": 3.2434257329441607,
|
| 733 |
+
"inter_token_latency_avg": 0.005120691123423978,
|
| 734 |
+
"chunk_inter_token_latency_avg": 0.014221049804726372,
|
| 735 |
+
"input_tokens": 32768,
|
| 736 |
+
"output_tokens": 512,
|
| 737 |
+
"output_tps_per_user": 195.28613929194475,
|
| 738 |
+
"e2e_output_tps_per_user": 157.8577843788769,
|
| 739 |
+
"completed": true
|
| 740 |
+
},
|
| 741 |
+
{
|
| 742 |
+
"ttft": 0.6241466680075973,
|
| 743 |
+
"time_to_second_token": 0.012666418915614486,
|
| 744 |
+
"latency": 3.2506541050970554,
|
| 745 |
+
"inter_token_latency_avg": 0.005139936276104614,
|
| 746 |
+
"chunk_inter_token_latency_avg": 0.014197337497780854,
|
| 747 |
+
"input_tokens": 32768,
|
| 748 |
+
"output_tokens": 512,
|
| 749 |
+
"output_tps_per_user": 194.5549412059767,
|
| 750 |
+
"e2e_output_tps_per_user": 157.50676123835487,
|
| 751 |
+
"completed": true
|
| 752 |
+
},
|
| 753 |
+
{
|
| 754 |
+
"ttft": 0.6298008449375629,
|
| 755 |
+
"time_to_second_token": 0.01369377807714045,
|
| 756 |
+
"latency": 3.0510415688622743,
|
| 757 |
+
"inter_token_latency_avg": 0.004738240164236226,
|
| 758 |
+
"chunk_inter_token_latency_avg": 0.014242592493674773,
|
| 759 |
+
"input_tokens": 32768,
|
| 760 |
+
"output_tokens": 512,
|
| 761 |
+
"output_tps_per_user": 211.0488209415602,
|
| 762 |
+
"e2e_output_tps_per_user": 167.81154515404506,
|
| 763 |
+
"completed": true
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"ttft": 0.6331783039495349,
|
| 767 |
+
"time_to_second_token": 0.00986177520826459,
|
| 768 |
+
"latency": 3.1286156300920993,
|
| 769 |
+
"inter_token_latency_avg": 0.00488343899440815,
|
| 770 |
+
"chunk_inter_token_latency_avg": 0.01417862117126457,
|
| 771 |
+
"input_tokens": 32768,
|
| 772 |
+
"output_tokens": 512,
|
| 773 |
+
"output_tps_per_user": 204.77372629105514,
|
| 774 |
+
"e2e_output_tps_per_user": 163.6506559244313,
|
| 775 |
+
"completed": true
|
| 776 |
+
},
|
| 777 |
+
{
|
| 778 |
+
"ttft": 0.6357619210612029,
|
| 779 |
+
"time_to_second_token": 0.007340142969042063,
|
| 780 |
+
"latency": 0.0,
|
| 781 |
+
"inter_token_latency_avg": 0.004978376419473111,
|
| 782 |
+
"chunk_inter_token_latency_avg": 0.014242485582666553,
|
| 783 |
+
"input_tokens": 32768,
|
| 784 |
+
"output_tokens": 330,
|
| 785 |
+
"output_tps_per_user": 200.86870010239915,
|
| 786 |
+
"e2e_output_tps_per_user": 0.0,
|
| 787 |
+
"completed": false
|
| 788 |
+
}
|
| 789 |
+
],
|
| 790 |
+
"total_tokens": 3258,
|
| 791 |
+
"wall_time": 29.116754403105006,
|
| 792 |
+
"num_completed": 1,
|
| 793 |
+
"num_errors": 0,
|
| 794 |
+
"server_gen_throughput": 162.8569571510368,
|
| 795 |
+
"server_utilization": 0.006979341150195384,
|
| 796 |
+
"server_spec_accept_rate": 0.6607142857142857,
|
| 797 |
+
"server_spec_accept_length": 0.0,
|
| 798 |
+
"avg_running_reqs": 0.9,
|
| 799 |
+
"max_running_reqs": 1,
|
| 800 |
+
"effective_concurrency": 0.9,
|
| 801 |
+
"avg_queue_reqs": 0,
|
| 802 |
+
"max_queue_reqs": 0,
|
| 803 |
+
"queue_fraction": 0.0,
|
| 804 |
+
"underfilled": false,
|
| 805 |
+
"warmup_timed_out": false,
|
| 806 |
+
"warmup_duration": 9.107,
|
| 807 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 808 |
+
"timeout_reason": "",
|
| 809 |
+
"capacity_limited": false,
|
| 810 |
+
"hardware_summary": {
|
| 811 |
+
"samples": 8,
|
| 812 |
+
"duration_seconds": 16.912,
|
| 813 |
+
"gpu_count": 4,
|
| 814 |
+
"cpu_util_avg_pct": 11.53,
|
| 815 |
+
"cpu_temp_max_c": 76.5,
|
| 816 |
+
"gpu_util_avg_pct": 97.97,
|
| 817 |
+
"gpu_util_max_pct": 100.0,
|
| 818 |
+
"mem_util_avg_pct": 38.0,
|
| 819 |
+
"mem_util_max_pct": 56.0,
|
| 820 |
+
"temp_avg_c": 68.16,
|
| 821 |
+
"temp_max_c": 84.0,
|
| 822 |
+
"power_total_avg_w": 1149.9,
|
| 823 |
+
"power_total_max_w": 1155.94,
|
| 824 |
+
"power_limit_total_w": 1200.0,
|
| 825 |
+
"vram_used_avg_mb": 384778.0,
|
| 826 |
+
"vram_used_max_mb": 384778.0,
|
| 827 |
+
"vram_total_mb": 391548.0,
|
| 828 |
+
"vram_used_avg_pct": 98.27,
|
| 829 |
+
"vram_used_max_pct": 98.27,
|
| 830 |
+
"pcie_rx_avg_mb_s": 8327.0,
|
| 831 |
+
"pcie_rx_max_mb_s": 8635.0,
|
| 832 |
+
"pcie_tx_avg_mb_s": 8132.62,
|
| 833 |
+
"pcie_tx_max_mb_s": 8389.0
|
| 834 |
+
}
|
| 835 |
+
},
|
| 836 |
+
{
|
| 837 |
+
"concurrency": 2,
|
| 838 |
+
"context_tokens": 0,
|
| 839 |
+
"benchmark_mode": "duration",
|
| 840 |
+
"request_count_target": 0,
|
| 841 |
+
"warmup_request_count": 0,
|
| 842 |
+
"measurement_seconds": 19.997106,
|
| 843 |
+
"measurement_wall_seconds": 20.0002,
|
| 844 |
+
"client_output_tokens": 5145,
|
| 845 |
+
"server_output_tokens": 5145,
|
| 846 |
+
"aggregate_source": "openai_continuous_usage",
|
| 847 |
+
"aggregate_tps": 257.2872334123046,
|
| 848 |
+
"per_request_avg_tps": 128.6436167061523,
|
| 849 |
+
"ttft_avg": 0.12196867128035851,
|
| 850 |
+
"ttft_p50": 0.1225746686104685,
|
| 851 |
+
"ttft_p90": 0.14186890744604172,
|
| 852 |
+
"ttft_p99": 0.16953610270284114,
|
| 853 |
+
"time_to_second_token_avg": 0.01940517939094986,
|
| 854 |
+
"time_to_second_token_p50": 0.02044149092398584,
|
| 855 |
+
"time_to_second_token_p90": 0.020944337057881058,
|
| 856 |
+
"time_to_second_token_p99": 0.021031919752713294,
|
| 857 |
+
"request_latency_avg": 3.9966156358908242,
|
| 858 |
+
"request_latency_p50": 3.9488427881151438,
|
| 859 |
+
"request_latency_p90": 4.166194258979521,
|
| 860 |
+
"request_latency_p99": 4.237951860819012,
|
| 861 |
+
"inter_token_latency_avg": 0.007504969642701137,
|
| 862 |
+
"inter_token_latency_p50": 0.00753824443737712,
|
| 863 |
+
"inter_token_latency_p90": 0.007882004104182853,
|
| 864 |
+
"inter_token_latency_p99": 0.008046831693049199,
|
| 865 |
+
"output_tps_per_user_avg": 133.69662528358077,
|
| 866 |
+
"output_tps_per_user_p50": 132.65955785556645,
|
| 867 |
+
"output_tps_per_user_p90": 139.45345815466882,
|
| 868 |
+
"output_tps_per_user_p99": 155.8549841566213,
|
| 869 |
+
"e2e_output_tps_per_user_avg": 128.2813132485478,
|
| 870 |
+
"e2e_output_tps_per_user_p50": 129.6588243765429,
|
| 871 |
+
"e2e_output_tps_per_user_p90": 134.02921814320612,
|
| 872 |
+
"e2e_output_tps_per_user_p99": 135.85115936191062,
|
| 873 |
+
"chunk_inter_token_latency_avg": 0.02096237495683361,
|
| 874 |
+
"chunk_inter_token_latency_p50": 0.02092384556948202,
|
| 875 |
+
"chunk_inter_token_latency_p90": 0.021176504729596514,
|
| 876 |
+
"chunk_inter_token_latency_p99": 0.021264344922454187,
|
| 877 |
+
"input_seq_len_avg": 78.0,
|
| 878 |
+
"output_seq_len_avg": 512.0,
|
| 879 |
+
"output_seq_len_p50": 512.0,
|
| 880 |
+
"output_seq_len_p90": 512.0,
|
| 881 |
+
"output_seq_len_p99": 512.0,
|
| 882 |
+
"request_count": 14,
|
| 883 |
+
"completed_request_count": 12,
|
| 884 |
+
"request_samples": [
|
| 885 |
+
{
|
| 886 |
+
"ttft": 0.07110429811291397,
|
| 887 |
+
"time_to_second_token": 0.013481322908774018,
|
| 888 |
+
"latency": 3.940448695095256,
|
| 889 |
+
"inter_token_latency_avg": 0.007572102538125914,
|
| 890 |
+
"chunk_inter_token_latency_avg": 0.020915375118823472,
|
| 891 |
+
"input_tokens": 78,
|
| 892 |
+
"output_tokens": 512,
|
| 893 |
+
"output_tps_per_user": 132.06371611648814,
|
| 894 |
+
"e2e_output_tps_per_user": 129.93444138412337,
|
| 895 |
+
"completed": true
|
| 896 |
+
},
|
| 897 |
+
{
|
| 898 |
+
"ttft": 0.17250841204077005,
|
| 899 |
+
"time_to_second_token": 0.020715914899483323,
|
| 900 |
+
"latency": 4.167202410986647,
|
| 901 |
+
"inter_token_latency_avg": 0.007817405085999759,
|
| 902 |
+
"chunk_inter_token_latency_avg": 0.021135947084369718,
|
| 903 |
+
"input_tokens": 78,
|
| 904 |
+
"output_tokens": 512,
|
| 905 |
+
"output_tps_per_user": 127.91968549652191,
|
| 906 |
+
"e2e_output_tps_per_user": 122.86420228835883,
|
| 907 |
+
"completed": true
|
| 908 |
+
},
|
| 909 |
+
{
|
| 910 |
+
"ttft": 0.12372587202116847,
|
| 911 |
+
"time_to_second_token": 0.020451861899346113,
|
| 912 |
+
"latency": 4.157120890915394,
|
| 913 |
+
"inter_token_latency_avg": 0.007893140937170695,
|
| 914 |
+
"chunk_inter_token_latency_avg": 0.02089841978701671,
|
| 915 |
+
"input_tokens": 78,
|
| 916 |
+
"output_tokens": 512,
|
| 917 |
+
"output_tps_per_user": 126.69227725185547,
|
| 918 |
+
"e2e_output_tps_per_user": 123.16216281294098,
|
| 919 |
+
"completed": true
|
| 920 |
+
},
|
| 921 |
+
{
|
| 922 |
+
"ttft": 0.11786476895213127,
|
| 923 |
+
"time_to_second_token": 0.02033530385233462,
|
| 924 |
+
"latency": 3.869324626866728,
|
| 925 |
+
"inter_token_latency_avg": 0.007341408723903321,
|
| 926 |
+
"chunk_inter_token_latency_avg": 0.020841443655081095,
|
| 927 |
+
"input_tokens": 78,
|
| 928 |
+
"output_tokens": 512,
|
| 929 |
+
"output_tps_per_user": 136.21363931748436,
|
| 930 |
+
"e2e_output_tps_per_user": 132.32283392427672,
|
| 931 |
+
"completed": true
|
| 932 |
+
},
|
| 933 |
+
{
|
| 934 |
+
"ttft": 0.12338922801427543,
|
| 935 |
+
"time_to_second_token": 0.020431119948625565,
|
| 936 |
+
"latency": 4.137814508052543,
|
| 937 |
+
"inter_token_latency_avg": 0.007856018160544554,
|
| 938 |
+
"chunk_inter_token_latency_avg": 0.02090846500019931,
|
| 939 |
+
"input_tokens": 78,
|
| 940 |
+
"output_tokens": 512,
|
| 941 |
+
"output_tps_per_user": 127.2909481068057,
|
| 942 |
+
"e2e_output_tps_per_user": 123.73681783066978,
|
| 943 |
+
"completed": true
|
| 944 |
+
},
|
| 945 |
+
{
|
| 946 |
+
"ttft": 0.12249546311795712,
|
| 947 |
+
"time_to_second_token": 0.02092183893546462,
|
| 948 |
+
"latency": 3.9572368811350316,
|
| 949 |
+
"inter_token_latency_avg": 0.007504386336628326,
|
| 950 |
+
"chunk_inter_token_latency_avg": 0.02107000779130261,
|
| 951 |
+
"input_tokens": 78,
|
| 952 |
+
"output_tokens": 512,
|
| 953 |
+
"output_tps_per_user": 133.25539959464479,
|
| 954 |
+
"e2e_output_tps_per_user": 129.38320736896245,
|
| 955 |
+
"completed": true
|
| 956 |
+
},
|
| 957 |
+
{
|
| 958 |
+
"ttft": 0.1226538741029799,
|
| 959 |
+
"time_to_second_token": 0.020244625862687826,
|
| 960 |
+
"latency": 0.0,
|
| 961 |
+
"inter_token_latency_avg": 0.0063211789832794095,
|
| 962 |
+
"chunk_inter_token_latency_avg": 0.020737902980232446,
|
| 963 |
+
"input_tokens": 78,
|
| 964 |
+
"output_tokens": 188,
|
| 965 |
+
"output_tps_per_user": 158.198336520002,
|
| 966 |
+
"e2e_output_tps_per_user": 0.0,
|
| 967 |
+
"completed": false
|
| 968 |
+
},
|
| 969 |
+
{
|
| 970 |
+
"ttft": 0.14964449405670166,
|
| 971 |
+
"time_to_second_token": 0.017545362003147602,
|
| 972 |
+
"latency": 3.9196609100326896,
|
| 973 |
+
"inter_token_latency_avg": 0.007377722927545964,
|
| 974 |
+
"chunk_inter_token_latency_avg": 0.020714375911955976,
|
| 975 |
+
"input_tokens": 78,
|
| 976 |
+
"output_tokens": 512,
|
| 977 |
+
"output_tps_per_user": 135.54317637306931,
|
| 978 |
+
"e2e_output_tps_per_user": 130.62354416666363,
|
| 979 |
+
"completed": true
|
| 980 |
+
},
|
| 981 |
+
{
|
| 982 |
+
"ttft": 0.10573618090711534,
|
| 983 |
+
"time_to_second_token": 0.013638033997267485,
|
| 984 |
+
"latency": 3.814666331978515,
|
| 985 |
+
"inter_token_latency_avg": 0.0072581803347776894,
|
| 986 |
+
"chunk_inter_token_latency_avg": 0.021193886577550853,
|
| 987 |
+
"input_tokens": 78,
|
| 988 |
+
"output_tokens": 512,
|
| 989 |
+
"output_tps_per_user": 137.77557926033936,
|
| 990 |
+
"e2e_output_tps_per_user": 134.21881638975384,
|
| 991 |
+
"completed": true
|
| 992 |
+
},
|
| 993 |
+
{
|
| 994 |
+
"ttft": 0.12302991887554526,
|
| 995 |
+
"time_to_second_token": 0.020953979110345244,
|
| 996 |
+
"latency": 4.246696174843237,
|
| 997 |
+
"inter_token_latency_avg": 0.008069796978410355,
|
| 998 |
+
"chunk_inter_token_latency_avg": 0.020932316020140566,
|
| 999 |
+
"input_tokens": 78,
|
| 1000 |
+
"output_tokens": 512,
|
| 1001 |
+
"output_tps_per_user": 123.91885479589686,
|
| 1002 |
+
"e2e_output_tps_per_user": 120.56431138940616,
|
| 1003 |
+
"completed": true
|
| 1004 |
+
},
|
| 1005 |
+
{
|
| 1006 |
+
"ttft": 0.11646403884515166,
|
| 1007 |
+
"time_to_second_token": 0.021043566055595875,
|
| 1008 |
+
"latency": 3.930458223912865,
|
| 1009 |
+
"inter_token_latency_avg": 0.00746378509797987,
|
| 1010 |
+
"chunk_inter_token_latency_avg": 0.020841498279058544,
|
| 1011 |
+
"input_tokens": 78,
|
| 1012 |
+
"output_tokens": 512,
|
| 1013 |
+
"output_tps_per_user": 133.98027768386012,
|
| 1014 |
+
"e2e_output_tps_per_user": 130.26470982059993,
|
| 1015 |
+
"completed": true
|
| 1016 |
+
},
|
| 1017 |
+
{
|
| 1018 |
+
"ttft": 0.11775453994050622,
|
| 1019 |
+
"time_to_second_token": 0.020746542140841484,
|
| 1020 |
+
"latency": 4.055516151012853,
|
| 1021 |
+
"inter_token_latency_avg": 0.007705991411100482,
|
| 1022 |
+
"chunk_inter_token_latency_avg": 0.021057548722312015,
|
| 1023 |
+
"input_tokens": 78,
|
| 1024 |
+
"output_tokens": 512,
|
| 1025 |
+
"output_tps_per_user": 129.7691557973319,
|
| 1026 |
+
"e2e_output_tps_per_user": 126.2478019899217,
|
| 1027 |
+
"completed": true
|
| 1028 |
+
},
|
| 1029 |
+
{
|
| 1030 |
+
"ttft": 0.11773488996550441,
|
| 1031 |
+
"time_to_second_token": 0.020856699906289577,
|
| 1032 |
+
"latency": 3.763241825858131,
|
| 1033 |
+
"inter_token_latency_avg": 0.007134064453801618,
|
| 1034 |
+
"chunk_inter_token_latency_avg": 0.020951189286739235,
|
| 1035 |
+
"input_tokens": 78,
|
| 1036 |
+
"output_tokens": 512,
|
| 1037 |
+
"output_tps_per_user": 140.17254910938146,
|
| 1038 |
+
"e2e_output_tps_per_user": 136.05290961689627,
|
| 1039 |
+
"completed": true
|
| 1040 |
+
},
|
| 1041 |
+
{
|
| 1042 |
+
"ttft": 0.1234554189722985,
|
| 1043 |
+
"time_to_second_token": 0.02030633995309472,
|
| 1044 |
+
"latency": 0.0,
|
| 1045 |
+
"inter_token_latency_avg": 0.00775439302854797,
|
| 1046 |
+
"chunk_inter_token_latency_avg": 0.02127487318088802,
|
| 1047 |
+
"input_tokens": 78,
|
| 1048 |
+
"output_tokens": 215,
|
| 1049 |
+
"output_tps_per_user": 128.95915854644946,
|
| 1050 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1051 |
+
"completed": false
|
| 1052 |
+
}
|
| 1053 |
+
],
|
| 1054 |
+
"total_tokens": 5145,
|
| 1055 |
+
"wall_time": 25.557628852082416,
|
| 1056 |
+
"num_completed": 2,
|
| 1057 |
+
"num_errors": 0,
|
| 1058 |
+
"server_gen_throughput": 257.1817350436559,
|
| 1059 |
+
"server_utilization": 0.011725293132328285,
|
| 1060 |
+
"server_spec_accept_rate": 0.7541666666666667,
|
| 1061 |
+
"server_spec_accept_length": 0.0,
|
| 1062 |
+
"avg_running_reqs": 2,
|
| 1063 |
+
"max_running_reqs": 2,
|
| 1064 |
+
"effective_concurrency": 2,
|
| 1065 |
+
"avg_queue_reqs": 0,
|
| 1066 |
+
"max_queue_reqs": 0,
|
| 1067 |
+
"queue_fraction": 0.0,
|
| 1068 |
+
"underfilled": false,
|
| 1069 |
+
"warmup_timed_out": false,
|
| 1070 |
+
"warmup_duration": 5.539,
|
| 1071 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1072 |
+
"timeout_reason": "",
|
| 1073 |
+
"capacity_limited": false,
|
| 1074 |
+
"hardware_summary": {
|
| 1075 |
+
"samples": 8,
|
| 1076 |
+
"duration_seconds": 16.896,
|
| 1077 |
+
"gpu_count": 4,
|
| 1078 |
+
"cpu_util_avg_pct": 11.54,
|
| 1079 |
+
"cpu_temp_max_c": 76.38,
|
| 1080 |
+
"gpu_util_avg_pct": 100.0,
|
| 1081 |
+
"gpu_util_max_pct": 100.0,
|
| 1082 |
+
"mem_util_avg_pct": 40.28,
|
| 1083 |
+
"mem_util_max_pct": 50.0,
|
| 1084 |
+
"temp_avg_c": 68.16,
|
| 1085 |
+
"temp_max_c": 84.0,
|
| 1086 |
+
"power_total_avg_w": 1172.61,
|
| 1087 |
+
"power_total_max_w": 1175.0,
|
| 1088 |
+
"power_limit_total_w": 1200.0,
|
| 1089 |
+
"vram_used_avg_mb": 384778.0,
|
| 1090 |
+
"vram_used_max_mb": 384778.0,
|
| 1091 |
+
"vram_total_mb": 391548.0,
|
| 1092 |
+
"vram_used_avg_pct": 98.27,
|
| 1093 |
+
"vram_used_max_pct": 98.27,
|
| 1094 |
+
"pcie_rx_avg_mb_s": 11187.12,
|
| 1095 |
+
"pcie_rx_max_mb_s": 11336.0,
|
| 1096 |
+
"pcie_tx_avg_mb_s": 11117.38,
|
| 1097 |
+
"pcie_tx_max_mb_s": 11239.0
|
| 1098 |
+
}
|
| 1099 |
+
},
|
| 1100 |
+
{
|
| 1101 |
+
"concurrency": 4,
|
| 1102 |
+
"context_tokens": 0,
|
| 1103 |
+
"benchmark_mode": "duration",
|
| 1104 |
+
"request_count_target": 0,
|
| 1105 |
+
"warmup_request_count": 0,
|
| 1106 |
+
"measurement_seconds": 19.982047,
|
| 1107 |
+
"measurement_wall_seconds": 20.00121,
|
| 1108 |
+
"client_output_tokens": 6174,
|
| 1109 |
+
"server_output_tokens": 6174,
|
| 1110 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1111 |
+
"aggregate_tps": 308.9773476463454,
|
| 1112 |
+
"per_request_avg_tps": 77.24433691158634,
|
| 1113 |
+
"ttft_avg": 0.17778973274535553,
|
| 1114 |
+
"ttft_p50": 0.1807985259220004,
|
| 1115 |
+
"ttft_p90": 0.1864776147995144,
|
| 1116 |
+
"ttft_p99": 0.21677800566889344,
|
| 1117 |
+
"time_to_second_token_avg": 0.03174264398694504,
|
| 1118 |
+
"time_to_second_token_p50": 0.03366913797799498,
|
| 1119 |
+
"time_to_second_token_p90": 0.03429299045819789,
|
| 1120 |
+
"time_to_second_token_p99": 0.03469313668319955,
|
| 1121 |
+
"request_latency_avg": 6.506094520983215,
|
| 1122 |
+
"request_latency_p50": 6.382211998803541,
|
| 1123 |
+
"request_latency_p90": 6.8716935135424135,
|
| 1124 |
+
"request_latency_p99": 7.440296533070504,
|
| 1125 |
+
"inter_token_latency_avg": 0.01244116428728022,
|
| 1126 |
+
"inter_token_latency_p50": 0.012306649429466642,
|
| 1127 |
+
"inter_token_latency_p90": 0.01313604974973618,
|
| 1128 |
+
"inter_token_latency_p99": 0.014173560375473144,
|
| 1129 |
+
"output_tps_per_user_avg": 80.63692147986394,
|
| 1130 |
+
"output_tps_per_user_p50": 81.27236798491623,
|
| 1131 |
+
"output_tps_per_user_p90": 85.68934634961276,
|
| 1132 |
+
"output_tps_per_user_p99": 87.43239080011531,
|
| 1133 |
+
"e2e_output_tps_per_user_avg": 78.95186391949123,
|
| 1134 |
+
"e2e_output_tps_per_user_p50": 80.2229697314949,
|
| 1135 |
+
"e2e_output_tps_per_user_p90": 83.75147031670433,
|
| 1136 |
+
"e2e_output_tps_per_user_p99": 84.93555512907336,
|
| 1137 |
+
"chunk_inter_token_latency_avg": 0.03454775971555825,
|
| 1138 |
+
"chunk_inter_token_latency_p50": 0.034643082201032946,
|
| 1139 |
+
"chunk_inter_token_latency_p90": 0.035144349182500076,
|
| 1140 |
+
"chunk_inter_token_latency_p99": 0.035562162856296174,
|
| 1141 |
+
"input_seq_len_avg": 78.0,
|
| 1142 |
+
"output_seq_len_avg": 512.0,
|
| 1143 |
+
"output_seq_len_p50": 512.0,
|
| 1144 |
+
"output_seq_len_p90": 512.0,
|
| 1145 |
+
"output_seq_len_p99": 512.0,
|
| 1146 |
+
"request_count": 17,
|
| 1147 |
+
"completed_request_count": 13,
|
| 1148 |
+
"request_samples": [
|
| 1149 |
+
{
|
| 1150 |
+
"ttft": 0.0723429499194026,
|
| 1151 |
+
"time_to_second_token": 0.013155136024579406,
|
| 1152 |
+
"latency": 6.662627859041095,
|
| 1153 |
+
"inter_token_latency_avg": 0.012896839352488634,
|
| 1154 |
+
"chunk_inter_token_latency_avg": 0.03450410947184132,
|
| 1155 |
+
"input_tokens": 78,
|
| 1156 |
+
"output_tokens": 512,
|
| 1157 |
+
"output_tps_per_user": 77.53837763413215,
|
| 1158 |
+
"e2e_output_tps_per_user": 76.84655526801231,
|
| 1159 |
+
"completed": true
|
| 1160 |
+
},
|
| 1161 |
+
{
|
| 1162 |
+
"ttft": 0.1807985259220004,
|
| 1163 |
+
"time_to_second_token": 0.03360512410290539,
|
| 1164 |
+
"latency": 6.19808776397258,
|
| 1165 |
+
"inter_token_latency_avg": 0.011775517099903288,
|
| 1166 |
+
"chunk_inter_token_latency_avg": 0.03399598439576599,
|
| 1167 |
+
"input_tokens": 78,
|
| 1168 |
+
"output_tokens": 512,
|
| 1169 |
+
"output_tps_per_user": 84.92196066771564,
|
| 1170 |
+
"e2e_output_tps_per_user": 82.60612296845576,
|
| 1171 |
+
"completed": true
|
| 1172 |
+
},
|
| 1173 |
+
{
|
| 1174 |
+
"ttft": 0.18202503910288215,
|
| 1175 |
+
"time_to_second_token": 0.033671696903184056,
|
| 1176 |
+
"latency": 6.092495953198522,
|
| 1177 |
+
"inter_token_latency_avg": 0.011566479283944501,
|
| 1178 |
+
"chunk_inter_token_latency_avg": 0.03456415739237217,
|
| 1179 |
+
"input_tokens": 78,
|
| 1180 |
+
"output_tokens": 512,
|
| 1181 |
+
"output_tps_per_user": 86.45673203150989,
|
| 1182 |
+
"e2e_output_tps_per_user": 84.03780715376647,
|
| 1183 |
+
"completed": true
|
| 1184 |
+
},
|
| 1185 |
+
{
|
| 1186 |
+
"ttft": 0.18031293293461204,
|
| 1187 |
+
"time_to_second_token": 0.03369922889396548,
|
| 1188 |
+
"latency": 6.382211998803541,
|
| 1189 |
+
"inter_token_latency_avg": 0.012136788778608472,
|
| 1190 |
+
"chunk_inter_token_latency_avg": 0.03503897777327079,
|
| 1191 |
+
"input_tokens": 78,
|
| 1192 |
+
"output_tokens": 512,
|
| 1193 |
+
"output_tps_per_user": 82.39411744254264,
|
| 1194 |
+
"e2e_output_tps_per_user": 80.2229697314949,
|
| 1195 |
+
"completed": true
|
| 1196 |
+
},
|
| 1197 |
+
{
|
| 1198 |
+
"ttft": 0.1809463920071721,
|
| 1199 |
+
"time_to_second_token": 0.0,
|
| 1200 |
+
"latency": 0.0,
|
| 1201 |
+
"inter_token_latency_avg": 0.0,
|
| 1202 |
+
"chunk_inter_token_latency_avg": 0.0,
|
| 1203 |
+
"input_tokens": 78,
|
| 1204 |
+
"output_tokens": 1,
|
| 1205 |
+
"output_tps_per_user": 0.0,
|
| 1206 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1207 |
+
"completed": false
|
| 1208 |
+
},
|
| 1209 |
+
{
|
| 1210 |
+
"ttft": 0.18660231097601354,
|
| 1211 |
+
"time_to_second_token": 0.02942012087441981,
|
| 1212 |
+
"latency": 6.2684185188263655,
|
| 1213 |
+
"inter_token_latency_avg": 0.01190179297035294,
|
| 1214 |
+
"chunk_inter_token_latency_avg": 0.03378786782139084,
|
| 1215 |
+
"input_tokens": 78,
|
| 1216 |
+
"output_tokens": 512,
|
| 1217 |
+
"output_tps_per_user": 84.0209540269247,
|
| 1218 |
+
"e2e_output_tps_per_user": 81.67929414130147,
|
| 1219 |
+
"completed": true
|
| 1220 |
+
},
|
| 1221 |
+
{
|
| 1222 |
+
"ttft": 0.17968551511876285,
|
| 1223 |
+
"time_to_second_token": 0.03475157590582967,
|
| 1224 |
+
"latency": 6.310488264076412,
|
| 1225 |
+
"inter_token_latency_avg": 0.01199765704296996,
|
| 1226 |
+
"chunk_inter_token_latency_avg": 0.034637303666427394,
|
| 1227 |
+
"input_tokens": 78,
|
| 1228 |
+
"output_tokens": 512,
|
| 1229 |
+
"output_tps_per_user": 83.34960704564823,
|
| 1230 |
+
"e2e_output_tps_per_user": 81.13476779834168,
|
| 1231 |
+
"completed": true
|
| 1232 |
+
},
|
| 1233 |
+
{
|
| 1234 |
+
"ttft": 0.17955420492216945,
|
| 1235 |
+
"time_to_second_token": 0.033856919035315514,
|
| 1236 |
+
"latency": 6.5550508559681475,
|
| 1237 |
+
"inter_token_latency_avg": 0.01247651008032481,
|
| 1238 |
+
"chunk_inter_token_latency_avg": 0.03561729972651385,
|
| 1239 |
+
"input_tokens": 78,
|
| 1240 |
+
"output_tokens": 512,
|
| 1241 |
+
"output_tps_per_user": 80.15061852728982,
|
| 1242 |
+
"e2e_output_tps_per_user": 78.10770827717403,
|
| 1243 |
+
"completed": true
|
| 1244 |
+
},
|
| 1245 |
+
{
|
| 1246 |
+
"ttft": 0.17957319412380457,
|
| 1247 |
+
"time_to_second_token": 0.034223999828100204,
|
| 1248 |
+
"latency": 0.0,
|
| 1249 |
+
"inter_token_latency_avg": 0.012843022881727473,
|
| 1250 |
+
"chunk_inter_token_latency_avg": 0.03484932613412567,
|
| 1251 |
+
"input_tokens": 78,
|
| 1252 |
+
"output_tokens": 484,
|
| 1253 |
+
"output_tps_per_user": 77.86328882297322,
|
| 1254 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1255 |
+
"completed": false
|
| 1256 |
+
},
|
| 1257 |
+
{
|
| 1258 |
+
"ttft": 0.18639448401518166,
|
| 1259 |
+
"time_to_second_token": 0.029375833924859762,
|
| 1260 |
+
"latency": 6.019423788879067,
|
| 1261 |
+
"inter_token_latency_avg": 0.011414930146504668,
|
| 1262 |
+
"chunk_inter_token_latency_avg": 0.03352315692450509,
|
| 1263 |
+
"input_tokens": 78,
|
| 1264 |
+
"output_tokens": 512,
|
| 1265 |
+
"output_tps_per_user": 87.60456587692804,
|
| 1266 |
+
"e2e_output_tps_per_user": 85.0579753075243,
|
| 1267 |
+
"completed": true
|
| 1268 |
+
},
|
| 1269 |
+
{
|
| 1270 |
+
"ttft": 0.1805800509173423,
|
| 1271 |
+
"time_to_second_token": 0.0336665790528059,
|
| 1272 |
+
"latency": 7.51252193399705,
|
| 1273 |
+
"inter_token_latency_avg": 0.014348222863169682,
|
| 1274 |
+
"chunk_inter_token_latency_avg": 0.035249720591729365,
|
| 1275 |
+
"input_tokens": 78,
|
| 1276 |
+
"output_tokens": 512,
|
| 1277 |
+
"output_tps_per_user": 69.69504234331978,
|
| 1278 |
+
"e2e_output_tps_per_user": 68.15287921929428,
|
| 1279 |
+
"completed": true
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"ttft": 0.18062497000209987,
|
| 1283 |
+
"time_to_second_token": 0.03364452510140836,
|
| 1284 |
+
"latency": 6.614234033040702,
|
| 1285 |
+
"inter_token_latency_avg": 0.012590233000075543,
|
| 1286 |
+
"chunk_inter_token_latency_avg": 0.03477626520561407,
|
| 1287 |
+
"input_tokens": 78,
|
| 1288 |
+
"output_tokens": 512,
|
| 1289 |
+
"output_tps_per_user": 79.42664762391608,
|
| 1290 |
+
"e2e_output_tps_per_user": 77.40881218329416,
|
| 1291 |
+
"completed": true
|
| 1292 |
+
},
|
| 1293 |
+
{
|
| 1294 |
+
"ttft": 0.18063039891421795,
|
| 1295 |
+
"time_to_second_token": 0.033818941097706556,
|
| 1296 |
+
"latency": 0.0,
|
| 1297 |
+
"inter_token_latency_avg": 0.013183806278526097,
|
| 1298 |
+
"chunk_inter_token_latency_avg": 0.03436578836602469,
|
| 1299 |
+
"input_tokens": 78,
|
| 1300 |
+
"output_tokens": 392,
|
| 1301 |
+
"output_tps_per_user": 75.8506290879599,
|
| 1302 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1303 |
+
"completed": false
|
| 1304 |
+
},
|
| 1305 |
+
{
|
| 1306 |
+
"ttft": 0.18626655801199377,
|
| 1307 |
+
"time_to_second_token": 0.029395153978839517,
|
| 1308 |
+
"latency": 6.337131014093757,
|
| 1309 |
+
"inter_token_latency_avg": 0.012036916743799928,
|
| 1310 |
+
"chunk_inter_token_latency_avg": 0.03379595854989979,
|
| 1311 |
+
"input_tokens": 78,
|
| 1312 |
+
"output_tokens": 512,
|
| 1313 |
+
"output_tps_per_user": 83.07775332209455,
|
| 1314 |
+
"e2e_output_tps_per_user": 80.79365865425756,
|
| 1315 |
+
"completed": true
|
| 1316 |
+
},
|
| 1317 |
+
{
|
| 1318 |
+
"ttft": 0.22252575703896582,
|
| 1319 |
+
"time_to_second_token": 0.03383295307867229,
|
| 1320 |
+
"latency": 6.910643592942506,
|
| 1321 |
+
"inter_token_latency_avg": 0.013088293220946262,
|
| 1322 |
+
"chunk_inter_token_latency_avg": 0.03465346028965565,
|
| 1323 |
+
"input_tokens": 78,
|
| 1324 |
+
"output_tokens": 512,
|
| 1325 |
+
"output_tps_per_user": 76.40415622715561,
|
| 1326 |
+
"e2e_output_tps_per_user": 74.0886131825522,
|
| 1327 |
+
"completed": true
|
| 1328 |
+
},
|
| 1329 |
+
{
|
| 1330 |
+
"ttft": 0.18178053596056998,
|
| 1331 |
+
"time_to_second_token": 0.03340253490023315,
|
| 1332 |
+
"latency": 6.715893195942044,
|
| 1333 |
+
"inter_token_latency_avg": 0.012786913228926564,
|
| 1334 |
+
"chunk_inter_token_latency_avg": 0.03475591840415678,
|
| 1335 |
+
"input_tokens": 78,
|
| 1336 |
+
"output_tokens": 512,
|
| 1337 |
+
"output_tps_per_user": 78.20495706014485,
|
| 1338 |
+
"e2e_output_tps_per_user": 76.23706706791684,
|
| 1339 |
+
"completed": true
|
| 1340 |
+
},
|
| 1341 |
+
{
|
| 1342 |
+
"ttft": 0.18178163678385317,
|
| 1343 |
+
"time_to_second_token": 0.03436198108829558,
|
| 1344 |
+
"latency": 0.0,
|
| 1345 |
+
"inter_token_latency_avg": 0.012014705624214687,
|
| 1346 |
+
"chunk_inter_token_latency_avg": 0.03464886073563849,
|
| 1347 |
+
"input_tokens": 78,
|
| 1348 |
+
"output_tokens": 448,
|
| 1349 |
+
"output_tps_per_user": 83.23133593756798,
|
| 1350 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1351 |
+
"completed": false
|
| 1352 |
+
}
|
| 1353 |
+
],
|
| 1354 |
+
"total_tokens": 6174,
|
| 1355 |
+
"wall_time": 25.552133698016405,
|
| 1356 |
+
"num_completed": 4,
|
| 1357 |
+
"num_errors": 0,
|
| 1358 |
+
"server_gen_throughput": 308.60564141811994,
|
| 1359 |
+
"server_utilization": 0.02345058626465657,
|
| 1360 |
+
"server_spec_accept_rate": 0.603448275862069,
|
| 1361 |
+
"server_spec_accept_length": 0.0,
|
| 1362 |
+
"avg_running_reqs": 3.9,
|
| 1363 |
+
"max_running_reqs": 4,
|
| 1364 |
+
"effective_concurrency": 3.9,
|
| 1365 |
+
"avg_queue_reqs": 0,
|
| 1366 |
+
"max_queue_reqs": 0,
|
| 1367 |
+
"queue_fraction": 0.0,
|
| 1368 |
+
"underfilled": true,
|
| 1369 |
+
"warmup_timed_out": false,
|
| 1370 |
+
"warmup_duration": 5.536,
|
| 1371 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1372 |
+
"timeout_reason": "",
|
| 1373 |
+
"capacity_limited": false,
|
| 1374 |
+
"hardware_summary": {
|
| 1375 |
+
"samples": 8,
|
| 1376 |
+
"duration_seconds": 16.827,
|
| 1377 |
+
"gpu_count": 4,
|
| 1378 |
+
"cpu_util_avg_pct": 11.47,
|
| 1379 |
+
"cpu_temp_max_c": 77.12,
|
| 1380 |
+
"gpu_util_avg_pct": 100.0,
|
| 1381 |
+
"gpu_util_max_pct": 100.0,
|
| 1382 |
+
"mem_util_avg_pct": 35.47,
|
| 1383 |
+
"mem_util_max_pct": 45.0,
|
| 1384 |
+
"temp_avg_c": 68.72,
|
| 1385 |
+
"temp_max_c": 84.0,
|
| 1386 |
+
"power_total_avg_w": 1170.25,
|
| 1387 |
+
"power_total_max_w": 1172.14,
|
| 1388 |
+
"power_limit_total_w": 1200.0,
|
| 1389 |
+
"vram_used_avg_mb": 384778.0,
|
| 1390 |
+
"vram_used_max_mb": 384778.0,
|
| 1391 |
+
"vram_total_mb": 391548.0,
|
| 1392 |
+
"vram_used_avg_pct": 98.27,
|
| 1393 |
+
"vram_used_max_pct": 98.27,
|
| 1394 |
+
"pcie_rx_avg_mb_s": 7813.12,
|
| 1395 |
+
"pcie_rx_max_mb_s": 8161.0,
|
| 1396 |
+
"pcie_tx_avg_mb_s": 8053.88,
|
| 1397 |
+
"pcie_tx_max_mb_s": 8437.0
|
| 1398 |
+
}
|
| 1399 |
+
},
|
| 1400 |
+
{
|
| 1401 |
+
"concurrency": 2,
|
| 1402 |
+
"context_tokens": 8192,
|
| 1403 |
+
"benchmark_mode": "duration",
|
| 1404 |
+
"request_count_target": 0,
|
| 1405 |
+
"warmup_request_count": 0,
|
| 1406 |
+
"measurement_seconds": 19.985498,
|
| 1407 |
+
"measurement_wall_seconds": 20.000656,
|
| 1408 |
+
"client_output_tokens": 3587,
|
| 1409 |
+
"server_output_tokens": 3587,
|
| 1410 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1411 |
+
"aggregate_tps": 179.48013878635928,
|
| 1412 |
+
"per_request_avg_tps": 89.74006939317964,
|
| 1413 |
+
"ttft_avg": 1.1894093764014542,
|
| 1414 |
+
"ttft_p50": 1.2212169540580362,
|
| 1415 |
+
"ttft_p90": 1.3684029981726777,
|
| 1416 |
+
"ttft_p99": 1.9125716026616284,
|
| 1417 |
+
"time_to_second_token_avg": 0.025312776444479823,
|
| 1418 |
+
"time_to_second_token_p50": 0.027226490550674498,
|
| 1419 |
+
"time_to_second_token_p90": 0.02769280921202153,
|
| 1420 |
+
"time_to_second_token_p99": 0.0277038907376118,
|
| 1421 |
+
"request_latency_avg": 5.413165779871633,
|
| 1422 |
+
"request_latency_p50": 5.438335195998661,
|
| 1423 |
+
"request_latency_p90": 5.877898134151473,
|
| 1424 |
+
"request_latency_p99": 6.255758887748234,
|
| 1425 |
+
"inter_token_latency_avg": 0.00830825860560843,
|
| 1426 |
+
"inter_token_latency_p50": 0.008347879003080923,
|
| 1427 |
+
"inter_token_latency_p90": 0.008825519827072598,
|
| 1428 |
+
"inter_token_latency_p99": 0.009230215194777226,
|
| 1429 |
+
"output_tps_per_user_avg": 120.7893245972912,
|
| 1430 |
+
"output_tps_per_user_p50": 119.79434916732279,
|
| 1431 |
+
"output_tps_per_user_p90": 129.7361418336343,
|
| 1432 |
+
"output_tps_per_user_p99": 131.83454344392672,
|
| 1433 |
+
"e2e_output_tps_per_user_avg": 95.20339521438798,
|
| 1434 |
+
"e2e_output_tps_per_user_p50": 94.14836874853962,
|
| 1435 |
+
"e2e_output_tps_per_user_p90": 103.85715540642657,
|
| 1436 |
+
"e2e_output_tps_per_user_p99": 108.00400096911441,
|
| 1437 |
+
"chunk_inter_token_latency_avg": 0.023168869852685122,
|
| 1438 |
+
"chunk_inter_token_latency_p50": 0.023156017746843782,
|
| 1439 |
+
"chunk_inter_token_latency_p90": 0.024457862475522812,
|
| 1440 |
+
"chunk_inter_token_latency_p99": 0.025339872482635258,
|
| 1441 |
+
"input_seq_len_avg": 8192.0,
|
| 1442 |
+
"output_seq_len_avg": 512.0,
|
| 1443 |
+
"output_seq_len_p50": 512.0,
|
| 1444 |
+
"output_seq_len_p90": 512.0,
|
| 1445 |
+
"output_seq_len_p99": 512.0,
|
| 1446 |
+
"request_count": 10,
|
| 1447 |
+
"completed_request_count": 8,
|
| 1448 |
+
"request_samples": [
|
| 1449 |
+
{
|
| 1450 |
+
"ttft": 0.5938855609856546,
|
| 1451 |
+
"time_to_second_token": 0.01751627796329558,
|
| 1452 |
+
"latency": 5.02539852890186,
|
| 1453 |
+
"inter_token_latency_avg": 0.008672236727820363,
|
| 1454 |
+
"chunk_inter_token_latency_avg": 0.024348972351187943,
|
| 1455 |
+
"input_tokens": 8192,
|
| 1456 |
+
"output_tokens": 512,
|
| 1457 |
+
"output_tps_per_user": 115.31050539614768,
|
| 1458 |
+
"e2e_output_tps_per_user": 101.88246704324189,
|
| 1459 |
+
"completed": true
|
| 1460 |
+
},
|
| 1461 |
+
{
|
| 1462 |
+
"ttft": 0.7053975961171091,
|
| 1463 |
+
"time_to_second_token": 0.01768357283435762,
|
| 1464 |
+
"latency": 4.720427100080997,
|
| 1465 |
+
"inter_token_latency_avg": 0.007857200594841268,
|
| 1466 |
+
"chunk_inter_token_latency_avg": 0.02206060167013125,
|
| 1467 |
+
"input_tokens": 8192,
|
| 1468 |
+
"output_tokens": 512,
|
| 1469 |
+
"output_tps_per_user": 127.27179202431984,
|
| 1470 |
+
"e2e_output_tps_per_user": 108.46476158719085,
|
| 1471 |
+
"completed": true
|
| 1472 |
+
},
|
| 1473 |
+
{
|
| 1474 |
+
"ttft": 1.2253350547980517,
|
| 1475 |
+
"time_to_second_token": 0.02671587117947638,
|
| 1476 |
+
"latency": 5.5139663990121335,
|
| 1477 |
+
"inter_token_latency_avg": 0.008392624939753585,
|
| 1478 |
+
"chunk_inter_token_latency_avg": 0.02330777904464175,
|
| 1479 |
+
"input_tokens": 8192,
|
| 1480 |
+
"output_tokens": 512,
|
| 1481 |
+
"output_tps_per_user": 119.1522327255769,
|
| 1482 |
+
"e2e_output_tps_per_user": 92.85511788605177,
|
| 1483 |
+
"completed": true
|
| 1484 |
+
},
|
| 1485 |
+
{
|
| 1486 |
+
"ttft": 1.2199293570593,
|
| 1487 |
+
"time_to_second_token": 0.027305281022563577,
|
| 1488 |
+
"latency": 5.4628303539939225,
|
| 1489 |
+
"inter_token_latency_avg": 0.008303133066408263,
|
| 1490 |
+
"chunk_inter_token_latency_avg": 0.023185251349369523,
|
| 1491 |
+
"input_tokens": 8192,
|
| 1492 |
+
"output_tokens": 512,
|
| 1493 |
+
"output_tps_per_user": 120.4364656090687,
|
| 1494 |
+
"e2e_output_tps_per_user": 93.72430897944183,
|
| 1495 |
+
"completed": true
|
| 1496 |
+
},
|
| 1497 |
+
{
|
| 1498 |
+
"ttft": 1.2225045510567725,
|
| 1499 |
+
"time_to_second_token": 0.02714770007878542,
|
| 1500 |
+
"latency": 0.0,
|
| 1501 |
+
"inter_token_latency_avg": 0.009275181346744406,
|
| 1502 |
+
"chunk_inter_token_latency_avg": 0.02543787359453664,
|
| 1503 |
+
"input_tokens": 8192,
|
| 1504 |
+
"output_tokens": 278,
|
| 1505 |
+
"output_tps_per_user": 107.81460357656518,
|
| 1506 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1507 |
+
"completed": false
|
| 1508 |
+
},
|
| 1509 |
+
{
|
| 1510 |
+
"ttft": 1.3012216889765114,
|
| 1511 |
+
"time_to_second_token": 0.02770512201823294,
|
| 1512 |
+
"latency": 5.4138400380034,
|
| 1513 |
+
"inter_token_latency_avg": 0.008048176808271797,
|
| 1514 |
+
"chunk_inter_token_latency_avg": 0.022596804115532356,
|
| 1515 |
+
"input_tokens": 8192,
|
| 1516 |
+
"output_tokens": 512,
|
| 1517 |
+
"output_tps_per_user": 124.25174344731278,
|
| 1518 |
+
"e2e_output_tps_per_user": 94.57242851763742,
|
| 1519 |
+
"completed": true
|
| 1520 |
+
},
|
| 1521 |
+
{
|
| 1522 |
+
"ttft": 1.9730347809381783,
|
| 1523 |
+
"time_to_second_token": 0.02644847217015922,
|
| 1524 |
+
"latency": 6.297743415925652,
|
| 1525 |
+
"inter_token_latency_avg": 0.008463226291560613,
|
| 1526 |
+
"chunk_inter_token_latency_avg": 0.02312678414431804,
|
| 1527 |
+
"input_tokens": 8192,
|
| 1528 |
+
"output_tokens": 512,
|
| 1529 |
+
"output_tps_per_user": 118.15824905889414,
|
| 1530 |
+
"e2e_output_tps_per_user": 81.2989615780886,
|
| 1531 |
+
"completed": true
|
| 1532 |
+
},
|
| 1533 |
+
{
|
| 1534 |
+
"ttft": 1.213654592167586,
|
| 1535 |
+
"time_to_second_token": 0.027514670975506306,
|
| 1536 |
+
"latency": 5.69796444196254,
|
| 1537 |
+
"inter_token_latency_avg": 0.008775557435997953,
|
| 1538 |
+
"chunk_inter_token_latency_avg": 0.023114999225747185,
|
| 1539 |
+
"input_tokens": 8192,
|
| 1540 |
+
"output_tokens": 512,
|
| 1541 |
+
"output_tps_per_user": 113.9528750501854,
|
| 1542 |
+
"e2e_output_tps_per_user": 89.85665060128959,
|
| 1543 |
+
"completed": true
|
| 1544 |
+
},
|
| 1545 |
+
{
|
| 1546 |
+
"ttft": 1.2265115019399673,
|
| 1547 |
+
"time_to_second_token": 0.027691441122442484,
|
| 1548 |
+
"latency": 5.1731559610925615,
|
| 1549 |
+
"inter_token_latency_avg": 0.00772337467544539,
|
| 1550 |
+
"chunk_inter_token_latency_avg": 0.023352925793802333,
|
| 1551 |
+
"input_tokens": 8192,
|
| 1552 |
+
"output_tokens": 512,
|
| 1553 |
+
"output_tps_per_user": 129.4770799064377,
|
| 1554 |
+
"e2e_output_tps_per_user": 98.97246552216193,
|
| 1555 |
+
"completed": true
|
| 1556 |
+
},
|
| 1557 |
+
{
|
| 1558 |
+
"ttft": 1.2126190799754113,
|
| 1559 |
+
"time_to_second_token": 0.027399355079978704,
|
| 1560 |
+
"latency": 0.0,
|
| 1561 |
+
"inter_token_latency_avg": 0.007571874169240656,
|
| 1562 |
+
"chunk_inter_token_latency_avg": 0.02115670723758419,
|
| 1563 |
+
"input_tokens": 8192,
|
| 1564 |
+
"output_tokens": 96,
|
| 1565 |
+
"output_tps_per_user": 132.06769917840364,
|
| 1566 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1567 |
+
"completed": false
|
| 1568 |
+
}
|
| 1569 |
+
],
|
| 1570 |
+
"total_tokens": 3587,
|
| 1571 |
+
"wall_time": 25.556549122091383,
|
| 1572 |
+
"num_completed": 2,
|
| 1573 |
+
"num_errors": 0,
|
| 1574 |
+
"server_gen_throughput": 179.29840638216805,
|
| 1575 |
+
"server_utilization": 0.012283640424343933,
|
| 1576 |
+
"server_spec_accept_rate": 0.6568627450980392,
|
| 1577 |
+
"server_spec_accept_length": 0.0,
|
| 1578 |
+
"avg_running_reqs": 1.9,
|
| 1579 |
+
"max_running_reqs": 2,
|
| 1580 |
+
"effective_concurrency": 1.9,
|
| 1581 |
+
"avg_queue_reqs": 0,
|
| 1582 |
+
"max_queue_reqs": 0,
|
| 1583 |
+
"queue_fraction": 0.0,
|
| 1584 |
+
"underfilled": true,
|
| 1585 |
+
"warmup_timed_out": false,
|
| 1586 |
+
"warmup_duration": 5.55,
|
| 1587 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1588 |
+
"timeout_reason": "",
|
| 1589 |
+
"capacity_limited": false,
|
| 1590 |
+
"hardware_summary": {
|
| 1591 |
+
"samples": 8,
|
| 1592 |
+
"duration_seconds": 16.921,
|
| 1593 |
+
"gpu_count": 4,
|
| 1594 |
+
"cpu_util_avg_pct": 11.51,
|
| 1595 |
+
"cpu_temp_max_c": 76.38,
|
| 1596 |
+
"gpu_util_avg_pct": 99.88,
|
| 1597 |
+
"gpu_util_max_pct": 100.0,
|
| 1598 |
+
"mem_util_avg_pct": 36.44,
|
| 1599 |
+
"mem_util_max_pct": 55.0,
|
| 1600 |
+
"temp_avg_c": 68.5,
|
| 1601 |
+
"temp_max_c": 84.0,
|
| 1602 |
+
"power_total_avg_w": 1162.41,
|
| 1603 |
+
"power_total_max_w": 1177.27,
|
| 1604 |
+
"power_limit_total_w": 1200.0,
|
| 1605 |
+
"vram_used_avg_mb": 384778.0,
|
| 1606 |
+
"vram_used_max_mb": 384778.0,
|
| 1607 |
+
"vram_total_mb": 391548.0,
|
| 1608 |
+
"vram_used_avg_pct": 98.27,
|
| 1609 |
+
"vram_used_max_pct": 98.27,
|
| 1610 |
+
"pcie_rx_avg_mb_s": 29258.88,
|
| 1611 |
+
"pcie_rx_max_mb_s": 70866.0,
|
| 1612 |
+
"pcie_tx_avg_mb_s": 28010.62,
|
| 1613 |
+
"pcie_tx_max_mb_s": 69989.0
|
| 1614 |
+
}
|
| 1615 |
+
},
|
| 1616 |
+
{
|
| 1617 |
+
"concurrency": 4,
|
| 1618 |
+
"context_tokens": 8192,
|
| 1619 |
+
"benchmark_mode": "duration",
|
| 1620 |
+
"request_count_target": 0,
|
| 1621 |
+
"warmup_request_count": 0,
|
| 1622 |
+
"measurement_seconds": 19.930302,
|
| 1623 |
+
"measurement_wall_seconds": 20.000584,
|
| 1624 |
+
"client_output_tokens": 4342,
|
| 1625 |
+
"server_output_tokens": 4342,
|
| 1626 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1627 |
+
"aggregate_tps": 217.85922059414168,
|
| 1628 |
+
"per_request_avg_tps": 54.46480514853542,
|
| 1629 |
+
"ttft_avg": 1.6845163684144306,
|
| 1630 |
+
"ttft_p50": 1.2621239490108564,
|
| 1631 |
+
"ttft_p90": 2.2127018794883044,
|
| 1632 |
+
"ttft_p99": 4.023498326067348,
|
| 1633 |
+
"time_to_second_token_avg": 0.03870494751026854,
|
| 1634 |
+
"time_to_second_token_p50": 0.04049360693898052,
|
| 1635 |
+
"time_to_second_token_p90": 0.041569333686493334,
|
| 1636 |
+
"time_to_second_token_p99": 0.044618300595320765,
|
| 1637 |
+
"request_latency_avg": 9.701175137112537,
|
| 1638 |
+
"request_latency_p50": 9.277128694113344,
|
| 1639 |
+
"request_latency_p90": 11.318623020872474,
|
| 1640 |
+
"request_latency_p99": 12.65666121724993,
|
| 1641 |
+
"inter_token_latency_avg": 0.015277014782169696,
|
| 1642 |
+
"inter_token_latency_p50": 0.015349018777003027,
|
| 1643 |
+
"inter_token_latency_p90": 0.016694702791212282,
|
| 1644 |
+
"inter_token_latency_p99": 0.017054498783689094,
|
| 1645 |
+
"output_tps_per_user_avg": 65.75017254999644,
|
| 1646 |
+
"output_tps_per_user_p50": 65.15961146932285,
|
| 1647 |
+
"output_tps_per_user_p90": 71.0132044249438,
|
| 1648 |
+
"output_tps_per_user_p99": 72.7397198355044,
|
| 1649 |
+
"e2e_output_tps_per_user_avg": 53.62687357686466,
|
| 1650 |
+
"e2e_output_tps_per_user_p50": 55.18948986068087,
|
| 1651 |
+
"e2e_output_tps_per_user_p90": 60.13891332324956,
|
| 1652 |
+
"e2e_output_tps_per_user_p99": 62.607782503263365,
|
| 1653 |
+
"chunk_inter_token_latency_avg": 0.04311462391925306,
|
| 1654 |
+
"chunk_inter_token_latency_p50": 0.04431330813678551,
|
| 1655 |
+
"chunk_inter_token_latency_p90": 0.04551846648306814,
|
| 1656 |
+
"chunk_inter_token_latency_p99": 0.0457929443786719,
|
| 1657 |
+
"input_seq_len_avg": 8192.0,
|
| 1658 |
+
"output_seq_len_avg": 512.0,
|
| 1659 |
+
"output_seq_len_p50": 512.0,
|
| 1660 |
+
"output_seq_len_p90": 512.0,
|
| 1661 |
+
"output_seq_len_p99": 512.0,
|
| 1662 |
+
"request_count": 12,
|
| 1663 |
+
"completed_request_count": 9,
|
| 1664 |
+
"request_samples": [
|
| 1665 |
+
{
|
| 1666 |
+
"ttft": 0.5955398578662425,
|
| 1667 |
+
"time_to_second_token": 0.016524713020771742,
|
| 1668 |
+
"latency": 8.142221544869244,
|
| 1669 |
+
"inter_token_latency_avg": 0.01476845731311742,
|
| 1670 |
+
"chunk_inter_token_latency_avg": 0.04035658656151338,
|
| 1671 |
+
"input_tokens": 8192,
|
| 1672 |
+
"output_tokens": 512,
|
| 1673 |
+
"output_tps_per_user": 67.71187936547678,
|
| 1674 |
+
"e2e_output_tps_per_user": 62.882101301042674,
|
| 1675 |
+
"completed": true
|
| 1676 |
+
},
|
| 1677 |
+
{
|
| 1678 |
+
"ttft": 1.2590207229368389,
|
| 1679 |
+
"time_to_second_token": 0.0395864921156317,
|
| 1680 |
+
"latency": 9.238898925017565,
|
| 1681 |
+
"inter_token_latency_avg": 0.015616200004071872,
|
| 1682 |
+
"chunk_inter_token_latency_avg": 0.045599304011889864,
|
| 1683 |
+
"input_tokens": 8192,
|
| 1684 |
+
"output_tokens": 512,
|
| 1685 |
+
"output_tps_per_user": 64.03606509517381,
|
| 1686 |
+
"e2e_output_tps_per_user": 55.41785922276735,
|
| 1687 |
+
"completed": true
|
| 1688 |
+
},
|
| 1689 |
+
{
|
| 1690 |
+
"ttft": 1.2618258679285645,
|
| 1691 |
+
"time_to_second_token": 0.03996979701332748,
|
| 1692 |
+
"latency": 9.196670216973871,
|
| 1693 |
+
"inter_token_latency_avg": 0.01552807113316107,
|
| 1694 |
+
"chunk_inter_token_latency_avg": 0.044577777241827564,
|
| 1695 |
+
"input_tokens": 8192,
|
| 1696 |
+
"output_tokens": 512,
|
| 1697 |
+
"output_tps_per_user": 64.39949890907208,
|
| 1698 |
+
"e2e_output_tps_per_user": 55.67232356065407,
|
| 1699 |
+
"completed": true
|
| 1700 |
+
},
|
| 1701 |
+
{
|
| 1702 |
+
"ttft": 2.2127146429847926,
|
| 1703 |
+
"time_to_second_token": 0.040582613088190556,
|
| 1704 |
+
"latency": 10.946945744100958,
|
| 1705 |
+
"inter_token_latency_avg": 0.017092428769307565,
|
| 1706 |
+
"chunk_inter_token_latency_avg": 0.04479092872367264,
|
| 1707 |
+
"input_tokens": 8192,
|
| 1708 |
+
"output_tokens": 512,
|
| 1709 |
+
"output_tps_per_user": 58.50543614934785,
|
| 1710 |
+
"e2e_output_tps_per_user": 46.771036594924595,
|
| 1711 |
+
"completed": true
|
| 1712 |
+
},
|
| 1713 |
+
{
|
| 1714 |
+
"ttft": 1.2621886630076915,
|
| 1715 |
+
"time_to_second_token": 0.04047246486879885,
|
| 1716 |
+
"latency": 8.611827800050378,
|
| 1717 |
+
"inter_token_latency_avg": 0.014382855454095277,
|
| 1718 |
+
"chunk_inter_token_latency_avg": 0.0445432674972284,
|
| 1719 |
+
"input_tokens": 8192,
|
| 1720 |
+
"output_tokens": 512,
|
| 1721 |
+
"output_tps_per_user": 69.52722310195134,
|
| 1722 |
+
"e2e_output_tps_per_user": 59.453116328801286,
|
| 1723 |
+
"completed": true
|
| 1724 |
+
},
|
| 1725 |
+
{
|
| 1726 |
+
"ttft": 1.923778808210045,
|
| 1727 |
+
"time_to_second_token": 0.04072440997697413,
|
| 1728 |
+
"latency": 0.0,
|
| 1729 |
+
"inter_token_latency_avg": 0.014053212739941631,
|
| 1730 |
+
"chunk_inter_token_latency_avg": 0.04110083452024025,
|
| 1731 |
+
"input_tokens": 8192,
|
| 1732 |
+
"output_tokens": 428,
|
| 1733 |
+
"output_tps_per_user": 71.15810587267559,
|
| 1734 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1735 |
+
"completed": false
|
| 1736 |
+
},
|
| 1737 |
+
{
|
| 1738 |
+
"ttft": 2.2125870080199093,
|
| 1739 |
+
"time_to_second_token": 0.04051474900916219,
|
| 1740 |
+
"latency": 9.543051223969087,
|
| 1741 |
+
"inter_token_latency_avg": 0.01434533114667158,
|
| 1742 |
+
"chunk_inter_token_latency_avg": 0.04188836694828101,
|
| 1743 |
+
"input_tokens": 8192,
|
| 1744 |
+
"output_tokens": 512,
|
| 1745 |
+
"output_tps_per_user": 69.70909139535765,
|
| 1746 |
+
"e2e_output_tps_per_user": 53.65160345299416,
|
| 1747 |
+
"completed": true
|
| 1748 |
+
},
|
| 1749 |
+
{
|
| 1750 |
+
"ttft": 1.2591751390136778,
|
| 1751 |
+
"time_to_second_token": 0.040605568094179034,
|
| 1752 |
+
"latency": 9.277128694113344,
|
| 1753 |
+
"inter_token_latency_avg": 0.015690711458120676,
|
| 1754 |
+
"chunk_inter_token_latency_avg": 0.04581687745771238,
|
| 1755 |
+
"input_tokens": 8192,
|
| 1756 |
+
"output_tokens": 512,
|
| 1757 |
+
"output_tps_per_user": 63.7319730637487,
|
| 1758 |
+
"e2e_output_tps_per_user": 55.18948986068087,
|
| 1759 |
+
"completed": true
|
| 1760 |
+
},
|
| 1761 |
+
{
|
| 1762 |
+
"ttft": 1.4571730380412191,
|
| 1763 |
+
"time_to_second_token": 0.04498353600502014,
|
| 1764 |
+
"latency": 0.0,
|
| 1765 |
+
"inter_token_latency_avg": 0.015169966420844982,
|
| 1766 |
+
"chunk_inter_token_latency_avg": 0.042138795613458284,
|
| 1767 |
+
"input_tokens": 8192,
|
| 1768 |
+
"output_tokens": 476,
|
| 1769 |
+
"output_tps_per_user": 65.91972402957363,
|
| 1770 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1771 |
+
"completed": false
|
| 1772 |
+
},
|
| 1773 |
+
{
|
| 1774 |
+
"ttft": 4.247303050942719,
|
| 1775 |
+
"time_to_second_token": 0.039115041960030794,
|
| 1776 |
+
"latency": 12.805332127958536,
|
| 1777 |
+
"inter_token_latency_avg": 0.01674761071823056,
|
| 1778 |
+
"chunk_inter_token_latency_avg": 0.04457306810945738,
|
| 1779 |
+
"input_tokens": 8192,
|
| 1780 |
+
"output_tokens": 512,
|
| 1781 |
+
"output_tps_per_user": 59.71000979330461,
|
| 1782 |
+
"e2e_output_tps_per_user": 39.98334403854502,
|
| 1783 |
+
"completed": true
|
| 1784 |
+
},
|
| 1785 |
+
{
|
| 1786 |
+
"ttft": 1.260830387007445,
|
| 1787 |
+
"time_to_second_token": 0.03971677087247372,
|
| 1788 |
+
"latency": 9.548499956959859,
|
| 1789 |
+
"inter_token_latency_avg": 0.016218531448047777,
|
| 1790 |
+
"chunk_inter_token_latency_avg": 0.044083348776342623,
|
| 1791 |
+
"input_tokens": 8192,
|
| 1792 |
+
"output_tokens": 512,
|
| 1793 |
+
"output_tps_per_user": 61.65786361134256,
|
| 1794 |
+
"e2e_output_tps_per_user": 53.620987831371934,
|
| 1795 |
+
"completed": true
|
| 1796 |
+
},
|
| 1797 |
+
{
|
| 1798 |
+
"ttft": 1.2620592350140214,
|
| 1799 |
+
"time_to_second_token": 0.04166321409866214,
|
| 1800 |
+
"latency": 0.0,
|
| 1801 |
+
"inter_token_latency_avg": 0.013710800780425945,
|
| 1802 |
+
"chunk_inter_token_latency_avg": 0.03790633156941291,
|
| 1803 |
+
"input_tokens": 8192,
|
| 1804 |
+
"output_tokens": 283,
|
| 1805 |
+
"output_tps_per_user": 72.93520021293268,
|
| 1806 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1807 |
+
"completed": false
|
| 1808 |
+
}
|
| 1809 |
+
],
|
| 1810 |
+
"total_tokens": 4342,
|
| 1811 |
+
"wall_time": 28.853528001811355,
|
| 1812 |
+
"num_completed": 4,
|
| 1813 |
+
"num_errors": 0,
|
| 1814 |
+
"server_gen_throughput": 217.03898889784327,
|
| 1815 |
+
"server_utilization": 0.027359017308765998,
|
| 1816 |
+
"server_spec_accept_rate": 0.6283185840707964,
|
| 1817 |
+
"server_spec_accept_length": 0.0,
|
| 1818 |
+
"avg_running_reqs": 3.9,
|
| 1819 |
+
"max_running_reqs": 4,
|
| 1820 |
+
"effective_concurrency": 3.9,
|
| 1821 |
+
"avg_queue_reqs": 0.1,
|
| 1822 |
+
"max_queue_reqs": 1,
|
| 1823 |
+
"queue_fraction": 0.05,
|
| 1824 |
+
"underfilled": true,
|
| 1825 |
+
"warmup_timed_out": false,
|
| 1826 |
+
"warmup_duration": 8.573,
|
| 1827 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1828 |
+
"timeout_reason": "",
|
| 1829 |
+
"capacity_limited": true,
|
| 1830 |
+
"hardware_summary": {
|
| 1831 |
+
"samples": 8,
|
| 1832 |
+
"duration_seconds": 16.883,
|
| 1833 |
+
"gpu_count": 4,
|
| 1834 |
+
"cpu_util_avg_pct": 11.5,
|
| 1835 |
+
"cpu_temp_max_c": 76.25,
|
| 1836 |
+
"gpu_util_avg_pct": 100.0,
|
| 1837 |
+
"gpu_util_max_pct": 100.0,
|
| 1838 |
+
"mem_util_avg_pct": 34.25,
|
| 1839 |
+
"mem_util_max_pct": 46.0,
|
| 1840 |
+
"temp_avg_c": 68.66,
|
| 1841 |
+
"temp_max_c": 84.0,
|
| 1842 |
+
"power_total_avg_w": 1159.22,
|
| 1843 |
+
"power_total_max_w": 1172.44,
|
| 1844 |
+
"power_limit_total_w": 1200.0,
|
| 1845 |
+
"vram_used_avg_mb": 384778.0,
|
| 1846 |
+
"vram_used_max_mb": 384778.0,
|
| 1847 |
+
"vram_total_mb": 391548.0,
|
| 1848 |
+
"vram_used_avg_pct": 98.27,
|
| 1849 |
+
"vram_used_max_pct": 98.27,
|
| 1850 |
+
"pcie_rx_avg_mb_s": 27724.5,
|
| 1851 |
+
"pcie_rx_max_mb_s": 67581.0,
|
| 1852 |
+
"pcie_tx_avg_mb_s": 27474.12,
|
| 1853 |
+
"pcie_tx_max_mb_s": 67821.0
|
| 1854 |
+
}
|
| 1855 |
+
},
|
| 1856 |
+
{
|
| 1857 |
+
"concurrency": 2,
|
| 1858 |
+
"context_tokens": 32768,
|
| 1859 |
+
"benchmark_mode": "duration",
|
| 1860 |
+
"request_count_target": 0,
|
| 1861 |
+
"warmup_request_count": 0,
|
| 1862 |
+
"measurement_seconds": 19.997661,
|
| 1863 |
+
"measurement_wall_seconds": 20.00078,
|
| 1864 |
+
"client_output_tokens": 3749,
|
| 1865 |
+
"server_output_tokens": 3749,
|
| 1866 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1867 |
+
"aggregate_tps": 187.47192708215627,
|
| 1868 |
+
"per_request_avg_tps": 93.73596354107814,
|
| 1869 |
+
"ttft_avg": 1.1601904219947756,
|
| 1870 |
+
"ttft_p50": 1.2142878150334582,
|
| 1871 |
+
"ttft_p90": 1.3290542589034884,
|
| 1872 |
+
"ttft_p99": 1.4677723066555337,
|
| 1873 |
+
"time_to_second_token_avg": 0.017825737223029138,
|
| 1874 |
+
"time_to_second_token_p50": 0.019167354563251138,
|
| 1875 |
+
"time_to_second_token_p90": 0.020459615555591882,
|
| 1876 |
+
"time_to_second_token_p99": 0.021644204219337552,
|
| 1877 |
+
"request_latency_avg": 5.313494802772766,
|
| 1878 |
+
"request_latency_p50": 5.375003230990842,
|
| 1879 |
+
"request_latency_p90": 5.586414952063933,
|
| 1880 |
+
"request_latency_p99": 5.701329016662203,
|
| 1881 |
+
"inter_token_latency_avg": 0.008015108251874231,
|
| 1882 |
+
"inter_token_latency_p50": 0.008146798233829671,
|
| 1883 |
+
"inter_token_latency_p90": 0.008546021327498477,
|
| 1884 |
+
"inter_token_latency_p99": 0.009447031634392257,
|
| 1885 |
+
"output_tps_per_user_avg": 125.91902973995667,
|
| 1886 |
+
"output_tps_per_user_p50": 122.74874938988106,
|
| 1887 |
+
"output_tps_per_user_p90": 143.33476272223038,
|
| 1888 |
+
"output_tps_per_user_p99": 150.46171413409994,
|
| 1889 |
+
"e2e_output_tps_per_user_avg": 96.63340093514198,
|
| 1890 |
+
"e2e_output_tps_per_user_p50": 95.25646775312458,
|
| 1891 |
+
"e2e_output_tps_per_user_p90": 104.75414728581389,
|
| 1892 |
+
"e2e_output_tps_per_user_p99": 105.26579246796483,
|
| 1893 |
+
"chunk_inter_token_latency_avg": 0.023044757916286292,
|
| 1894 |
+
"chunk_inter_token_latency_p50": 0.02316345668564709,
|
| 1895 |
+
"chunk_inter_token_latency_p90": 0.02447127468927797,
|
| 1896 |
+
"chunk_inter_token_latency_p99": 0.025079763939748784,
|
| 1897 |
+
"input_seq_len_avg": 32768.0,
|
| 1898 |
+
"output_seq_len_avg": 512.0,
|
| 1899 |
+
"output_seq_len_p50": 512.0,
|
| 1900 |
+
"output_seq_len_p90": 512.0,
|
| 1901 |
+
"output_seq_len_p99": 512.0,
|
| 1902 |
+
"request_count": 10,
|
| 1903 |
+
"completed_request_count": 8,
|
| 1904 |
+
"request_samples": [
|
| 1905 |
+
{
|
| 1906 |
+
"ttft": 0.6090631559491158,
|
| 1907 |
+
"time_to_second_token": 0.01044343295507133,
|
| 1908 |
+
"latency": 5.4876536841038615,
|
| 1909 |
+
"inter_token_latency_avg": 0.009547143890713788,
|
| 1910 |
+
"chunk_inter_token_latency_avg": 0.025147373856467762,
|
| 1911 |
+
"input_tokens": 32768,
|
| 1912 |
+
"output_tokens": 512,
|
| 1913 |
+
"output_tps_per_user": 104.74336738264405,
|
| 1914 |
+
"e2e_output_tps_per_user": 93.30034828602892,
|
| 1915 |
+
"completed": true
|
| 1916 |
+
},
|
| 1917 |
+
{
|
| 1918 |
+
"ttft": 1.4831854230724275,
|
| 1919 |
+
"time_to_second_token": 0.016494586830958724,
|
| 1920 |
+
"latency": 5.7140972460620105,
|
| 1921 |
+
"inter_token_latency_avg": 0.00827967088647668,
|
| 1922 |
+
"chunk_inter_token_latency_avg": 0.023119736737647993,
|
| 1923 |
+
"input_tokens": 32768,
|
| 1924 |
+
"output_tokens": 512,
|
| 1925 |
+
"output_tps_per_user": 120.77774753502777,
|
| 1926 |
+
"e2e_output_tps_per_user": 89.60295527921852,
|
| 1927 |
+
"completed": true
|
| 1928 |
+
},
|
| 1929 |
+
{
|
| 1930 |
+
"ttft": 1.2215185849927366,
|
| 1931 |
+
"time_to_second_token": 0.01897827093489468,
|
| 1932 |
+
"latency": 5.5316939689219,
|
| 1933 |
+
"inter_token_latency_avg": 0.00843478548714122,
|
| 1934 |
+
"chunk_inter_token_latency_avg": 0.023049066224220125,
|
| 1935 |
+
"input_tokens": 32768,
|
| 1936 |
+
"output_tokens": 512,
|
| 1937 |
+
"output_tps_per_user": 118.55666057239915,
|
| 1938 |
+
"e2e_output_tps_per_user": 92.55754256770396,
|
| 1939 |
+
"completed": true
|
| 1940 |
+
},
|
| 1941 |
+
{
|
| 1942 |
+
"ttft": 1.213982854038477,
|
| 1943 |
+
"time_to_second_token": 0.017720089061185718,
|
| 1944 |
+
"latency": 5.389688584022224,
|
| 1945 |
+
"inter_token_latency_avg": 0.008171635479420248,
|
| 1946 |
+
"chunk_inter_token_latency_avg": 0.02332796497197624,
|
| 1947 |
+
"input_tokens": 32768,
|
| 1948 |
+
"output_tokens": 512,
|
| 1949 |
+
"output_tps_per_user": 122.37452374355627,
|
| 1950 |
+
"e2e_output_tps_per_user": 94.99621212213043,
|
| 1951 |
+
"completed": true
|
| 1952 |
+
},
|
| 1953 |
+
{
|
| 1954 |
+
"ttft": 1.2081716759130359,
|
| 1955 |
+
"time_to_second_token": 0.02177582518197596,
|
| 1956 |
+
"latency": 0.0,
|
| 1957 |
+
"inter_token_latency_avg": 0.0066114129892226245,
|
| 1958 |
+
"chunk_inter_token_latency_avg": 0.020994135983320967,
|
| 1959 |
+
"input_tokens": 32768,
|
| 1960 |
+
"output_tokens": 182,
|
| 1961 |
+
"output_tps_per_user": 151.25359762430767,
|
| 1962 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1963 |
+
"completed": false
|
| 1964 |
+
},
|
| 1965 |
+
{
|
| 1966 |
+
"ttft": 1.3119285739958286,
|
| 1967 |
+
"time_to_second_token": 0.020313370041549206,
|
| 1968 |
+
"latency": 4.8990289689973,
|
| 1969 |
+
"inter_token_latency_avg": 0.0070197659393375165,
|
| 1970 |
+
"chunk_inter_token_latency_avg": 0.020734684364170353,
|
| 1971 |
+
"input_tokens": 32768,
|
| 1972 |
+
"output_tokens": 512,
|
| 1973 |
+
"output_tps_per_user": 142.45489217755514,
|
| 1974 |
+
"e2e_output_tps_per_user": 104.51050672288487,
|
| 1975 |
+
"completed": true
|
| 1976 |
+
},
|
| 1977 |
+
{
|
| 1978 |
+
"ttft": 0.9148407790344208,
|
| 1979 |
+
"time_to_second_token": 0.013460942078381777,
|
| 1980 |
+
"latency": 4.861252913950011,
|
| 1981 |
+
"inter_token_latency_avg": 0.007722920029189022,
|
| 1982 |
+
"chunk_inter_token_latency_avg": 0.02335155109417509,
|
| 1983 |
+
"input_tokens": 32768,
|
| 1984 |
+
"output_tokens": 512,
|
| 1985 |
+
"output_tps_per_user": 129.48470218783416,
|
| 1986 |
+
"e2e_output_tps_per_user": 105.32264193264827,
|
| 1987 |
+
"completed": true
|
| 1988 |
+
},
|
| 1989 |
+
{
|
| 1990 |
+
"ttft": 1.2099958129692823,
|
| 1991 |
+
"time_to_second_token": 0.02024669898673892,
|
| 1992 |
+
"latency": 5.36031787795946,
|
| 1993 |
+
"inter_token_latency_avg": 0.008121960988239096,
|
| 1994 |
+
"chunk_inter_token_latency_avg": 0.023186156787654625,
|
| 1995 |
+
"input_tokens": 32768,
|
| 1996 |
+
"output_tokens": 512,
|
| 1997 |
+
"output_tps_per_user": 123.12297503620586,
|
| 1998 |
+
"e2e_output_tps_per_user": 95.51672338411872,
|
| 1999 |
+
"completed": true
|
| 2000 |
+
},
|
| 2001 |
+
{
|
| 2002 |
+
"ttft": 1.2145927760284394,
|
| 2003 |
+
"time_to_second_token": 0.019467717967927456,
|
| 2004 |
+
"latency": 5.264225178165361,
|
| 2005 |
+
"inter_token_latency_avg": 0.007924916638232724,
|
| 2006 |
+
"chunk_inter_token_latency_avg": 0.023140756583639555,
|
| 2007 |
+
"input_tokens": 32768,
|
| 2008 |
+
"output_tokens": 512,
|
| 2009 |
+
"output_tps_per_user": 126.18429261143653,
|
| 2010 |
+
"e2e_output_tps_per_user": 97.26027718640209,
|
| 2011 |
+
"completed": true
|
| 2012 |
+
},
|
| 2013 |
+
{
|
| 2014 |
+
"ttft": 1.2146245839539915,
|
| 2015 |
+
"time_to_second_token": 0.019356438191607594,
|
| 2016 |
+
"latency": 0.0,
|
| 2017 |
+
"inter_token_latency_avg": 0.008316870190769392,
|
| 2018 |
+
"chunk_inter_token_latency_avg": 0.024396152559590215,
|
| 2019 |
+
"input_tokens": 32768,
|
| 2020 |
+
"output_tokens": 353,
|
| 2021 |
+
"output_tps_per_user": 120.23753852860004,
|
| 2022 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2023 |
+
"completed": false
|
| 2024 |
+
}
|
| 2025 |
+
],
|
| 2026 |
+
"total_tokens": 3749,
|
| 2027 |
+
"wall_time": 25.56879776297137,
|
| 2028 |
+
"num_completed": 2,
|
| 2029 |
+
"num_errors": 0,
|
| 2030 |
+
"server_gen_throughput": 187.39629296317187,
|
| 2031 |
+
"server_utilization": 0.013121161362367406,
|
| 2032 |
+
"server_spec_accept_rate": 0.6424242424242425,
|
| 2033 |
+
"server_spec_accept_length": 0.0,
|
| 2034 |
+
"avg_running_reqs": 1.9,
|
| 2035 |
+
"max_running_reqs": 2,
|
| 2036 |
+
"effective_concurrency": 1.9,
|
| 2037 |
+
"avg_queue_reqs": 0,
|
| 2038 |
+
"max_queue_reqs": 0,
|
| 2039 |
+
"queue_fraction": 0.0,
|
| 2040 |
+
"underfilled": true,
|
| 2041 |
+
"warmup_timed_out": false,
|
| 2042 |
+
"warmup_duration": 5.55,
|
| 2043 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 2044 |
+
"timeout_reason": "",
|
| 2045 |
+
"capacity_limited": false,
|
| 2046 |
+
"hardware_summary": {
|
| 2047 |
+
"samples": 9,
|
| 2048 |
+
"duration_seconds": 19.323,
|
| 2049 |
+
"gpu_count": 4,
|
| 2050 |
+
"cpu_util_avg_pct": 11.59,
|
| 2051 |
+
"cpu_temp_max_c": 75.75,
|
| 2052 |
+
"gpu_util_avg_pct": 99.92,
|
| 2053 |
+
"gpu_util_max_pct": 100.0,
|
| 2054 |
+
"mem_util_avg_pct": 36.47,
|
| 2055 |
+
"mem_util_max_pct": 55.0,
|
| 2056 |
+
"temp_avg_c": 68.5,
|
| 2057 |
+
"temp_max_c": 84.0,
|
| 2058 |
+
"power_total_avg_w": 1160.0,
|
| 2059 |
+
"power_total_max_w": 1177.71,
|
| 2060 |
+
"power_limit_total_w": 1200.0,
|
| 2061 |
+
"vram_used_avg_mb": 384778.0,
|
| 2062 |
+
"vram_used_max_mb": 384778.0,
|
| 2063 |
+
"vram_total_mb": 391548.0,
|
| 2064 |
+
"vram_used_avg_pct": 98.27,
|
| 2065 |
+
"vram_used_max_pct": 98.27,
|
| 2066 |
+
"pcie_rx_avg_mb_s": 22504.67,
|
| 2067 |
+
"pcie_rx_max_mb_s": 74746.0,
|
| 2068 |
+
"pcie_tx_avg_mb_s": 23453.89,
|
| 2069 |
+
"pcie_tx_max_mb_s": 70866.0
|
| 2070 |
+
}
|
| 2071 |
+
},
|
| 2072 |
+
{
|
| 2073 |
+
"concurrency": 4,
|
| 2074 |
+
"context_tokens": 32768,
|
| 2075 |
+
"benchmark_mode": "duration",
|
| 2076 |
+
"request_count_target": 0,
|
| 2077 |
+
"warmup_request_count": 0,
|
| 2078 |
+
"measurement_seconds": 19.777601,
|
| 2079 |
+
"measurement_wall_seconds": 20.001801,
|
| 2080 |
+
"client_output_tokens": 4295,
|
| 2081 |
+
"server_output_tokens": 4295,
|
| 2082 |
+
"aggregate_source": "openai_continuous_usage",
|
| 2083 |
+
"aggregate_tps": 217.16486009639328,
|
| 2084 |
+
"per_request_avg_tps": 54.29121502409832,
|
| 2085 |
+
"ttft_avg": 1.7167683002503158,
|
| 2086 |
+
"ttft_p50": 1.2517880250234157,
|
| 2087 |
+
"ttft_p90": 2.2701463484205306,
|
| 2088 |
+
"ttft_p99": 3.9549784664250893,
|
| 2089 |
+
"time_to_second_token_avg": 0.03188446167713174,
|
| 2090 |
+
"time_to_second_token_p50": 0.03449397184886038,
|
| 2091 |
+
"time_to_second_token_p90": 0.03738179220817983,
|
| 2092 |
+
"time_to_second_token_p99": 0.039646569741889834,
|
| 2093 |
+
"request_latency_avg": 9.444077699189075,
|
| 2094 |
+
"request_latency_p50": 9.303670058376156,
|
| 2095 |
+
"request_latency_p90": 10.359732999047264,
|
| 2096 |
+
"request_latency_p99": 11.179385469323025,
|
| 2097 |
+
"inter_token_latency_avg": 0.014419010797828693,
|
| 2098 |
+
"inter_token_latency_p50": 0.014763854274074246,
|
| 2099 |
+
"inter_token_latency_p90": 0.015734248097034612,
|
| 2100 |
+
"inter_token_latency_p99": 0.01580286538425228,
|
| 2101 |
+
"output_tps_per_user_avg": 70.02365221779164,
|
| 2102 |
+
"output_tps_per_user_p50": 67.73299041267488,
|
| 2103 |
+
"output_tps_per_user_p90": 73.71836215116325,
|
| 2104 |
+
"output_tps_per_user_p99": 90.54101765838234,
|
| 2105 |
+
"e2e_output_tps_per_user_avg": 54.73059414564044,
|
| 2106 |
+
"e2e_output_tps_per_user_p50": 55.03247278818422,
|
| 2107 |
+
"e2e_output_tps_per_user_p90": 60.046589541051475,
|
| 2108 |
+
"e2e_output_tps_per_user_p99": 65.10818536189939,
|
| 2109 |
+
"chunk_inter_token_latency_avg": 0.042334877835426166,
|
| 2110 |
+
"chunk_inter_token_latency_p50": 0.042865508716204204,
|
| 2111 |
+
"chunk_inter_token_latency_p90": 0.04529354933856685,
|
| 2112 |
+
"chunk_inter_token_latency_p99": 0.0453712748584173,
|
| 2113 |
+
"input_seq_len_avg": 32768.0,
|
| 2114 |
+
"output_seq_len_avg": 512.0,
|
| 2115 |
+
"output_seq_len_p50": 512.0,
|
| 2116 |
+
"output_seq_len_p90": 512.0,
|
| 2117 |
+
"output_seq_len_p99": 512.0,
|
| 2118 |
+
"request_count": 13,
|
| 2119 |
+
"completed_request_count": 10,
|
| 2120 |
+
"request_samples": [
|
| 2121 |
+
{
|
| 2122 |
+
"ttft": 0.6077568109612912,
|
| 2123 |
+
"time_to_second_token": 0.010776221984997392,
|
| 2124 |
+
"latency": 7.796489109983668,
|
| 2125 |
+
"inter_token_latency_avg": 0.014067969274016393,
|
| 2126 |
+
"chunk_inter_token_latency_avg": 0.04016051563699652,
|
| 2127 |
+
"input_tokens": 32768,
|
| 2128 |
+
"output_tokens": 512,
|
| 2129 |
+
"output_tps_per_user": 71.08346489261992,
|
| 2130 |
+
"e2e_output_tps_per_user": 65.67058489754916,
|
| 2131 |
+
"completed": true
|
| 2132 |
+
},
|
| 2133 |
+
{
|
| 2134 |
+
"ttft": 1.2430980298668146,
|
| 2135 |
+
"time_to_second_token": 0.032302224077284336,
|
| 2136 |
+
"latency": 8.787427563918754,
|
| 2137 |
+
"inter_token_latency_avg": 0.014763854274074246,
|
| 2138 |
+
"chunk_inter_token_latency_avg": 0.042865508716204204,
|
| 2139 |
+
"input_tokens": 32768,
|
| 2140 |
+
"output_tokens": 512,
|
| 2141 |
+
"output_tps_per_user": 67.73299041267488,
|
| 2142 |
+
"e2e_output_tps_per_user": 58.2650606535041,
|
| 2143 |
+
"completed": true
|
| 2144 |
+
},
|
| 2145 |
+
{
|
| 2146 |
+
"ttft": 1.24380440893583,
|
| 2147 |
+
"time_to_second_token": 0.03317554807290435,
|
| 2148 |
+
"latency": 9.216824874980375,
|
| 2149 |
+
"inter_token_latency_avg": 0.015602779776995196,
|
| 2150 |
+
"chunk_inter_token_latency_avg": 0.04530125264798037,
|
| 2151 |
+
"input_tokens": 32768,
|
| 2152 |
+
"output_tokens": 512,
|
| 2153 |
+
"output_tps_per_user": 64.09114364828787,
|
| 2154 |
+
"e2e_output_tps_per_user": 55.55058351926104,
|
| 2155 |
+
"completed": true
|
| 2156 |
+
},
|
| 2157 |
+
{
|
| 2158 |
+
"ttft": 1.2160316940862685,
|
| 2159 |
+
"time_to_second_token": 0.03449397184886038,
|
| 2160 |
+
"latency": 0.0,
|
| 2161 |
+
"inter_token_latency_avg": 0.010777328099156248,
|
| 2162 |
+
"chunk_inter_token_latency_avg": 0.03472694609728125,
|
| 2163 |
+
"input_tokens": 32768,
|
| 2164 |
+
"output_tokens": 30,
|
| 2165 |
+
"output_tps_per_user": 92.78737649995917,
|
| 2166 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2167 |
+
"completed": false
|
| 2168 |
+
},
|
| 2169 |
+
{
|
| 2170 |
+
"ttft": 2.2123095341958106,
|
| 2171 |
+
"time_to_second_token": 0.026011183857917786,
|
| 2172 |
+
"latency": 9.824777495115995,
|
| 2173 |
+
"inter_token_latency_avg": 0.014897197575186271,
|
| 2174 |
+
"chunk_inter_token_latency_avg": 0.04349981691954392,
|
| 2175 |
+
"input_tokens": 32768,
|
| 2176 |
+
"output_tokens": 512,
|
| 2177 |
+
"output_tps_per_user": 67.1267193009284,
|
| 2178 |
+
"e2e_output_tps_per_user": 52.1131394837716,
|
| 2179 |
+
"completed": true
|
| 2180 |
+
},
|
| 2181 |
+
{
|
| 2182 |
+
"ttft": 2.2846055519767106,
|
| 2183 |
+
"time_to_second_token": 0.03989882906898856,
|
| 2184 |
+
"latency": 10.258541336050257,
|
| 2185 |
+
"inter_token_latency_avg": 0.015604571006014768,
|
| 2186 |
+
"chunk_inter_token_latency_avg": 0.04333660752213884,
|
| 2187 |
+
"input_tokens": 32768,
|
| 2188 |
+
"output_tokens": 512,
|
| 2189 |
+
"output_tps_per_user": 64.08378670676373,
|
| 2190 |
+
"e2e_output_tps_per_user": 49.909629763906594,
|
| 2191 |
+
"completed": true
|
| 2192 |
+
},
|
| 2193 |
+
{
|
| 2194 |
+
"ttft": 1.214527043979615,
|
| 2195 |
+
"time_to_second_token": 0.03522572107613087,
|
| 2196 |
+
"latency": 0.0,
|
| 2197 |
+
"inter_token_latency_avg": 0.014891711289040101,
|
| 2198 |
+
"chunk_inter_token_latency_avg": 0.0437039353047916,
|
| 2199 |
+
"input_tokens": 32768,
|
| 2200 |
+
"output_tokens": 406,
|
| 2201 |
+
"output_tps_per_user": 67.15144959437758,
|
| 2202 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2203 |
+
"completed": false
|
| 2204 |
+
},
|
| 2205 |
+
{
|
| 2206 |
+
"ttft": 2.212038089055568,
|
| 2207 |
+
"time_to_second_token": 0.025989510817453265,
|
| 2208 |
+
"latency": 9.277765536913648,
|
| 2209 |
+
"inter_token_latency_avg": 0.013827255279565714,
|
| 2210 |
+
"chunk_inter_token_latency_avg": 0.04180903815300639,
|
| 2211 |
+
"input_tokens": 32768,
|
| 2212 |
+
"output_tokens": 512,
|
| 2213 |
+
"output_tps_per_user": 72.32093280853985,
|
| 2214 |
+
"e2e_output_tps_per_user": 55.185701553126606,
|
| 2215 |
+
"completed": true
|
| 2216 |
+
},
|
| 2217 |
+
{
|
| 2218 |
+
"ttft": 1.4273314119782299,
|
| 2219 |
+
"time_to_second_token": 0.0377966680098325,
|
| 2220 |
+
"latency": 8.616380715044215,
|
| 2221 |
+
"inter_token_latency_avg": 0.01406858963418001,
|
| 2222 |
+
"chunk_inter_token_latency_avg": 0.0425387532725798,
|
| 2223 |
+
"input_tokens": 32768,
|
| 2224 |
+
"output_tokens": 512,
|
| 2225 |
+
"output_tps_per_user": 71.08033043841677,
|
| 2226 |
+
"e2e_output_tps_per_user": 59.42170116810729,
|
| 2227 |
+
"completed": true
|
| 2228 |
+
},
|
| 2229 |
+
{
|
| 2230 |
+
"ttft": 1.2517880250234157,
|
| 2231 |
+
"time_to_second_token": 0.03299012500792742,
|
| 2232 |
+
"latency": 9.329574579838663,
|
| 2233 |
+
"inter_token_latency_avg": 0.015807801477133558,
|
| 2234 |
+
"chunk_inter_token_latency_avg": 0.045380823341658695,
|
| 2235 |
+
"input_tokens": 32768,
|
| 2236 |
+
"output_tokens": 512,
|
| 2237 |
+
"output_tps_per_user": 63.25990375363259,
|
| 2238 |
+
"e2e_output_tps_per_user": 54.879244023241846,
|
| 2239 |
+
"completed": true
|
| 2240 |
+
},
|
| 2241 |
+
{
|
| 2242 |
+
"ttft": 4.1827565911225975,
|
| 2243 |
+
"time_to_second_token": 0.03485451405867934,
|
| 2244 |
+
"latency": 11.27045796602033,
|
| 2245 |
+
"inter_token_latency_avg": 0.013870257093733334,
|
| 2246 |
+
"chunk_inter_token_latency_avg": 0.041939061389927416,
|
| 2247 |
+
"input_tokens": 32768,
|
| 2248 |
+
"output_tokens": 512,
|
| 2249 |
+
"output_tps_per_user": 72.09671697086321,
|
| 2250 |
+
"e2e_output_tps_per_user": 45.42850002578825,
|
| 2251 |
+
"completed": true
|
| 2252 |
+
},
|
| 2253 |
+
{
|
| 2254 |
+
"ttft": 2.005770788062364,
|
| 2255 |
+
"time_to_second_token": 0.03572228900156915,
|
| 2256 |
+
"latency": 10.062537814024836,
|
| 2257 |
+
"inter_token_latency_avg": 0.015766667369789572,
|
| 2258 |
+
"chunk_inter_token_latency_avg": 0.04526273610091276,
|
| 2259 |
+
"input_tokens": 32768,
|
| 2260 |
+
"output_tokens": 512,
|
| 2261 |
+
"output_tps_per_user": 63.424944317408176,
|
| 2262 |
+
"e2e_output_tps_per_user": 50.88179636814792,
|
| 2263 |
+
"completed": true
|
| 2264 |
+
},
|
| 2265 |
+
{
|
| 2266 |
+
"ttft": 1.2161699240095913,
|
| 2267 |
+
"time_to_second_token": 0.03526119492016733,
|
| 2268 |
+
"latency": 0.0,
|
| 2269 |
+
"inter_token_latency_avg": 0.013501158222887602,
|
| 2270 |
+
"chunk_inter_token_latency_avg": 0.03982841675751843,
|
| 2271 |
+
"input_tokens": 32768,
|
| 2272 |
+
"output_tokens": 355,
|
| 2273 |
+
"output_tps_per_user": 74.0677194868191,
|
| 2274 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2275 |
+
"completed": false
|
| 2276 |
+
}
|
| 2277 |
+
],
|
| 2278 |
+
"total_tokens": 4295,
|
| 2279 |
+
"wall_time": 29.076672724913806,
|
| 2280 |
+
"num_completed": 4,
|
| 2281 |
+
"num_errors": 0,
|
| 2282 |
+
"server_gen_throughput": 214.67688879215112,
|
| 2283 |
+
"server_utilization": 0.009771077610273626,
|
| 2284 |
+
"server_spec_accept_rate": 0.625,
|
| 2285 |
+
"server_spec_accept_length": 0.0,
|
| 2286 |
+
"avg_running_reqs": 4.0,
|
| 2287 |
+
"max_running_reqs": 4,
|
| 2288 |
+
"effective_concurrency": 4.0,
|
| 2289 |
+
"avg_queue_reqs": 0,
|
| 2290 |
+
"max_queue_reqs": 0,
|
| 2291 |
+
"queue_fraction": 0.0,
|
| 2292 |
+
"underfilled": false,
|
| 2293 |
+
"warmup_timed_out": false,
|
| 2294 |
+
"warmup_duration": 8.573,
|
| 2295 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 2296 |
+
"timeout_reason": "",
|
| 2297 |
+
"capacity_limited": false,
|
| 2298 |
+
"hardware_summary": {
|
| 2299 |
+
"samples": 9,
|
| 2300 |
+
"duration_seconds": 19.242,
|
| 2301 |
+
"gpu_count": 4,
|
| 2302 |
+
"cpu_util_avg_pct": 11.54,
|
| 2303 |
+
"cpu_temp_max_c": 76.75,
|
| 2304 |
+
"gpu_util_avg_pct": 100.0,
|
| 2305 |
+
"gpu_util_max_pct": 100.0,
|
| 2306 |
+
"mem_util_avg_pct": 31.81,
|
| 2307 |
+
"mem_util_max_pct": 46.0,
|
| 2308 |
+
"temp_avg_c": 68.78,
|
| 2309 |
+
"temp_max_c": 84.0,
|
| 2310 |
+
"power_total_avg_w": 1160.18,
|
| 2311 |
+
"power_total_max_w": 1172.54,
|
| 2312 |
+
"power_limit_total_w": 1200.0,
|
| 2313 |
+
"vram_used_avg_mb": 384778.0,
|
| 2314 |
+
"vram_used_max_mb": 384778.0,
|
| 2315 |
+
"vram_total_mb": 391548.0,
|
| 2316 |
+
"vram_used_avg_pct": 98.27,
|
| 2317 |
+
"vram_used_max_pct": 98.27,
|
| 2318 |
+
"pcie_rx_avg_mb_s": 19024.67,
|
| 2319 |
+
"pcie_rx_max_mb_s": 74971.0,
|
| 2320 |
+
"pcie_tx_avg_mb_s": 20175.22,
|
| 2321 |
+
"pcie_tx_max_mb_s": 67212.0
|
| 2322 |
+
}
|
| 2323 |
+
}
|
| 2324 |
+
],
|
| 2325 |
+
"summary_table": {
|
| 2326 |
+
"0": {
|
| 2327 |
+
"1": 181.37569616773123,
|
| 2328 |
+
"2": 257.2872334123046,
|
| 2329 |
+
"4": 308.9773476463454
|
| 2330 |
+
},
|
| 2331 |
+
"8192": {
|
| 2332 |
+
"1": 161.46688548018994,
|
| 2333 |
+
"2": 179.48013878635928,
|
| 2334 |
+
"4": 217.85922059414168
|
| 2335 |
+
},
|
| 2336 |
+
"32768": {
|
| 2337 |
+
"1": 162.9385363574482,
|
| 2338 |
+
"2": 187.47192708215627,
|
| 2339 |
+
"4": 217.16486009639328
|
| 2340 |
+
}
|
| 2341 |
+
},
|
| 2342 |
+
"burst_results": [],
|
| 2343 |
+
"burst_summary_table": {},
|
| 2344 |
+
"methodology": {
|
| 2345 |
+
"prefill": {
|
| 2346 |
+
"name": "Prefill",
|
| 2347 |
+
"present": false,
|
| 2348 |
+
"mode": "skipped",
|
| 2349 |
+
"formula": "prompt_tokens / TTFT",
|
| 2350 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 2351 |
+
},
|
| 2352 |
+
"sustained_decode": {
|
| 2353 |
+
"name": "Sustained Decode",
|
| 2354 |
+
"present": true,
|
| 2355 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 2356 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 2357 |
+
},
|
| 2358 |
+
"burst_e2e_decode": {
|
| 2359 |
+
"name": "Burst / E2E Decode",
|
| 2360 |
+
"present": false,
|
| 2361 |
+
"status": "not run; use --run-burst",
|
| 2362 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 2363 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 2364 |
+
}
|
| 2365 |
+
}
|
| 2366 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap512.log
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0', '8k', '32k'] │
|
| 16 |
+
│ Duration: 20.0s per decode test | Max tokens: 512 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill: skipped | Sustained decode: 9 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
|
| 23 |
+
7,337,984 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: skipped
|
| 26 |
+
Calibrating padding text (run=ftbfeppyynkm, up to 32k)...
|
| 27 |
+
8k: 50,558 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,152 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
Token targeting: /tokenize exact
|
| 30 |
+
Done.
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
llm-decode-bench v0.4.29
|
| 35 |
+
╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
|
| 36 |
+
│ Sustained Decode │
|
| 37 |
+
│ Steady-state decode throughput after the engine has admitted the requested │
|
| 38 |
+
│ concurrency and passed warmup. Use this as the main tuning/regression signal │
|
| 39 |
+
│ for kernels, NCCL, DCP, MTP, and scheduler changes. │
|
| 40 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 41 |
+
Aggregate tok/s + TTFT/ITL
|
| 42 |
+
╭────────────┬─────────────┬──────────────────┬────────────────────╮
|
| 43 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 44 |
+
├────────────┼─────────────┼──────────────────┼────────────────────┤
|
| 45 |
+
│ 0 │ 181.4 84/5 │ 257.3 123/8 │ 309.0 (4/4) 181/12 │
|
| 46 |
+
│ 8k │ 161.5 625/5 │ 179.5 (2/2) 1k/8 │ ∅ (4/4)* 1k/15 │
|
| 47 |
+
│ 32k │ 162.9 628/5 │ 187.5 (2/2) 1k/8 │ 217.2 1k/15 │
|
| 48 |
+
╰────────────┴─────────────┴──────────────────┴────────────────────╯
|
| 49 |
+
Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
|
| 50 |
+
(continuous completion_tokens when the server supports it). Prometheus is kept
|
| 51 |
+
as validation/scheduler data.
|
| 52 |
+
Aggregate source(s): openai_continuous_usage
|
| 53 |
+
∅ = skipped/hidden because the cell does not fit in KV cache; exact deficit is
|
| 54 |
+
kept in JSON timeout_reason
|
| 55 |
+
(X/Y) = avg running / requested concurrency from Prometheus; * =
|
| 56 |
+
capacity-limited or warmup timed out
|
| 57 |
+
Per-Request tok/s
|
| 58 |
+
╭────────────┬───────┬────────────┬────────────╮
|
| 59 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 60 |
+
├────────────┼───────┼──���─────────┼────────────┤
|
| 61 |
+
│ 0 │ 181.4 │ 128.6 │ 77.2 (4/4) │
|
| 62 |
+
│ 8k │ 161.5 │ 89.7 (2/2) │ ∅ (4/4)* │
|
| 63 |
+
│ 32k │ 162.9 │ 93.7 (2/2) │ 54.3 │
|
| 64 |
+
╰────────────┴───────┴────────────┴────────────╯
|
| 65 |
+
Client request latency: p50 / p90 ms
|
| 66 |
+
╭────────────┬───────────┬───────────┬────────────╮
|
| 67 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 68 |
+
├────────────┼───────────┼───────────┼────────────┤
|
| 69 |
+
│ 0 │ 2.8k/2.9k │ 3.9k/4.2k │ 6.4k/6.9k │
|
| 70 |
+
│ 8k │ 3.2k/3.3k │ 5.4k/5.9k │ 9.3k/11.3k │
|
| 71 |
+
│ 32k │ 3.2k/3.2k │ 5.4k/5.6k │ 9.3k/10.4k │
|
| 72 |
+
╰────────────┴───────────┴───────────┴────────────╯
|
| 73 |
+
Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
|
| 74 |
+
coordinate. ITL is computed from observed generated tokens, including streams
|
| 75 |
+
stopped at the measurement boundary; a missing ITL means no stream produced at
|
| 76 |
+
least two measured output tokens. Per-request tok/s and request latency are
|
| 77 |
+
shown in separate per-cell matrices. Completion/sample counts and full
|
| 78 |
+
request-level distributions remain in JSON under request_samples.
|
| 79 |
+
Sustained mode: client latency metrics explain request UX variance; aggregate
|
| 80 |
+
tok/s remains the primary throughput signal.
|
| 81 |
+
ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
|
| 82 |
+
Hardware Summary
|
| 83 |
+
╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
|
| 84 |
+
│ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
|
| 85 |
+
├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
|
| 86 |
+
│ 0 │ │ sust… │ 99/99% │ 44% │ 1151/1… │ 82C │ 76C │ 98… │ 8369/8254 │
|
| 87 |
+
│ … │ │ sust… │ 99/100% │ 40% │ 1153/1… │ 83C │ 76C │ 98… │ 12913/11281 │
|
| 88 |
+
│ … │ │ sust… │ 98/100% │ 38% │ 1150/1… │ 84C │ 76C │ 98… │ 8327/8133 │
|
| 89 |
+
│ 0 │ │ sust… │ 100/100% │ 40% │ 1173/1… │ 84C │ 76C │ 98… │ 11187/11117 │
|
| 90 |
+
│ 0 │ │ sust… │ 100/100% │ 35% │ 1170/1… │ 84C │ 77C │ 98… │ 7813/8054 │
|
| 91 |
+
│ … │ │ sust… │ 100/100% │ 36% │ 1162/1… │ 84C │ 76C │ 98… │ 29259/28011 │
|
| 92 |
+
│ … │ │ sust… │ 100/100% │ 34% │ 1159/1… │ 84C │ 76C │ 98… │ 27724/27474 │
|
| 93 |
+
│ … │ │ sust… │ 100/100% │ 36% │ 1160/1… │ 84C │ 76C │ 98… │ 22505/23454 │
|
| 94 |
+
│ … │ │ sust… │ 100/100% │ 32% │ 1160/1… │ 84C │ 77C │ 98… │ 19025/20175 │
|
| 95 |
+
╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
|
| 96 |
+
╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
|
| 97 |
+
│ avg 1,098 W | max 1,178 W | limit 1,200 W | over 4m 30s | 113 samples │
|
| 98 |
+
╰───────────────────────────────────────────────────────────────────────╯
|
| 99 |
+
Hardware summary is sampled from nvidia-smi during the measured part of each
|
| 100 |
+
cell. Whole-run GPU power is the sampled sum of GPU power draw across the
|
| 101 |
+
complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
|
| 102 |
+
a coarse live diagnostic, not a per-kernel NCCL profiler.
|
| 103 |
+
|
| 104 |
+
╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
|
| 105 |
+
│ Burst / E2E Decode │
|
| 106 |
+
│ Not run. Re-run with --run-burst to append a finite client-facing request │
|
| 107 |
+
│ burst after Sustained Decode. This is intentionally disabled by default │
|
| 108 |
+
│ because it adds another full decode matrix. │
|
| 109 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 110 |
+
|
| 111 |
+
╭────────────────────────────── Primary Summary ───────────────────────────────╮
|
| 112 |
+
│ Primary matrices repeated last so the important numbers are visible without │
|
| 113 |
+
│ scrolling back through diagnostics. │
|
| 114 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 115 |
+
Aggregate decode tok/s
|
| 116 |
+
╭────────────┬───────┬─────────────┬─────────────╮
|
| 117 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 118 |
+
├────────────┼───────┼─────────────┼─────────────┤
|
| 119 |
+
│ 0 │ 181.4 │ 257.3 │ 309.0 (4/4) │
|
| 120 |
+
│ 8k │ 161.5 │ 179.5 (2/2) │ ∅ (4/4)* │
|
| 121 |
+
│ 32k │ 162.9 │ 187.5 (2/2) │ 217.2 │
|
| 122 |
+
╰────────────┴───────┴─────────────┴─────────────╯
|
| 123 |
+
|
| 124 |
+
Results saved to
|
| 125 |
+
<campaign>/candidate-speed-wi
|
| 126 |
+
ndow-01/results-01/decode-warp-quant/rep-1/decode-cap512.json
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-command.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0,8k,32k",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--skip-prefill",
|
| 25 |
+
"--cell-warmup-timeout-seconds",
|
| 26 |
+
"180"
|
| 27 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "b236cf123d887ede3ee9c5ba2a6ef16597fbf3e99450397dcb46488a24dd99b3"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json
ADDED
|
@@ -0,0 +1,1394 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:22:18.084779",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 32768,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0,
|
| 42 |
+
8192,
|
| 43 |
+
32768
|
| 44 |
+
],
|
| 45 |
+
"startup_diagnostics_available": true,
|
| 46 |
+
"nvidia_p2p_override_effective": true,
|
| 47 |
+
"p2pmark_status": "not_run",
|
| 48 |
+
"amd_fabric_status": "not_run"
|
| 49 |
+
},
|
| 50 |
+
"startup_diagnostics": {
|
| 51 |
+
"version": "0.4.29",
|
| 52 |
+
"server_url": "http://127.0.0.1:8001",
|
| 53 |
+
"hostname": "<host>",
|
| 54 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 55 |
+
"env": {},
|
| 56 |
+
"args": {
|
| 57 |
+
"concurrency": "1,2,4",
|
| 58 |
+
"contexts": "0,8k,32k",
|
| 59 |
+
"max_tokens": 8192,
|
| 60 |
+
"duration": 20.0,
|
| 61 |
+
"request_count": 0,
|
| 62 |
+
"run_burst": false,
|
| 63 |
+
"standalone_prefill": false,
|
| 64 |
+
"prefill_only": false,
|
| 65 |
+
"skip_prefill": true,
|
| 66 |
+
"prefill_contexts": "8k,64k,128k",
|
| 67 |
+
"prefill_metric": "client",
|
| 68 |
+
"dcp_size": 0,
|
| 69 |
+
"kv_budget": 0
|
| 70 |
+
},
|
| 71 |
+
"nvidia_p2p_override": {
|
| 72 |
+
"effective": true,
|
| 73 |
+
"configured": true,
|
| 74 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 75 |
+
"params_available": true,
|
| 76 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 77 |
+
"modprobe_available": true,
|
| 78 |
+
"runtime": {
|
| 79 |
+
"ForceP2P": "0x11",
|
| 80 |
+
"RMForceP2PType": "1",
|
| 81 |
+
"RMPcieP2PType": "2",
|
| 82 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 83 |
+
"EnableResizableBar": "1",
|
| 84 |
+
"DmaRemapPeerMmio": "1"
|
| 85 |
+
},
|
| 86 |
+
"expected": {
|
| 87 |
+
"ForceP2P": "0x11",
|
| 88 |
+
"RMForceP2PType": "1",
|
| 89 |
+
"RMPcieP2PType": "2",
|
| 90 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 91 |
+
"EnableResizableBar": "1"
|
| 92 |
+
},
|
| 93 |
+
"missing": [],
|
| 94 |
+
"mismatched": {},
|
| 95 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 96 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 97 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 98 |
+
},
|
| 99 |
+
"p2pmark": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"amd_fabric": {
|
| 103 |
+
"status": "not_run"
|
| 104 |
+
},
|
| 105 |
+
"nvidia_smi_query": {
|
| 106 |
+
"cmd": [
|
| 107 |
+
"nvidia-smi",
|
| 108 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 109 |
+
"--format=csv,noheader,nounits"
|
| 110 |
+
],
|
| 111 |
+
"returncode": 0,
|
| 112 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 113 |
+
"stderr": ""
|
| 114 |
+
},
|
| 115 |
+
"nvidia_smi_topo": {
|
| 116 |
+
"cmd": [
|
| 117 |
+
"nvidia-smi",
|
| 118 |
+
"topo",
|
| 119 |
+
"-m"
|
| 120 |
+
],
|
| 121 |
+
"returncode": 0,
|
| 122 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 123 |
+
"stderr": ""
|
| 124 |
+
}
|
| 125 |
+
},
|
| 126 |
+
"nvidia_p2p_override": {
|
| 127 |
+
"effective": true,
|
| 128 |
+
"configured": true,
|
| 129 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 130 |
+
"params_available": true,
|
| 131 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 132 |
+
"modprobe_available": true,
|
| 133 |
+
"runtime": {
|
| 134 |
+
"ForceP2P": "0x11",
|
| 135 |
+
"RMForceP2PType": "1",
|
| 136 |
+
"RMPcieP2PType": "2",
|
| 137 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 138 |
+
"EnableResizableBar": "1",
|
| 139 |
+
"DmaRemapPeerMmio": "1"
|
| 140 |
+
},
|
| 141 |
+
"expected": {
|
| 142 |
+
"ForceP2P": "0x11",
|
| 143 |
+
"RMForceP2PType": "1",
|
| 144 |
+
"RMPcieP2PType": "2",
|
| 145 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 146 |
+
"EnableResizableBar": "1"
|
| 147 |
+
},
|
| 148 |
+
"missing": [],
|
| 149 |
+
"mismatched": {},
|
| 150 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 151 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 152 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 153 |
+
},
|
| 154 |
+
"p2pmark": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"amd_fabric": {
|
| 158 |
+
"status": "not_run"
|
| 159 |
+
},
|
| 160 |
+
"hardware_run_summary": {
|
| 161 |
+
"samples": 114,
|
| 162 |
+
"duration_seconds": 272.443,
|
| 163 |
+
"gpu_count": 4,
|
| 164 |
+
"cpu_util_avg_pct": 10.97,
|
| 165 |
+
"cpu_temp_max_c": 76.25,
|
| 166 |
+
"gpu_util_avg_pct": 90.62,
|
| 167 |
+
"gpu_util_max_pct": 100.0,
|
| 168 |
+
"mem_util_avg_pct": 35.29,
|
| 169 |
+
"mem_util_max_pct": 57.0,
|
| 170 |
+
"temp_avg_c": 66.66,
|
| 171 |
+
"temp_max_c": 84.0,
|
| 172 |
+
"power_total_avg_w": 1094.77,
|
| 173 |
+
"power_total_max_w": 1178.23,
|
| 174 |
+
"power_limit_total_w": 1200.0,
|
| 175 |
+
"vram_used_avg_mb": 384774.21,
|
| 176 |
+
"vram_used_max_mb": 384778.0,
|
| 177 |
+
"vram_total_mb": 391548.0,
|
| 178 |
+
"vram_used_avg_pct": 98.27,
|
| 179 |
+
"vram_used_max_pct": 98.27,
|
| 180 |
+
"pcie_rx_avg_mb_s": 12350.85,
|
| 181 |
+
"pcie_rx_max_mb_s": 60894.0,
|
| 182 |
+
"pcie_tx_avg_mb_s": 12097.24,
|
| 183 |
+
"pcie_tx_max_mb_s": 58766.0
|
| 184 |
+
},
|
| 185 |
+
"event_log": [
|
| 186 |
+
"02:17:43 benchmark start engine=vllm",
|
| 187 |
+
"02:17:43 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 188 |
+
"02:17:43 startup decode concurrency=1,2,4 contexts=0,8k,32k",
|
| 189 |
+
"02:17:43 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 190 |
+
"02:17:43 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 191 |
+
"02:17:43 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
|
| 192 |
+
"02:17:43 startup model context length: 1,000,000 tokens",
|
| 193 |
+
"02:17:43 startup prefill tests: skipped",
|
| 194 |
+
"02:17:43 startup calibrating padding text run=rumxqbvzafby up_to=32k",
|
| 195 |
+
"02:17:43 startup context 8k: 50,540 chars (8,192 prompt tokens via /tokenize)",
|
| 196 |
+
"02:17:43 startup context 32k: 205,139 chars (32,768 prompt tokens via /tokenize)",
|
| 197 |
+
"02:17:43 startup token targeting: /tokenize exact",
|
| 198 |
+
"02:17:43 startup startup preparation done",
|
| 199 |
+
"02:17:43 hardware monitor interval=2s",
|
| 200 |
+
"02:17:43 decode warmup start",
|
| 201 |
+
"02:17:43 decode warmup start C=1 ctx=32k 3s",
|
| 202 |
+
"02:17:43 cell start C=1 ctx=32k",
|
| 203 |
+
"02:17:51 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 204 |
+
"02:17:54 cell done C=1 ctx=32k 187.9 tok/s",
|
| 205 |
+
"02:17:54 decode warmup done C=1 ctx=32k",
|
| 206 |
+
"02:17:56 cell start C=1 ctx=0",
|
| 207 |
+
"02:18:02 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 208 |
+
"02:18:22 cell done C=1 ctx=0 185.7 tok/s",
|
| 209 |
+
"02:18:24 cell start C=1 ctx=8k",
|
| 210 |
+
"02:18:30 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 211 |
+
"02:18:50 cell done C=1 ctx=8k 166.9 tok/s",
|
| 212 |
+
"02:18:52 cell start C=1 ctx=32k",
|
| 213 |
+
"02:19:01 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 214 |
+
"02:19:21 cell done C=1 ctx=32k 173.8 tok/s",
|
| 215 |
+
"02:19:23 cell start C=2 ctx=0",
|
| 216 |
+
"02:19:28 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 217 |
+
"02:19:49 cell done C=2 ctx=0 246.2 tok/s",
|
| 218 |
+
"02:19:51 cell start C=4 ctx=0",
|
| 219 |
+
"02:19:56 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 220 |
+
"02:20:16 cell done C=4 ctx=0 303.4 tok/s",
|
| 221 |
+
"02:20:18 cell start C=2 ctx=8k",
|
| 222 |
+
"02:20:24 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 223 |
+
"02:20:44 cell done C=2 ctx=8k 236.9 tok/s",
|
| 224 |
+
"02:20:46 cell start C=4 ctx=8k",
|
| 225 |
+
"02:20:57 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 226 |
+
"02:21:17 cell done C=4 ctx=8k 291.1 tok/s",
|
| 227 |
+
"02:21:19 cell start C=2 ctx=32k",
|
| 228 |
+
"02:21:25 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 229 |
+
"02:21:45 cell done C=2 ctx=32k 231.8 tok/s",
|
| 230 |
+
"02:21:47 cell start C=4 ctx=32k",
|
| 231 |
+
"02:21:56 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 232 |
+
"02:22:16 cell done C=4 ctx=32k 303.9 tok/s"
|
| 233 |
+
],
|
| 234 |
+
"prefill": {},
|
| 235 |
+
"results": [
|
| 236 |
+
{
|
| 237 |
+
"concurrency": 1,
|
| 238 |
+
"context_tokens": 0,
|
| 239 |
+
"benchmark_mode": "duration",
|
| 240 |
+
"request_count_target": 0,
|
| 241 |
+
"warmup_request_count": 0,
|
| 242 |
+
"measurement_seconds": 19.990511,
|
| 243 |
+
"measurement_wall_seconds": 20.000618,
|
| 244 |
+
"client_output_tokens": 3712,
|
| 245 |
+
"server_output_tokens": 3712,
|
| 246 |
+
"aggregate_source": "openai_continuous_usage",
|
| 247 |
+
"aggregate_tps": 185.68810232786203,
|
| 248 |
+
"per_request_avg_tps": 185.68810232786203,
|
| 249 |
+
"ttft_avg": 0.07161088101565838,
|
| 250 |
+
"ttft_p50": 0.07161088101565838,
|
| 251 |
+
"ttft_p90": 0.07161088101565838,
|
| 252 |
+
"ttft_p99": 0.07161088101565838,
|
| 253 |
+
"time_to_second_token_avg": 0.01281460584141314,
|
| 254 |
+
"time_to_second_token_p50": 0.01281460584141314,
|
| 255 |
+
"time_to_second_token_p90": 0.01281460584141314,
|
| 256 |
+
"time_to_second_token_p99": 0.01281460584141314,
|
| 257 |
+
"request_latency_avg": 0.0,
|
| 258 |
+
"request_latency_p50": 0.0,
|
| 259 |
+
"request_latency_p90": 0.0,
|
| 260 |
+
"request_latency_p99": 0.0,
|
| 261 |
+
"inter_token_latency_avg": 0.005329198802730078,
|
| 262 |
+
"inter_token_latency_p50": 0.005329198802730078,
|
| 263 |
+
"inter_token_latency_p90": 0.005329198802730078,
|
| 264 |
+
"inter_token_latency_p99": 0.005329198802730078,
|
| 265 |
+
"output_tps_per_user_avg": 187.64546736138146,
|
| 266 |
+
"output_tps_per_user_p50": 187.64546736138146,
|
| 267 |
+
"output_tps_per_user_p90": 187.64546736138146,
|
| 268 |
+
"output_tps_per_user_p99": 187.64546736138146,
|
| 269 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 270 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 271 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 272 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.014062018498253509,
|
| 274 |
+
"chunk_inter_token_latency_p50": 0.014062018498253509,
|
| 275 |
+
"chunk_inter_token_latency_p90": 0.014062018498253509,
|
| 276 |
+
"chunk_inter_token_latency_p99": 0.014062018498253509,
|
| 277 |
+
"input_seq_len_avg": 78.0,
|
| 278 |
+
"output_seq_len_avg": 4777.0,
|
| 279 |
+
"output_seq_len_p50": 4777.0,
|
| 280 |
+
"output_seq_len_p90": 4777.0,
|
| 281 |
+
"output_seq_len_p99": 4777.0,
|
| 282 |
+
"request_count": 1,
|
| 283 |
+
"completed_request_count": 0,
|
| 284 |
+
"request_samples": [
|
| 285 |
+
{
|
| 286 |
+
"ttft": 0.07161088101565838,
|
| 287 |
+
"time_to_second_token": 0.01281460584141314,
|
| 288 |
+
"latency": 0.0,
|
| 289 |
+
"inter_token_latency_avg": 0.005329198802730078,
|
| 290 |
+
"chunk_inter_token_latency_avg": 0.014062018498253509,
|
| 291 |
+
"input_tokens": 78,
|
| 292 |
+
"output_tokens": 4777,
|
| 293 |
+
"output_tps_per_user": 187.64546736138146,
|
| 294 |
+
"e2e_output_tps_per_user": 0.0,
|
| 295 |
+
"completed": false
|
| 296 |
+
}
|
| 297 |
+
],
|
| 298 |
+
"total_tokens": 3712,
|
| 299 |
+
"wall_time": 25.54038013308309,
|
| 300 |
+
"num_completed": 1,
|
| 301 |
+
"num_errors": 0,
|
| 302 |
+
"server_gen_throughput": 185.54062187903287,
|
| 303 |
+
"server_utilization": 0.005862646566164198,
|
| 304 |
+
"server_spec_accept_rate": 0.5648148148148148,
|
| 305 |
+
"server_spec_accept_length": 0.0,
|
| 306 |
+
"avg_running_reqs": 1,
|
| 307 |
+
"max_running_reqs": 1,
|
| 308 |
+
"effective_concurrency": 1,
|
| 309 |
+
"avg_queue_reqs": 0,
|
| 310 |
+
"max_queue_reqs": 0,
|
| 311 |
+
"queue_fraction": 0.0,
|
| 312 |
+
"underfilled": false,
|
| 313 |
+
"warmup_timed_out": false,
|
| 314 |
+
"warmup_duration": 5.533,
|
| 315 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 316 |
+
"timeout_reason": "",
|
| 317 |
+
"capacity_limited": false,
|
| 318 |
+
"hardware_summary": {
|
| 319 |
+
"samples": 9,
|
| 320 |
+
"duration_seconds": 19.316,
|
| 321 |
+
"gpu_count": 4,
|
| 322 |
+
"cpu_util_avg_pct": 11.53,
|
| 323 |
+
"cpu_temp_max_c": 75.25,
|
| 324 |
+
"gpu_util_avg_pct": 99.0,
|
| 325 |
+
"gpu_util_max_pct": 99.0,
|
| 326 |
+
"mem_util_avg_pct": 43.92,
|
| 327 |
+
"mem_util_max_pct": 56.0,
|
| 328 |
+
"temp_avg_c": 65.25,
|
| 329 |
+
"temp_max_c": 80.0,
|
| 330 |
+
"power_total_avg_w": 1149.49,
|
| 331 |
+
"power_total_max_w": 1153.43,
|
| 332 |
+
"power_limit_total_w": 1200.0,
|
| 333 |
+
"vram_used_avg_mb": 384770.0,
|
| 334 |
+
"vram_used_max_mb": 384770.0,
|
| 335 |
+
"vram_total_mb": 391548.0,
|
| 336 |
+
"vram_used_avg_pct": 98.27,
|
| 337 |
+
"vram_used_max_pct": 98.27,
|
| 338 |
+
"pcie_rx_avg_mb_s": 8501.89,
|
| 339 |
+
"pcie_rx_max_mb_s": 8783.0,
|
| 340 |
+
"pcie_tx_avg_mb_s": 8302.22,
|
| 341 |
+
"pcie_tx_max_mb_s": 8578.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
{
|
| 345 |
+
"concurrency": 1,
|
| 346 |
+
"context_tokens": 8192,
|
| 347 |
+
"benchmark_mode": "duration",
|
| 348 |
+
"request_count_target": 0,
|
| 349 |
+
"warmup_request_count": 0,
|
| 350 |
+
"measurement_seconds": 20.000286,
|
| 351 |
+
"measurement_wall_seconds": 20.000336,
|
| 352 |
+
"client_output_tokens": 3338,
|
| 353 |
+
"server_output_tokens": 3338,
|
| 354 |
+
"aggregate_source": "openai_continuous_usage",
|
| 355 |
+
"aggregate_tps": 166.89761277085296,
|
| 356 |
+
"per_request_avg_tps": 166.89761277085296,
|
| 357 |
+
"ttft_avg": 0.5802665068767965,
|
| 358 |
+
"ttft_p50": 0.5802665068767965,
|
| 359 |
+
"ttft_p90": 0.5802665068767965,
|
| 360 |
+
"ttft_p99": 0.5802665068767965,
|
| 361 |
+
"time_to_second_token_avg": 0.016474336152896285,
|
| 362 |
+
"time_to_second_token_p50": 0.016474336152896285,
|
| 363 |
+
"time_to_second_token_p90": 0.016474336152896285,
|
| 364 |
+
"time_to_second_token_p99": 0.016474336152896285,
|
| 365 |
+
"request_latency_avg": 0.0,
|
| 366 |
+
"request_latency_p50": 0.0,
|
| 367 |
+
"request_latency_p90": 0.0,
|
| 368 |
+
"request_latency_p99": 0.0,
|
| 369 |
+
"inter_token_latency_avg": 0.0059050409250968015,
|
| 370 |
+
"inter_token_latency_p50": 0.0059050409250968015,
|
| 371 |
+
"inter_token_latency_p90": 0.0059050409250968015,
|
| 372 |
+
"inter_token_latency_p99": 0.0059050409250968015,
|
| 373 |
+
"output_tps_per_user_avg": 169.34683648845447,
|
| 374 |
+
"output_tps_per_user_p50": 169.34683648845447,
|
| 375 |
+
"output_tps_per_user_p90": 169.34683648845447,
|
| 376 |
+
"output_tps_per_user_p99": 169.34683648845447,
|
| 377 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 378 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 379 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 380 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 381 |
+
"chunk_inter_token_latency_avg": 0.014231043370286765,
|
| 382 |
+
"chunk_inter_token_latency_p50": 0.014231043370286765,
|
| 383 |
+
"chunk_inter_token_latency_p90": 0.014231043370286765,
|
| 384 |
+
"chunk_inter_token_latency_p99": 0.014231043370286765,
|
| 385 |
+
"input_seq_len_avg": 8192.0,
|
| 386 |
+
"output_seq_len_avg": 4057.0,
|
| 387 |
+
"output_seq_len_p50": 4057.0,
|
| 388 |
+
"output_seq_len_p90": 4057.0,
|
| 389 |
+
"output_seq_len_p99": 4057.0,
|
| 390 |
+
"request_count": 1,
|
| 391 |
+
"completed_request_count": 0,
|
| 392 |
+
"request_samples": [
|
| 393 |
+
{
|
| 394 |
+
"ttft": 0.5802665068767965,
|
| 395 |
+
"time_to_second_token": 0.016474336152896285,
|
| 396 |
+
"latency": 0.0,
|
| 397 |
+
"inter_token_latency_avg": 0.0059050409250968015,
|
| 398 |
+
"chunk_inter_token_latency_avg": 0.014231043370286765,
|
| 399 |
+
"input_tokens": 8192,
|
| 400 |
+
"output_tokens": 4057,
|
| 401 |
+
"output_tps_per_user": 169.34683648845447,
|
| 402 |
+
"e2e_output_tps_per_user": 0.0,
|
| 403 |
+
"completed": false
|
| 404 |
+
}
|
| 405 |
+
],
|
| 406 |
+
"total_tokens": 3338,
|
| 407 |
+
"wall_time": 26.075741773936898,
|
| 408 |
+
"num_completed": 1,
|
| 409 |
+
"num_errors": 0,
|
| 410 |
+
"server_gen_throughput": 166.8524947323479,
|
| 411 |
+
"server_utilization": 0.006141820212172022,
|
| 412 |
+
"server_spec_accept_rate": 0.3286384976525822,
|
| 413 |
+
"server_spec_accept_length": 0.0,
|
| 414 |
+
"avg_running_reqs": 1,
|
| 415 |
+
"max_running_reqs": 1,
|
| 416 |
+
"effective_concurrency": 1,
|
| 417 |
+
"avg_queue_reqs": 0,
|
| 418 |
+
"max_queue_reqs": 0,
|
| 419 |
+
"queue_fraction": 0.0,
|
| 420 |
+
"underfilled": false,
|
| 421 |
+
"warmup_timed_out": false,
|
| 422 |
+
"warmup_duration": 6.06,
|
| 423 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 424 |
+
"timeout_reason": "",
|
| 425 |
+
"capacity_limited": false,
|
| 426 |
+
"hardware_summary": {
|
| 427 |
+
"samples": 8,
|
| 428 |
+
"duration_seconds": 16.922,
|
| 429 |
+
"gpu_count": 4,
|
| 430 |
+
"cpu_util_avg_pct": 11.59,
|
| 431 |
+
"cpu_temp_max_c": 75.0,
|
| 432 |
+
"gpu_util_avg_pct": 99.0,
|
| 433 |
+
"gpu_util_max_pct": 99.0,
|
| 434 |
+
"mem_util_avg_pct": 44.31,
|
| 435 |
+
"mem_util_max_pct": 56.0,
|
| 436 |
+
"temp_avg_c": 65.97,
|
| 437 |
+
"temp_max_c": 81.0,
|
| 438 |
+
"power_total_avg_w": 1152.41,
|
| 439 |
+
"power_total_max_w": 1153.89,
|
| 440 |
+
"power_limit_total_w": 1200.0,
|
| 441 |
+
"vram_used_avg_mb": 384770.0,
|
| 442 |
+
"vram_used_max_mb": 384770.0,
|
| 443 |
+
"vram_total_mb": 391548.0,
|
| 444 |
+
"vram_used_avg_pct": 98.27,
|
| 445 |
+
"vram_used_max_pct": 98.27,
|
| 446 |
+
"pcie_rx_avg_mb_s": 8303.25,
|
| 447 |
+
"pcie_rx_max_mb_s": 8596.0,
|
| 448 |
+
"pcie_tx_avg_mb_s": 8386.88,
|
| 449 |
+
"pcie_tx_max_mb_s": 8513.0
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"concurrency": 1,
|
| 454 |
+
"context_tokens": 32768,
|
| 455 |
+
"benchmark_mode": "duration",
|
| 456 |
+
"request_count_target": 0,
|
| 457 |
+
"warmup_request_count": 0,
|
| 458 |
+
"measurement_seconds": 19.989714,
|
| 459 |
+
"measurement_wall_seconds": 20.000821,
|
| 460 |
+
"client_output_tokens": 3475,
|
| 461 |
+
"server_output_tokens": 3475,
|
| 462 |
+
"aggregate_source": "openai_continuous_usage",
|
| 463 |
+
"aggregate_tps": 173.8394059270179,
|
| 464 |
+
"per_request_avg_tps": 173.8394059270179,
|
| 465 |
+
"ttft_avg": 0.5945898578502238,
|
| 466 |
+
"ttft_p50": 0.5945898578502238,
|
| 467 |
+
"ttft_p90": 0.5945898578502238,
|
| 468 |
+
"ttft_p99": 0.5945898578502238,
|
| 469 |
+
"time_to_second_token_avg": 0.011255914112553,
|
| 470 |
+
"time_to_second_token_p50": 0.011255914112553,
|
| 471 |
+
"time_to_second_token_p90": 0.011255914112553,
|
| 472 |
+
"time_to_second_token_p99": 0.011255914112553,
|
| 473 |
+
"request_latency_avg": 0.0,
|
| 474 |
+
"request_latency_p50": 0.0,
|
| 475 |
+
"request_latency_p90": 0.0,
|
| 476 |
+
"request_latency_p99": 0.0,
|
| 477 |
+
"inter_token_latency_avg": 0.005599262254130903,
|
| 478 |
+
"inter_token_latency_p50": 0.005599262254130903,
|
| 479 |
+
"inter_token_latency_p90": 0.005599262254130903,
|
| 480 |
+
"inter_token_latency_p99": 0.005599262254130903,
|
| 481 |
+
"output_tps_per_user_avg": 178.59495673063742,
|
| 482 |
+
"output_tps_per_user_p50": 178.59495673063742,
|
| 483 |
+
"output_tps_per_user_p90": 178.59495673063742,
|
| 484 |
+
"output_tps_per_user_p99": 178.59495673063742,
|
| 485 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 486 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 487 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 488 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 489 |
+
"chunk_inter_token_latency_avg": 0.014177278953883576,
|
| 490 |
+
"chunk_inter_token_latency_p50": 0.014177278953883576,
|
| 491 |
+
"chunk_inter_token_latency_p90": 0.014177278953883576,
|
| 492 |
+
"chunk_inter_token_latency_p99": 0.014177278953883576,
|
| 493 |
+
"input_seq_len_avg": 32768.0,
|
| 494 |
+
"output_seq_len_avg": 4275.0,
|
| 495 |
+
"output_seq_len_p50": 4275.0,
|
| 496 |
+
"output_seq_len_p90": 4275.0,
|
| 497 |
+
"output_seq_len_p99": 4275.0,
|
| 498 |
+
"request_count": 1,
|
| 499 |
+
"completed_request_count": 0,
|
| 500 |
+
"request_samples": [
|
| 501 |
+
{
|
| 502 |
+
"ttft": 0.5945898578502238,
|
| 503 |
+
"time_to_second_token": 0.011255914112553,
|
| 504 |
+
"latency": 0.0,
|
| 505 |
+
"inter_token_latency_avg": 0.005599262254130903,
|
| 506 |
+
"chunk_inter_token_latency_avg": 0.014177278953883576,
|
| 507 |
+
"input_tokens": 32768,
|
| 508 |
+
"output_tokens": 4275,
|
| 509 |
+
"output_tps_per_user": 178.59495673063742,
|
| 510 |
+
"e2e_output_tps_per_user": 0.0,
|
| 511 |
+
"completed": false
|
| 512 |
+
}
|
| 513 |
+
],
|
| 514 |
+
"total_tokens": 3475,
|
| 515 |
+
"wall_time": 29.124992428114638,
|
| 516 |
+
"num_completed": 1,
|
| 517 |
+
"num_errors": 0,
|
| 518 |
+
"server_gen_throughput": 173.70034429070287,
|
| 519 |
+
"server_utilization": 0.006979341150195384,
|
| 520 |
+
"server_spec_accept_rate": 0.38967136150234744,
|
| 521 |
+
"server_spec_accept_length": 0.0,
|
| 522 |
+
"avg_running_reqs": 1,
|
| 523 |
+
"max_running_reqs": 1,
|
| 524 |
+
"effective_concurrency": 1,
|
| 525 |
+
"avg_queue_reqs": 0,
|
| 526 |
+
"max_queue_reqs": 0,
|
| 527 |
+
"queue_fraction": 0.0,
|
| 528 |
+
"underfilled": false,
|
| 529 |
+
"warmup_timed_out": false,
|
| 530 |
+
"warmup_duration": 9.119,
|
| 531 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 532 |
+
"timeout_reason": "",
|
| 533 |
+
"capacity_limited": false,
|
| 534 |
+
"hardware_summary": {
|
| 535 |
+
"samples": 8,
|
| 536 |
+
"duration_seconds": 16.869,
|
| 537 |
+
"gpu_count": 4,
|
| 538 |
+
"cpu_util_avg_pct": 11.56,
|
| 539 |
+
"cpu_temp_max_c": 74.88,
|
| 540 |
+
"gpu_util_avg_pct": 99.0,
|
| 541 |
+
"gpu_util_max_pct": 99.0,
|
| 542 |
+
"mem_util_avg_pct": 43.78,
|
| 543 |
+
"mem_util_max_pct": 55.0,
|
| 544 |
+
"temp_avg_c": 66.56,
|
| 545 |
+
"temp_max_c": 82.0,
|
| 546 |
+
"power_total_avg_w": 1153.2,
|
| 547 |
+
"power_total_max_w": 1153.93,
|
| 548 |
+
"power_limit_total_w": 1200.0,
|
| 549 |
+
"vram_used_avg_mb": 384770.0,
|
| 550 |
+
"vram_used_max_mb": 384770.0,
|
| 551 |
+
"vram_total_mb": 391548.0,
|
| 552 |
+
"vram_used_avg_pct": 98.27,
|
| 553 |
+
"vram_used_max_pct": 98.27,
|
| 554 |
+
"pcie_rx_avg_mb_s": 8358.0,
|
| 555 |
+
"pcie_rx_max_mb_s": 8664.0,
|
| 556 |
+
"pcie_tx_avg_mb_s": 8305.5,
|
| 557 |
+
"pcie_tx_max_mb_s": 8516.0
|
| 558 |
+
}
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"concurrency": 2,
|
| 562 |
+
"context_tokens": 0,
|
| 563 |
+
"benchmark_mode": "duration",
|
| 564 |
+
"request_count_target": 0,
|
| 565 |
+
"warmup_request_count": 0,
|
| 566 |
+
"measurement_seconds": 19.998344,
|
| 567 |
+
"measurement_wall_seconds": 20.000475,
|
| 568 |
+
"client_output_tokens": 4924,
|
| 569 |
+
"server_output_tokens": 4924,
|
| 570 |
+
"aggregate_source": "openai_continuous_usage",
|
| 571 |
+
"aggregate_tps": 246.22038564672286,
|
| 572 |
+
"per_request_avg_tps": 123.11019282336143,
|
| 573 |
+
"ttft_avg": 0.11326407559681684,
|
| 574 |
+
"ttft_p50": 0.11326407559681684,
|
| 575 |
+
"ttft_p90": 0.14774754794780165,
|
| 576 |
+
"ttft_p99": 0.15550632922677324,
|
| 577 |
+
"time_to_second_token_avg": 0.015398422023281455,
|
| 578 |
+
"time_to_second_token_p50": 0.015398422023281455,
|
| 579 |
+
"time_to_second_token_p90": 0.017018200410529972,
|
| 580 |
+
"time_to_second_token_p99": 0.01738265054766089,
|
| 581 |
+
"request_latency_avg": 0.0,
|
| 582 |
+
"request_latency_p50": 0.0,
|
| 583 |
+
"request_latency_p90": 0.0,
|
| 584 |
+
"request_latency_p99": 0.0,
|
| 585 |
+
"inter_token_latency_avg": 0.007996202209365196,
|
| 586 |
+
"inter_token_latency_p50": 0.007996202209365196,
|
| 587 |
+
"inter_token_latency_p90": 0.008062418200969368,
|
| 588 |
+
"inter_token_latency_p99": 0.008077316799080308,
|
| 589 |
+
"output_tps_per_user_avg": 125.07276978039832,
|
| 590 |
+
"output_tps_per_user_p50": 125.07276978039832,
|
| 591 |
+
"output_tps_per_user_p90": 126.10848864503505,
|
| 592 |
+
"output_tps_per_user_p99": 126.34152538957832,
|
| 593 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 594 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 595 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 596 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 597 |
+
"chunk_inter_token_latency_avg": 0.02072697419690292,
|
| 598 |
+
"chunk_inter_token_latency_p50": 0.02072697419690292,
|
| 599 |
+
"chunk_inter_token_latency_p90": 0.020734919510937286,
|
| 600 |
+
"chunk_inter_token_latency_p99": 0.02073670720659502,
|
| 601 |
+
"input_seq_len_avg": 78.0,
|
| 602 |
+
"output_seq_len_avg": 3180.5,
|
| 603 |
+
"output_seq_len_p50": 3180.5,
|
| 604 |
+
"output_seq_len_p90": 3202.5,
|
| 605 |
+
"output_seq_len_p99": 3207.45,
|
| 606 |
+
"request_count": 2,
|
| 607 |
+
"completed_request_count": 0,
|
| 608 |
+
"request_samples": [
|
| 609 |
+
{
|
| 610 |
+
"ttft": 0.07015973515808582,
|
| 611 |
+
"time_to_second_token": 0.01337369903922081,
|
| 612 |
+
"latency": 0.0,
|
| 613 |
+
"inter_token_latency_avg": 0.008078972198870412,
|
| 614 |
+
"chunk_inter_token_latency_avg": 0.020736905839445877,
|
| 615 |
+
"input_tokens": 78,
|
| 616 |
+
"output_tokens": 3153,
|
| 617 |
+
"output_tps_per_user": 123.77812119960238,
|
| 618 |
+
"e2e_output_tps_per_user": 0.0,
|
| 619 |
+
"completed": false
|
| 620 |
+
},
|
| 621 |
+
{
|
| 622 |
+
"ttft": 0.15636841603554785,
|
| 623 |
+
"time_to_second_token": 0.0174231450073421,
|
| 624 |
+
"latency": 0.0,
|
| 625 |
+
"inter_token_latency_avg": 0.007913432219859979,
|
| 626 |
+
"chunk_inter_token_latency_avg": 0.02071704255435996,
|
| 627 |
+
"input_tokens": 78,
|
| 628 |
+
"output_tokens": 3208,
|
| 629 |
+
"output_tps_per_user": 126.36741836119424,
|
| 630 |
+
"e2e_output_tps_per_user": 0.0,
|
| 631 |
+
"completed": false
|
| 632 |
+
}
|
| 633 |
+
],
|
| 634 |
+
"total_tokens": 4924,
|
| 635 |
+
"wall_time": 25.577398049877957,
|
| 636 |
+
"num_completed": 2,
|
| 637 |
+
"num_errors": 0,
|
| 638 |
+
"server_gen_throughput": 246.13329155338783,
|
| 639 |
+
"server_utilization": 0.005862646566164198,
|
| 640 |
+
"server_spec_accept_rate": 0.47959183673469385,
|
| 641 |
+
"server_spec_accept_length": 0.0,
|
| 642 |
+
"avg_running_reqs": 2,
|
| 643 |
+
"max_running_reqs": 2,
|
| 644 |
+
"effective_concurrency": 2,
|
| 645 |
+
"avg_queue_reqs": 0,
|
| 646 |
+
"max_queue_reqs": 0,
|
| 647 |
+
"queue_fraction": 0.0,
|
| 648 |
+
"underfilled": false,
|
| 649 |
+
"warmup_timed_out": false,
|
| 650 |
+
"warmup_duration": 5.537,
|
| 651 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 652 |
+
"timeout_reason": "",
|
| 653 |
+
"capacity_limited": false,
|
| 654 |
+
"hardware_summary": {
|
| 655 |
+
"samples": 9,
|
| 656 |
+
"duration_seconds": 19.327,
|
| 657 |
+
"gpu_count": 4,
|
| 658 |
+
"cpu_util_avg_pct": 11.58,
|
| 659 |
+
"cpu_temp_max_c": 75.75,
|
| 660 |
+
"gpu_util_avg_pct": 100.0,
|
| 661 |
+
"gpu_util_max_pct": 100.0,
|
| 662 |
+
"mem_util_avg_pct": 40.89,
|
| 663 |
+
"mem_util_max_pct": 50.0,
|
| 664 |
+
"temp_avg_c": 67.36,
|
| 665 |
+
"temp_max_c": 83.0,
|
| 666 |
+
"power_total_avg_w": 1176.5,
|
| 667 |
+
"power_total_max_w": 1178.11,
|
| 668 |
+
"power_limit_total_w": 1200.0,
|
| 669 |
+
"vram_used_avg_mb": 384770.0,
|
| 670 |
+
"vram_used_max_mb": 384770.0,
|
| 671 |
+
"vram_total_mb": 391548.0,
|
| 672 |
+
"vram_used_avg_pct": 98.27,
|
| 673 |
+
"vram_used_max_pct": 98.27,
|
| 674 |
+
"pcie_rx_avg_mb_s": 11309.78,
|
| 675 |
+
"pcie_rx_max_mb_s": 11397.0,
|
| 676 |
+
"pcie_tx_avg_mb_s": 11179.56,
|
| 677 |
+
"pcie_tx_max_mb_s": 11268.0
|
| 678 |
+
}
|
| 679 |
+
},
|
| 680 |
+
{
|
| 681 |
+
"concurrency": 4,
|
| 682 |
+
"context_tokens": 0,
|
| 683 |
+
"benchmark_mode": "duration",
|
| 684 |
+
"request_count_target": 0,
|
| 685 |
+
"warmup_request_count": 0,
|
| 686 |
+
"measurement_seconds": 19.993715,
|
| 687 |
+
"measurement_wall_seconds": 20.000812,
|
| 688 |
+
"client_output_tokens": 6067,
|
| 689 |
+
"server_output_tokens": 6067,
|
| 690 |
+
"aggregate_source": "openai_continuous_usage",
|
| 691 |
+
"aggregate_tps": 303.4453614548168,
|
| 692 |
+
"per_request_avg_tps": 75.8613403637042,
|
| 693 |
+
"ttft_avg": 0.15646358660887927,
|
| 694 |
+
"ttft_p50": 0.18395527265965939,
|
| 695 |
+
"ttft_p90": 0.18397697466425597,
|
| 696 |
+
"ttft_p99": 0.1839802248729393,
|
| 697 |
+
"time_to_second_token_avg": 0.025640497857239097,
|
| 698 |
+
"time_to_second_token_p50": 0.029807523358613253,
|
| 699 |
+
"time_to_second_token_p90": 0.029867575969547033,
|
| 700 |
+
"time_to_second_token_p99": 0.029868205869570376,
|
| 701 |
+
"request_latency_avg": 0.0,
|
| 702 |
+
"request_latency_p50": 0.0,
|
| 703 |
+
"request_latency_p90": 0.0,
|
| 704 |
+
"request_latency_p99": 0.0,
|
| 705 |
+
"inter_token_latency_avg": 0.012890471746641244,
|
| 706 |
+
"inter_token_latency_p50": 0.01291707436819987,
|
| 707 |
+
"inter_token_latency_p90": 0.013106307107593229,
|
| 708 |
+
"inter_token_latency_p99": 0.013131838856624441,
|
| 709 |
+
"output_tps_per_user_avg": 77.59775588315449,
|
| 710 |
+
"output_tps_per_user_p50": 77.42393606470978,
|
| 711 |
+
"output_tps_per_user_p90": 79.03458812531339,
|
| 712 |
+
"output_tps_per_user_p99": 79.37137995360604,
|
| 713 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 714 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 715 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 716 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 717 |
+
"chunk_inter_token_latency_avg": 0.033961205786559916,
|
| 718 |
+
"chunk_inter_token_latency_p50": 0.03394124054211881,
|
| 719 |
+
"chunk_inter_token_latency_p90": 0.03400282599744794,
|
| 720 |
+
"chunk_inter_token_latency_p99": 0.034024420723019345,
|
| 721 |
+
"input_seq_len_avg": 78.0,
|
| 722 |
+
"output_seq_len_avg": 1970.25,
|
| 723 |
+
"output_seq_len_p50": 1968.0,
|
| 724 |
+
"output_seq_len_p90": 2007.1,
|
| 725 |
+
"output_seq_len_p99": 2013.31,
|
| 726 |
+
"request_count": 4,
|
| 727 |
+
"completed_request_count": 0,
|
| 728 |
+
"request_samples": [
|
| 729 |
+
{
|
| 730 |
+
"ttft": 0.07396321510896087,
|
| 731 |
+
"time_to_second_token": 0.013078668853268027,
|
| 732 |
+
"latency": 0.0,
|
| 733 |
+
"inter_token_latency_avg": 0.012794035052220763,
|
| 734 |
+
"chunk_inter_token_latency_avg": 0.03394683967189242,
|
| 735 |
+
"input_tokens": 78,
|
| 736 |
+
"output_tokens": 1991,
|
| 737 |
+
"output_tps_per_user": 78.16142412603614,
|
| 738 |
+
"e2e_output_tps_per_user": 0.0,
|
| 739 |
+
"completed": false
|
| 740 |
+
},
|
| 741 |
+
{
|
| 742 |
+
"ttft": 0.18394199712201953,
|
| 743 |
+
"time_to_second_token": 0.029868275858461857,
|
| 744 |
+
"latency": 0.0,
|
| 745 |
+
"inter_token_latency_avg": 0.013040113684178978,
|
| 746 |
+
"chunk_inter_token_latency_avg": 0.034026820136971725,
|
| 747 |
+
"input_tokens": 78,
|
| 748 |
+
"output_tokens": 1945,
|
| 749 |
+
"output_tps_per_user": 76.68644800338343,
|
| 750 |
+
"e2e_output_tps_per_user": 0.0,
|
| 751 |
+
"completed": false
|
| 752 |
+
},
|
| 753 |
+
{
|
| 754 |
+
"ttft": 0.18398058600723743,
|
| 755 |
+
"time_to_second_token": 0.029865942895412445,
|
| 756 |
+
"latency": 0.0,
|
| 757 |
+
"inter_token_latency_avg": 0.013134675717627909,
|
| 758 |
+
"chunk_inter_token_latency_avg": 0.0339356414123452,
|
| 759 |
+
"input_tokens": 78,
|
| 760 |
+
"output_tokens": 1931,
|
| 761 |
+
"output_tps_per_user": 76.13435013533761,
|
| 762 |
+
"e2e_output_tps_per_user": 0.0,
|
| 763 |
+
"completed": false
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"ttft": 0.18396854819729924,
|
| 767 |
+
"time_to_second_token": 0.02974910382181406,
|
| 768 |
+
"latency": 0.0,
|
| 769 |
+
"inter_token_latency_avg": 0.012593062532537325,
|
| 770 |
+
"chunk_inter_token_latency_avg": 0.033935521925030306,
|
| 771 |
+
"input_tokens": 78,
|
| 772 |
+
"output_tokens": 2014,
|
| 773 |
+
"output_tps_per_user": 79.40880126786078,
|
| 774 |
+
"e2e_output_tps_per_user": 0.0,
|
| 775 |
+
"completed": false
|
| 776 |
+
}
|
| 777 |
+
],
|
| 778 |
+
"total_tokens": 6067,
|
| 779 |
+
"wall_time": 25.569467527093366,
|
| 780 |
+
"num_completed": 4,
|
| 781 |
+
"num_errors": 0,
|
| 782 |
+
"server_gen_throughput": 303.24736236685254,
|
| 783 |
+
"server_utilization": 0.02345058626465657,
|
| 784 |
+
"server_spec_accept_rate": 0.5431034482758621,
|
| 785 |
+
"server_spec_accept_length": 0.0,
|
| 786 |
+
"avg_running_reqs": 4,
|
| 787 |
+
"max_running_reqs": 4,
|
| 788 |
+
"effective_concurrency": 4,
|
| 789 |
+
"avg_queue_reqs": 0,
|
| 790 |
+
"max_queue_reqs": 0,
|
| 791 |
+
"queue_fraction": 0.0,
|
| 792 |
+
"underfilled": false,
|
| 793 |
+
"warmup_timed_out": false,
|
| 794 |
+
"warmup_duration": 5.541,
|
| 795 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 796 |
+
"timeout_reason": "",
|
| 797 |
+
"capacity_limited": false,
|
| 798 |
+
"hardware_summary": {
|
| 799 |
+
"samples": 8,
|
| 800 |
+
"duration_seconds": 16.826,
|
| 801 |
+
"gpu_count": 4,
|
| 802 |
+
"cpu_util_avg_pct": 11.47,
|
| 803 |
+
"cpu_temp_max_c": 75.88,
|
| 804 |
+
"gpu_util_avg_pct": 100.0,
|
| 805 |
+
"gpu_util_max_pct": 100.0,
|
| 806 |
+
"mem_util_avg_pct": 35.31,
|
| 807 |
+
"mem_util_max_pct": 43.0,
|
| 808 |
+
"temp_avg_c": 68.16,
|
| 809 |
+
"temp_max_c": 83.0,
|
| 810 |
+
"power_total_avg_w": 1172.73,
|
| 811 |
+
"power_total_max_w": 1173.45,
|
| 812 |
+
"power_limit_total_w": 1200.0,
|
| 813 |
+
"vram_used_avg_mb": 384778.0,
|
| 814 |
+
"vram_used_max_mb": 384778.0,
|
| 815 |
+
"vram_total_mb": 391548.0,
|
| 816 |
+
"vram_used_avg_pct": 98.27,
|
| 817 |
+
"vram_used_max_pct": 98.27,
|
| 818 |
+
"pcie_rx_avg_mb_s": 7893.62,
|
| 819 |
+
"pcie_rx_max_mb_s": 8279.0,
|
| 820 |
+
"pcie_tx_avg_mb_s": 7755.88,
|
| 821 |
+
"pcie_tx_max_mb_s": 8153.0
|
| 822 |
+
}
|
| 823 |
+
},
|
| 824 |
+
{
|
| 825 |
+
"concurrency": 2,
|
| 826 |
+
"context_tokens": 8192,
|
| 827 |
+
"benchmark_mode": "duration",
|
| 828 |
+
"request_count_target": 0,
|
| 829 |
+
"warmup_request_count": 0,
|
| 830 |
+
"measurement_seconds": 19.988454,
|
| 831 |
+
"measurement_wall_seconds": 20.00057,
|
| 832 |
+
"client_output_tokens": 4735,
|
| 833 |
+
"server_output_tokens": 4735,
|
| 834 |
+
"aggregate_source": "openai_continuous_usage",
|
| 835 |
+
"aggregate_tps": 236.8867590379722,
|
| 836 |
+
"per_request_avg_tps": 118.4433795189861,
|
| 837 |
+
"ttft_avg": 0.9413391120033339,
|
| 838 |
+
"ttft_p50": 0.9413391120033339,
|
| 839 |
+
"ttft_p90": 1.2234342327574268,
|
| 840 |
+
"ttft_p99": 1.2869056349270978,
|
| 841 |
+
"time_to_second_token_avg": 0.021329568000510335,
|
| 842 |
+
"time_to_second_token_p50": 0.021329568000510335,
|
| 843 |
+
"time_to_second_token_p90": 0.02503214799799025,
|
| 844 |
+
"time_to_second_token_p99": 0.02586522849742323,
|
| 845 |
+
"request_latency_avg": 0.0,
|
| 846 |
+
"request_latency_p50": 0.0,
|
| 847 |
+
"request_latency_p90": 0.0,
|
| 848 |
+
"request_latency_p99": 0.0,
|
| 849 |
+
"inter_token_latency_avg": 0.008409807194832745,
|
| 850 |
+
"inter_token_latency_p50": 0.008409807194832745,
|
| 851 |
+
"inter_token_latency_p90": 0.00844567690517398,
|
| 852 |
+
"inter_token_latency_p99": 0.00845374759000076,
|
| 853 |
+
"output_tps_per_user_avg": 118.91217038040975,
|
| 854 |
+
"output_tps_per_user_p50": 118.91217038040975,
|
| 855 |
+
"output_tps_per_user_p90": 119.41935740726734,
|
| 856 |
+
"output_tps_per_user_p99": 119.5334744883103,
|
| 857 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 858 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 859 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 860 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 861 |
+
"chunk_inter_token_latency_avg": 0.02114018666727937,
|
| 862 |
+
"chunk_inter_token_latency_p50": 0.02114018666727937,
|
| 863 |
+
"chunk_inter_token_latency_p90": 0.02137045316589387,
|
| 864 |
+
"chunk_inter_token_latency_p99": 0.021422263128082132,
|
| 865 |
+
"input_seq_len_avg": 8192.0,
|
| 866 |
+
"output_seq_len_avg": 2805.0,
|
| 867 |
+
"output_seq_len_p50": 2805.0,
|
| 868 |
+
"output_seq_len_p90": 2826.6,
|
| 869 |
+
"output_seq_len_p99": 2831.46,
|
| 870 |
+
"request_count": 2,
|
| 871 |
+
"completed_request_count": 0,
|
| 872 |
+
"request_samples": [
|
| 873 |
+
{
|
| 874 |
+
"ttft": 0.5887202110607177,
|
| 875 |
+
"time_to_second_token": 0.01670134300366044,
|
| 876 |
+
"latency": 0.0,
|
| 877 |
+
"inter_token_latency_avg": 0.00845464433275929,
|
| 878 |
+
"chunk_inter_token_latency_avg": 0.021428019790547495,
|
| 879 |
+
"input_tokens": 8192,
|
| 880 |
+
"output_tokens": 2832,
|
| 881 |
+
"output_tps_per_user": 118.27818659683774,
|
| 882 |
+
"e2e_output_tps_per_user": 0.0,
|
| 883 |
+
"completed": false
|
| 884 |
+
},
|
| 885 |
+
{
|
| 886 |
+
"ttft": 1.29395801294595,
|
| 887 |
+
"time_to_second_token": 0.02595779299736023,
|
| 888 |
+
"latency": 0.0,
|
| 889 |
+
"inter_token_latency_avg": 0.008364970056906203,
|
| 890 |
+
"chunk_inter_token_latency_avg": 0.020852353544011243,
|
| 891 |
+
"input_tokens": 8192,
|
| 892 |
+
"output_tokens": 2778,
|
| 893 |
+
"output_tps_per_user": 119.54615416398174,
|
| 894 |
+
"e2e_output_tps_per_user": 0.0,
|
| 895 |
+
"completed": false
|
| 896 |
+
}
|
| 897 |
+
],
|
| 898 |
+
"total_tokens": 4735,
|
| 899 |
+
"wall_time": 25.56432215613313,
|
| 900 |
+
"num_completed": 2,
|
| 901 |
+
"num_errors": 0,
|
| 902 |
+
"server_gen_throughput": 236.6879632349728,
|
| 903 |
+
"server_utilization": 0.012283640424343933,
|
| 904 |
+
"server_spec_accept_rate": 0.4791666666666667,
|
| 905 |
+
"server_spec_accept_length": 0.0,
|
| 906 |
+
"avg_running_reqs": 2,
|
| 907 |
+
"max_running_reqs": 2,
|
| 908 |
+
"effective_concurrency": 2,
|
| 909 |
+
"avg_queue_reqs": 0,
|
| 910 |
+
"max_queue_reqs": 0,
|
| 911 |
+
"queue_fraction": 0.0,
|
| 912 |
+
"underfilled": false,
|
| 913 |
+
"warmup_timed_out": false,
|
| 914 |
+
"warmup_duration": 5.554,
|
| 915 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 916 |
+
"timeout_reason": "",
|
| 917 |
+
"capacity_limited": false,
|
| 918 |
+
"hardware_summary": {
|
| 919 |
+
"samples": 8,
|
| 920 |
+
"duration_seconds": 16.882,
|
| 921 |
+
"gpu_count": 4,
|
| 922 |
+
"cpu_util_avg_pct": 11.54,
|
| 923 |
+
"cpu_temp_max_c": 75.62,
|
| 924 |
+
"gpu_util_avg_pct": 100.0,
|
| 925 |
+
"gpu_util_max_pct": 100.0,
|
| 926 |
+
"mem_util_avg_pct": 40.84,
|
| 927 |
+
"mem_util_max_pct": 50.0,
|
| 928 |
+
"temp_avg_c": 67.75,
|
| 929 |
+
"temp_max_c": 83.0,
|
| 930 |
+
"power_total_avg_w": 1177.72,
|
| 931 |
+
"power_total_max_w": 1178.14,
|
| 932 |
+
"power_limit_total_w": 1200.0,
|
| 933 |
+
"vram_used_avg_mb": 384778.0,
|
| 934 |
+
"vram_used_max_mb": 384778.0,
|
| 935 |
+
"vram_total_mb": 391548.0,
|
| 936 |
+
"vram_used_avg_pct": 98.27,
|
| 937 |
+
"vram_used_max_pct": 98.27,
|
| 938 |
+
"pcie_rx_avg_mb_s": 11234.38,
|
| 939 |
+
"pcie_rx_max_mb_s": 11286.0,
|
| 940 |
+
"pcie_tx_avg_mb_s": 11126.88,
|
| 941 |
+
"pcie_tx_max_mb_s": 11190.0
|
| 942 |
+
}
|
| 943 |
+
},
|
| 944 |
+
{
|
| 945 |
+
"concurrency": 4,
|
| 946 |
+
"context_tokens": 8192,
|
| 947 |
+
"benchmark_mode": "duration",
|
| 948 |
+
"request_count_target": 0,
|
| 949 |
+
"warmup_request_count": 0,
|
| 950 |
+
"measurement_seconds": 19.988009,
|
| 951 |
+
"measurement_wall_seconds": 20.000154,
|
| 952 |
+
"client_output_tokens": 5819,
|
| 953 |
+
"server_output_tokens": 5819,
|
| 954 |
+
"aggregate_source": "openai_continuous_usage",
|
| 955 |
+
"aggregate_tps": 291.1245424388467,
|
| 956 |
+
"per_request_avg_tps": 72.78113560971167,
|
| 957 |
+
"ttft_avg": 3.0831757867126726,
|
| 958 |
+
"ttft_p50": 2.1997361014364287,
|
| 959 |
+
"ttft_p90": 5.800122947827914,
|
| 960 |
+
"ttft_p99": 7.188826314958277,
|
| 961 |
+
"time_to_second_token_avg": 0.03507792699383572,
|
| 962 |
+
"time_to_second_token_p50": 0.0406363804358989,
|
| 963 |
+
"time_to_second_token_p90": 0.0417290014680475,
|
| 964 |
+
"time_to_second_token_p99": 0.041733411899767814,
|
| 965 |
+
"request_latency_avg": 0.0,
|
| 966 |
+
"request_latency_p50": 0.0,
|
| 967 |
+
"request_latency_p90": 0.0,
|
| 968 |
+
"request_latency_p99": 0.0,
|
| 969 |
+
"inter_token_latency_avg": 0.014156019616279113,
|
| 970 |
+
"inter_token_latency_p50": 0.014055135648646407,
|
| 971 |
+
"inter_token_latency_p90": 0.01461799139621679,
|
| 972 |
+
"inter_token_latency_p99": 0.014785406939632685,
|
| 973 |
+
"output_tps_per_user_avg": 70.69962462194486,
|
| 974 |
+
"output_tps_per_user_p50": 71.15434738255996,
|
| 975 |
+
"output_tps_per_user_p90": 72.6003158823885,
|
| 976 |
+
"output_tps_per_user_p99": 72.90651063203018,
|
| 977 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 978 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 979 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 980 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 981 |
+
"chunk_inter_token_latency_avg": 0.03618710908102364,
|
| 982 |
+
"chunk_inter_token_latency_p50": 0.03679464568694915,
|
| 983 |
+
"chunk_inter_token_latency_p90": 0.03703470265231493,
|
| 984 |
+
"chunk_inter_token_latency_p99": 0.03705366284251491,
|
| 985 |
+
"input_seq_len_avg": 8192.0,
|
| 986 |
+
"output_seq_len_avg": 1940.0,
|
| 987 |
+
"output_seq_len_p50": 2013.5,
|
| 988 |
+
"output_seq_len_p90": 2034.4,
|
| 989 |
+
"output_seq_len_p99": 2037.64,
|
| 990 |
+
"request_count": 4,
|
| 991 |
+
"completed_request_count": 0,
|
| 992 |
+
"request_samples": [
|
| 993 |
+
{
|
| 994 |
+
"ttft": 0.5901042548939586,
|
| 995 |
+
"time_to_second_token": 0.01730504515580833,
|
| 996 |
+
"latency": 0.0,
|
| 997 |
+
"inter_token_latency_avg": 0.014804008666678895,
|
| 998 |
+
"chunk_inter_token_latency_avg": 0.03705576953031491,
|
| 999 |
+
"input_tokens": 8192,
|
| 1000 |
+
"output_tokens": 2026,
|
| 1001 |
+
"output_tps_per_user": 67.54927145178024,
|
| 1002 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1003 |
+
"completed": false
|
| 1004 |
+
},
|
| 1005 |
+
{
|
| 1006 |
+
"ttft": 2.1997808848973364,
|
| 1007 |
+
"time_to_second_token": 0.04173390194773674,
|
| 1008 |
+
"latency": 0.0,
|
| 1009 |
+
"inter_token_latency_avg": 0.01418395109847188,
|
| 1010 |
+
"chunk_inter_token_latency_avg": 0.03660374477025001,
|
| 1011 |
+
"input_tokens": 8192,
|
| 1012 |
+
"output_tokens": 2001,
|
| 1013 |
+
"output_tps_per_user": 70.50221712254323,
|
| 1014 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1015 |
+
"completed": false
|
| 1016 |
+
},
|
| 1017 |
+
{
|
| 1018 |
+
"ttft": 2.199691317975521,
|
| 1019 |
+
"time_to_second_token": 0.04171756701543927,
|
| 1020 |
+
"latency": 0.0,
|
| 1021 |
+
"inter_token_latency_avg": 0.013926320198820936,
|
| 1022 |
+
"chunk_inter_token_latency_avg": 0.0369855466036483,
|
| 1023 |
+
"input_tokens": 8192,
|
| 1024 |
+
"output_tokens": 2038,
|
| 1025 |
+
"output_tps_per_user": 71.80647764257671,
|
| 1026 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1027 |
+
"completed": false
|
| 1028 |
+
},
|
| 1029 |
+
{
|
| 1030 |
+
"ttft": 7.343126689083874,
|
| 1031 |
+
"time_to_second_token": 0.03955519385635853,
|
| 1032 |
+
"latency": 0.0,
|
| 1033 |
+
"inter_token_latency_avg": 0.013709798501144739,
|
| 1034 |
+
"chunk_inter_token_latency_avg": 0.03410337541988133,
|
| 1035 |
+
"input_tokens": 8192,
|
| 1036 |
+
"output_tokens": 1695,
|
| 1037 |
+
"output_tps_per_user": 72.94053227087926,
|
| 1038 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1039 |
+
"completed": false
|
| 1040 |
+
}
|
| 1041 |
+
],
|
| 1042 |
+
"total_tokens": 5819,
|
| 1043 |
+
"wall_time": 31.621210746001452,
|
| 1044 |
+
"num_completed": 4,
|
| 1045 |
+
"num_errors": 0,
|
| 1046 |
+
"server_gen_throughput": 290.8514927893891,
|
| 1047 |
+
"server_utilization": 0.024567280848687867,
|
| 1048 |
+
"server_spec_accept_rate": 0.5,
|
| 1049 |
+
"server_spec_accept_length": 0.0,
|
| 1050 |
+
"avg_running_reqs": 4,
|
| 1051 |
+
"max_running_reqs": 4,
|
| 1052 |
+
"effective_concurrency": 4,
|
| 1053 |
+
"avg_queue_reqs": 0,
|
| 1054 |
+
"max_queue_reqs": 0,
|
| 1055 |
+
"queue_fraction": 0.0,
|
| 1056 |
+
"underfilled": false,
|
| 1057 |
+
"warmup_timed_out": false,
|
| 1058 |
+
"warmup_duration": 11.597,
|
| 1059 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1060 |
+
"timeout_reason": "",
|
| 1061 |
+
"capacity_limited": false,
|
| 1062 |
+
"hardware_summary": {
|
| 1063 |
+
"samples": 8,
|
| 1064 |
+
"duration_seconds": 16.872,
|
| 1065 |
+
"gpu_count": 4,
|
| 1066 |
+
"cpu_util_avg_pct": 11.55,
|
| 1067 |
+
"cpu_temp_max_c": 75.88,
|
| 1068 |
+
"gpu_util_avg_pct": 100.0,
|
| 1069 |
+
"gpu_util_max_pct": 100.0,
|
| 1070 |
+
"mem_util_avg_pct": 34.75,
|
| 1071 |
+
"mem_util_max_pct": 43.0,
|
| 1072 |
+
"temp_avg_c": 68.12,
|
| 1073 |
+
"temp_max_c": 83.0,
|
| 1074 |
+
"power_total_avg_w": 1173.04,
|
| 1075 |
+
"power_total_max_w": 1173.65,
|
| 1076 |
+
"power_limit_total_w": 1200.0,
|
| 1077 |
+
"vram_used_avg_mb": 384778.0,
|
| 1078 |
+
"vram_used_max_mb": 384778.0,
|
| 1079 |
+
"vram_total_mb": 391548.0,
|
| 1080 |
+
"vram_used_avg_pct": 98.27,
|
| 1081 |
+
"vram_used_max_pct": 98.27,
|
| 1082 |
+
"pcie_rx_avg_mb_s": 8010.0,
|
| 1083 |
+
"pcie_rx_max_mb_s": 8332.0,
|
| 1084 |
+
"pcie_tx_avg_mb_s": 7806.5,
|
| 1085 |
+
"pcie_tx_max_mb_s": 8129.0
|
| 1086 |
+
}
|
| 1087 |
+
},
|
| 1088 |
+
{
|
| 1089 |
+
"concurrency": 2,
|
| 1090 |
+
"context_tokens": 32768,
|
| 1091 |
+
"benchmark_mode": "duration",
|
| 1092 |
+
"request_count_target": 0,
|
| 1093 |
+
"warmup_request_count": 0,
|
| 1094 |
+
"measurement_seconds": 19.989635,
|
| 1095 |
+
"measurement_wall_seconds": 20.000757,
|
| 1096 |
+
"client_output_tokens": 4633,
|
| 1097 |
+
"server_output_tokens": 4633,
|
| 1098 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1099 |
+
"aggregate_tps": 231.7701102106262,
|
| 1100 |
+
"per_request_avg_tps": 115.8850551053131,
|
| 1101 |
+
"ttft_avg": 0.9621059750206769,
|
| 1102 |
+
"ttft_p50": 0.9621059750206769,
|
| 1103 |
+
"ttft_p90": 1.2434888477437198,
|
| 1104 |
+
"ttft_p99": 1.3067999941064046,
|
| 1105 |
+
"time_to_second_token_avg": 0.01357436552643776,
|
| 1106 |
+
"time_to_second_token_p50": 0.01357436552643776,
|
| 1107 |
+
"time_to_second_token_p90": 0.01572811957448721,
|
| 1108 |
+
"time_to_second_token_p99": 0.016212714235298336,
|
| 1109 |
+
"request_latency_avg": 0.0,
|
| 1110 |
+
"request_latency_p50": 0.0,
|
| 1111 |
+
"request_latency_p90": 0.0,
|
| 1112 |
+
"request_latency_p99": 0.0,
|
| 1113 |
+
"inter_token_latency_avg": 0.008537162788289372,
|
| 1114 |
+
"inter_token_latency_p50": 0.008537162788289372,
|
| 1115 |
+
"inter_token_latency_p90": 0.00858358805811946,
|
| 1116 |
+
"inter_token_latency_p99": 0.00859403374383123,
|
| 1117 |
+
"output_tps_per_user_avg": 117.14034665810209,
|
| 1118 |
+
"output_tps_per_user_p50": 117.14034665810209,
|
| 1119 |
+
"output_tps_per_user_p90": 117.77735831366668,
|
| 1120 |
+
"output_tps_per_user_p99": 117.92068593616872,
|
| 1121 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1122 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1123 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1124 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1125 |
+
"chunk_inter_token_latency_avg": 0.021224602006348896,
|
| 1126 |
+
"chunk_inter_token_latency_p50": 0.021224602006348896,
|
| 1127 |
+
"chunk_inter_token_latency_p90": 0.021463160367322105,
|
| 1128 |
+
"chunk_inter_token_latency_p99": 0.021516835998541078,
|
| 1129 |
+
"input_seq_len_avg": 32768.0,
|
| 1130 |
+
"output_seq_len_avg": 2760.5,
|
| 1131 |
+
"output_seq_len_p50": 2760.5,
|
| 1132 |
+
"output_seq_len_p90": 2778.5,
|
| 1133 |
+
"output_seq_len_p99": 2782.55,
|
| 1134 |
+
"request_count": 2,
|
| 1135 |
+
"completed_request_count": 0,
|
| 1136 |
+
"request_samples": [
|
| 1137 |
+
{
|
| 1138 |
+
"ttft": 0.6103773841168731,
|
| 1139 |
+
"time_to_second_token": 0.010882172966375947,
|
| 1140 |
+
"latency": 0.0,
|
| 1141 |
+
"inter_token_latency_avg": 0.008595194375576983,
|
| 1142 |
+
"chunk_inter_token_latency_avg": 0.021522799957565408,
|
| 1143 |
+
"input_tokens": 32768,
|
| 1144 |
+
"output_tokens": 2783,
|
| 1145 |
+
"output_tps_per_user": 116.34408208864636,
|
| 1146 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1147 |
+
"completed": false
|
| 1148 |
+
},
|
| 1149 |
+
{
|
| 1150 |
+
"ttft": 1.3138345659244806,
|
| 1151 |
+
"time_to_second_token": 0.016266558086499572,
|
| 1152 |
+
"latency": 0.0,
|
| 1153 |
+
"inter_token_latency_avg": 0.00847913120100176,
|
| 1154 |
+
"chunk_inter_token_latency_avg": 0.020926404055132387,
|
| 1155 |
+
"input_tokens": 32768,
|
| 1156 |
+
"output_tokens": 2738,
|
| 1157 |
+
"output_tps_per_user": 117.93661122755783,
|
| 1158 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1159 |
+
"completed": false
|
| 1160 |
+
}
|
| 1161 |
+
],
|
| 1162 |
+
"total_tokens": 4633,
|
| 1163 |
+
"wall_time": 25.564056790899485,
|
| 1164 |
+
"num_completed": 2,
|
| 1165 |
+
"num_errors": 0,
|
| 1166 |
+
"server_gen_throughput": 231.5835038411289,
|
| 1167 |
+
"server_utilization": 0.013121161362367406,
|
| 1168 |
+
"server_spec_accept_rate": 0.5,
|
| 1169 |
+
"server_spec_accept_length": 0.0,
|
| 1170 |
+
"avg_running_reqs": 2,
|
| 1171 |
+
"max_running_reqs": 2,
|
| 1172 |
+
"effective_concurrency": 2,
|
| 1173 |
+
"avg_queue_reqs": 0,
|
| 1174 |
+
"max_queue_reqs": 0,
|
| 1175 |
+
"queue_fraction": 0.0,
|
| 1176 |
+
"underfilled": false,
|
| 1177 |
+
"warmup_timed_out": false,
|
| 1178 |
+
"warmup_duration": 5.553,
|
| 1179 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1180 |
+
"timeout_reason": "",
|
| 1181 |
+
"capacity_limited": false,
|
| 1182 |
+
"hardware_summary": {
|
| 1183 |
+
"samples": 8,
|
| 1184 |
+
"duration_seconds": 16.876,
|
| 1185 |
+
"gpu_count": 4,
|
| 1186 |
+
"cpu_util_avg_pct": 11.51,
|
| 1187 |
+
"cpu_temp_max_c": 75.25,
|
| 1188 |
+
"gpu_util_avg_pct": 100.0,
|
| 1189 |
+
"gpu_util_max_pct": 100.0,
|
| 1190 |
+
"mem_util_avg_pct": 40.84,
|
| 1191 |
+
"mem_util_max_pct": 51.0,
|
| 1192 |
+
"temp_avg_c": 67.97,
|
| 1193 |
+
"temp_max_c": 83.0,
|
| 1194 |
+
"power_total_avg_w": 1177.56,
|
| 1195 |
+
"power_total_max_w": 1178.23,
|
| 1196 |
+
"power_limit_total_w": 1200.0,
|
| 1197 |
+
"vram_used_avg_mb": 384778.0,
|
| 1198 |
+
"vram_used_max_mb": 384778.0,
|
| 1199 |
+
"vram_total_mb": 391548.0,
|
| 1200 |
+
"vram_used_avg_pct": 98.27,
|
| 1201 |
+
"vram_used_max_pct": 98.27,
|
| 1202 |
+
"pcie_rx_avg_mb_s": 11116.0,
|
| 1203 |
+
"pcie_rx_max_mb_s": 11361.0,
|
| 1204 |
+
"pcie_tx_avg_mb_s": 11002.38,
|
| 1205 |
+
"pcie_tx_max_mb_s": 11184.0
|
| 1206 |
+
}
|
| 1207 |
+
},
|
| 1208 |
+
{
|
| 1209 |
+
"concurrency": 4,
|
| 1210 |
+
"context_tokens": 32768,
|
| 1211 |
+
"benchmark_mode": "duration",
|
| 1212 |
+
"request_count_target": 0,
|
| 1213 |
+
"warmup_request_count": 0,
|
| 1214 |
+
"measurement_seconds": 19.997266,
|
| 1215 |
+
"measurement_wall_seconds": 20.000362,
|
| 1216 |
+
"client_output_tokens": 6078,
|
| 1217 |
+
"server_output_tokens": 6078,
|
| 1218 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1219 |
+
"aggregate_tps": 303.94154153181165,
|
| 1220 |
+
"per_request_avg_tps": 75.98538538295291,
|
| 1221 |
+
"ttft_avg": 2.3015168351703323,
|
| 1222 |
+
"ttft_p50": 2.210984098375775,
|
| 1223 |
+
"ttft_p90": 3.5852746270596985,
|
| 1224 |
+
"ttft_p99": 4.115192952086217,
|
| 1225 |
+
"time_to_second_token_avg": 0.023911484284326434,
|
| 1226 |
+
"time_to_second_token_p50": 0.024445773102343082,
|
| 1227 |
+
"time_to_second_token_p90": 0.03160055840853602,
|
| 1228 |
+
"time_to_second_token_p99": 0.03434951798757538,
|
| 1229 |
+
"request_latency_avg": 0.0,
|
| 1230 |
+
"request_latency_p50": 0.0,
|
| 1231 |
+
"request_latency_p90": 0.0,
|
| 1232 |
+
"request_latency_p99": 0.0,
|
| 1233 |
+
"inter_token_latency_avg": 0.013270635558348034,
|
| 1234 |
+
"inter_token_latency_p50": 0.01317594037834293,
|
| 1235 |
+
"inter_token_latency_p90": 0.013953309896230092,
|
| 1236 |
+
"inter_token_latency_p99": 0.014080148007312025,
|
| 1237 |
+
"output_tps_per_user_avg": 75.5138067713092,
|
| 1238 |
+
"output_tps_per_user_p50": 75.98396361855555,
|
| 1239 |
+
"output_tps_per_user_p90": 78.96660843355386,
|
| 1240 |
+
"output_tps_per_user_p99": 79.11936279225931,
|
| 1241 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1242 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1243 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1244 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1245 |
+
"chunk_inter_token_latency_avg": 0.03530397237485847,
|
| 1246 |
+
"chunk_inter_token_latency_p50": 0.03529481439149047,
|
| 1247 |
+
"chunk_inter_token_latency_p90": 0.03563218072342417,
|
| 1248 |
+
"chunk_inter_token_latency_p99": 0.03572440514104369,
|
| 1249 |
+
"input_seq_len_avg": 32768.0,
|
| 1250 |
+
"output_seq_len_avg": 1907.25,
|
| 1251 |
+
"output_seq_len_p50": 1856.0,
|
| 1252 |
+
"output_seq_len_p90": 2040.9,
|
| 1253 |
+
"output_seq_len_p99": 2110.29,
|
| 1254 |
+
"request_count": 4,
|
| 1255 |
+
"completed_request_count": 0,
|
| 1256 |
+
"request_samples": [
|
| 1257 |
+
{
|
| 1258 |
+
"ttft": 0.6100263779517263,
|
| 1259 |
+
"time_to_second_token": 0.012099432991817594,
|
| 1260 |
+
"latency": 0.0,
|
| 1261 |
+
"inter_token_latency_avg": 0.012727410407705224,
|
| 1262 |
+
"chunk_inter_token_latency_avg": 0.03573465229855697,
|
| 1263 |
+
"input_tokens": 32768,
|
| 1264 |
+
"output_tokens": 2118,
|
| 1265 |
+
"output_tps_per_user": 78.57057861468788,
|
| 1266 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1267 |
+
"completed": false
|
| 1268 |
+
},
|
| 1269 |
+
{
|
| 1270 |
+
"ttft": 2.2114123029168695,
|
| 1271 |
+
"time_to_second_token": 0.024417920038104057,
|
| 1272 |
+
"latency": 0.0,
|
| 1273 |
+
"inter_token_latency_avg": 0.014094241130765572,
|
| 1274 |
+
"chunk_inter_token_latency_avg": 0.035393080381447624,
|
| 1275 |
+
"input_tokens": 32768,
|
| 1276 |
+
"output_tokens": 1799,
|
| 1277 |
+
"output_tps_per_user": 70.95096434934358,
|
| 1278 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1279 |
+
"completed": false
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"ttft": 2.2105558938346803,
|
| 1283 |
+
"time_to_second_token": 0.024473626166582108,
|
| 1284 |
+
"latency": 0.0,
|
| 1285 |
+
"inter_token_latency_avg": 0.013624470348980637,
|
| 1286 |
+
"chunk_inter_token_latency_avg": 0.035196548401533315,
|
| 1287 |
+
"input_tokens": 32768,
|
| 1288 |
+
"output_tokens": 1861,
|
| 1289 |
+
"output_tps_per_user": 73.39734862242322,
|
| 1290 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1291 |
+
"completed": false
|
| 1292 |
+
},
|
| 1293 |
+
{
|
| 1294 |
+
"ttft": 4.174072765978053,
|
| 1295 |
+
"time_to_second_token": 0.03465495794080198,
|
| 1296 |
+
"latency": 0.0,
|
| 1297 |
+
"inter_token_latency_avg": 0.012636420345940702,
|
| 1298 |
+
"chunk_inter_token_latency_avg": 0.03489160841789597,
|
| 1299 |
+
"input_tokens": 32768,
|
| 1300 |
+
"output_tokens": 1851,
|
| 1301 |
+
"output_tps_per_user": 79.13633549878213,
|
| 1302 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1303 |
+
"completed": false
|
| 1304 |
+
}
|
| 1305 |
+
],
|
| 1306 |
+
"total_tokens": 6078,
|
| 1307 |
+
"wall_time": 28.609053145861253,
|
| 1308 |
+
"num_completed": 4,
|
| 1309 |
+
"num_errors": 0,
|
| 1310 |
+
"server_gen_throughput": 303.824900318235,
|
| 1311 |
+
"server_utilization": 0.02540480178671134,
|
| 1312 |
+
"server_spec_accept_rate": 0.5316091954022989,
|
| 1313 |
+
"server_spec_accept_length": 0.0,
|
| 1314 |
+
"avg_running_reqs": 4,
|
| 1315 |
+
"max_running_reqs": 4,
|
| 1316 |
+
"effective_concurrency": 4,
|
| 1317 |
+
"avg_queue_reqs": 0,
|
| 1318 |
+
"max_queue_reqs": 0,
|
| 1319 |
+
"queue_fraction": 0.0,
|
| 1320 |
+
"underfilled": false,
|
| 1321 |
+
"warmup_timed_out": false,
|
| 1322 |
+
"warmup_duration": 8.576,
|
| 1323 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1324 |
+
"timeout_reason": "",
|
| 1325 |
+
"capacity_limited": false,
|
| 1326 |
+
"hardware_summary": {
|
| 1327 |
+
"samples": 9,
|
| 1328 |
+
"duration_seconds": 19.268,
|
| 1329 |
+
"gpu_count": 4,
|
| 1330 |
+
"cpu_util_avg_pct": 11.48,
|
| 1331 |
+
"cpu_temp_max_c": 76.25,
|
| 1332 |
+
"gpu_util_avg_pct": 100.0,
|
| 1333 |
+
"gpu_util_max_pct": 100.0,
|
| 1334 |
+
"mem_util_avg_pct": 35.08,
|
| 1335 |
+
"mem_util_max_pct": 44.0,
|
| 1336 |
+
"temp_avg_c": 68.44,
|
| 1337 |
+
"temp_max_c": 84.0,
|
| 1338 |
+
"power_total_avg_w": 1172.67,
|
| 1339 |
+
"power_total_max_w": 1174.13,
|
| 1340 |
+
"power_limit_total_w": 1200.0,
|
| 1341 |
+
"vram_used_avg_mb": 384778.0,
|
| 1342 |
+
"vram_used_max_mb": 384778.0,
|
| 1343 |
+
"vram_total_mb": 391548.0,
|
| 1344 |
+
"vram_used_avg_pct": 98.27,
|
| 1345 |
+
"vram_used_max_pct": 98.27,
|
| 1346 |
+
"pcie_rx_avg_mb_s": 7773.78,
|
| 1347 |
+
"pcie_rx_max_mb_s": 8029.0,
|
| 1348 |
+
"pcie_tx_avg_mb_s": 7899.0,
|
| 1349 |
+
"pcie_tx_max_mb_s": 8283.0
|
| 1350 |
+
}
|
| 1351 |
+
}
|
| 1352 |
+
],
|
| 1353 |
+
"summary_table": {
|
| 1354 |
+
"0": {
|
| 1355 |
+
"1": 185.68810232786203,
|
| 1356 |
+
"2": 246.22038564672286,
|
| 1357 |
+
"4": 303.4453614548168
|
| 1358 |
+
},
|
| 1359 |
+
"8192": {
|
| 1360 |
+
"1": 166.89761277085296,
|
| 1361 |
+
"2": 236.8867590379722,
|
| 1362 |
+
"4": 291.1245424388467
|
| 1363 |
+
},
|
| 1364 |
+
"32768": {
|
| 1365 |
+
"1": 173.8394059270179,
|
| 1366 |
+
"2": 231.7701102106262,
|
| 1367 |
+
"4": 303.94154153181165
|
| 1368 |
+
}
|
| 1369 |
+
},
|
| 1370 |
+
"burst_results": [],
|
| 1371 |
+
"burst_summary_table": {},
|
| 1372 |
+
"methodology": {
|
| 1373 |
+
"prefill": {
|
| 1374 |
+
"name": "Prefill",
|
| 1375 |
+
"present": false,
|
| 1376 |
+
"mode": "skipped",
|
| 1377 |
+
"formula": "prompt_tokens / TTFT",
|
| 1378 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 1379 |
+
},
|
| 1380 |
+
"sustained_decode": {
|
| 1381 |
+
"name": "Sustained Decode",
|
| 1382 |
+
"present": true,
|
| 1383 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 1384 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 1385 |
+
},
|
| 1386 |
+
"burst_e2e_decode": {
|
| 1387 |
+
"name": "Burst / E2E Decode",
|
| 1388 |
+
"present": false,
|
| 1389 |
+
"status": "not run; use --run-burst",
|
| 1390 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 1391 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 1392 |
+
}
|
| 1393 |
+
}
|
| 1394 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/decode-cap8192.log
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0', '8k', '32k'] │
|
| 16 |
+
│ Duration: 20.0s per decode test | Max tokens: 8192 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill: skipped | Sustained decode: 9 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
|
| 23 |
+
7,337,984 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: skipped
|
| 26 |
+
Calibrating padding text (run=rumxqbvzafby, up to 32k)...
|
| 27 |
+
8k: 50,540 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,139 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
Token targeting: /tokenize exact
|
| 30 |
+
Done.
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
llm-decode-bench v0.4.29
|
| 35 |
+
╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
|
| 36 |
+
│ Sustained Decode │
|
| 37 |
+
│ Steady-state decode throughput after the engine has admitted the requested │
|
| 38 |
+
│ concurrency and passed warmup. Use this as the main tuning/regression signal │
|
| 39 |
+
│ for kernels, NCCL, DCP, MTP, and scheduler changes. │
|
| 40 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 41 |
+
Aggregate tok/s + TTFT/ITL
|
| 42 |
+
╭────────────┬─────────────┬─────────────┬──────────────╮
|
| 43 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 44 |
+
├────────────┼─────────────┼─────────────┼──────────────┤
|
| 45 |
+
│ 0 │ 185.7 72/5 │ 246.2 113/8 │ 303.4 184/13 │
|
| 46 |
+
│ 8k │ 166.9 580/6 │ 236.9 941/8 │ 291.1 2k/14 │
|
| 47 |
+
│ 32k │ 173.8 595/6 │ 231.8 962/9 │ 303.9 2k/13 │
|
| 48 |
+
╰────────────┴─────────────┴─────────────┴──────────────╯
|
| 49 |
+
Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
|
| 50 |
+
(continuous completion_tokens when the server supports it). Prometheus is kept
|
| 51 |
+
as validation/scheduler data.
|
| 52 |
+
Aggregate source(s): openai_continuous_usage
|
| 53 |
+
Per-Request tok/s
|
| 54 |
+
╭────────────┬───────┬───────┬──────╮
|
| 55 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 56 |
+
├────────────┼───────┼───────┼──────┤
|
| 57 |
+
│ 0 │ 185.7 │ 123.1 │ 75.9 │
|
| 58 |
+
│ 8k │ 166.9 │ 118.4 │ 72.8 │
|
| 59 |
+
│ 32k │ 173.8 │ 115.9 │ 76.0 │
|
| 60 |
+
╰────────────┴───────┴───────┴──────╯
|
| 61 |
+
Client request latency: p50 /
|
| 62 |
+
p90 ms
|
| 63 |
+
╭────────────┬─────┬─────┬─────╮
|
| 64 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 65 |
+
├────────────┼─────┼─────┼─────┤
|
| 66 |
+
│ 0 │ —/— │ —/— │ —/— │
|
| 67 |
+
│ 8k │ —/— │ —/— │ —/— │
|
| 68 |
+
│ 32k │ —/— │ —/— │ —/— │
|
| 69 |
+
╰────────────┴─────┴─────┴─────╯
|
| 70 |
+
Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
|
| 71 |
+
coordinate. ITL is computed from observed generated tokens, including streams
|
| 72 |
+
stopped at the measurement boundary; a missing ITL means no stream produced at
|
| 73 |
+
least two measured output tokens. Per-request tok/s and request latency are
|
| 74 |
+
shown in separate per-cell matrices. Completion/sample counts and full
|
| 75 |
+
request-level distributions remain in JSON under request_samples.
|
| 76 |
+
Sustained mode: client latency metrics explain request UX variance; aggregate
|
| 77 |
+
tok/s remains the primary throughput signal.
|
| 78 |
+
ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
|
| 79 |
+
Hardware Summary
|
| 80 |
+
╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
|
| 81 |
+
│ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
|
| 82 |
+
├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
|
| 83 |
+
│ 0 │ │ sust… │ 99/99% │ 44% │ 1149/1… │ 80C │ 75C │ 98… │ 8502/8302 │
|
| 84 |
+
│ … │ │ sust… │ 99/99% │ 44% │ 1152/1… │ 81C │ 75C │ 98… │ 8303/8387 │
|
| 85 |
+
│ … │ │ sust… │ 99/99% │ 44% │ 1153/1… │ 82C │ 75C │ 98… │ 8358/8306 │
|
| 86 |
+
│ 0 │ │ sust… │ 100/100% │ 41% │ 1176/1… │ 83C │ 76C │ 98… │ 11310/11180 │
|
| 87 |
+
│ 0 │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 83C │ 76C │ 98… │ 7894/7756 │
|
| 88 |
+
│ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 83C │ 76C │ 98… │ 11234/11127 │
|
| 89 |
+
│ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 83C │ 76C │ 98… │ 8010/7806 │
|
| 90 |
+
│ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 83C │ 75C │ 98… │ 11116/11002 │
|
| 91 |
+
│ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 76C │ 98… │ 7774/7899 │
|
| 92 |
+
╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
|
| 93 |
+
╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
|
| 94 |
+
│ avg 1,095 W | max 1,178 W | limit 1,200 W | over 4m 32s | 114 samples │
|
| 95 |
+
╰───────────────────────────────────────────────────────────────────────╯
|
| 96 |
+
Hardware summary is sampled from nvidia-smi during the measured part of each
|
| 97 |
+
cell. Whole-run GPU power is the sampled sum of GPU power draw across the
|
| 98 |
+
complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
|
| 99 |
+
a coarse live diagnostic, not a per-kernel NCCL profiler.
|
| 100 |
+
|
| 101 |
+
╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
|
| 102 |
+
│ Burst / E2E Decode │
|
| 103 |
+
│ Not run. Re-run with --run-burst to append a finite client-facing request │
|
| 104 |
+
│ burst after Sustained Decode. This is intentionally disabled by default │
|
| 105 |
+
│ because it adds another full decode matrix. │
|
| 106 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 107 |
+
|
| 108 |
+
╭────────────────────────────── Primary Summary ───────────────────────────────╮
|
| 109 |
+
│ Primary matrices repeated last so the important numbers are visible without │
|
| 110 |
+
│ scrolling back through diagnostics. │
|
| 111 |
+
╰─────────────────────────────────────��────────────────────────────────────────╯
|
| 112 |
+
Aggregate decode tok/s
|
| 113 |
+
╭────────────┬───────┬───────┬───────╮
|
| 114 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 115 |
+
├────────────┼───────┼───────┼───────┤
|
| 116 |
+
│ 0 │ 185.7 │ 246.2 │ 303.4 │
|
| 117 |
+
│ 8k │ 166.9 │ 236.9 │ 291.1 │
|
| 118 |
+
│ 32k │ 173.8 │ 231.8 │ 303.9 │
|
| 119 |
+
╰────────────┴───────┴───────┴───────╯
|
| 120 |
+
|
| 121 |
+
Results saved to
|
| 122 |
+
<campaign>/candidate-speed-wi
|
| 123 |
+
ndow-01/results-01/decode-warp-quant/rep-1/decode-cap8192.json
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-command.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--prefill-only",
|
| 25 |
+
"--prefill-contexts",
|
| 26 |
+
"8k,32k,64k,128k",
|
| 27 |
+
"--prefill-duration",
|
| 28 |
+
"20",
|
| 29 |
+
"--cell-warmup-timeout-seconds",
|
| 30 |
+
"180"
|
| 31 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "67ce4838168f0c3c85e10e8484e79c3aa9878c8aa48713fceae52113000a0cf5"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.json
ADDED
|
@@ -0,0 +1,396 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:17:41.737509",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "standalone_cold",
|
| 15 |
+
"standalone_prefill": true,
|
| 16 |
+
"prefill_only": true,
|
| 17 |
+
"skip_prefill": false,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 0,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0
|
| 42 |
+
],
|
| 43 |
+
"startup_diagnostics_available": true,
|
| 44 |
+
"nvidia_p2p_override_effective": true,
|
| 45 |
+
"p2pmark_status": "not_run",
|
| 46 |
+
"amd_fabric_status": "not_run"
|
| 47 |
+
},
|
| 48 |
+
"startup_diagnostics": {
|
| 49 |
+
"version": "0.4.29",
|
| 50 |
+
"server_url": "http://127.0.0.1:8001",
|
| 51 |
+
"hostname": "<host>",
|
| 52 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 53 |
+
"env": {},
|
| 54 |
+
"args": {
|
| 55 |
+
"concurrency": "1,2,4",
|
| 56 |
+
"contexts": "0",
|
| 57 |
+
"max_tokens": 8192,
|
| 58 |
+
"duration": 20.0,
|
| 59 |
+
"request_count": 0,
|
| 60 |
+
"run_burst": false,
|
| 61 |
+
"standalone_prefill": true,
|
| 62 |
+
"prefill_only": true,
|
| 63 |
+
"skip_prefill": false,
|
| 64 |
+
"prefill_contexts": "8k,32k,64k,128k",
|
| 65 |
+
"prefill_metric": "client",
|
| 66 |
+
"dcp_size": 0,
|
| 67 |
+
"kv_budget": 0
|
| 68 |
+
},
|
| 69 |
+
"nvidia_p2p_override": {
|
| 70 |
+
"effective": true,
|
| 71 |
+
"configured": true,
|
| 72 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 73 |
+
"params_available": true,
|
| 74 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 75 |
+
"modprobe_available": true,
|
| 76 |
+
"runtime": {
|
| 77 |
+
"ForceP2P": "0x11",
|
| 78 |
+
"RMForceP2PType": "1",
|
| 79 |
+
"RMPcieP2PType": "2",
|
| 80 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 81 |
+
"EnableResizableBar": "1",
|
| 82 |
+
"DmaRemapPeerMmio": "1"
|
| 83 |
+
},
|
| 84 |
+
"expected": {
|
| 85 |
+
"ForceP2P": "0x11",
|
| 86 |
+
"RMForceP2PType": "1",
|
| 87 |
+
"RMPcieP2PType": "2",
|
| 88 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 89 |
+
"EnableResizableBar": "1"
|
| 90 |
+
},
|
| 91 |
+
"missing": [],
|
| 92 |
+
"mismatched": {},
|
| 93 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 94 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 95 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 96 |
+
},
|
| 97 |
+
"p2pmark": {
|
| 98 |
+
"status": "not_run"
|
| 99 |
+
},
|
| 100 |
+
"amd_fabric": {
|
| 101 |
+
"status": "not_run"
|
| 102 |
+
},
|
| 103 |
+
"nvidia_smi_query": {
|
| 104 |
+
"cmd": [
|
| 105 |
+
"nvidia-smi",
|
| 106 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 107 |
+
"--format=csv,noheader,nounits"
|
| 108 |
+
],
|
| 109 |
+
"returncode": 0,
|
| 110 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 111 |
+
"stderr": ""
|
| 112 |
+
},
|
| 113 |
+
"nvidia_smi_topo": {
|
| 114 |
+
"cmd": [
|
| 115 |
+
"nvidia-smi",
|
| 116 |
+
"topo",
|
| 117 |
+
"-m"
|
| 118 |
+
],
|
| 119 |
+
"returncode": 0,
|
| 120 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 121 |
+
"stderr": ""
|
| 122 |
+
}
|
| 123 |
+
},
|
| 124 |
+
"nvidia_p2p_override": {
|
| 125 |
+
"effective": true,
|
| 126 |
+
"configured": true,
|
| 127 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 128 |
+
"params_available": true,
|
| 129 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 130 |
+
"modprobe_available": true,
|
| 131 |
+
"runtime": {
|
| 132 |
+
"ForceP2P": "0x11",
|
| 133 |
+
"RMForceP2PType": "1",
|
| 134 |
+
"RMPcieP2PType": "2",
|
| 135 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 136 |
+
"EnableResizableBar": "1",
|
| 137 |
+
"DmaRemapPeerMmio": "1"
|
| 138 |
+
},
|
| 139 |
+
"expected": {
|
| 140 |
+
"ForceP2P": "0x11",
|
| 141 |
+
"RMForceP2PType": "1",
|
| 142 |
+
"RMPcieP2PType": "2",
|
| 143 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 144 |
+
"EnableResizableBar": "1"
|
| 145 |
+
},
|
| 146 |
+
"missing": [],
|
| 147 |
+
"mismatched": {},
|
| 148 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 149 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 150 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 151 |
+
},
|
| 152 |
+
"p2pmark": {
|
| 153 |
+
"status": "not_run"
|
| 154 |
+
},
|
| 155 |
+
"amd_fabric": {
|
| 156 |
+
"status": "not_run"
|
| 157 |
+
},
|
| 158 |
+
"hardware_run_summary": {
|
| 159 |
+
"samples": 47,
|
| 160 |
+
"duration_seconds": 110.814,
|
| 161 |
+
"gpu_count": 4,
|
| 162 |
+
"cpu_util_avg_pct": 10.87,
|
| 163 |
+
"cpu_temp_max_c": 75.0,
|
| 164 |
+
"gpu_util_avg_pct": 85.39,
|
| 165 |
+
"gpu_util_max_pct": 100.0,
|
| 166 |
+
"mem_util_avg_pct": 19.4,
|
| 167 |
+
"mem_util_max_pct": 35.0,
|
| 168 |
+
"temp_avg_c": 60.56,
|
| 169 |
+
"temp_max_c": 81.0,
|
| 170 |
+
"power_total_avg_w": 1029.23,
|
| 171 |
+
"power_total_max_w": 1167.61,
|
| 172 |
+
"power_limit_total_w": 1200.0,
|
| 173 |
+
"vram_used_avg_mb": 384658.34,
|
| 174 |
+
"vram_used_max_mb": 384770.0,
|
| 175 |
+
"vram_total_mb": 391548.0,
|
| 176 |
+
"vram_used_avg_pct": 98.24,
|
| 177 |
+
"vram_used_max_pct": 98.27,
|
| 178 |
+
"pcie_rx_avg_mb_s": 43654.87,
|
| 179 |
+
"pcie_rx_max_mb_s": 56329.0,
|
| 180 |
+
"pcie_tx_avg_mb_s": 42837.81,
|
| 181 |
+
"pcie_tx_max_mb_s": 56583.0
|
| 182 |
+
},
|
| 183 |
+
"event_log": [],
|
| 184 |
+
"prefill": {
|
| 185 |
+
"8192": {
|
| 186 |
+
"ttft_seconds": 1.095,
|
| 187 |
+
"prefill_seconds": 1.095,
|
| 188 |
+
"tok_per_sec": 7485.0,
|
| 189 |
+
"client_ttft_seconds": 1.095,
|
| 190 |
+
"client_tok_per_sec": 7485.0,
|
| 191 |
+
"prompt_tokens": 8194,
|
| 192 |
+
"samples": 14,
|
| 193 |
+
"method": "client",
|
| 194 |
+
"server_validation": {
|
| 195 |
+
"method": "",
|
| 196 |
+
"tok_per_sec": 0.0,
|
| 197 |
+
"prefill_seconds": 0.0,
|
| 198 |
+
"prompt_tokens": 0,
|
| 199 |
+
"request_prompt_tokens": 0,
|
| 200 |
+
"cached_tokens": 0,
|
| 201 |
+
"token_source": "",
|
| 202 |
+
"samples": 0,
|
| 203 |
+
"invalid_reason": ""
|
| 204 |
+
},
|
| 205 |
+
"hardware_summary": {
|
| 206 |
+
"samples": 9,
|
| 207 |
+
"duration_seconds": 19.243,
|
| 208 |
+
"gpu_count": 4,
|
| 209 |
+
"cpu_util_avg_pct": 10.9,
|
| 210 |
+
"cpu_temp_max_c": 73.38,
|
| 211 |
+
"gpu_util_avg_pct": 74.67,
|
| 212 |
+
"gpu_util_max_pct": 100.0,
|
| 213 |
+
"mem_util_avg_pct": 18.72,
|
| 214 |
+
"mem_util_max_pct": 34.0,
|
| 215 |
+
"temp_avg_c": 54.17,
|
| 216 |
+
"temp_max_c": 69.0,
|
| 217 |
+
"power_total_avg_w": 973.23,
|
| 218 |
+
"power_total_max_w": 1134.69,
|
| 219 |
+
"power_limit_total_w": 1200.0,
|
| 220 |
+
"vram_used_avg_mb": 384770.0,
|
| 221 |
+
"vram_used_max_mb": 384770.0,
|
| 222 |
+
"vram_total_mb": 391548.0,
|
| 223 |
+
"vram_used_avg_pct": 98.27,
|
| 224 |
+
"vram_used_max_pct": 98.27,
|
| 225 |
+
"pcie_rx_avg_mb_s": 31974.11,
|
| 226 |
+
"pcie_rx_max_mb_s": 55296.0,
|
| 227 |
+
"pcie_tx_avg_mb_s": 32134.78,
|
| 228 |
+
"pcie_tx_max_mb_s": 49579.0
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"32768": {
|
| 232 |
+
"ttft_seconds": 4.247,
|
| 233 |
+
"prefill_seconds": 4.247,
|
| 234 |
+
"tok_per_sec": 7716.0,
|
| 235 |
+
"client_ttft_seconds": 4.247,
|
| 236 |
+
"client_tok_per_sec": 7716.0,
|
| 237 |
+
"prompt_tokens": 32770,
|
| 238 |
+
"samples": 5,
|
| 239 |
+
"method": "client",
|
| 240 |
+
"server_validation": {
|
| 241 |
+
"method": "",
|
| 242 |
+
"tok_per_sec": 0.0,
|
| 243 |
+
"prefill_seconds": 0.0,
|
| 244 |
+
"prompt_tokens": 0,
|
| 245 |
+
"request_prompt_tokens": 0,
|
| 246 |
+
"cached_tokens": 0,
|
| 247 |
+
"token_source": "",
|
| 248 |
+
"samples": 0,
|
| 249 |
+
"invalid_reason": ""
|
| 250 |
+
},
|
| 251 |
+
"hardware_summary": {
|
| 252 |
+
"samples": 10,
|
| 253 |
+
"duration_seconds": 21.686,
|
| 254 |
+
"gpu_count": 4,
|
| 255 |
+
"cpu_util_avg_pct": 11.06,
|
| 256 |
+
"cpu_temp_max_c": 75.0,
|
| 257 |
+
"gpu_util_avg_pct": 86.97,
|
| 258 |
+
"gpu_util_max_pct": 100.0,
|
| 259 |
+
"mem_util_avg_pct": 19.82,
|
| 260 |
+
"mem_util_max_pct": 30.0,
|
| 261 |
+
"temp_avg_c": 58.9,
|
| 262 |
+
"temp_max_c": 75.0,
|
| 263 |
+
"power_total_avg_w": 997.14,
|
| 264 |
+
"power_total_max_w": 1146.5,
|
| 265 |
+
"power_limit_total_w": 1200.0,
|
| 266 |
+
"vram_used_avg_mb": 384770.0,
|
| 267 |
+
"vram_used_max_mb": 384770.0,
|
| 268 |
+
"vram_total_mb": 391548.0,
|
| 269 |
+
"vram_used_avg_pct": 98.27,
|
| 270 |
+
"vram_used_max_pct": 98.27,
|
| 271 |
+
"pcie_rx_avg_mb_s": 48453.9,
|
| 272 |
+
"pcie_rx_max_mb_s": 55948.0,
|
| 273 |
+
"pcie_tx_avg_mb_s": 46628.3,
|
| 274 |
+
"pcie_tx_max_mb_s": 53063.0
|
| 275 |
+
}
|
| 276 |
+
},
|
| 277 |
+
"65536": {
|
| 278 |
+
"ttft_seconds": 8.558,
|
| 279 |
+
"prefill_seconds": 8.558,
|
| 280 |
+
"tok_per_sec": 7659.0,
|
| 281 |
+
"client_ttft_seconds": 8.558,
|
| 282 |
+
"client_tok_per_sec": 7659.0,
|
| 283 |
+
"prompt_tokens": 65538,
|
| 284 |
+
"samples": 3,
|
| 285 |
+
"method": "client",
|
| 286 |
+
"server_validation": {
|
| 287 |
+
"method": "",
|
| 288 |
+
"tok_per_sec": 0.0,
|
| 289 |
+
"prefill_seconds": 0.0,
|
| 290 |
+
"prompt_tokens": 0,
|
| 291 |
+
"request_prompt_tokens": 0,
|
| 292 |
+
"cached_tokens": 0,
|
| 293 |
+
"token_source": "",
|
| 294 |
+
"samples": 0,
|
| 295 |
+
"invalid_reason": ""
|
| 296 |
+
},
|
| 297 |
+
"hardware_summary": {
|
| 298 |
+
"samples": 11,
|
| 299 |
+
"duration_seconds": 24.072,
|
| 300 |
+
"gpu_count": 4,
|
| 301 |
+
"cpu_util_avg_pct": 11.24,
|
| 302 |
+
"cpu_temp_max_c": 74.88,
|
| 303 |
+
"gpu_util_avg_pct": 97.55,
|
| 304 |
+
"gpu_util_max_pct": 100.0,
|
| 305 |
+
"mem_util_avg_pct": 21.66,
|
| 306 |
+
"mem_util_max_pct": 35.0,
|
| 307 |
+
"temp_avg_c": 62.7,
|
| 308 |
+
"temp_max_c": 79.0,
|
| 309 |
+
"power_total_avg_w": 1097.1,
|
| 310 |
+
"power_total_max_w": 1167.61,
|
| 311 |
+
"power_limit_total_w": 1200.0,
|
| 312 |
+
"vram_used_avg_mb": 384770.0,
|
| 313 |
+
"vram_used_max_mb": 384770.0,
|
| 314 |
+
"vram_total_mb": 391548.0,
|
| 315 |
+
"vram_used_avg_pct": 98.27,
|
| 316 |
+
"vram_used_max_pct": 98.27,
|
| 317 |
+
"pcie_rx_avg_mb_s": 51029.18,
|
| 318 |
+
"pcie_rx_max_mb_s": 56329.0,
|
| 319 |
+
"pcie_tx_avg_mb_s": 47833.18,
|
| 320 |
+
"pcie_tx_max_mb_s": 53350.0
|
| 321 |
+
}
|
| 322 |
+
},
|
| 323 |
+
"131072": {
|
| 324 |
+
"ttft_seconds": 17.398,
|
| 325 |
+
"prefill_seconds": 17.398,
|
| 326 |
+
"tok_per_sec": 7534.0,
|
| 327 |
+
"client_ttft_seconds": 17.398,
|
| 328 |
+
"client_tok_per_sec": 7534.0,
|
| 329 |
+
"prompt_tokens": 131074,
|
| 330 |
+
"samples": 2,
|
| 331 |
+
"method": "client",
|
| 332 |
+
"server_validation": {
|
| 333 |
+
"method": "",
|
| 334 |
+
"tok_per_sec": 0.0,
|
| 335 |
+
"prefill_seconds": 0.0,
|
| 336 |
+
"prompt_tokens": 0,
|
| 337 |
+
"request_prompt_tokens": 0,
|
| 338 |
+
"cached_tokens": 0,
|
| 339 |
+
"token_source": "",
|
| 340 |
+
"samples": 0,
|
| 341 |
+
"invalid_reason": ""
|
| 342 |
+
},
|
| 343 |
+
"hardware_summary": {
|
| 344 |
+
"samples": 15,
|
| 345 |
+
"duration_seconds": 33.754,
|
| 346 |
+
"gpu_count": 4,
|
| 347 |
+
"cpu_util_avg_pct": 11.3,
|
| 348 |
+
"cpu_temp_max_c": 75.0,
|
| 349 |
+
"gpu_util_avg_pct": 93.25,
|
| 350 |
+
"gpu_util_max_pct": 100.0,
|
| 351 |
+
"mem_util_avg_pct": 20.45,
|
| 352 |
+
"mem_util_max_pct": 28.0,
|
| 353 |
+
"temp_avg_c": 64.9,
|
| 354 |
+
"temp_max_c": 81.0,
|
| 355 |
+
"power_total_avg_w": 1108.87,
|
| 356 |
+
"power_total_max_w": 1149.51,
|
| 357 |
+
"power_limit_total_w": 1200.0,
|
| 358 |
+
"vram_used_avg_mb": 384770.0,
|
| 359 |
+
"vram_used_max_mb": 384770.0,
|
| 360 |
+
"vram_total_mb": 391548.0,
|
| 361 |
+
"vram_used_avg_pct": 98.27,
|
| 362 |
+
"vram_used_max_pct": 98.27,
|
| 363 |
+
"pcie_rx_avg_mb_s": 47286.07,
|
| 364 |
+
"pcie_rx_max_mb_s": 55550.0,
|
| 365 |
+
"pcie_tx_avg_mb_s": 48228.93,
|
| 366 |
+
"pcie_tx_max_mb_s": 56583.0
|
| 367 |
+
}
|
| 368 |
+
}
|
| 369 |
+
},
|
| 370 |
+
"results": [],
|
| 371 |
+
"summary_table": {},
|
| 372 |
+
"burst_results": [],
|
| 373 |
+
"burst_summary_table": {},
|
| 374 |
+
"methodology": {
|
| 375 |
+
"prefill": {
|
| 376 |
+
"name": "Prefill",
|
| 377 |
+
"present": true,
|
| 378 |
+
"mode": "standalone_cold",
|
| 379 |
+
"formula": "prompt_tokens / TTFT",
|
| 380 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 381 |
+
},
|
| 382 |
+
"sustained_decode": {
|
| 383 |
+
"name": "Sustained Decode",
|
| 384 |
+
"present": false,
|
| 385 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 386 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 387 |
+
},
|
| 388 |
+
"burst_e2e_decode": {
|
| 389 |
+
"name": "Burst / E2E Decode",
|
| 390 |
+
"present": false,
|
| 391 |
+
"status": "not run; use --run-burst",
|
| 392 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 393 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 394 |
+
}
|
| 395 |
+
}
|
| 396 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-1/prefill.log
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0'] │
|
| 16 |
+
│ Decode: skipped (--prefill-only) | Max tokens: 8192 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill-only: standalone cold profile (client) | Sustained decode: 0 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
|
| 23 |
+
7,337,984 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: standalone cold profile ['8k', '32k', '64k', '128k']
|
| 26 |
+
Calibrating padding text (run=cniaedazxusz, up to 128k)...
|
| 27 |
+
8k: 50,558 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,152 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
64k: 411,264 chars (65,536 prompt tokens via /tokenize)
|
| 30 |
+
128k: 823,408 chars (131,072 prompt tokens via /tokenize)
|
| 31 |
+
Token targeting: /tokenize exact
|
| 32 |
+
Done.
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
llm-decode-bench v0.4.29
|
| 37 |
+
Prefill Speed (C=1, client ISL / TTFT)
|
| 38 |
+
|
| 39 |
+
Client PCIe rx/tx
|
| 40 |
+
Context Tokens TTFT (s) tok/s Server tok/s avg N
|
| 41 |
+
──────────────────────────────────────────────────────────────────────────────
|
| 42 |
+
8k 8,194 1.09 7,485 — 31974/32135 14
|
| 43 |
+
32k 32,770 4.25 7,716 — 48454/46628 5
|
| 44 |
+
64k 65,538 8.56 7,659 — 51029/47833 3
|
| 45 |
+
128k 131,074 17.40 7,534 — 47286/48229 2
|
| 46 |
+
|
| 47 |
+
Client tok/s = prompt_tokens / TTFT. Integrated scout rows come from the
|
| 48 |
+
prefix-cache scout request that decode needs anyway. Server tok/s is optional
|
| 49 |
+
Prometheus validation when the engine exports prefill counters and the exact
|
| 50 |
+
counter delta is uncontaminated; for vLLM this uses newly computed KV tokens,
|
| 51 |
+
not request prompt tokens.
|
| 52 |
+
|
| 53 |
+
|
| 54 |
+
Results saved to
|
| 55 |
+
<campaign>/candidate-speed-wi
|
| 56 |
+
ndow-01/results-01/decode-warp-quant/rep-1/prefill.json
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-command.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"512",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0,8k,32k",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--skip-prefill",
|
| 25 |
+
"--cell-warmup-timeout-seconds",
|
| 26 |
+
"180"
|
| 27 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "46f539494020e5442078334b0cb382d3f3c063fdd3312f4e2fe67775555155ce"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.json
ADDED
|
@@ -0,0 +1,2342 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:37:59.150763",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 32768,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 512,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0,
|
| 42 |
+
8192,
|
| 43 |
+
32768
|
| 44 |
+
],
|
| 45 |
+
"startup_diagnostics_available": true,
|
| 46 |
+
"nvidia_p2p_override_effective": true,
|
| 47 |
+
"p2pmark_status": "not_run",
|
| 48 |
+
"amd_fabric_status": "not_run"
|
| 49 |
+
},
|
| 50 |
+
"startup_diagnostics": {
|
| 51 |
+
"version": "0.4.29",
|
| 52 |
+
"server_url": "http://127.0.0.1:8001",
|
| 53 |
+
"hostname": "<host>",
|
| 54 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 55 |
+
"env": {},
|
| 56 |
+
"args": {
|
| 57 |
+
"concurrency": "1,2,4",
|
| 58 |
+
"contexts": "0,8k,32k",
|
| 59 |
+
"max_tokens": 512,
|
| 60 |
+
"duration": 20.0,
|
| 61 |
+
"request_count": 0,
|
| 62 |
+
"run_burst": false,
|
| 63 |
+
"standalone_prefill": false,
|
| 64 |
+
"prefill_only": false,
|
| 65 |
+
"skip_prefill": true,
|
| 66 |
+
"prefill_contexts": "8k,64k,128k",
|
| 67 |
+
"prefill_metric": "client",
|
| 68 |
+
"dcp_size": 0,
|
| 69 |
+
"kv_budget": 0
|
| 70 |
+
},
|
| 71 |
+
"nvidia_p2p_override": {
|
| 72 |
+
"effective": true,
|
| 73 |
+
"configured": true,
|
| 74 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 75 |
+
"params_available": true,
|
| 76 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 77 |
+
"modprobe_available": true,
|
| 78 |
+
"runtime": {
|
| 79 |
+
"ForceP2P": "0x11",
|
| 80 |
+
"RMForceP2PType": "1",
|
| 81 |
+
"RMPcieP2PType": "2",
|
| 82 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 83 |
+
"EnableResizableBar": "1",
|
| 84 |
+
"DmaRemapPeerMmio": "1"
|
| 85 |
+
},
|
| 86 |
+
"expected": {
|
| 87 |
+
"ForceP2P": "0x11",
|
| 88 |
+
"RMForceP2PType": "1",
|
| 89 |
+
"RMPcieP2PType": "2",
|
| 90 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 91 |
+
"EnableResizableBar": "1"
|
| 92 |
+
},
|
| 93 |
+
"missing": [],
|
| 94 |
+
"mismatched": {},
|
| 95 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 96 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 97 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 98 |
+
},
|
| 99 |
+
"p2pmark": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"amd_fabric": {
|
| 103 |
+
"status": "not_run"
|
| 104 |
+
},
|
| 105 |
+
"nvidia_smi_query": {
|
| 106 |
+
"cmd": [
|
| 107 |
+
"nvidia-smi",
|
| 108 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 109 |
+
"--format=csv,noheader,nounits"
|
| 110 |
+
],
|
| 111 |
+
"returncode": 0,
|
| 112 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 113 |
+
"stderr": ""
|
| 114 |
+
},
|
| 115 |
+
"nvidia_smi_topo": {
|
| 116 |
+
"cmd": [
|
| 117 |
+
"nvidia-smi",
|
| 118 |
+
"topo",
|
| 119 |
+
"-m"
|
| 120 |
+
],
|
| 121 |
+
"returncode": 0,
|
| 122 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 123 |
+
"stderr": ""
|
| 124 |
+
}
|
| 125 |
+
},
|
| 126 |
+
"nvidia_p2p_override": {
|
| 127 |
+
"effective": true,
|
| 128 |
+
"configured": true,
|
| 129 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 130 |
+
"params_available": true,
|
| 131 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 132 |
+
"modprobe_available": true,
|
| 133 |
+
"runtime": {
|
| 134 |
+
"ForceP2P": "0x11",
|
| 135 |
+
"RMForceP2PType": "1",
|
| 136 |
+
"RMPcieP2PType": "2",
|
| 137 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 138 |
+
"EnableResizableBar": "1",
|
| 139 |
+
"DmaRemapPeerMmio": "1"
|
| 140 |
+
},
|
| 141 |
+
"expected": {
|
| 142 |
+
"ForceP2P": "0x11",
|
| 143 |
+
"RMForceP2PType": "1",
|
| 144 |
+
"RMPcieP2PType": "2",
|
| 145 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 146 |
+
"EnableResizableBar": "1"
|
| 147 |
+
},
|
| 148 |
+
"missing": [],
|
| 149 |
+
"mismatched": {},
|
| 150 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 151 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 152 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 153 |
+
},
|
| 154 |
+
"p2pmark": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"amd_fabric": {
|
| 158 |
+
"status": "not_run"
|
| 159 |
+
},
|
| 160 |
+
"hardware_run_summary": {
|
| 161 |
+
"samples": 113,
|
| 162 |
+
"duration_seconds": 269.93,
|
| 163 |
+
"gpu_count": 4,
|
| 164 |
+
"cpu_util_avg_pct": 10.96,
|
| 165 |
+
"cpu_temp_max_c": 77.38,
|
| 166 |
+
"gpu_util_avg_pct": 91.76,
|
| 167 |
+
"gpu_util_max_pct": 100.0,
|
| 168 |
+
"mem_util_avg_pct": 33.91,
|
| 169 |
+
"mem_util_max_pct": 56.0,
|
| 170 |
+
"temp_avg_c": 68.4,
|
| 171 |
+
"temp_max_c": 84.0,
|
| 172 |
+
"power_total_avg_w": 1097.05,
|
| 173 |
+
"power_total_max_w": 1178.78,
|
| 174 |
+
"power_limit_total_w": 1200.0,
|
| 175 |
+
"vram_used_avg_mb": 384778.0,
|
| 176 |
+
"vram_used_max_mb": 384778.0,
|
| 177 |
+
"vram_total_mb": 391548.0,
|
| 178 |
+
"vram_used_avg_pct": 98.27,
|
| 179 |
+
"vram_used_max_pct": 98.27,
|
| 180 |
+
"pcie_rx_avg_mb_s": 17221.25,
|
| 181 |
+
"pcie_rx_max_mb_s": 74341.0,
|
| 182 |
+
"pcie_tx_avg_mb_s": 16568.58,
|
| 183 |
+
"pcie_tx_max_mb_s": 70985.0
|
| 184 |
+
},
|
| 185 |
+
"event_log": [
|
| 186 |
+
"02:33:26 benchmark start engine=vllm",
|
| 187 |
+
"02:33:26 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 188 |
+
"02:33:26 startup decode concurrency=1,2,4 contexts=0,8k,32k",
|
| 189 |
+
"02:33:26 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 190 |
+
"02:33:26 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 191 |
+
"02:33:26 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
|
| 192 |
+
"02:33:26 startup model context length: 1,000,000 tokens",
|
| 193 |
+
"02:33:26 startup prefill tests: skipped",
|
| 194 |
+
"02:33:26 startup calibrating padding text run=mkkehtszimwn up_to=32k",
|
| 195 |
+
"02:33:26 startup context 8k: 50,558 chars (8,192 prompt tokens via /tokenize)",
|
| 196 |
+
"02:33:26 startup context 32k: 205,152 chars (32,768 prompt tokens via /tokenize)",
|
| 197 |
+
"02:33:26 startup token targeting: /tokenize exact",
|
| 198 |
+
"02:33:26 startup startup preparation done",
|
| 199 |
+
"02:33:26 hardware monitor interval=2s",
|
| 200 |
+
"02:33:26 decode warmup start",
|
| 201 |
+
"02:33:26 decode warmup start C=1 ctx=32k 3s",
|
| 202 |
+
"02:33:26 cell start C=1 ctx=32k",
|
| 203 |
+
"02:33:35 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 204 |
+
"02:33:38 cell done C=1 ctx=32k 180.4 tok/s",
|
| 205 |
+
"02:33:38 decode warmup done C=1 ctx=32k",
|
| 206 |
+
"02:33:40 cell start C=1 ctx=0",
|
| 207 |
+
"02:33:45 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 208 |
+
"02:34:05 cell done C=1 ctx=0 191.1 tok/s",
|
| 209 |
+
"02:34:07 cell start C=1 ctx=8k",
|
| 210 |
+
"02:34:13 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 211 |
+
"02:34:33 cell done C=1 ctx=8k 165.2 tok/s",
|
| 212 |
+
"02:34:35 cell start C=1 ctx=32k",
|
| 213 |
+
"02:34:44 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 214 |
+
"02:35:04 cell done C=1 ctx=32k 162.4 tok/s",
|
| 215 |
+
"02:35:06 cell start C=2 ctx=0",
|
| 216 |
+
"02:35:12 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 217 |
+
"02:35:32 cell done C=2 ctx=0 257.8 tok/s",
|
| 218 |
+
"02:35:34 cell start C=4 ctx=0",
|
| 219 |
+
"02:35:39 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 220 |
+
"02:35:59 cell done C=4 ctx=0 305.6 tok/s",
|
| 221 |
+
"02:36:01 cell start C=2 ctx=8k",
|
| 222 |
+
"02:36:07 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 223 |
+
"02:36:27 cell done C=2 ctx=8k 183.5 tok/s",
|
| 224 |
+
"02:36:29 cell start C=4 ctx=8k",
|
| 225 |
+
"02:36:38 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 226 |
+
"02:36:58 cell done C=4 ctx=8k 214.9 tok/s",
|
| 227 |
+
"02:37:00 cell start C=2 ctx=32k",
|
| 228 |
+
"02:37:06 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 229 |
+
"02:37:26 cell done C=2 ctx=32k 186.4 tok/s",
|
| 230 |
+
"02:37:28 cell start C=4 ctx=32k",
|
| 231 |
+
"02:37:36 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 232 |
+
"02:37:57 cell done C=4 ctx=32k 213.0 tok/s"
|
| 233 |
+
],
|
| 234 |
+
"prefill": {},
|
| 235 |
+
"results": [
|
| 236 |
+
{
|
| 237 |
+
"concurrency": 1,
|
| 238 |
+
"context_tokens": 0,
|
| 239 |
+
"benchmark_mode": "duration",
|
| 240 |
+
"request_count_target": 0,
|
| 241 |
+
"warmup_request_count": 0,
|
| 242 |
+
"measurement_seconds": 20.000282,
|
| 243 |
+
"measurement_wall_seconds": 20.000333,
|
| 244 |
+
"client_output_tokens": 3823,
|
| 245 |
+
"server_output_tokens": 3823,
|
| 246 |
+
"aggregate_source": "openai_continuous_usage",
|
| 247 |
+
"aggregate_tps": 191.1473049181911,
|
| 248 |
+
"per_request_avg_tps": 191.1473049181911,
|
| 249 |
+
"ttft_avg": 0.07954429925885051,
|
| 250 |
+
"ttft_p50": 0.08016932185273618,
|
| 251 |
+
"ttft_p90": 0.08546189323533326,
|
| 252 |
+
"ttft_p99": 0.08585712626809254,
|
| 253 |
+
"time_to_second_token_avg": 0.013534792908467352,
|
| 254 |
+
"time_to_second_token_p50": 0.013614983530715108,
|
| 255 |
+
"time_to_second_token_p90": 0.01384659220930189,
|
| 256 |
+
"time_to_second_token_p99": 0.014208161800634117,
|
| 257 |
+
"request_latency_avg": 2.673719661180965,
|
| 258 |
+
"request_latency_p50": 2.642661032034084,
|
| 259 |
+
"request_latency_p90": 2.810000071441755,
|
| 260 |
+
"request_latency_p99": 2.8247497128229586,
|
| 261 |
+
"inter_token_latency_avg": 0.005110297649696225,
|
| 262 |
+
"inter_token_latency_p50": 0.005072282791254339,
|
| 263 |
+
"inter_token_latency_p90": 0.00536784812823367,
|
| 264 |
+
"inter_token_latency_p99": 0.005399117829612438,
|
| 265 |
+
"output_tps_per_user_avg": 196.0011309531592,
|
| 266 |
+
"output_tps_per_user_p50": 197.17019837912926,
|
| 267 |
+
"output_tps_per_user_p90": 204.7882193380745,
|
| 268 |
+
"output_tps_per_user_p99": 209.5508687453994,
|
| 269 |
+
"e2e_output_tps_per_user_avg": 191.73898029102156,
|
| 270 |
+
"e2e_output_tps_per_user_p50": 193.74410633584293,
|
| 271 |
+
"e2e_output_tps_per_user_p90": 198.93980762708756,
|
| 272 |
+
"e2e_output_tps_per_user_p99": 202.87404403324115,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.01418378739450653,
|
| 274 |
+
"chunk_inter_token_latency_p50": 0.014161504070741708,
|
| 275 |
+
"chunk_inter_token_latency_p90": 0.014271326914315551,
|
| 276 |
+
"chunk_inter_token_latency_p99": 0.014287257237805287,
|
| 277 |
+
"input_seq_len_avg": 78.0,
|
| 278 |
+
"output_seq_len_avg": 512.0,
|
| 279 |
+
"output_seq_len_p50": 512.0,
|
| 280 |
+
"output_seq_len_p90": 512.0,
|
| 281 |
+
"output_seq_len_p99": 512.0,
|
| 282 |
+
"request_count": 10,
|
| 283 |
+
"completed_request_count": 9,
|
| 284 |
+
"request_samples": [
|
| 285 |
+
{
|
| 286 |
+
"ttft": 0.0710762650705874,
|
| 287 |
+
"time_to_second_token": 0.012838942930102348,
|
| 288 |
+
"latency": 2.636708308942616,
|
| 289 |
+
"inter_token_latency_avg": 0.0050208063480861615,
|
| 290 |
+
"chunk_inter_token_latency_avg": 0.014174762673326124,
|
| 291 |
+
"input_tokens": 78,
|
| 292 |
+
"output_tokens": 512,
|
| 293 |
+
"output_tps_per_user": 199.17119495779428,
|
| 294 |
+
"e2e_output_tps_per_user": 194.18150967382675,
|
| 295 |
+
"completed": true
|
| 296 |
+
},
|
| 297 |
+
{
|
| 298 |
+
"ttft": 0.07547624688595533,
|
| 299 |
+
"time_to_second_token": 0.013636301038786769,
|
| 300 |
+
"latency": 2.739387070061639,
|
| 301 |
+
"inter_token_latency_avg": 0.005213132726371201,
|
| 302 |
+
"chunk_inter_token_latency_avg": 0.014169738421147254,
|
| 303 |
+
"input_tokens": 78,
|
| 304 |
+
"output_tokens": 512,
|
| 305 |
+
"output_tps_per_user": 191.8232380582583,
|
| 306 |
+
"e2e_output_tps_per_user": 186.90312354744358,
|
| 307 |
+
"completed": true
|
| 308 |
+
},
|
| 309 |
+
{
|
| 310 |
+
"ttft": 0.07437980198301375,
|
| 311 |
+
"time_to_second_token": 0.013298804871737957,
|
| 312 |
+
"latency": 2.6143259189557284,
|
| 313 |
+
"inter_token_latency_avg": 0.004970540346326251,
|
| 314 |
+
"chunk_inter_token_latency_avg": 0.01426936020771188,
|
| 315 |
+
"input_tokens": 78,
|
| 316 |
+
"output_tokens": 512,
|
| 317 |
+
"output_tps_per_user": 201.18537026645492,
|
| 318 |
+
"e2e_output_tps_per_user": 195.84398268312097,
|
| 319 |
+
"completed": true
|
| 320 |
+
},
|
| 321 |
+
{
|
| 322 |
+
"ttft": 0.08492515003308654,
|
| 323 |
+
"time_to_second_token": 0.013621869031339884,
|
| 324 |
+
"latency": 2.642661032034084,
|
| 325 |
+
"inter_token_latency_avg": 0.005005353976518586,
|
| 326 |
+
"chunk_inter_token_latency_avg": 0.01428902727374859,
|
| 327 |
+
"input_tokens": 78,
|
| 328 |
+
"output_tokens": 512,
|
| 329 |
+
"output_tps_per_user": 199.78607001448037,
|
| 330 |
+
"e2e_output_tps_per_user": 193.74410633584293,
|
| 331 |
+
"completed": true
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"ttft": 0.07432189281098545,
|
| 335 |
+
"time_to_second_token": 0.013672125991433859,
|
| 336 |
+
"latency": 2.8059029488358647,
|
| 337 |
+
"inter_token_latency_avg": 0.005345559796526182,
|
| 338 |
+
"chunk_inter_token_latency_avg": 0.014153269720336162,
|
| 339 |
+
"input_tokens": 78,
|
| 340 |
+
"output_tokens": 512,
|
| 341 |
+
"output_tps_per_user": 187.0711465335868,
|
| 342 |
+
"e2e_output_tps_per_user": 182.4724551547382,
|
| 343 |
+
"completed": true
|
| 344 |
+
},
|
| 345 |
+
{
|
| 346 |
+
"ttft": 0.08590104104951024,
|
| 347 |
+
"time_to_second_token": 0.013801953988149762,
|
| 348 |
+
"latency": 2.5183071410283446,
|
| 349 |
+
"inter_token_latency_avg": 0.004760090215222768,
|
| 350 |
+
"chunk_inter_token_latency_avg": 0.014141895930109503,
|
| 351 |
+
"input_tokens": 78,
|
| 352 |
+
"output_tokens": 512,
|
| 353 |
+
"output_tps_per_user": 210.08005201287995,
|
| 354 |
+
"e2e_output_tps_per_user": 203.31118141170265,
|
| 355 |
+
"completed": true
|
| 356 |
+
},
|
| 357 |
+
{
|
| 358 |
+
"ttft": 0.07369623705744743,
|
| 359 |
+
"time_to_second_token": 0.01322799688205123,
|
| 360 |
+
"latency": 2.6919372058473527,
|
| 361 |
+
"inter_token_latency_avg": 0.0051237592344225156,
|
| 362 |
+
"chunk_inter_token_latency_avg": 0.014152653885350839,
|
| 363 |
+
"input_tokens": 78,
|
| 364 |
+
"output_tokens": 512,
|
| 365 |
+
"output_tps_per_user": 195.1692018004642,
|
| 366 |
+
"e2e_output_tps_per_user": 190.19760152199967,
|
| 367 |
+
"completed": true
|
| 368 |
+
},
|
| 369 |
+
{
|
| 370 |
+
"ttft": 0.08539086184464395,
|
| 371 |
+
"time_to_second_token": 0.013393500121310353,
|
| 372 |
+
"latency": 2.826388561865315,
|
| 373 |
+
"inter_token_latency_avg": 0.005363987671273328,
|
| 374 |
+
"chunk_inter_token_latency_avg": 0.01420206062186876,
|
| 375 |
+
"input_tokens": 78,
|
| 376 |
+
"output_tokens": 512,
|
| 377 |
+
"output_tps_per_user": 186.42846726801207,
|
| 378 |
+
"e2e_output_tps_per_user": 181.14989810958562,
|
| 379 |
+
"completed": true
|
| 380 |
+
},
|
| 381 |
+
{
|
| 382 |
+
"ttft": 0.08541309903375804,
|
| 383 |
+
"time_to_second_token": 0.013608098030090332,
|
| 384 |
+
"latency": 2.5878587630577385,
|
| 385 |
+
"inter_token_latency_avg": 0.004897153941338514,
|
| 386 |
+
"chunk_inter_token_latency_avg": 0.014138111096180682,
|
| 387 |
+
"input_tokens": 78,
|
| 388 |
+
"output_tokens": 512,
|
| 389 |
+
"output_tps_per_user": 204.20023792976278,
|
| 390 |
+
"e2e_output_tps_per_user": 197.84696418093378,
|
| 391 |
+
"completed": true
|
| 392 |
+
},
|
| 393 |
+
{
|
| 394 |
+
"ttft": 0.08486239681951702,
|
| 395 |
+
"time_to_second_token": 0.01424833619967103,
|
| 396 |
+
"latency": 0.0,
|
| 397 |
+
"inter_token_latency_avg": 0.005402592240876745,
|
| 398 |
+
"chunk_inter_token_latency_avg": 0.0141469941152855,
|
| 399 |
+
"input_tokens": 78,
|
| 400 |
+
"output_tokens": 255,
|
| 401 |
+
"output_tps_per_user": 185.09633068989817,
|
| 402 |
+
"e2e_output_tps_per_user": 0.0,
|
| 403 |
+
"completed": false
|
| 404 |
+
}
|
| 405 |
+
],
|
| 406 |
+
"total_tokens": 3823,
|
| 407 |
+
"wall_time": 25.55206400505267,
|
| 408 |
+
"num_completed": 1,
|
| 409 |
+
"num_errors": 0,
|
| 410 |
+
"server_gen_throughput": 191.09785171771475,
|
| 411 |
+
"server_utilization": 0.005862646566164198,
|
| 412 |
+
"server_spec_accept_rate": 0.6,
|
| 413 |
+
"server_spec_accept_length": 0.0,
|
| 414 |
+
"avg_running_reqs": 1,
|
| 415 |
+
"max_running_reqs": 1,
|
| 416 |
+
"effective_concurrency": 1,
|
| 417 |
+
"avg_queue_reqs": 0,
|
| 418 |
+
"max_queue_reqs": 0,
|
| 419 |
+
"queue_fraction": 0.0,
|
| 420 |
+
"underfilled": false,
|
| 421 |
+
"warmup_timed_out": false,
|
| 422 |
+
"warmup_duration": 5.537,
|
| 423 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 424 |
+
"timeout_reason": "",
|
| 425 |
+
"capacity_limited": false,
|
| 426 |
+
"hardware_summary": {
|
| 427 |
+
"samples": 9,
|
| 428 |
+
"duration_seconds": 19.265,
|
| 429 |
+
"gpu_count": 4,
|
| 430 |
+
"cpu_util_avg_pct": 11.57,
|
| 431 |
+
"cpu_temp_max_c": 75.75,
|
| 432 |
+
"gpu_util_avg_pct": 98.97,
|
| 433 |
+
"gpu_util_max_pct": 100.0,
|
| 434 |
+
"mem_util_avg_pct": 43.47,
|
| 435 |
+
"mem_util_max_pct": 55.0,
|
| 436 |
+
"temp_avg_c": 68.11,
|
| 437 |
+
"temp_max_c": 83.0,
|
| 438 |
+
"power_total_avg_w": 1152.88,
|
| 439 |
+
"power_total_max_w": 1155.06,
|
| 440 |
+
"power_limit_total_w": 1200.0,
|
| 441 |
+
"vram_used_avg_mb": 384778.0,
|
| 442 |
+
"vram_used_max_mb": 384778.0,
|
| 443 |
+
"vram_total_mb": 391548.0,
|
| 444 |
+
"vram_used_avg_pct": 98.27,
|
| 445 |
+
"vram_used_max_pct": 98.27,
|
| 446 |
+
"pcie_rx_avg_mb_s": 8388.89,
|
| 447 |
+
"pcie_rx_max_mb_s": 8661.0,
|
| 448 |
+
"pcie_tx_avg_mb_s": 8246.33,
|
| 449 |
+
"pcie_tx_max_mb_s": 8504.0
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"concurrency": 1,
|
| 454 |
+
"context_tokens": 8192,
|
| 455 |
+
"benchmark_mode": "duration",
|
| 456 |
+
"request_count_target": 0,
|
| 457 |
+
"warmup_request_count": 0,
|
| 458 |
+
"measurement_seconds": 19.993423,
|
| 459 |
+
"measurement_wall_seconds": 20.000549,
|
| 460 |
+
"client_output_tokens": 3302,
|
| 461 |
+
"server_output_tokens": 3302,
|
| 462 |
+
"aggregate_source": "openai_continuous_usage",
|
| 463 |
+
"aggregate_tps": 165.15431387835127,
|
| 464 |
+
"per_request_avg_tps": 165.15431387835127,
|
| 465 |
+
"ttft_avg": 0.6249137402628548,
|
| 466 |
+
"ttft_p50": 0.6289092639926821,
|
| 467 |
+
"ttft_p90": 0.6339087889529764,
|
| 468 |
+
"ttft_p99": 0.6347594925714657,
|
| 469 |
+
"time_to_second_token_avg": 0.017741062707500532,
|
| 470 |
+
"time_to_second_token_p50": 0.01763475441839546,
|
| 471 |
+
"time_to_second_token_p90": 0.018197339260950685,
|
| 472 |
+
"time_to_second_token_p99": 0.018255747086368502,
|
| 473 |
+
"request_latency_avg": 3.151980838239459,
|
| 474 |
+
"request_latency_p50": 3.1515419960487634,
|
| 475 |
+
"request_latency_p90": 3.2448623499367386,
|
| 476 |
+
"request_latency_p99": 3.291342785186134,
|
| 477 |
+
"inter_token_latency_avg": 0.004930662211393978,
|
| 478 |
+
"inter_token_latency_p50": 0.004938842342353614,
|
| 479 |
+
"inter_token_latency_p90": 0.005095802539678877,
|
| 480 |
+
"inter_token_latency_p99": 0.0051998018340067296,
|
| 481 |
+
"output_tps_per_user_avg": 203.0459222201492,
|
| 482 |
+
"output_tps_per_user_p50": 202.51955283196384,
|
| 483 |
+
"output_tps_per_user_p90": 210.3772401347902,
|
| 484 |
+
"output_tps_per_user_p99": 215.44168752783455,
|
| 485 |
+
"e2e_output_tps_per_user_avg": 162.56228618550645,
|
| 486 |
+
"e2e_output_tps_per_user_p50": 162.46015462967605,
|
| 487 |
+
"e2e_output_tps_per_user_p90": 167.22336677689123,
|
| 488 |
+
"e2e_output_tps_per_user_p99": 170.41814081496065,
|
| 489 |
+
"chunk_inter_token_latency_avg": 0.01426418191410213,
|
| 490 |
+
"chunk_inter_token_latency_p50": 0.014250208142300946,
|
| 491 |
+
"chunk_inter_token_latency_p90": 0.014323315611798752,
|
| 492 |
+
"chunk_inter_token_latency_p99": 0.014336108877223197,
|
| 493 |
+
"input_seq_len_avg": 8192.0,
|
| 494 |
+
"output_seq_len_avg": 512.0,
|
| 495 |
+
"output_seq_len_p50": 512.0,
|
| 496 |
+
"output_seq_len_p90": 512.0,
|
| 497 |
+
"output_seq_len_p99": 512.0,
|
| 498 |
+
"request_count": 8,
|
| 499 |
+
"completed_request_count": 7,
|
| 500 |
+
"request_samples": [
|
| 501 |
+
{
|
| 502 |
+
"ttft": 0.591038762126118,
|
| 503 |
+
"time_to_second_token": 0.017253336030989885,
|
| 504 |
+
"latency": 3.1515419960487634,
|
| 505 |
+
"inter_token_latency_avg": 0.005010769538009091,
|
| 506 |
+
"chunk_inter_token_latency_avg": 0.01422501796623692,
|
| 507 |
+
"input_tokens": 8192,
|
| 508 |
+
"output_tokens": 512,
|
| 509 |
+
"output_tps_per_user": 199.57014434899074,
|
| 510 |
+
"e2e_output_tps_per_user": 162.46015462967605,
|
| 511 |
+
"completed": true
|
| 512 |
+
},
|
| 513 |
+
{
|
| 514 |
+
"ttft": 0.6249128989875317,
|
| 515 |
+
"time_to_second_token": 0.017716374015435576,
|
| 516 |
+
"latency": 3.1119065389502794,
|
| 517 |
+
"inter_token_latency_avg": 0.004866915146698137,
|
| 518 |
+
"chunk_inter_token_latency_avg": 0.014211392228358558,
|
| 519 |
+
"input_tokens": 8192,
|
| 520 |
+
"output_tokens": 512,
|
| 521 |
+
"output_tps_per_user": 205.46896131493693,
|
| 522 |
+
"e2e_output_tps_per_user": 164.52936281714614,
|
| 523 |
+
"completed": true
|
| 524 |
+
},
|
| 525 |
+
{
|
| 526 |
+
"ttft": 0.6324374838732183,
|
| 527 |
+
"time_to_second_token": 0.018169526010751724,
|
| 528 |
+
"latency": 2.998129991814494,
|
| 529 |
+
"inter_token_latency_avg": 0.004629535240589581,
|
| 530 |
+
"chunk_inter_token_latency_avg": 0.014337530351159247,
|
| 531 |
+
"input_tokens": 8192,
|
| 532 |
+
"output_tokens": 512,
|
| 533 |
+
"output_tps_per_user": 216.00440390483948,
|
| 534 |
+
"e2e_output_tps_per_user": 170.77311570807947,
|
| 535 |
+
"completed": true
|
| 536 |
+
},
|
| 537 |
+
{
|
| 538 |
+
"ttft": 0.6335036919917911,
|
| 539 |
+
"time_to_second_token": 0.01815098780207336,
|
| 540 |
+
"latency": 3.2965072779916227,
|
| 541 |
+
"inter_token_latency_avg": 0.005211357311154269,
|
| 542 |
+
"chunk_inter_token_latency_avg": 0.014317223580644255,
|
| 543 |
+
"input_tokens": 8192,
|
| 544 |
+
"output_tokens": 512,
|
| 545 |
+
"output_tps_per_user": 191.88858876738755,
|
| 546 |
+
"e2e_output_tps_per_user": 155.3159015962898,
|
| 547 |
+
"completed": true
|
| 548 |
+
},
|
| 549 |
+
{
|
| 550 |
+
"ttft": 0.6260347329080105,
|
| 551 |
+
"time_to_second_token": 0.017283352091908455,
|
| 552 |
+
"latency": 3.1057244250550866,
|
| 553 |
+
"inter_token_latency_avg": 0.004852621706745746,
|
| 554 |
+
"chunk_inter_token_latency_avg": 0.01425109018475331,
|
| 555 |
+
"input_tokens": 8192,
|
| 556 |
+
"output_tokens": 512,
|
| 557 |
+
"output_tps_per_user": 206.07417194912927,
|
| 558 |
+
"e2e_output_tps_per_user": 164.85686748943237,
|
| 559 |
+
"completed": true
|
| 560 |
+
},
|
| 561 |
+
{
|
| 562 |
+
"ttft": 0.6317837950773537,
|
| 563 |
+
"time_to_second_token": 0.01826223684474826,
|
| 564 |
+
"latency": 3.2104323979001492,
|
| 565 |
+
"inter_token_latency_avg": 0.005046279066189424,
|
| 566 |
+
"chunk_inter_token_latency_avg": 0.014246677363661853,
|
| 567 |
+
"input_tokens": 8192,
|
| 568 |
+
"output_tokens": 512,
|
| 569 |
+
"output_tps_per_user": 198.165814233323,
|
| 570 |
+
"e2e_output_tps_per_user": 159.48007512473532,
|
| 571 |
+
"completed": true
|
| 572 |
+
},
|
| 573 |
+
{
|
| 574 |
+
"ttft": 0.6247445419430733,
|
| 575 |
+
"time_to_second_token": 0.017539554042741656,
|
| 576 |
+
"latency": 3.189623239915818,
|
| 577 |
+
"inter_token_latency_avg": 0.005019332089966232,
|
| 578 |
+
"chunk_inter_token_latency_avg": 0.014249326099848582,
|
| 579 |
+
"input_tokens": 8192,
|
| 580 |
+
"output_tokens": 512,
|
| 581 |
+
"output_tps_per_user": 199.22969472353194,
|
| 582 |
+
"e2e_output_tps_per_user": 160.52052593318606,
|
| 583 |
+
"completed": true
|
| 584 |
+
},
|
| 585 |
+
{
|
| 586 |
+
"ttft": 0.6348540151957422,
|
| 587 |
+
"time_to_second_token": 0.017553134821355343,
|
| 588 |
+
"latency": 0.0,
|
| 589 |
+
"inter_token_latency_avg": 0.004808487591799348,
|
| 590 |
+
"chunk_inter_token_latency_avg": 0.014275197538154316,
|
| 591 |
+
"input_tokens": 8192,
|
| 592 |
+
"output_tokens": 381,
|
| 593 |
+
"output_tps_per_user": 207.96559851905482,
|
| 594 |
+
"e2e_output_tps_per_user": 0.0,
|
| 595 |
+
"completed": false
|
| 596 |
+
}
|
| 597 |
+
],
|
| 598 |
+
"total_tokens": 3302,
|
| 599 |
+
"wall_time": 26.068795799044892,
|
| 600 |
+
"num_completed": 1,
|
| 601 |
+
"num_errors": 0,
|
| 602 |
+
"server_gen_throughput": 165.0529013540346,
|
| 603 |
+
"server_utilization": 0.006141820212172022,
|
| 604 |
+
"server_spec_accept_rate": 0.6956521739130435,
|
| 605 |
+
"server_spec_accept_length": 0.0,
|
| 606 |
+
"avg_running_reqs": 0.9,
|
| 607 |
+
"max_running_reqs": 1,
|
| 608 |
+
"effective_concurrency": 0.9,
|
| 609 |
+
"avg_queue_reqs": 0,
|
| 610 |
+
"max_queue_reqs": 0,
|
| 611 |
+
"queue_fraction": 0.0,
|
| 612 |
+
"underfilled": false,
|
| 613 |
+
"warmup_timed_out": false,
|
| 614 |
+
"warmup_duration": 6.061,
|
| 615 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 616 |
+
"timeout_reason": "",
|
| 617 |
+
"capacity_limited": false,
|
| 618 |
+
"hardware_summary": {
|
| 619 |
+
"samples": 8,
|
| 620 |
+
"duration_seconds": 16.92,
|
| 621 |
+
"gpu_count": 4,
|
| 622 |
+
"cpu_util_avg_pct": 11.54,
|
| 623 |
+
"cpu_temp_max_c": 76.12,
|
| 624 |
+
"gpu_util_avg_pct": 98.81,
|
| 625 |
+
"gpu_util_max_pct": 100.0,
|
| 626 |
+
"mem_util_avg_pct": 39.97,
|
| 627 |
+
"mem_util_max_pct": 56.0,
|
| 628 |
+
"temp_avg_c": 68.56,
|
| 629 |
+
"temp_max_c": 84.0,
|
| 630 |
+
"power_total_avg_w": 1154.06,
|
| 631 |
+
"power_total_max_w": 1157.95,
|
| 632 |
+
"power_limit_total_w": 1200.0,
|
| 633 |
+
"vram_used_avg_mb": 384778.0,
|
| 634 |
+
"vram_used_max_mb": 384778.0,
|
| 635 |
+
"vram_total_mb": 391548.0,
|
| 636 |
+
"vram_used_avg_pct": 98.27,
|
| 637 |
+
"vram_used_max_pct": 98.27,
|
| 638 |
+
"pcie_rx_avg_mb_s": 11118.0,
|
| 639 |
+
"pcie_rx_max_mb_s": 30706.0,
|
| 640 |
+
"pcie_tx_avg_mb_s": 9898.38,
|
| 641 |
+
"pcie_tx_max_mb_s": 21763.0
|
| 642 |
+
}
|
| 643 |
+
},
|
| 644 |
+
{
|
| 645 |
+
"concurrency": 1,
|
| 646 |
+
"context_tokens": 32768,
|
| 647 |
+
"benchmark_mode": "duration",
|
| 648 |
+
"request_count_target": 0,
|
| 649 |
+
"warmup_request_count": 0,
|
| 650 |
+
"measurement_seconds": 19.993472,
|
| 651 |
+
"measurement_wall_seconds": 20.000595,
|
| 652 |
+
"client_output_tokens": 3246,
|
| 653 |
+
"server_output_tokens": 3246,
|
| 654 |
+
"aggregate_source": "openai_continuous_usage",
|
| 655 |
+
"aggregate_tps": 162.35299164317425,
|
| 656 |
+
"per_request_avg_tps": 162.35299164317425,
|
| 657 |
+
"ttft_avg": 0.6279244296310935,
|
| 658 |
+
"ttft_p50": 0.629666926455684,
|
| 659 |
+
"ttft_p90": 0.6346158204367385,
|
| 660 |
+
"ttft_p99": 0.6394360246905125,
|
| 661 |
+
"time_to_second_token_avg": 0.011822661821497604,
|
| 662 |
+
"time_to_second_token_p50": 0.012242211028933525,
|
| 663 |
+
"time_to_second_token_p90": 0.012837472674436867,
|
| 664 |
+
"time_to_second_token_p99": 0.012860276207793503,
|
| 665 |
+
"request_latency_avg": 3.184498446295038,
|
| 666 |
+
"request_latency_p50": 3.200756788952276,
|
| 667 |
+
"request_latency_p90": 3.258722552191466,
|
| 668 |
+
"request_latency_p99": 3.2730169181805104,
|
| 669 |
+
"inter_token_latency_avg": 0.005025481984653069,
|
| 670 |
+
"inter_token_latency_p50": 0.005032082509756118,
|
| 671 |
+
"inter_token_latency_p90": 0.005174755995160009,
|
| 672 |
+
"inter_token_latency_p99": 0.005182216347690941,
|
| 673 |
+
"output_tps_per_user_avg": 199.13235326934893,
|
| 674 |
+
"output_tps_per_user_p50": 198.72498796426024,
|
| 675 |
+
"output_tps_per_user_p90": 206.07410309939215,
|
| 676 |
+
"output_tps_per_user_p99": 208.73558083264146,
|
| 677 |
+
"e2e_output_tps_per_user_avg": 160.84431793668682,
|
| 678 |
+
"e2e_output_tps_per_user_p50": 159.96216949916902,
|
| 679 |
+
"e2e_output_tps_per_user_p90": 164.8638332468683,
|
| 680 |
+
"e2e_output_tps_per_user_p99": 166.41261273335627,
|
| 681 |
+
"chunk_inter_token_latency_avg": 0.014293934351271118,
|
| 682 |
+
"chunk_inter_token_latency_p50": 0.014270523142799533,
|
| 683 |
+
"chunk_inter_token_latency_p90": 0.014406589656055605,
|
| 684 |
+
"chunk_inter_token_latency_p99": 0.01443256986090686,
|
| 685 |
+
"input_seq_len_avg": 32768.0,
|
| 686 |
+
"output_seq_len_avg": 512.0,
|
| 687 |
+
"output_seq_len_p50": 512.0,
|
| 688 |
+
"output_seq_len_p90": 512.0,
|
| 689 |
+
"output_seq_len_p99": 512.0,
|
| 690 |
+
"request_count": 8,
|
| 691 |
+
"completed_request_count": 7,
|
| 692 |
+
"request_samples": [
|
| 693 |
+
{
|
| 694 |
+
"ttft": 0.6056491718627512,
|
| 695 |
+
"time_to_second_token": 0.01198451709933579,
|
| 696 |
+
"latency": 3.248134132940322,
|
| 697 |
+
"inter_token_latency_avg": 0.005171203446335755,
|
| 698 |
+
"chunk_inter_token_latency_avg": 0.014283702492311194,
|
| 699 |
+
"input_tokens": 32768,
|
| 700 |
+
"output_tokens": 512,
|
| 701 |
+
"output_tps_per_user": 193.37858399452193,
|
| 702 |
+
"e2e_output_tps_per_user": 157.62895836340357,
|
| 703 |
+
"completed": true
|
| 704 |
+
},
|
| 705 |
+
{
|
| 706 |
+
"ttft": 0.6288057561032474,
|
| 707 |
+
"time_to_second_token": 0.012826613849028945,
|
| 708 |
+
"latency": 3.2020828151144087,
|
| 709 |
+
"inter_token_latency_avg": 0.005035767238769396,
|
| 710 |
+
"chunk_inter_token_latency_avg": 0.014295983661173118,
|
| 711 |
+
"input_tokens": 32768,
|
| 712 |
+
"output_tokens": 512,
|
| 713 |
+
"output_tps_per_user": 198.5794721211882,
|
| 714 |
+
"e2e_output_tps_per_user": 159.89592698329588,
|
| 715 |
+
"completed": true
|
| 716 |
+
},
|
| 717 |
+
{
|
| 718 |
+
"ttft": 0.6289016020018607,
|
| 719 |
+
"time_to_second_token": 0.01273016701452434,
|
| 720 |
+
"latency": 3.073511565104127,
|
| 721 |
+
"inter_token_latency_avg": 0.0047839725305328104,
|
| 722 |
+
"chunk_inter_token_latency_avg": 0.014212848622687594,
|
| 723 |
+
"input_tokens": 32768,
|
| 724 |
+
"output_tokens": 512,
|
| 725 |
+
"output_tps_per_user": 209.03130058078028,
|
| 726 |
+
"e2e_output_tps_per_user": 166.58469934296605,
|
| 727 |
+
"completed": true
|
| 728 |
+
},
|
| 729 |
+
{
|
| 730 |
+
"ttft": 0.6260690451599658,
|
| 731 |
+
"time_to_second_token": 0.011439014924690127,
|
| 732 |
+
"latency": 3.274605181068182,
|
| 733 |
+
"inter_token_latency_avg": 0.005183045275749934,
|
| 734 |
+
"chunk_inter_token_latency_avg": 0.014394218129935958,
|
| 735 |
+
"input_tokens": 32768,
|
| 736 |
+
"output_tokens": 512,
|
| 737 |
+
"output_tps_per_user": 192.9367672473805,
|
| 738 |
+
"e2e_output_tps_per_user": 156.35472726913133,
|
| 739 |
+
"completed": true
|
| 740 |
+
},
|
| 741 |
+
{
|
| 742 |
+
"ttft": 0.6312455229926854,
|
| 743 |
+
"time_to_second_token": 0.011678099865093827,
|
| 744 |
+
"latency": 3.200756788952276,
|
| 745 |
+
"inter_token_latency_avg": 0.005028397780742839,
|
| 746 |
+
"chunk_inter_token_latency_avg": 0.014435456550334779,
|
| 747 |
+
"input_tokens": 32768,
|
| 748 |
+
"output_tokens": 512,
|
| 749 |
+
"output_tps_per_user": 198.87050380733228,
|
| 750 |
+
"e2e_output_tps_per_user": 159.96216949916902,
|
| 751 |
+
"completed": true
|
| 752 |
+
},
|
| 753 |
+
{
|
| 754 |
+
"ttft": 0.6304322509095073,
|
| 755 |
+
"time_to_second_token": 0.01249990495853126,
|
| 756 |
+
"latency": 3.165042991982773,
|
| 757 |
+
"inter_token_latency_avg": 0.004960099297599346,
|
| 758 |
+
"chunk_inter_token_latency_avg": 0.014239386185804863,
|
| 759 |
+
"input_tokens": 32768,
|
| 760 |
+
"output_tokens": 512,
|
| 761 |
+
"output_tps_per_user": 201.6088670813492,
|
| 762 |
+
"e2e_output_tps_per_user": 161.76715491603875,
|
| 763 |
+
"completed": true
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"ttft": 0.6323204850777984,
|
| 767 |
+
"time_to_second_token": 0.01286280993372202,
|
| 768 |
+
"latency": 3.127355648903176,
|
| 769 |
+
"inter_token_latency_avg": 0.004882651984002696,
|
| 770 |
+
"chunk_inter_token_latency_avg": 0.014257343793287873,
|
| 771 |
+
"input_tokens": 32768,
|
| 772 |
+
"output_tokens": 512,
|
| 773 |
+
"output_tps_per_user": 204.80673275022582,
|
| 774 |
+
"e2e_output_tps_per_user": 163.71658918280312,
|
| 775 |
+
"completed": true
|
| 776 |
+
},
|
| 777 |
+
{
|
| 778 |
+
"ttft": 0.6399716029409319,
|
| 779 |
+
"time_to_second_token": 0.008560166927054524,
|
| 780 |
+
"latency": 0.0,
|
| 781 |
+
"inter_token_latency_avg": 0.005158718323491779,
|
| 782 |
+
"chunk_inter_token_latency_avg": 0.014232535374633568,
|
| 783 |
+
"input_tokens": 32768,
|
| 784 |
+
"output_tokens": 310,
|
| 785 |
+
"output_tps_per_user": 193.84659857201325,
|
| 786 |
+
"e2e_output_tps_per_user": 0.0,
|
| 787 |
+
"completed": false
|
| 788 |
+
}
|
| 789 |
+
],
|
| 790 |
+
"total_tokens": 3246,
|
| 791 |
+
"wall_time": 29.109418482985348,
|
| 792 |
+
"num_completed": 1,
|
| 793 |
+
"num_errors": 0,
|
| 794 |
+
"server_gen_throughput": 162.2569315918152,
|
| 795 |
+
"server_utilization": 0.006979341150195384,
|
| 796 |
+
"server_spec_accept_rate": 0.6163522012578616,
|
| 797 |
+
"server_spec_accept_length": 0.0,
|
| 798 |
+
"avg_running_reqs": 0.9,
|
| 799 |
+
"max_running_reqs": 1,
|
| 800 |
+
"effective_concurrency": 0.9,
|
| 801 |
+
"avg_queue_reqs": 0,
|
| 802 |
+
"max_queue_reqs": 0,
|
| 803 |
+
"queue_fraction": 0.0,
|
| 804 |
+
"underfilled": false,
|
| 805 |
+
"warmup_timed_out": false,
|
| 806 |
+
"warmup_duration": 9.102,
|
| 807 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 808 |
+
"timeout_reason": "",
|
| 809 |
+
"capacity_limited": false,
|
| 810 |
+
"hardware_summary": {
|
| 811 |
+
"samples": 8,
|
| 812 |
+
"duration_seconds": 16.9,
|
| 813 |
+
"gpu_count": 4,
|
| 814 |
+
"cpu_util_avg_pct": 11.49,
|
| 815 |
+
"cpu_temp_max_c": 76.62,
|
| 816 |
+
"gpu_util_avg_pct": 95.94,
|
| 817 |
+
"gpu_util_max_pct": 100.0,
|
| 818 |
+
"mem_util_avg_pct": 38.56,
|
| 819 |
+
"mem_util_max_pct": 55.0,
|
| 820 |
+
"temp_avg_c": 68.72,
|
| 821 |
+
"temp_max_c": 84.0,
|
| 822 |
+
"power_total_avg_w": 1151.98,
|
| 823 |
+
"power_total_max_w": 1158.51,
|
| 824 |
+
"power_limit_total_w": 1200.0,
|
| 825 |
+
"vram_used_avg_mb": 384778.0,
|
| 826 |
+
"vram_used_max_mb": 384778.0,
|
| 827 |
+
"vram_total_mb": 391548.0,
|
| 828 |
+
"vram_used_avg_pct": 98.27,
|
| 829 |
+
"vram_used_max_pct": 98.27,
|
| 830 |
+
"pcie_rx_avg_mb_s": 12769.0,
|
| 831 |
+
"pcie_rx_max_mb_s": 39144.0,
|
| 832 |
+
"pcie_tx_avg_mb_s": 12689.38,
|
| 833 |
+
"pcie_tx_max_mb_s": 43337.0
|
| 834 |
+
}
|
| 835 |
+
},
|
| 836 |
+
{
|
| 837 |
+
"concurrency": 2,
|
| 838 |
+
"context_tokens": 0,
|
| 839 |
+
"benchmark_mode": "duration",
|
| 840 |
+
"request_count_target": 0,
|
| 841 |
+
"warmup_request_count": 0,
|
| 842 |
+
"measurement_seconds": 19.987074,
|
| 843 |
+
"measurement_wall_seconds": 20.001238,
|
| 844 |
+
"client_output_tokens": 5152,
|
| 845 |
+
"server_output_tokens": 5152,
|
| 846 |
+
"aggregate_source": "openai_continuous_usage",
|
| 847 |
+
"aggregate_tps": 257.76658819245483,
|
| 848 |
+
"per_request_avg_tps": 128.88329409622742,
|
| 849 |
+
"ttft_avg": 0.12264618660057229,
|
| 850 |
+
"ttft_p50": 0.12349911453202367,
|
| 851 |
+
"ttft_p90": 0.14274162366054954,
|
| 852 |
+
"ttft_p99": 0.16984736640239134,
|
| 853 |
+
"time_to_second_token_avg": 0.019459182529577186,
|
| 854 |
+
"time_to_second_token_p50": 0.02056812052614987,
|
| 855 |
+
"time_to_second_token_p90": 0.02087695982772857,
|
| 856 |
+
"time_to_second_token_p99": 0.020941995084285736,
|
| 857 |
+
"request_latency_avg": 4.002757317425373,
|
| 858 |
+
"request_latency_p50": 4.017580935033038,
|
| 859 |
+
"request_latency_p90": 4.2898590574972335,
|
| 860 |
+
"request_latency_p99": 4.336852466568817,
|
| 861 |
+
"inter_token_latency_avg": 0.007545030826191364,
|
| 862 |
+
"inter_token_latency_p50": 0.00755755253999135,
|
| 863 |
+
"inter_token_latency_p90": 0.008070413317256143,
|
| 864 |
+
"inter_token_latency_p99": 0.008259788365663738,
|
| 865 |
+
"output_tps_per_user_avg": 132.88272402904317,
|
| 866 |
+
"output_tps_per_user_p50": 132.32326777905013,
|
| 867 |
+
"output_tps_per_user_p90": 141.37995688471477,
|
| 868 |
+
"output_tps_per_user_p99": 145.6980556167255,
|
| 869 |
+
"e2e_output_tps_per_user_avg": 128.22354203464204,
|
| 870 |
+
"e2e_output_tps_per_user_p50": 127.43988690683825,
|
| 871 |
+
"e2e_output_tps_per_user_p90": 134.35973544624332,
|
| 872 |
+
"e2e_output_tps_per_user_p99": 140.69704751927173,
|
| 873 |
+
"chunk_inter_token_latency_avg": 0.02104700313164791,
|
| 874 |
+
"chunk_inter_token_latency_p50": 0.02098676121404283,
|
| 875 |
+
"chunk_inter_token_latency_p90": 0.0212750823967716,
|
| 876 |
+
"chunk_inter_token_latency_p99": 0.02139749144743477,
|
| 877 |
+
"input_seq_len_avg": 78.0,
|
| 878 |
+
"output_seq_len_avg": 512.0,
|
| 879 |
+
"output_seq_len_p50": 512.0,
|
| 880 |
+
"output_seq_len_p90": 512.0,
|
| 881 |
+
"output_seq_len_p99": 512.0,
|
| 882 |
+
"request_count": 14,
|
| 883 |
+
"completed_request_count": 12,
|
| 884 |
+
"request_samples": [
|
| 885 |
+
{
|
| 886 |
+
"ttft": 0.07193002197891474,
|
| 887 |
+
"time_to_second_token": 0.01364690694026649,
|
| 888 |
+
"latency": 4.016205613967031,
|
| 889 |
+
"inter_token_latency_avg": 0.0077187389275697,
|
| 890 |
+
"chunk_inter_token_latency_avg": 0.020869183026392152,
|
| 891 |
+
"input_tokens": 78,
|
| 892 |
+
"output_tokens": 512,
|
| 893 |
+
"output_tps_per_user": 129.55484171490914,
|
| 894 |
+
"e2e_output_tps_per_user": 127.48351285089433,
|
| 895 |
+
"completed": true
|
| 896 |
+
},
|
| 897 |
+
{
|
| 898 |
+
"ttft": 0.10589846200309694,
|
| 899 |
+
"time_to_second_token": 0.01401821500621736,
|
| 900 |
+
"latency": 3.8386515881866217,
|
| 901 |
+
"inter_token_latency_avg": 0.007304800638323923,
|
| 902 |
+
"chunk_inter_token_latency_avg": 0.02108900071290127,
|
| 903 |
+
"input_tokens": 78,
|
| 904 |
+
"output_tokens": 512,
|
| 905 |
+
"output_tps_per_user": 136.89627541011834,
|
| 906 |
+
"e2e_output_tps_per_user": 133.3801696344806,
|
| 907 |
+
"completed": true
|
| 908 |
+
},
|
| 909 |
+
{
|
| 910 |
+
"ttft": 0.12403434701263905,
|
| 911 |
+
"time_to_second_token": 0.020574419992044568,
|
| 912 |
+
"latency": 3.8075810340233147,
|
| 913 |
+
"inter_token_latency_avg": 0.007208506236811498,
|
| 914 |
+
"chunk_inter_token_latency_avg": 0.021048838211489576,
|
| 915 |
+
"input_tokens": 78,
|
| 916 |
+
"output_tokens": 512,
|
| 917 |
+
"output_tps_per_user": 138.72499615708523,
|
| 918 |
+
"e2e_output_tps_per_user": 134.46857609199472,
|
| 919 |
+
"completed": true
|
| 920 |
+
},
|
| 921 |
+
{
|
| 922 |
+
"ttft": 0.11724809114821255,
|
| 923 |
+
"time_to_second_token": 0.020686850883066654,
|
| 924 |
+
"latency": 3.954718264983967,
|
| 925 |
+
"inter_token_latency_avg": 0.007509726367584646,
|
| 926 |
+
"chunk_inter_token_latency_avg": 0.02096978237068718,
|
| 927 |
+
"input_tokens": 78,
|
| 928 |
+
"output_tokens": 512,
|
| 929 |
+
"output_tps_per_user": 133.16064408371113,
|
| 930 |
+
"e2e_output_tps_per_user": 129.46560682549045,
|
| 931 |
+
"completed": true
|
| 932 |
+
},
|
| 933 |
+
{
|
| 934 |
+
"ttft": 0.11513775610364974,
|
| 935 |
+
"time_to_second_token": 0.020763778826221824,
|
| 936 |
+
"latency": 4.340469955932349,
|
| 937 |
+
"inter_token_latency_avg": 0.008268751858764578,
|
| 938 |
+
"chunk_inter_token_latency_avg": 0.021126660999143496,
|
| 939 |
+
"input_tokens": 78,
|
| 940 |
+
"output_tokens": 512,
|
| 941 |
+
"output_tps_per_user": 120.93723660845332,
|
| 942 |
+
"e2e_output_tps_per_user": 117.95957700391926,
|
| 943 |
+
"completed": true
|
| 944 |
+
},
|
| 945 |
+
{
|
| 946 |
+
"ttft": 0.12337096291594207,
|
| 947 |
+
"time_to_second_token": 0.020409638062119484,
|
| 948 |
+
"latency": 3.619222234003246,
|
| 949 |
+
"inter_token_latency_avg": 0.006841196225219772,
|
| 950 |
+
"chunk_inter_token_latency_avg": 0.02093324114423535,
|
| 951 |
+
"input_tokens": 78,
|
| 952 |
+
"output_tokens": 512,
|
| 953 |
+
"output_tps_per_user": 146.1732666450267,
|
| 954 |
+
"e2e_output_tps_per_user": 141.46685859455317,
|
| 955 |
+
"completed": true
|
| 956 |
+
},
|
| 957 |
+
{
|
| 958 |
+
"ttft": 0.12318649212829769,
|
| 959 |
+
"time_to_second_token": 0.02092546597123146,
|
| 960 |
+
"latency": 0.0,
|
| 961 |
+
"inter_token_latency_avg": 0.007490993097976402,
|
| 962 |
+
"chunk_inter_token_latency_avg": 0.021415427327156067,
|
| 963 |
+
"input_tokens": 78,
|
| 964 |
+
"output_tokens": 244,
|
| 965 |
+
"output_tps_per_user": 133.4936485617825,
|
| 966 |
+
"e2e_output_tps_per_user": 0.0,
|
| 967 |
+
"completed": false
|
| 968 |
+
},
|
| 969 |
+
{
|
| 970 |
+
"ttft": 0.15055592195130885,
|
| 971 |
+
"time_to_second_token": 0.018002169905230403,
|
| 972 |
+
"latency": 4.036904443986714,
|
| 973 |
+
"inter_token_latency_avg": 0.007605378712398053,
|
| 974 |
+
"chunk_inter_token_latency_avg": 0.02089434689266347,
|
| 975 |
+
"input_tokens": 78,
|
| 976 |
+
"output_tokens": 512,
|
| 977 |
+
"output_tps_per_user": 131.48589147438915,
|
| 978 |
+
"e2e_output_tps_per_user": 126.82985369214379,
|
| 979 |
+
"completed": true
|
| 980 |
+
},
|
| 981 |
+
{
|
| 982 |
+
"ttft": 0.17272999603301287,
|
| 983 |
+
"time_to_second_token": 0.020944464951753616,
|
| 984 |
+
"latency": 4.130337374052033,
|
| 985 |
+
"inter_token_latency_avg": 0.0077448285284129545,
|
| 986 |
+
"chunk_inter_token_latency_avg": 0.021277459021607634,
|
| 987 |
+
"input_tokens": 78,
|
| 988 |
+
"output_tokens": 512,
|
| 989 |
+
"output_tps_per_user": 129.11841706131574,
|
| 990 |
+
"e2e_output_tps_per_user": 123.96081812021731,
|
| 991 |
+
"completed": true
|
| 992 |
+
},
|
| 993 |
+
{
|
| 994 |
+
"ttft": 0.12362726614810526,
|
| 995 |
+
"time_to_second_token": 0.020660570822656155,
|
| 996 |
+
"latency": 4.093334136996418,
|
| 997 |
+
"inter_token_latency_avg": 0.0077685065965720414,
|
| 998 |
+
"chunk_inter_token_latency_avg": 0.02100374005739848,
|
| 999 |
+
"input_tokens": 78,
|
| 1000 |
+
"output_tokens": 512,
|
| 1001 |
+
"output_tps_per_user": 128.72486977629183,
|
| 1002 |
+
"e2e_output_tps_per_user": 125.08140866694363,
|
| 1003 |
+
"completed": true
|
| 1004 |
+
},
|
| 1005 |
+
{
|
| 1006 |
+
"ttft": 0.12366992980241776,
|
| 1007 |
+
"time_to_second_token": 0.02057293802499771,
|
| 1008 |
+
"latency": 3.8691232178825885,
|
| 1009 |
+
"inter_token_latency_avg": 0.007329654184109923,
|
| 1010 |
+
"chunk_inter_token_latency_avg": 0.02092432004514062,
|
| 1011 |
+
"input_tokens": 78,
|
| 1012 |
+
"output_tokens": 512,
|
| 1013 |
+
"output_tps_per_user": 136.43208463612325,
|
| 1014 |
+
"e2e_output_tps_per_user": 132.32972205010222,
|
| 1015 |
+
"completed": true
|
| 1016 |
+
},
|
| 1017 |
+
{
|
| 1018 |
+
"ttft": 0.11748491204343736,
|
| 1019 |
+
"time_to_second_token": 0.020382387097924948,
|
| 1020 |
+
"latency": 4.307583688991144,
|
| 1021 |
+
"inter_token_latency_avg": 0.008199801911835043,
|
| 1022 |
+
"chunk_inter_token_latency_avg": 0.021269536938820846,
|
| 1023 |
+
"input_tokens": 78,
|
| 1024 |
+
"output_tokens": 512,
|
| 1025 |
+
"output_tps_per_user": 121.9541655703496,
|
| 1026 |
+
"e2e_output_tps_per_user": 118.86013992218285,
|
| 1027 |
+
"completed": true
|
| 1028 |
+
},
|
| 1029 |
+
{
|
| 1030 |
+
"ttft": 0.12366419215686619,
|
| 1031 |
+
"time_to_second_token": 0.02027744590304792,
|
| 1032 |
+
"latency": 4.018956256099045,
|
| 1033 |
+
"inter_token_latency_avg": 0.007622880751354558,
|
| 1034 |
+
"chunk_inter_token_latency_avg": 0.02094243045130204,
|
| 1035 |
+
"input_tokens": 78,
|
| 1036 |
+
"output_tokens": 512,
|
| 1037 |
+
"output_tps_per_user": 131.18400151049244,
|
| 1038 |
+
"e2e_output_tps_per_user": 127.39626096278218,
|
| 1039 |
+
"completed": true
|
| 1040 |
+
},
|
| 1041 |
+
{
|
| 1042 |
+
"ttft": 0.1245082609821111,
|
| 1043 |
+
"time_to_second_token": 0.020563303027302027,
|
| 1044 |
+
"latency": 0.0,
|
| 1045 |
+
"inter_token_latency_avg": 0.007016667529746,
|
| 1046 |
+
"chunk_inter_token_latency_avg": 0.020894076644132533,
|
| 1047 |
+
"input_tokens": 78,
|
| 1048 |
+
"output_tokens": 135,
|
| 1049 |
+
"output_tps_per_user": 142.517797196556,
|
| 1050 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1051 |
+
"completed": false
|
| 1052 |
+
}
|
| 1053 |
+
],
|
| 1054 |
+
"total_tokens": 5152,
|
| 1055 |
+
"wall_time": 25.544181779026985,
|
| 1056 |
+
"num_completed": 2,
|
| 1057 |
+
"num_errors": 0,
|
| 1058 |
+
"server_gen_throughput": 257.5183083679263,
|
| 1059 |
+
"server_utilization": 0.011725293132328285,
|
| 1060 |
+
"server_spec_accept_rate": 0.5490196078431373,
|
| 1061 |
+
"server_spec_accept_length": 0.0,
|
| 1062 |
+
"avg_running_reqs": 2,
|
| 1063 |
+
"max_running_reqs": 2,
|
| 1064 |
+
"effective_concurrency": 2,
|
| 1065 |
+
"avg_queue_reqs": 0,
|
| 1066 |
+
"max_queue_reqs": 0,
|
| 1067 |
+
"queue_fraction": 0.0,
|
| 1068 |
+
"underfilled": false,
|
| 1069 |
+
"warmup_timed_out": false,
|
| 1070 |
+
"warmup_duration": 5.536,
|
| 1071 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1072 |
+
"timeout_reason": "",
|
| 1073 |
+
"capacity_limited": false,
|
| 1074 |
+
"hardware_summary": {
|
| 1075 |
+
"samples": 9,
|
| 1076 |
+
"duration_seconds": 19.327,
|
| 1077 |
+
"gpu_count": 4,
|
| 1078 |
+
"cpu_util_avg_pct": 11.5,
|
| 1079 |
+
"cpu_temp_max_c": 76.5,
|
| 1080 |
+
"gpu_util_avg_pct": 100.0,
|
| 1081 |
+
"gpu_util_max_pct": 100.0,
|
| 1082 |
+
"mem_util_avg_pct": 40.53,
|
| 1083 |
+
"mem_util_max_pct": 50.0,
|
| 1084 |
+
"temp_avg_c": 68.83,
|
| 1085 |
+
"temp_max_c": 84.0,
|
| 1086 |
+
"power_total_avg_w": 1175.08,
|
| 1087 |
+
"power_total_max_w": 1176.23,
|
| 1088 |
+
"power_limit_total_w": 1200.0,
|
| 1089 |
+
"vram_used_avg_mb": 384778.0,
|
| 1090 |
+
"vram_used_max_mb": 384778.0,
|
| 1091 |
+
"vram_total_mb": 391548.0,
|
| 1092 |
+
"vram_used_avg_pct": 98.27,
|
| 1093 |
+
"vram_used_max_pct": 98.27,
|
| 1094 |
+
"pcie_rx_avg_mb_s": 11424.78,
|
| 1095 |
+
"pcie_rx_max_mb_s": 12660.0,
|
| 1096 |
+
"pcie_tx_avg_mb_s": 11473.0,
|
| 1097 |
+
"pcie_tx_max_mb_s": 12876.0
|
| 1098 |
+
}
|
| 1099 |
+
},
|
| 1100 |
+
{
|
| 1101 |
+
"concurrency": 4,
|
| 1102 |
+
"context_tokens": 0,
|
| 1103 |
+
"benchmark_mode": "duration",
|
| 1104 |
+
"request_count_target": 0,
|
| 1105 |
+
"warmup_request_count": 0,
|
| 1106 |
+
"measurement_seconds": 19.982261,
|
| 1107 |
+
"measurement_wall_seconds": 20.001434,
|
| 1108 |
+
"client_output_tokens": 6107,
|
| 1109 |
+
"server_output_tokens": 6107,
|
| 1110 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1111 |
+
"aggregate_tps": 305.62106309879454,
|
| 1112 |
+
"per_request_avg_tps": 76.40526577469863,
|
| 1113 |
+
"ttft_avg": 0.18411949259461835,
|
| 1114 |
+
"ttft_p50": 0.18233595008496195,
|
| 1115 |
+
"ttft_p90": 0.22579583001788706,
|
| 1116 |
+
"ttft_p99": 0.24202772771241143,
|
| 1117 |
+
"time_to_second_token_avg": 0.032063150822068565,
|
| 1118 |
+
"time_to_second_token_p50": 0.03419885353650898,
|
| 1119 |
+
"time_to_second_token_p90": 0.034818764543160796,
|
| 1120 |
+
"time_to_second_token_p99": 0.03491839349735528,
|
| 1121 |
+
"request_latency_avg": 6.813750679527099,
|
| 1122 |
+
"request_latency_p50": 6.7830935755046085,
|
| 1123 |
+
"request_latency_p90": 7.136929157585837,
|
| 1124 |
+
"request_latency_p99": 7.218653435369488,
|
| 1125 |
+
"inter_token_latency_avg": 0.01272961827279658,
|
| 1126 |
+
"inter_token_latency_p50": 0.012860421986189724,
|
| 1127 |
+
"inter_token_latency_p90": 0.013537522864770526,
|
| 1128 |
+
"inter_token_latency_p99": 0.013757404152471607,
|
| 1129 |
+
"output_tps_per_user_avg": 78.81967025679022,
|
| 1130 |
+
"output_tps_per_user_p50": 77.75794916087145,
|
| 1131 |
+
"output_tps_per_user_p90": 85.34425839032679,
|
| 1132 |
+
"output_tps_per_user_p99": 89.87592923645526,
|
| 1133 |
+
"e2e_output_tps_per_user_avg": 75.21735206235604,
|
| 1134 |
+
"e2e_output_tps_per_user_p50": 75.48202324847628,
|
| 1135 |
+
"e2e_output_tps_per_user_p90": 78.42642900563833,
|
| 1136 |
+
"e2e_output_tps_per_user_p99": 78.66327604603737,
|
| 1137 |
+
"chunk_inter_token_latency_avg": 0.03457812773162968,
|
| 1138 |
+
"chunk_inter_token_latency_p50": 0.03461816136042967,
|
| 1139 |
+
"chunk_inter_token_latency_p90": 0.03496178867127844,
|
| 1140 |
+
"chunk_inter_token_latency_p99": 0.035020915078215,
|
| 1141 |
+
"input_seq_len_avg": 78.0,
|
| 1142 |
+
"output_seq_len_avg": 512.0,
|
| 1143 |
+
"output_seq_len_p50": 512.0,
|
| 1144 |
+
"output_seq_len_p90": 512.0,
|
| 1145 |
+
"output_seq_len_p99": 512.0,
|
| 1146 |
+
"request_count": 16,
|
| 1147 |
+
"completed_request_count": 12,
|
| 1148 |
+
"request_samples": [
|
| 1149 |
+
{
|
| 1150 |
+
"ttft": 0.071592987049371,
|
| 1151 |
+
"time_to_second_token": 0.013179793022572994,
|
| 1152 |
+
"latency": 6.610689978115261,
|
| 1153 |
+
"inter_token_latency_avg": 0.012796667301498806,
|
| 1154 |
+
"chunk_inter_token_latency_avg": 0.03405779682846818,
|
| 1155 |
+
"input_tokens": 78,
|
| 1156 |
+
"output_tokens": 512,
|
| 1157 |
+
"output_tps_per_user": 78.14534647492752,
|
| 1158 |
+
"e2e_output_tps_per_user": 77.45031179725261,
|
| 1159 |
+
"completed": true
|
| 1160 |
+
},
|
| 1161 |
+
{
|
| 1162 |
+
"ttft": 0.22638577106408775,
|
| 1163 |
+
"time_to_second_token": 0.034233805956318974,
|
| 1164 |
+
"latency": 6.729027182096615,
|
| 1165 |
+
"inter_token_latency_avg": 0.012725325657597902,
|
| 1166 |
+
"chunk_inter_token_latency_avg": 0.03458851814379004,
|
| 1167 |
+
"input_tokens": 78,
|
| 1168 |
+
"output_tokens": 512,
|
| 1169 |
+
"output_tps_per_user": 78.5834505856383,
|
| 1170 |
+
"e2e_output_tps_per_user": 76.08826449122355,
|
| 1171 |
+
"completed": true
|
| 1172 |
+
},
|
| 1173 |
+
{
|
| 1174 |
+
"ttft": 0.18156861700117588,
|
| 1175 |
+
"time_to_second_token": 0.034193265018984675,
|
| 1176 |
+
"latency": 6.771002053981647,
|
| 1177 |
+
"inter_token_latency_avg": 0.012895173066497987,
|
| 1178 |
+
"chunk_inter_token_latency_avg": 0.03468122861568669,
|
| 1179 |
+
"input_tokens": 78,
|
| 1180 |
+
"output_tokens": 512,
|
| 1181 |
+
"output_tps_per_user": 77.5483969732851,
|
| 1182 |
+
"e2e_output_tps_per_user": 75.61657726848887,
|
| 1183 |
+
"completed": true
|
| 1184 |
+
},
|
| 1185 |
+
{
|
| 1186 |
+
"ttft": 0.16493634996004403,
|
| 1187 |
+
"time_to_second_token": 0.03424763702787459,
|
| 1188 |
+
"latency": 0.0,
|
| 1189 |
+
"inter_token_latency_avg": 0.01179988086244578,
|
| 1190 |
+
"chunk_inter_token_latency_avg": 0.03492764735283951,
|
| 1191 |
+
"input_tokens": 78,
|
| 1192 |
+
"output_tokens": 445,
|
| 1193 |
+
"output_tps_per_user": 84.74661834786765,
|
| 1194 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1195 |
+
"completed": false
|
| 1196 |
+
},
|
| 1197 |
+
{
|
| 1198 |
+
"ttft": 0.18636853992938995,
|
| 1199 |
+
"time_to_second_token": 0.029447735054418445,
|
| 1200 |
+
"latency": 7.226989238988608,
|
| 1201 |
+
"inter_token_latency_avg": 0.013778122698746023,
|
| 1202 |
+
"chunk_inter_token_latency_avg": 0.03468286058649861,
|
| 1203 |
+
"input_tokens": 78,
|
| 1204 |
+
"output_tokens": 512,
|
| 1205 |
+
"output_tps_per_user": 72.57882818034507,
|
| 1206 |
+
"e2e_output_tps_per_user": 70.84554619755497,
|
| 1207 |
+
"completed": true
|
| 1208 |
+
},
|
| 1209 |
+
{
|
| 1210 |
+
"ttft": 0.18095432897098362,
|
| 1211 |
+
"time_to_second_token": 0.03420444205403328,
|
| 1212 |
+
"latency": 6.79518509702757,
|
| 1213 |
+
"inter_token_latency_avg": 0.012943700133183144,
|
| 1214 |
+
"chunk_inter_token_latency_avg": 0.034995929989717386,
|
| 1215 |
+
"input_tokens": 78,
|
| 1216 |
+
"output_tokens": 512,
|
| 1217 |
+
"output_tps_per_user": 77.25766123369529,
|
| 1218 |
+
"e2e_output_tps_per_user": 75.34746922846371,
|
| 1219 |
+
"completed": true
|
| 1220 |
+
},
|
| 1221 |
+
{
|
| 1222 |
+
"ttft": 0.18133209715597332,
|
| 1223 |
+
"time_to_second_token": 0.034592753974720836,
|
| 1224 |
+
"latency": 6.89423265401274,
|
| 1225 |
+
"inter_token_latency_avg": 0.01313679169639289,
|
| 1226 |
+
"chunk_inter_token_latency_avg": 0.03460258018998333,
|
| 1227 |
+
"input_tokens": 78,
|
| 1228 |
+
"output_tokens": 512,
|
| 1229 |
+
"output_tps_per_user": 76.12208696850851,
|
| 1230 |
+
"e2e_output_tps_per_user": 74.26497272354074,
|
| 1231 |
+
"completed": true
|
| 1232 |
+
},
|
| 1233 |
+
{
|
| 1234 |
+
"ttft": 0.18180311610922217,
|
| 1235 |
+
"time_to_second_token": 0.03340313700027764,
|
| 1236 |
+
"latency": 0.0,
|
| 1237 |
+
"inter_token_latency_avg": 0.011041162894689477,
|
| 1238 |
+
"chunk_inter_token_latency_avg": 0.034236164014541014,
|
| 1239 |
+
"input_tokens": 78,
|
| 1240 |
+
"output_tokens": 401,
|
| 1241 |
+
"output_tps_per_user": 90.57016996651457,
|
| 1242 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1243 |
+
"completed": false
|
| 1244 |
+
},
|
| 1245 |
+
{
|
| 1246 |
+
"ttft": 0.18593904399313033,
|
| 1247 |
+
"time_to_second_token": 0.029689945047721267,
|
| 1248 |
+
"latency": 6.507442394969985,
|
| 1249 |
+
"inter_token_latency_avg": 0.01237084804496449,
|
| 1250 |
+
"chunk_inter_token_latency_avg": 0.033804830753886926,
|
| 1251 |
+
"input_tokens": 78,
|
| 1252 |
+
"output_tokens": 512,
|
| 1253 |
+
"output_tps_per_user": 80.83520194939638,
|
| 1254 |
+
"e2e_output_tps_per_user": 78.67914442020374,
|
| 1255 |
+
"completed": true
|
| 1256 |
+
},
|
| 1257 |
+
{
|
| 1258 |
+
"ttft": 0.1811696880031377,
|
| 1259 |
+
"time_to_second_token": 0.034721852047368884,
|
| 1260 |
+
"latency": 7.151209206087515,
|
| 1261 |
+
"inter_token_latency_avg": 0.01363999905691659,
|
| 1262 |
+
"chunk_inter_token_latency_avg": 0.03502532421147928,
|
| 1263 |
+
"input_tokens": 78,
|
| 1264 |
+
"output_tokens": 512,
|
| 1265 |
+
"output_tps_per_user": 73.31378806019188,
|
| 1266 |
+
"e2e_output_tps_per_user": 71.5962832641166,
|
| 1267 |
+
"completed": true
|
| 1268 |
+
},
|
| 1269 |
+
{
|
| 1270 |
+
"ttft": 0.18142080190591514,
|
| 1271 |
+
"time_to_second_token": 0.03491567703895271,
|
| 1272 |
+
"latency": 6.5193956850562245,
|
| 1273 |
+
"inter_token_latency_avg": 0.01240308196311215,
|
| 1274 |
+
"chunk_inter_token_latency_avg": 0.034633742530876005,
|
| 1275 |
+
"input_tokens": 78,
|
| 1276 |
+
"output_tokens": 512,
|
| 1277 |
+
"output_tps_per_user": 80.62512228606465,
|
| 1278 |
+
"e2e_output_tps_per_user": 78.53488647323674,
|
| 1279 |
+
"completed": true
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"ttft": 0.24478807300329208,
|
| 1283 |
+
"time_to_second_token": 0.03329922794364393,
|
| 1284 |
+
"latency": 0.0,
|
| 1285 |
+
"inter_token_latency_avg": 0.01343504667262446,
|
| 1286 |
+
"chunk_inter_token_latency_avg": 0.034875908828251166,
|
| 1287 |
+
"input_tokens": 78,
|
| 1288 |
+
"output_tokens": 380,
|
| 1289 |
+
"output_tps_per_user": 74.43219397500282,
|
| 1290 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1291 |
+
"completed": false
|
| 1292 |
+
},
|
| 1293 |
+
{
|
| 1294 |
+
"ttft": 0.18580231512896717,
|
| 1295 |
+
"time_to_second_token": 0.029555089073255658,
|
| 1296 |
+
"latency": 7.008408721070737,
|
| 1297 |
+
"inter_token_latency_avg": 0.013351480246461388,
|
| 1298 |
+
"chunk_inter_token_latency_avg": 0.034457608110817016,
|
| 1299 |
+
"input_tokens": 78,
|
| 1300 |
+
"output_tokens": 512,
|
| 1301 |
+
"output_tps_per_user": 74.89806235267697,
|
| 1302 |
+
"e2e_output_tps_per_user": 73.05510000589366,
|
| 1303 |
+
"completed": true
|
| 1304 |
+
},
|
| 1305 |
+
{
|
| 1306 |
+
"ttft": 0.18286878406070173,
|
| 1307 |
+
"time_to_second_token": 0.0349188728723675,
|
| 1308 |
+
"latency": 6.753246791893616,
|
| 1309 |
+
"inter_token_latency_avg": 0.012857882598498854,
|
| 1310 |
+
"chunk_inter_token_latency_avg": 0.03458093688333113,
|
| 1311 |
+
"input_tokens": 78,
|
| 1312 |
+
"output_tokens": 512,
|
| 1313 |
+
"output_tps_per_user": 77.7733030566595,
|
| 1314 |
+
"e2e_output_tps_per_user": 75.81538418151527,
|
| 1315 |
+
"completed": true
|
| 1316 |
+
},
|
| 1317 |
+
{
|
| 1318 |
+
"ttft": 0.22520588897168636,
|
| 1319 |
+
"time_to_second_token": 0.03444401710294187,
|
| 1320 |
+
"latency": 6.798179151024669,
|
| 1321 |
+
"inter_token_latency_avg": 0.012862961373880592,
|
| 1322 |
+
"chunk_inter_token_latency_avg": 0.03477763630715864,
|
| 1323 |
+
"input_tokens": 78,
|
| 1324 |
+
"output_tokens": 512,
|
| 1325 |
+
"output_tps_per_user": 77.7425952650834,
|
| 1326 |
+
"e2e_output_tps_per_user": 75.31428469678204,
|
| 1327 |
+
"completed": true
|
| 1328 |
+
},
|
| 1329 |
+
{
|
| 1330 |
+
"ttft": 0.18377547920681536,
|
| 1331 |
+
"time_to_second_token": 0.033963162917643785,
|
| 1332 |
+
"latency": 0.0,
|
| 1333 |
+
"inter_token_latency_avg": 0.011635768097234754,
|
| 1334 |
+
"chunk_inter_token_latency_avg": 0.03432133035874999,
|
| 1335 |
+
"input_tokens": 78,
|
| 1336 |
+
"output_tokens": 411,
|
| 1337 |
+
"output_tps_per_user": 85.94189843278592,
|
| 1338 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1339 |
+
"completed": false
|
| 1340 |
+
}
|
| 1341 |
+
],
|
| 1342 |
+
"total_tokens": 6107,
|
| 1343 |
+
"wall_time": 25.55054651084356,
|
| 1344 |
+
"num_completed": 4,
|
| 1345 |
+
"num_errors": 0,
|
| 1346 |
+
"server_gen_throughput": 305.247463685858,
|
| 1347 |
+
"server_utilization": 0.02345058626465657,
|
| 1348 |
+
"server_spec_accept_rate": 0.5805555555555556,
|
| 1349 |
+
"server_spec_accept_length": 0.0,
|
| 1350 |
+
"avg_running_reqs": 3.9,
|
| 1351 |
+
"max_running_reqs": 4,
|
| 1352 |
+
"effective_concurrency": 3.9,
|
| 1353 |
+
"avg_queue_reqs": 0.1,
|
| 1354 |
+
"max_queue_reqs": 1,
|
| 1355 |
+
"queue_fraction": 0.05,
|
| 1356 |
+
"underfilled": true,
|
| 1357 |
+
"warmup_timed_out": false,
|
| 1358 |
+
"warmup_duration": 5.534,
|
| 1359 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1360 |
+
"timeout_reason": "",
|
| 1361 |
+
"capacity_limited": true,
|
| 1362 |
+
"hardware_summary": {
|
| 1363 |
+
"samples": 8,
|
| 1364 |
+
"duration_seconds": 16.89,
|
| 1365 |
+
"gpu_count": 4,
|
| 1366 |
+
"cpu_util_avg_pct": 11.53,
|
| 1367 |
+
"cpu_temp_max_c": 77.38,
|
| 1368 |
+
"gpu_util_avg_pct": 100.0,
|
| 1369 |
+
"gpu_util_max_pct": 100.0,
|
| 1370 |
+
"mem_util_avg_pct": 35.53,
|
| 1371 |
+
"mem_util_max_pct": 45.0,
|
| 1372 |
+
"temp_avg_c": 69.31,
|
| 1373 |
+
"temp_max_c": 84.0,
|
| 1374 |
+
"power_total_avg_w": 1171.59,
|
| 1375 |
+
"power_total_max_w": 1174.02,
|
| 1376 |
+
"power_limit_total_w": 1200.0,
|
| 1377 |
+
"vram_used_avg_mb": 384778.0,
|
| 1378 |
+
"vram_used_max_mb": 384778.0,
|
| 1379 |
+
"vram_total_mb": 391548.0,
|
| 1380 |
+
"vram_used_avg_pct": 98.27,
|
| 1381 |
+
"vram_used_max_pct": 98.27,
|
| 1382 |
+
"pcie_rx_avg_mb_s": 8123.88,
|
| 1383 |
+
"pcie_rx_max_mb_s": 11132.0,
|
| 1384 |
+
"pcie_tx_avg_mb_s": 8456.62,
|
| 1385 |
+
"pcie_tx_max_mb_s": 11467.0
|
| 1386 |
+
}
|
| 1387 |
+
},
|
| 1388 |
+
{
|
| 1389 |
+
"concurrency": 2,
|
| 1390 |
+
"context_tokens": 8192,
|
| 1391 |
+
"benchmark_mode": "duration",
|
| 1392 |
+
"request_count_target": 0,
|
| 1393 |
+
"warmup_request_count": 0,
|
| 1394 |
+
"measurement_seconds": 19.996131,
|
| 1395 |
+
"measurement_wall_seconds": 20.000268,
|
| 1396 |
+
"client_output_tokens": 3670,
|
| 1397 |
+
"server_output_tokens": 3670,
|
| 1398 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1399 |
+
"aggregate_tps": 183.53550127953065,
|
| 1400 |
+
"per_request_avg_tps": 91.76775063976532,
|
| 1401 |
+
"ttft_avg": 1.2004322615684941,
|
| 1402 |
+
"ttft_p50": 1.2173257099930197,
|
| 1403 |
+
"ttft_p90": 1.3659410797525195,
|
| 1404 |
+
"ttft_p99": 1.9023066108068454,
|
| 1405 |
+
"time_to_second_token_avg": 0.02541845981031656,
|
| 1406 |
+
"time_to_second_token_p50": 0.027192589943297207,
|
| 1407 |
+
"time_to_second_token_p90": 0.027629395388066767,
|
| 1408 |
+
"time_to_second_token_p99": 0.027753254855051635,
|
| 1409 |
+
"request_latency_avg": 5.37285651997081,
|
| 1410 |
+
"request_latency_p50": 5.446895623463206,
|
| 1411 |
+
"request_latency_p90": 5.665212600212544,
|
| 1412 |
+
"request_latency_p99": 5.969511628868058,
|
| 1413 |
+
"inter_token_latency_avg": 0.00809208952899205,
|
| 1414 |
+
"inter_token_latency_p50": 0.00815642245274234,
|
| 1415 |
+
"inter_token_latency_p90": 0.008585907641263088,
|
| 1416 |
+
"inter_token_latency_p99": 0.009444491741223483,
|
| 1417 |
+
"output_tps_per_user_avg": 124.44614817528017,
|
| 1418 |
+
"output_tps_per_user_p50": 122.61098226109307,
|
| 1419 |
+
"output_tps_per_user_p90": 140.4927553359884,
|
| 1420 |
+
"output_tps_per_user_p99": 142.21413373382023,
|
| 1421 |
+
"e2e_output_tps_per_user_avg": 95.63015924091229,
|
| 1422 |
+
"e2e_output_tps_per_user_p50": 93.9996349349982,
|
| 1423 |
+
"e2e_output_tps_per_user_p90": 103.25204134689379,
|
| 1424 |
+
"e2e_output_tps_per_user_p99": 103.55866690719752,
|
| 1425 |
+
"chunk_inter_token_latency_avg": 0.023212736865606577,
|
| 1426 |
+
"chunk_inter_token_latency_p50": 0.02336988428975669,
|
| 1427 |
+
"chunk_inter_token_latency_p90": 0.0253701743817706,
|
| 1428 |
+
"chunk_inter_token_latency_p99": 0.02550767397899725,
|
| 1429 |
+
"input_seq_len_avg": 8192.0,
|
| 1430 |
+
"output_seq_len_avg": 512.0,
|
| 1431 |
+
"output_seq_len_p50": 512.0,
|
| 1432 |
+
"output_seq_len_p90": 512.0,
|
| 1433 |
+
"output_seq_len_p99": 512.0,
|
| 1434 |
+
"request_count": 10,
|
| 1435 |
+
"completed_request_count": 8,
|
| 1436 |
+
"request_samples": [
|
| 1437 |
+
{
|
| 1438 |
+
"ttft": 0.5909663320053369,
|
| 1439 |
+
"time_to_second_token": 0.018111236859112978,
|
| 1440 |
+
"latency": 5.465850109001622,
|
| 1441 |
+
"inter_token_latency_avg": 0.009539889974552416,
|
| 1442 |
+
"chunk_inter_token_latency_avg": 0.025522951712022433,
|
| 1443 |
+
"input_tokens": 8192,
|
| 1444 |
+
"output_tokens": 512,
|
| 1445 |
+
"output_tps_per_user": 104.82301186570206,
|
| 1446 |
+
"e2e_output_tps_per_user": 93.67252847947574,
|
| 1447 |
+
"completed": true
|
| 1448 |
+
},
|
| 1449 |
+
{
|
| 1450 |
+
"ttft": 1.9619027809239924,
|
| 1451 |
+
"time_to_second_token": 0.027767017018049955,
|
| 1452 |
+
"latency": 6.003322632052004,
|
| 1453 |
+
"inter_token_latency_avg": 0.007908845109839554,
|
| 1454 |
+
"chunk_inter_token_latency_avg": 0.023634034217122877,
|
| 1455 |
+
"input_tokens": 8192,
|
| 1456 |
+
"output_tokens": 512,
|
| 1457 |
+
"output_tps_per_user": 126.44071114199465,
|
| 1458 |
+
"e2e_output_tps_per_user": 85.28610427605696,
|
| 1459 |
+
"completed": true
|
| 1460 |
+
},
|
| 1461 |
+
{
|
| 1462 |
+
"ttft": 1.2156082668807358,
|
| 1463 |
+
"time_to_second_token": 0.026928086066618562,
|
| 1464 |
+
"latency": 5.520308300852776,
|
| 1465 |
+
"inter_token_latency_avg": 0.008424070516579334,
|
| 1466 |
+
"chunk_inter_token_latency_avg": 0.023395108880282824,
|
| 1467 |
+
"input_tokens": 8192,
|
| 1468 |
+
"output_tokens": 512,
|
| 1469 |
+
"output_tps_per_user": 118.70745835186321,
|
| 1470 |
+
"e2e_output_tps_per_user": 92.74844303911547,
|
| 1471 |
+
"completed": true
|
| 1472 |
+
},
|
| 1473 |
+
{
|
| 1474 |
+
"ttft": 1.2087725747842342,
|
| 1475 |
+
"time_to_second_token": 0.027614104095846415,
|
| 1476 |
+
"latency": 5.18546292395331,
|
| 1477 |
+
"inter_token_latency_avg": 0.007782172894655725,
|
| 1478 |
+
"chunk_inter_token_latency_avg": 0.0235307121252608,
|
| 1479 |
+
"input_tokens": 8192,
|
| 1480 |
+
"output_tokens": 512,
|
| 1481 |
+
"output_tps_per_user": 128.4988156311373,
|
| 1482 |
+
"e2e_output_tps_per_user": 98.73756837309712,
|
| 1483 |
+
"completed": true
|
| 1484 |
+
},
|
| 1485 |
+
{
|
| 1486 |
+
"ttft": 1.2152251708321273,
|
| 1487 |
+
"time_to_second_token": 0.02757256105542183,
|
| 1488 |
+
"latency": 0.0,
|
| 1489 |
+
"inter_token_latency_avg": 0.007022205717217775,
|
| 1490 |
+
"chunk_inter_token_latency_avg": 0.021066617151653325,
|
| 1491 |
+
"input_tokens": 8192,
|
| 1492 |
+
"output_tokens": 163,
|
| 1493 |
+
"output_tps_per_user": 142.40539800024598,
|
| 1494 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1495 |
+
"completed": false
|
| 1496 |
+
},
|
| 1497 |
+
{
|
| 1498 |
+
"ttft": 1.2997231129556894,
|
| 1499 |
+
"time_to_second_token": 0.02719552395865321,
|
| 1500 |
+
"latency": 4.942431465024129,
|
| 1501 |
+
"inter_token_latency_avg": 0.007128587773128061,
|
| 1502 |
+
"chunk_inter_token_latency_avg": 0.020935105471657695,
|
| 1503 |
+
"input_tokens": 8192,
|
| 1504 |
+
"output_tokens": 512,
|
| 1505 |
+
"output_tps_per_user": 140.2802394844042,
|
| 1506 |
+
"e2e_output_tps_per_user": 103.59273641389794,
|
| 1507 |
+
"completed": true
|
| 1508 |
+
},
|
| 1509 |
+
{
|
| 1510 |
+
"ttft": 0.8319369831588119,
|
| 1511 |
+
"time_to_second_token": 0.017664924962446094,
|
| 1512 |
+
"latency": 4.9657619839999825,
|
| 1513 |
+
"inter_token_latency_avg": 0.008089677105364327,
|
| 1514 |
+
"chunk_inter_token_latency_avg": 0.022106016047278985,
|
| 1515 |
+
"input_tokens": 8192,
|
| 1516 |
+
"output_tokens": 512,
|
| 1517 |
+
"output_tps_per_user": 123.61432810920134,
|
| 1518 |
+
"e2e_output_tps_per_user": 103.10602917532059,
|
| 1519 |
+
"completed": true
|
| 1520 |
+
},
|
| 1521 |
+
{
|
| 1522 |
+
"ttft": 1.2190431531053036,
|
| 1523 |
+
"time_to_second_token": 0.027477154973894358,
|
| 1524 |
+
"latency": 5.471773606957868,
|
| 1525 |
+
"inter_token_latency_avg": 0.008322368794232024,
|
| 1526 |
+
"chunk_inter_token_latency_avg": 0.023238964228702537,
|
| 1527 |
+
"input_tokens": 8192,
|
| 1528 |
+
"output_tokens": 512,
|
| 1529 |
+
"output_tps_per_user": 120.15809737884591,
|
| 1530 |
+
"e2e_output_tps_per_user": 93.57112277981393,
|
| 1531 |
+
"completed": true
|
| 1532 |
+
},
|
| 1533 |
+
{
|
| 1534 |
+
"ttft": 1.2259023920632899,
|
| 1535 |
+
"time_to_second_token": 0.027189655927941203,
|
| 1536 |
+
"latency": 5.42794113792479,
|
| 1537 |
+
"inter_token_latency_avg": 0.008223167800120354,
|
| 1538 |
+
"chunk_inter_token_latency_avg": 0.02334465969923056,
|
| 1539 |
+
"input_tokens": 8192,
|
| 1540 |
+
"output_tokens": 512,
|
| 1541 |
+
"output_tps_per_user": 121.6076364129848,
|
| 1542 |
+
"e2e_output_tps_per_user": 94.32674139052064,
|
| 1543 |
+
"completed": true
|
| 1544 |
+
},
|
| 1545 |
+
{
|
| 1546 |
+
"ttft": 1.23524184897542,
|
| 1547 |
+
"time_to_second_token": 0.02666433318518102,
|
| 1548 |
+
"latency": 0.0,
|
| 1549 |
+
"inter_token_latency_avg": 0.00847990960423094,
|
| 1550 |
+
"chunk_inter_token_latency_avg": 0.02535319912285373,
|
| 1551 |
+
"input_tokens": 8192,
|
| 1552 |
+
"output_tokens": 294,
|
| 1553 |
+
"output_tps_per_user": 117.9257853764223,
|
| 1554 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1555 |
+
"completed": false
|
| 1556 |
+
}
|
| 1557 |
+
],
|
| 1558 |
+
"total_tokens": 3670,
|
| 1559 |
+
"wall_time": 25.57071568304673,
|
| 1560 |
+
"num_completed": 2,
|
| 1561 |
+
"num_errors": 0,
|
| 1562 |
+
"server_gen_throughput": 183.44782855657667,
|
| 1563 |
+
"server_utilization": 0.012283640424343933,
|
| 1564 |
+
"server_spec_accept_rate": 0.6352201257861635,
|
| 1565 |
+
"server_spec_accept_length": 0.0,
|
| 1566 |
+
"avg_running_reqs": 1.9,
|
| 1567 |
+
"max_running_reqs": 2,
|
| 1568 |
+
"effective_concurrency": 1.9,
|
| 1569 |
+
"avg_queue_reqs": 0,
|
| 1570 |
+
"max_queue_reqs": 0,
|
| 1571 |
+
"queue_fraction": 0.0,
|
| 1572 |
+
"underfilled": true,
|
| 1573 |
+
"warmup_timed_out": false,
|
| 1574 |
+
"warmup_duration": 5.553,
|
| 1575 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1576 |
+
"timeout_reason": "",
|
| 1577 |
+
"capacity_limited": false,
|
| 1578 |
+
"hardware_summary": {
|
| 1579 |
+
"samples": 8,
|
| 1580 |
+
"duration_seconds": 16.958,
|
| 1581 |
+
"gpu_count": 4,
|
| 1582 |
+
"cpu_util_avg_pct": 11.59,
|
| 1583 |
+
"cpu_temp_max_c": 76.75,
|
| 1584 |
+
"gpu_util_avg_pct": 99.81,
|
| 1585 |
+
"gpu_util_max_pct": 100.0,
|
| 1586 |
+
"mem_util_avg_pct": 37.69,
|
| 1587 |
+
"mem_util_max_pct": 55.0,
|
| 1588 |
+
"temp_avg_c": 68.81,
|
| 1589 |
+
"temp_max_c": 84.0,
|
| 1590 |
+
"power_total_avg_w": 1163.89,
|
| 1591 |
+
"power_total_max_w": 1178.06,
|
| 1592 |
+
"power_limit_total_w": 1200.0,
|
| 1593 |
+
"vram_used_avg_mb": 384778.0,
|
| 1594 |
+
"vram_used_max_mb": 384778.0,
|
| 1595 |
+
"vram_total_mb": 391548.0,
|
| 1596 |
+
"vram_used_avg_pct": 98.27,
|
| 1597 |
+
"vram_used_max_pct": 98.27,
|
| 1598 |
+
"pcie_rx_avg_mb_s": 37319.5,
|
| 1599 |
+
"pcie_rx_max_mb_s": 67250.0,
|
| 1600 |
+
"pcie_tx_avg_mb_s": 33538.0,
|
| 1601 |
+
"pcie_tx_max_mb_s": 67525.0
|
| 1602 |
+
}
|
| 1603 |
+
},
|
| 1604 |
+
{
|
| 1605 |
+
"concurrency": 4,
|
| 1606 |
+
"context_tokens": 8192,
|
| 1607 |
+
"benchmark_mode": "duration",
|
| 1608 |
+
"request_count_target": 0,
|
| 1609 |
+
"warmup_request_count": 0,
|
| 1610 |
+
"measurement_seconds": 19.987172,
|
| 1611 |
+
"measurement_wall_seconds": 20.001291,
|
| 1612 |
+
"client_output_tokens": 4295,
|
| 1613 |
+
"server_output_tokens": 4295,
|
| 1614 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1615 |
+
"aggregate_tps": 214.8878284941506,
|
| 1616 |
+
"per_request_avg_tps": 53.72195712353765,
|
| 1617 |
+
"ttft_avg": 1.7763419252975534,
|
| 1618 |
+
"ttft_p50": 1.3619820600142702,
|
| 1619 |
+
"ttft_p90": 2.3120769894914703,
|
| 1620 |
+
"ttft_p99": 4.039579109621701,
|
| 1621 |
+
"time_to_second_token_avg": 0.03882854864544546,
|
| 1622 |
+
"time_to_second_token_p50": 0.04027673741802573,
|
| 1623 |
+
"time_to_second_token_p90": 0.043103844858706,
|
| 1624 |
+
"time_to_second_token_p99": 0.043685059298295525,
|
| 1625 |
+
"request_latency_avg": 9.80318702897057,
|
| 1626 |
+
"request_latency_p50": 9.880283242091537,
|
| 1627 |
+
"request_latency_p90": 10.619067505188287,
|
| 1628 |
+
"request_latency_p99": 11.727601487580687,
|
| 1629 |
+
"inter_token_latency_avg": 0.015157077239068617,
|
| 1630 |
+
"inter_token_latency_p50": 0.01507125252269121,
|
| 1631 |
+
"inter_token_latency_p90": 0.016277488439941666,
|
| 1632 |
+
"inter_token_latency_p99": 0.01681107227458143,
|
| 1633 |
+
"output_tps_per_user_avg": 66.20683975505182,
|
| 1634 |
+
"output_tps_per_user_p50": 66.35252027526646,
|
| 1635 |
+
"output_tps_per_user_p90": 70.24470636649606,
|
| 1636 |
+
"output_tps_per_user_p99": 74.73678565140237,
|
| 1637 |
+
"e2e_output_tps_per_user_avg": 52.8389794149083,
|
| 1638 |
+
"e2e_output_tps_per_user_p50": 51.82037674980822,
|
| 1639 |
+
"e2e_output_tps_per_user_p90": 58.62424331130355,
|
| 1640 |
+
"e2e_output_tps_per_user_p99": 64.67365860038261,
|
| 1641 |
+
"chunk_inter_token_latency_avg": 0.04252216480840284,
|
| 1642 |
+
"chunk_inter_token_latency_p50": 0.042666952384107254,
|
| 1643 |
+
"chunk_inter_token_latency_p90": 0.044934111945601436,
|
| 1644 |
+
"chunk_inter_token_latency_p99": 0.04507588112417429,
|
| 1645 |
+
"input_seq_len_avg": 8192.0,
|
| 1646 |
+
"output_seq_len_avg": 512.0,
|
| 1647 |
+
"output_seq_len_p50": 512.0,
|
| 1648 |
+
"output_seq_len_p90": 512.0,
|
| 1649 |
+
"output_seq_len_p99": 512.0,
|
| 1650 |
+
"request_count": 12,
|
| 1651 |
+
"completed_request_count": 9,
|
| 1652 |
+
"request_samples": [
|
| 1653 |
+
{
|
| 1654 |
+
"ttft": 0.5923555630724877,
|
| 1655 |
+
"time_to_second_token": 0.017299477010965347,
|
| 1656 |
+
"latency": 7.835237701190636,
|
| 1657 |
+
"inter_token_latency_avg": 0.01417393764798072,
|
| 1658 |
+
"chunk_inter_token_latency_avg": 0.04069034909055139,
|
| 1659 |
+
"input_tokens": 8192,
|
| 1660 |
+
"output_tokens": 512,
|
| 1661 |
+
"output_tps_per_user": 70.55202476797012,
|
| 1662 |
+
"e2e_output_tps_per_user": 65.34581585472473,
|
| 1663 |
+
"completed": true
|
| 1664 |
+
},
|
| 1665 |
+
{
|
| 1666 |
+
"ttft": 1.2594965409953147,
|
| 1667 |
+
"time_to_second_token": 0.03946907399222255,
|
| 1668 |
+
"latency": 8.991313345031813,
|
| 1669 |
+
"inter_token_latency_avg": 0.01513075695506164,
|
| 1670 |
+
"chunk_inter_token_latency_avg": 0.042717219911803855,
|
| 1671 |
+
"input_tokens": 8192,
|
| 1672 |
+
"output_tokens": 512,
|
| 1673 |
+
"output_tps_per_user": 66.09054675651726,
|
| 1674 |
+
"e2e_output_tps_per_user": 56.943850175448254,
|
| 1675 |
+
"completed": true
|
| 1676 |
+
},
|
| 1677 |
+
{
|
| 1678 |
+
"ttft": 1.2591414290945977,
|
| 1679 |
+
"time_to_second_token": 0.04004574380815029,
|
| 1680 |
+
"latency": 9.880283242091537,
|
| 1681 |
+
"inter_token_latency_avg": 0.0168711190078218,
|
| 1682 |
+
"chunk_inter_token_latency_avg": 0.044901780276025725,
|
| 1683 |
+
"input_tokens": 8192,
|
| 1684 |
+
"output_tokens": 512,
|
| 1685 |
+
"output_tps_per_user": 59.27289111862582,
|
| 1686 |
+
"e2e_output_tps_per_user": 51.82037674980822,
|
| 1687 |
+
"completed": true
|
| 1688 |
+
},
|
| 1689 |
+
{
|
| 1690 |
+
"ttft": 2.212953792186454,
|
| 1691 |
+
"time_to_second_token": 0.04086994403041899,
|
| 1692 |
+
"latency": 9.785697993123904,
|
| 1693 |
+
"inter_token_latency_avg": 0.014819460275807142,
|
| 1694 |
+
"chunk_inter_token_latency_avg": 0.04160848462053544,
|
| 1695 |
+
"input_tokens": 8192,
|
| 1696 |
+
"output_tokens": 512,
|
| 1697 |
+
"output_tps_per_user": 67.47884075322945,
|
| 1698 |
+
"e2e_output_tps_per_user": 52.32125499476542,
|
| 1699 |
+
"completed": true
|
| 1700 |
+
},
|
| 1701 |
+
{
|
| 1702 |
+
"ttft": 1.4577314089983702,
|
| 1703 |
+
"time_to_second_token": 0.043352056061849,
|
| 1704 |
+
"latency": 9.128734683152288,
|
| 1705 |
+
"inter_token_latency_avg": 0.015011748090320779,
|
| 1706 |
+
"chunk_inter_token_latency_avg": 0.04261668485641065,
|
| 1707 |
+
"input_tokens": 8192,
|
| 1708 |
+
"output_tokens": 512,
|
| 1709 |
+
"output_tps_per_user": 66.61449379401566,
|
| 1710 |
+
"e2e_output_tps_per_user": 56.08663388420428,
|
| 1711 |
+
"completed": true
|
| 1712 |
+
},
|
| 1713 |
+
{
|
| 1714 |
+
"ttft": 1.2618389299605042,
|
| 1715 |
+
"time_to_second_token": 0.04062130395323038,
|
| 1716 |
+
"latency": 0.0,
|
| 1717 |
+
"inter_token_latency_avg": 0.015226825442038138,
|
| 1718 |
+
"chunk_inter_token_latency_avg": 0.04493770435333207,
|
| 1719 |
+
"input_tokens": 8192,
|
| 1720 |
+
"output_tokens": 485,
|
| 1721 |
+
"output_tps_per_user": 65.67357088360686,
|
| 1722 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1723 |
+
"completed": false
|
| 1724 |
+
},
|
| 1725 |
+
{
|
| 1726 |
+
"ttft": 2.2129524589981884,
|
| 1727 |
+
"time_to_second_token": 0.04079655185341835,
|
| 1728 |
+
"latency": 10.311141398968175,
|
| 1729 |
+
"inter_token_latency_avg": 0.015847727866868857,
|
| 1730 |
+
"chunk_inter_token_latency_avg": 0.04307547308494674,
|
| 1731 |
+
"input_tokens": 8192,
|
| 1732 |
+
"output_tokens": 512,
|
| 1733 |
+
"output_tps_per_user": 63.100528252418606,
|
| 1734 |
+
"e2e_output_tps_per_user": 49.65502655712153,
|
| 1735 |
+
"completed": true
|
| 1736 |
+
},
|
| 1737 |
+
{
|
| 1738 |
+
"ttft": 2.3230906780809164,
|
| 1739 |
+
"time_to_second_token": 0.04372621700167656,
|
| 1740 |
+
"latency": 10.149147124029696,
|
| 1741 |
+
"inter_token_latency_avg": 0.01531517895488998,
|
| 1742 |
+
"chunk_inter_token_latency_avg": 0.04372098573155743,
|
| 1743 |
+
"input_tokens": 8192,
|
| 1744 |
+
"output_tokens": 512,
|
| 1745 |
+
"output_tps_per_user": 65.29469900060882,
|
| 1746 |
+
"e2e_output_tps_per_user": 50.447588722776494,
|
| 1747 |
+
"completed": true
|
| 1748 |
+
},
|
| 1749 |
+
{
|
| 1750 |
+
"ttft": 1.2644218259956688,
|
| 1751 |
+
"time_to_second_token": 0.03977838600985706,
|
| 1752 |
+
"latency": 0.0,
|
| 1753 |
+
"inter_token_latency_avg": 0.015003678621124169,
|
| 1754 |
+
"chunk_inter_token_latency_avg": 0.04158162360711556,
|
| 1755 |
+
"input_tokens": 8192,
|
| 1756 |
+
"output_tokens": 389,
|
| 1757 |
+
"output_tps_per_user": 66.65032124802163,
|
| 1758 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1759 |
+
"completed": false
|
| 1760 |
+
},
|
| 1761 |
+
{
|
| 1762 |
+
"ttft": 4.251729365205392,
|
| 1763 |
+
"time_to_second_token": 0.03976707882247865,
|
| 1764 |
+
"latency": 11.850771930068731,
|
| 1765 |
+
"inter_token_latency_avg": 0.014870924784468375,
|
| 1766 |
+
"chunk_inter_token_latency_avg": 0.04175298112562274,
|
| 1767 |
+
"input_tokens": 8192,
|
| 1768 |
+
"output_tokens": 512,
|
| 1769 |
+
"output_tps_per_user": 67.24531355604925,
|
| 1770 |
+
"e2e_output_tps_per_user": 43.203936673602875,
|
| 1771 |
+
"completed": true
|
| 1772 |
+
},
|
| 1773 |
+
{
|
| 1774 |
+
"ttft": 1.9541583999525756,
|
| 1775 |
+
"time_to_second_token": 0.03970902017317712,
|
| 1776 |
+
"latency": 10.296355843078345,
|
| 1777 |
+
"inter_token_latency_avg": 0.016325239614727535,
|
| 1778 |
+
"chunk_inter_token_latency_avg": 0.04509295915203119,
|
| 1779 |
+
"input_tokens": 8192,
|
| 1780 |
+
"output_tokens": 512,
|
| 1781 |
+
"output_tps_per_user": 61.254843640877844,
|
| 1782 |
+
"e2e_output_tps_per_user": 49.726331121722886,
|
| 1783 |
+
"completed": true
|
| 1784 |
+
},
|
| 1785 |
+
{
|
| 1786 |
+
"ttft": 1.2662327110301703,
|
| 1787 |
+
"time_to_second_token": 0.04050773102790117,
|
| 1788 |
+
"latency": 0.0,
|
| 1789 |
+
"inter_token_latency_avg": 0.013288329607714268,
|
| 1790 |
+
"chunk_inter_token_latency_avg": 0.037569731890901244,
|
| 1791 |
+
"input_tokens": 8192,
|
| 1792 |
+
"output_tokens": 312,
|
| 1793 |
+
"output_tps_per_user": 75.25400328868051,
|
| 1794 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1795 |
+
"completed": false
|
| 1796 |
+
}
|
| 1797 |
+
],
|
| 1798 |
+
"total_tokens": 4295,
|
| 1799 |
+
"wall_time": 28.98594278888777,
|
| 1800 |
+
"num_completed": 4,
|
| 1801 |
+
"num_errors": 0,
|
| 1802 |
+
"server_gen_throughput": 214.68362406405674,
|
| 1803 |
+
"server_utilization": 0.008933556672250154,
|
| 1804 |
+
"server_spec_accept_rate": 0.603448275862069,
|
| 1805 |
+
"server_spec_accept_length": 0.0,
|
| 1806 |
+
"avg_running_reqs": 3.9,
|
| 1807 |
+
"max_running_reqs": 4,
|
| 1808 |
+
"effective_concurrency": 3.9,
|
| 1809 |
+
"avg_queue_reqs": 0.1,
|
| 1810 |
+
"max_queue_reqs": 1,
|
| 1811 |
+
"queue_fraction": 0.05,
|
| 1812 |
+
"underfilled": true,
|
| 1813 |
+
"warmup_timed_out": false,
|
| 1814 |
+
"warmup_duration": 8.579,
|
| 1815 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1816 |
+
"timeout_reason": "",
|
| 1817 |
+
"capacity_limited": true,
|
| 1818 |
+
"hardware_summary": {
|
| 1819 |
+
"samples": 8,
|
| 1820 |
+
"duration_seconds": 16.84,
|
| 1821 |
+
"gpu_count": 4,
|
| 1822 |
+
"cpu_util_avg_pct": 11.51,
|
| 1823 |
+
"cpu_temp_max_c": 76.62,
|
| 1824 |
+
"gpu_util_avg_pct": 100.0,
|
| 1825 |
+
"gpu_util_max_pct": 100.0,
|
| 1826 |
+
"mem_util_avg_pct": 31.03,
|
| 1827 |
+
"mem_util_max_pct": 46.0,
|
| 1828 |
+
"temp_avg_c": 69.19,
|
| 1829 |
+
"temp_max_c": 84.0,
|
| 1830 |
+
"power_total_avg_w": 1159.62,
|
| 1831 |
+
"power_total_max_w": 1173.08,
|
| 1832 |
+
"power_limit_total_w": 1200.0,
|
| 1833 |
+
"vram_used_avg_mb": 384778.0,
|
| 1834 |
+
"vram_used_max_mb": 384778.0,
|
| 1835 |
+
"vram_total_mb": 391548.0,
|
| 1836 |
+
"vram_used_avg_pct": 98.27,
|
| 1837 |
+
"vram_used_max_pct": 98.27,
|
| 1838 |
+
"pcie_rx_avg_mb_s": 9737.0,
|
| 1839 |
+
"pcie_rx_max_mb_s": 24076.0,
|
| 1840 |
+
"pcie_tx_avg_mb_s": 7863.75,
|
| 1841 |
+
"pcie_tx_max_mb_s": 8476.0
|
| 1842 |
+
}
|
| 1843 |
+
},
|
| 1844 |
+
{
|
| 1845 |
+
"concurrency": 2,
|
| 1846 |
+
"context_tokens": 32768,
|
| 1847 |
+
"benchmark_mode": "duration",
|
| 1848 |
+
"request_count_target": 0,
|
| 1849 |
+
"warmup_request_count": 0,
|
| 1850 |
+
"measurement_seconds": 19.995457,
|
| 1851 |
+
"measurement_wall_seconds": 20.000576,
|
| 1852 |
+
"client_output_tokens": 3727,
|
| 1853 |
+
"server_output_tokens": 3727,
|
| 1854 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1855 |
+
"aggregate_tps": 186.39233603677485,
|
| 1856 |
+
"per_request_avg_tps": 93.19616801838743,
|
| 1857 |
+
"ttft_avg": 1.1562239681370556,
|
| 1858 |
+
"ttft_p50": 1.213384915026836,
|
| 1859 |
+
"ttft_p90": 1.3303045589243994,
|
| 1860 |
+
"ttft_p99": 1.4650318499957211,
|
| 1861 |
+
"time_to_second_token_avg": 0.01826943955384195,
|
| 1862 |
+
"time_to_second_token_p50": 0.020064805983565748,
|
| 1863 |
+
"time_to_second_token_p90": 0.021722023980692028,
|
| 1864 |
+
"time_to_second_token_p99": 0.021918757599778474,
|
| 1865 |
+
"request_latency_avg": 5.276185235736193,
|
| 1866 |
+
"request_latency_p50": 5.3531620495487005,
|
| 1867 |
+
"request_latency_p90": 5.51592188810464,
|
| 1868 |
+
"request_latency_p99": 5.72075475651538,
|
| 1869 |
+
"inter_token_latency_avg": 0.008087649016423303,
|
| 1870 |
+
"inter_token_latency_p50": 0.00809587391185412,
|
| 1871 |
+
"inter_token_latency_p90": 0.008533725732085115,
|
| 1872 |
+
"inter_token_latency_p99": 0.009295477393811626,
|
| 1873 |
+
"output_tps_per_user_avg": 124.29876816215655,
|
| 1874 |
+
"output_tps_per_user_p50": 123.51971941775874,
|
| 1875 |
+
"output_tps_per_user_p90": 131.3783276316068,
|
| 1876 |
+
"output_tps_per_user_p99": 142.50807229575008,
|
| 1877 |
+
"e2e_output_tps_per_user_avg": 97.32706467298749,
|
| 1878 |
+
"e2e_output_tps_per_user_p50": 95.64440520102602,
|
| 1879 |
+
"e2e_output_tps_per_user_p90": 105.49679616444378,
|
| 1880 |
+
"e2e_output_tps_per_user_p99": 106.19561216563397,
|
| 1881 |
+
"chunk_inter_token_latency_avg": 0.02302849329169416,
|
| 1882 |
+
"chunk_inter_token_latency_p50": 0.02311558809813452,
|
| 1883 |
+
"chunk_inter_token_latency_p90": 0.02417786236213047,
|
| 1884 |
+
"chunk_inter_token_latency_p99": 0.02524273630673258,
|
| 1885 |
+
"input_seq_len_avg": 32768.0,
|
| 1886 |
+
"output_seq_len_avg": 512.0,
|
| 1887 |
+
"output_seq_len_p50": 512.0,
|
| 1888 |
+
"output_seq_len_p90": 512.0,
|
| 1889 |
+
"output_seq_len_p99": 512.0,
|
| 1890 |
+
"request_count": 10,
|
| 1891 |
+
"completed_request_count": 8,
|
| 1892 |
+
"request_samples": [
|
| 1893 |
+
{
|
| 1894 |
+
"ttft": 0.6123396621551365,
|
| 1895 |
+
"time_to_second_token": 0.007417730987071991,
|
| 1896 |
+
"latency": 5.405579176964238,
|
| 1897 |
+
"inter_token_latency_avg": 0.009380116467336793,
|
| 1898 |
+
"chunk_inter_token_latency_avg": 0.02536105563391059,
|
| 1899 |
+
"input_tokens": 32768,
|
| 1900 |
+
"output_tokens": 512,
|
| 1901 |
+
"output_tps_per_user": 106.60848439165707,
|
| 1902 |
+
"e2e_output_tps_per_user": 94.71695506410806,
|
| 1903 |
+
"completed": true
|
| 1904 |
+
},
|
| 1905 |
+
{
|
| 1906 |
+
"ttft": 1.4800015490036458,
|
| 1907 |
+
"time_to_second_token": 0.0212986059486866,
|
| 1908 |
+
"latency": 5.743513964116573,
|
| 1909 |
+
"inter_token_latency_avg": 0.008343468522725885,
|
| 1910 |
+
"chunk_inter_token_latency_avg": 0.023046013054664475,
|
| 1911 |
+
"input_tokens": 32768,
|
| 1912 |
+
"output_tokens": 512,
|
| 1913 |
+
"output_tps_per_user": 119.85423056085207,
|
| 1914 |
+
"e2e_output_tps_per_user": 89.14403328672888,
|
| 1915 |
+
"completed": true
|
| 1916 |
+
},
|
| 1917 |
+
{
|
| 1918 |
+
"ttft": 1.2117794880177826,
|
| 1919 |
+
"time_to_second_token": 0.018087500939145684,
|
| 1920 |
+
"latency": 5.4183824269566685,
|
| 1921 |
+
"inter_token_latency_avg": 0.00823209968481191,
|
| 1922 |
+
"chunk_inter_token_latency_avg": 0.02311320296120267,
|
| 1923 |
+
"input_tokens": 32768,
|
| 1924 |
+
"output_tokens": 512,
|
| 1925 |
+
"output_tps_per_user": 121.4756912923423,
|
| 1926 |
+
"e2e_output_tps_per_user": 94.49314567624086,
|
| 1927 |
+
"completed": true
|
| 1928 |
+
},
|
| 1929 |
+
{
|
| 1930 |
+
"ttft": 1.2173947461415082,
|
| 1931 |
+
"time_to_second_token": 0.021697735879570246,
|
| 1932 |
+
"latency": 5.353260674979538,
|
| 1933 |
+
"inter_token_latency_avg": 0.008093671093616497,
|
| 1934 |
+
"chunk_inter_token_latency_avg": 0.023105396250491784,
|
| 1935 |
+
"input_tokens": 32768,
|
| 1936 |
+
"output_tokens": 512,
|
| 1937 |
+
"output_tps_per_user": 123.55332807985033,
|
| 1938 |
+
"e2e_output_tps_per_user": 95.64264307042306,
|
| 1939 |
+
"completed": true
|
| 1940 |
+
},
|
| 1941 |
+
{
|
| 1942 |
+
"ttft": 1.2256406019441783,
|
| 1943 |
+
"time_to_second_token": 0.02125983708538115,
|
| 1944 |
+
"latency": 0.0,
|
| 1945 |
+
"inter_token_latency_avg": 0.0076920541456072695,
|
| 1946 |
+
"chunk_inter_token_latency_avg": 0.020978329488019826,
|
| 1947 |
+
"input_tokens": 32768,
|
| 1948 |
+
"output_tokens": 181,
|
| 1949 |
+
"output_tps_per_user": 130.00428508047798,
|
| 1950 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1951 |
+
"completed": false
|
| 1952 |
+
},
|
| 1953 |
+
{
|
| 1954 |
+
"ttft": 1.3136715600267053,
|
| 1955 |
+
"time_to_second_token": 0.019344910979270935,
|
| 1956 |
+
"latency": 4.868584974901751,
|
| 1957 |
+
"inter_token_latency_avg": 0.006956777719912027,
|
| 1958 |
+
"chunk_inter_token_latency_avg": 0.020911255381617914,
|
| 1959 |
+
"input_tokens": 32768,
|
| 1960 |
+
"output_tokens": 512,
|
| 1961 |
+
"output_tps_per_user": 143.744710591766,
|
| 1962 |
+
"e2e_output_tps_per_user": 105.16402664006749,
|
| 1963 |
+
"completed": true
|
| 1964 |
+
},
|
| 1965 |
+
{
|
| 1966 |
+
"ttft": 0.8629559089895338,
|
| 1967 |
+
"time_to_second_token": 0.013778487918898463,
|
| 1968 |
+
"latency": 4.81776890787296,
|
| 1969 |
+
"inter_token_latency_avg": 0.007739360076092811,
|
| 1970 |
+
"chunk_inter_token_latency_avg": 0.02340126034842264,
|
| 1971 |
+
"input_tokens": 32768,
|
| 1972 |
+
"output_tokens": 512,
|
| 1973 |
+
"output_tps_per_user": 129.2096491399902,
|
| 1974 |
+
"e2e_output_tps_per_user": 106.27325838798845,
|
| 1975 |
+
"completed": true
|
| 1976 |
+
},
|
| 1977 |
+
{
|
| 1978 |
+
"ttft": 1.2149462150409818,
|
| 1979 |
+
"time_to_second_token": 0.017084267921745777,
|
| 1980 |
+
"latency": 5.353063424117863,
|
| 1981 |
+
"inter_token_latency_avg": 0.008098076730091745,
|
| 1982 |
+
"chunk_inter_token_latency_avg": 0.023117973235066376,
|
| 1983 |
+
"input_tokens": 32768,
|
| 1984 |
+
"output_tokens": 512,
|
| 1985 |
+
"output_tps_per_user": 123.48611075566714,
|
| 1986 |
+
"e2e_output_tps_per_user": 95.64616733162899,
|
| 1987 |
+
"completed": true
|
| 1988 |
+
},
|
| 1989 |
+
{
|
| 1990 |
+
"ttft": 1.2118236150126904,
|
| 1991 |
+
"time_to_second_token": 0.02194061689078808,
|
| 1992 |
+
"latency": 5.249328335979953,
|
| 1993 |
+
"inter_token_latency_avg": 0.007901183406980945,
|
| 1994 |
+
"chunk_inter_token_latency_avg": 0.023204050120501512,
|
| 1995 |
+
"input_tokens": 32768,
|
| 1996 |
+
"output_tokens": 512,
|
| 1997 |
+
"output_tps_per_user": 126.56331950432494,
|
| 1998 |
+
"e2e_output_tps_per_user": 97.53628792671415,
|
| 1999 |
+
"completed": true
|
| 2000 |
+
},
|
| 2001 |
+
{
|
| 2002 |
+
"ttft": 1.2116863350383937,
|
| 2003 |
+
"time_to_second_token": 0.02078470098786056,
|
| 2004 |
+
"latency": 0.0,
|
| 2005 |
+
"inter_token_latency_avg": 0.008439682317057152,
|
| 2006 |
+
"chunk_inter_token_latency_avg": 0.02404639644304379,
|
| 2007 |
+
"input_tokens": 32768,
|
| 2008 |
+
"output_tokens": 360,
|
| 2009 |
+
"output_tps_per_user": 118.48787222463746,
|
| 2010 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2011 |
+
"completed": false
|
| 2012 |
+
}
|
| 2013 |
+
],
|
| 2014 |
+
"total_tokens": 3727,
|
| 2015 |
+
"wall_time": 25.575839941157028,
|
| 2016 |
+
"num_completed": 2,
|
| 2017 |
+
"num_errors": 0,
|
| 2018 |
+
"server_gen_throughput": 186.28349244400528,
|
| 2019 |
+
"server_utilization": 0.013121161362367406,
|
| 2020 |
+
"server_spec_accept_rate": 0.55,
|
| 2021 |
+
"server_spec_accept_length": 0.0,
|
| 2022 |
+
"avg_running_reqs": 1.9,
|
| 2023 |
+
"max_running_reqs": 2,
|
| 2024 |
+
"effective_concurrency": 1.9,
|
| 2025 |
+
"avg_queue_reqs": 0,
|
| 2026 |
+
"max_queue_reqs": 0,
|
| 2027 |
+
"queue_fraction": 0.0,
|
| 2028 |
+
"underfilled": true,
|
| 2029 |
+
"warmup_timed_out": false,
|
| 2030 |
+
"warmup_duration": 5.559,
|
| 2031 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 2032 |
+
"timeout_reason": "",
|
| 2033 |
+
"capacity_limited": false,
|
| 2034 |
+
"hardware_summary": {
|
| 2035 |
+
"samples": 9,
|
| 2036 |
+
"duration_seconds": 19.296,
|
| 2037 |
+
"gpu_count": 4,
|
| 2038 |
+
"cpu_util_avg_pct": 11.52,
|
| 2039 |
+
"cpu_temp_max_c": 76.75,
|
| 2040 |
+
"gpu_util_avg_pct": 99.89,
|
| 2041 |
+
"gpu_util_max_pct": 100.0,
|
| 2042 |
+
"mem_util_avg_pct": 38.5,
|
| 2043 |
+
"mem_util_max_pct": 55.0,
|
| 2044 |
+
"temp_avg_c": 68.69,
|
| 2045 |
+
"temp_max_c": 84.0,
|
| 2046 |
+
"power_total_avg_w": 1159.62,
|
| 2047 |
+
"power_total_max_w": 1178.78,
|
| 2048 |
+
"power_limit_total_w": 1200.0,
|
| 2049 |
+
"vram_used_avg_mb": 384778.0,
|
| 2050 |
+
"vram_used_max_mb": 384778.0,
|
| 2051 |
+
"vram_total_mb": 391548.0,
|
| 2052 |
+
"vram_used_avg_pct": 98.27,
|
| 2053 |
+
"vram_used_max_pct": 98.27,
|
| 2054 |
+
"pcie_rx_avg_mb_s": 18465.67,
|
| 2055 |
+
"pcie_rx_max_mb_s": 56770.0,
|
| 2056 |
+
"pcie_tx_avg_mb_s": 18393.11,
|
| 2057 |
+
"pcie_tx_max_mb_s": 51076.0
|
| 2058 |
+
}
|
| 2059 |
+
},
|
| 2060 |
+
{
|
| 2061 |
+
"concurrency": 4,
|
| 2062 |
+
"context_tokens": 32768,
|
| 2063 |
+
"benchmark_mode": "duration",
|
| 2064 |
+
"request_count_target": 0,
|
| 2065 |
+
"warmup_request_count": 0,
|
| 2066 |
+
"measurement_seconds": 19.996813,
|
| 2067 |
+
"measurement_wall_seconds": 20.000936,
|
| 2068 |
+
"client_output_tokens": 4259,
|
| 2069 |
+
"server_output_tokens": 4259,
|
| 2070 |
+
"aggregate_source": "openai_continuous_usage",
|
| 2071 |
+
"aggregate_tps": 212.98394100240475,
|
| 2072 |
+
"per_request_avg_tps": 53.24598525060119,
|
| 2073 |
+
"ttft_avg": 1.6594751148368232,
|
| 2074 |
+
"ttft_p50": 1.2507346520433202,
|
| 2075 |
+
"ttft_p90": 2.214244986465201,
|
| 2076 |
+
"ttft_p99": 3.976403836400715,
|
| 2077 |
+
"time_to_second_token_avg": 0.03094565647188574,
|
| 2078 |
+
"time_to_second_token_p50": 0.03346484643407166,
|
| 2079 |
+
"time_to_second_token_p90": 0.03496504717040807,
|
| 2080 |
+
"time_to_second_token_p99": 0.035161831779405475,
|
| 2081 |
+
"request_latency_avg": 9.732150099524814,
|
| 2082 |
+
"request_latency_p50": 9.580612420104444,
|
| 2083 |
+
"request_latency_p90": 11.209756568027661,
|
| 2084 |
+
"request_latency_p99": 12.458838488040492,
|
| 2085 |
+
"inter_token_latency_avg": 0.015378300425214705,
|
| 2086 |
+
"inter_token_latency_p50": 0.01540227101955366,
|
| 2087 |
+
"inter_token_latency_p90": 0.016775147308305694,
|
| 2088 |
+
"inter_token_latency_p99": 0.016912279111026093,
|
| 2089 |
+
"output_tps_per_user_avg": 65.35975868873867,
|
| 2090 |
+
"output_tps_per_user_p50": 64.9280008387403,
|
| 2091 |
+
"output_tps_per_user_p90": 70.36340913782152,
|
| 2092 |
+
"output_tps_per_user_p99": 74.69250108569062,
|
| 2093 |
+
"e2e_output_tps_per_user_avg": 53.42647790415391,
|
| 2094 |
+
"e2e_output_tps_per_user_p50": 53.44126007285225,
|
| 2095 |
+
"e2e_output_tps_per_user_p90": 58.48582251783244,
|
| 2096 |
+
"e2e_output_tps_per_user_p99": 64.43718529934995,
|
| 2097 |
+
"chunk_inter_token_latency_avg": 0.043139480462465546,
|
| 2098 |
+
"chunk_inter_token_latency_p50": 0.04417947091825035,
|
| 2099 |
+
"chunk_inter_token_latency_p90": 0.04521715045374133,
|
| 2100 |
+
"chunk_inter_token_latency_p99": 0.04545031231990535,
|
| 2101 |
+
"input_seq_len_avg": 32768.0,
|
| 2102 |
+
"output_seq_len_avg": 512.0,
|
| 2103 |
+
"output_seq_len_p50": 512.0,
|
| 2104 |
+
"output_seq_len_p90": 512.0,
|
| 2105 |
+
"output_seq_len_p99": 512.0,
|
| 2106 |
+
"request_count": 12,
|
| 2107 |
+
"completed_request_count": 9,
|
| 2108 |
+
"request_samples": [
|
| 2109 |
+
{
|
| 2110 |
+
"ttft": 0.6141200510319322,
|
| 2111 |
+
"time_to_second_token": 0.011225768830627203,
|
| 2112 |
+
"latency": 7.865010872948915,
|
| 2113 |
+
"inter_token_latency_avg": 0.014189610219015622,
|
| 2114 |
+
"chunk_inter_token_latency_avg": 0.04028272678842768,
|
| 2115 |
+
"input_tokens": 32768,
|
| 2116 |
+
"output_tokens": 512,
|
| 2117 |
+
"output_tps_per_user": 70.474099327964,
|
| 2118 |
+
"e2e_output_tps_per_user": 65.09844783062967,
|
| 2119 |
+
"completed": true
|
| 2120 |
+
},
|
| 2121 |
+
{
|
| 2122 |
+
"ttft": 1.275223245844245,
|
| 2123 |
+
"time_to_second_token": 0.030873473035171628,
|
| 2124 |
+
"latency": 9.194723297841847,
|
| 2125 |
+
"inter_token_latency_avg": 0.015498043154594134,
|
| 2126 |
+
"chunk_inter_token_latency_avg": 0.04525428601141487,
|
| 2127 |
+
"input_tokens": 32768,
|
| 2128 |
+
"output_tokens": 512,
|
| 2129 |
+
"output_tps_per_user": 64.52427509879315,
|
| 2130 |
+
"e2e_output_tps_per_user": 55.68411179052825,
|
| 2131 |
+
"completed": true
|
| 2132 |
+
},
|
| 2133 |
+
{
|
| 2134 |
+
"ttft": 1.2505115061067045,
|
| 2135 |
+
"time_to_second_token": 0.035012715961784124,
|
| 2136 |
+
"latency": 9.598736566957086,
|
| 2137 |
+
"inter_token_latency_avg": 0.016337035344129905,
|
| 2138 |
+
"chunk_inter_token_latency_avg": 0.044882930434679474,
|
| 2139 |
+
"input_tokens": 32768,
|
| 2140 |
+
"output_tokens": 512,
|
| 2141 |
+
"output_tps_per_user": 61.21061618192019,
|
| 2142 |
+
"e2e_output_tps_per_user": 53.3403533296789,
|
| 2143 |
+
"completed": true
|
| 2144 |
+
},
|
| 2145 |
+
{
|
| 2146 |
+
"ttft": 2.2142701919656247,
|
| 2147 |
+
"time_to_second_token": 0.02940852497704327,
|
| 2148 |
+
"latency": 10.862789368024096,
|
| 2149 |
+
"inter_token_latency_avg": 0.016924695060779787,
|
| 2150 |
+
"chunk_inter_token_latency_avg": 0.044351380390043445,
|
| 2151 |
+
"input_tokens": 32768,
|
| 2152 |
+
"output_tokens": 512,
|
| 2153 |
+
"output_tps_per_user": 59.08525952218403,
|
| 2154 |
+
"e2e_output_tps_per_user": 47.13338192003727,
|
| 2155 |
+
"completed": true
|
| 2156 |
+
},
|
| 2157 |
+
{
|
| 2158 |
+
"ttft": 1.250957797979936,
|
| 2159 |
+
"time_to_second_token": 0.03281243494711816,
|
| 2160 |
+
"latency": 9.841799243818969,
|
| 2161 |
+
"inter_token_latency_avg": 0.01681182279029165,
|
| 2162 |
+
"chunk_inter_token_latency_avg": 0.044282687865149654,
|
| 2163 |
+
"input_tokens": 32768,
|
| 2164 |
+
"output_tokens": 512,
|
| 2165 |
+
"output_tps_per_user": 59.4819498440985,
|
| 2166 |
+
"e2e_output_tps_per_user": 52.023007919162325,
|
| 2167 |
+
"completed": true
|
| 2168 |
+
},
|
| 2169 |
+
{
|
| 2170 |
+
"ttft": 1.2194926510564983,
|
| 2171 |
+
"time_to_second_token": 0.03447063802741468,
|
| 2172 |
+
"latency": 0.0,
|
| 2173 |
+
"inter_token_latency_avg": 0.014444936651048233,
|
| 2174 |
+
"chunk_inter_token_latency_avg": 0.041717839432504976,
|
| 2175 |
+
"input_tokens": 32768,
|
| 2176 |
+
"output_tokens": 388,
|
| 2177 |
+
"output_tps_per_user": 69.22841021441465,
|
| 2178 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2179 |
+
"completed": false
|
| 2180 |
+
},
|
| 2181 |
+
{
|
| 2182 |
+
"ttft": 2.2140181369613856,
|
| 2183 |
+
"time_to_second_token": 0.02934236405417323,
|
| 2184 |
+
"latency": 9.580612420104444,
|
| 2185 |
+
"inter_token_latency_avg": 0.014416035779144928,
|
| 2186 |
+
"chunk_inter_token_latency_avg": 0.04138536114125314,
|
| 2187 |
+
"input_tokens": 32768,
|
| 2188 |
+
"output_tokens": 512,
|
| 2189 |
+
"output_tps_per_user": 69.36719742653926,
|
| 2190 |
+
"e2e_output_tps_per_user": 53.44126007285225,
|
| 2191 |
+
"completed": true
|
| 2192 |
+
},
|
| 2193 |
+
{
|
| 2194 |
+
"ttft": 1.2175294221378863,
|
| 2195 |
+
"time_to_second_token": 0.029971704818308353,
|
| 2196 |
+
"latency": 9.039150352124125,
|
| 2197 |
+
"inter_token_latency_avg": 0.015306498884513187,
|
| 2198 |
+
"chunk_inter_token_latency_avg": 0.04547454029061766,
|
| 2199 |
+
"input_tokens": 32768,
|
| 2200 |
+
"output_tokens": 512,
|
| 2201 |
+
"output_tps_per_user": 65.33172657868745,
|
| 2202 |
+
"e2e_output_tps_per_user": 56.64249183328213,
|
| 2203 |
+
"completed": true
|
| 2204 |
+
},
|
| 2205 |
+
{
|
| 2206 |
+
"ttft": 1.2166177490726113,
|
| 2207 |
+
"time_to_second_token": 0.03411725792102516,
|
| 2208 |
+
"latency": 0.0,
|
| 2209 |
+
"inter_token_latency_avg": 0.015614057580944196,
|
| 2210 |
+
"chunk_inter_token_latency_avg": 0.044076253971351044,
|
| 2211 |
+
"input_tokens": 32768,
|
| 2212 |
+
"output_tokens": 495,
|
| 2213 |
+
"output_tps_per_user": 64.04485155866378,
|
| 2214 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2215 |
+
"completed": false
|
| 2216 |
+
},
|
| 2217 |
+
{
|
| 2218 |
+
"ttft": 4.194195635151118,
|
| 2219 |
+
"time_to_second_token": 0.035180261824280024,
|
| 2220 |
+
"latency": 12.597625368041918,
|
| 2221 |
+
"inter_token_latency_avg": 0.016445067970432093,
|
| 2222 |
+
"chunk_inter_token_latency_avg": 0.0444625911793164,
|
| 2223 |
+
"input_tokens": 32768,
|
| 2224 |
+
"output_tokens": 512,
|
| 2225 |
+
"output_tps_per_user": 60.80850512737194,
|
| 2226 |
+
"e2e_output_tps_per_user": 40.64258025158129,
|
| 2227 |
+
"completed": true
|
| 2228 |
+
},
|
| 2229 |
+
{
|
| 2230 |
+
"ttft": 1.2128918368835002,
|
| 2231 |
+
"time_to_second_token": 0.03453602804802358,
|
| 2232 |
+
"latency": 9.008903405861929,
|
| 2233 |
+
"inter_token_latency_avg": 0.015256382718157395,
|
| 2234 |
+
"chunk_inter_token_latency_avg": 0.04355313725686273,
|
| 2235 |
+
"input_tokens": 32768,
|
| 2236 |
+
"output_tokens": 512,
|
| 2237 |
+
"output_tps_per_user": 65.54633680039039,
|
| 2238 |
+
"e2e_output_tps_per_user": 56.83266618963313,
|
| 2239 |
+
"completed": true
|
| 2240 |
+
},
|
| 2241 |
+
{
|
| 2242 |
+
"ttft": 2.033873153850436,
|
| 2243 |
+
"time_to_second_token": 0.03439670521765947,
|
| 2244 |
+
"latency": 0.0,
|
| 2245 |
+
"inter_token_latency_avg": 0.013295418949525321,
|
| 2246 |
+
"chunk_inter_token_latency_avg": 0.03795003078796548,
|
| 2247 |
+
"input_tokens": 32768,
|
| 2248 |
+
"output_tokens": 295,
|
| 2249 |
+
"output_tps_per_user": 75.21387658383661,
|
| 2250 |
+
"e2e_output_tps_per_user": 0.0,
|
| 2251 |
+
"completed": false
|
| 2252 |
+
}
|
| 2253 |
+
],
|
| 2254 |
+
"total_tokens": 4259,
|
| 2255 |
+
"wall_time": 28.930821696063504,
|
| 2256 |
+
"num_completed": 4,
|
| 2257 |
+
"num_errors": 0,
|
| 2258 |
+
"server_gen_throughput": 212.86426983114268,
|
| 2259 |
+
"server_utilization": 0.02819653824678947,
|
| 2260 |
+
"server_spec_accept_rate": 0.5884057971014492,
|
| 2261 |
+
"server_spec_accept_length": 0.0,
|
| 2262 |
+
"avg_running_reqs": 3.8,
|
| 2263 |
+
"max_running_reqs": 4,
|
| 2264 |
+
"effective_concurrency": 3.8,
|
| 2265 |
+
"avg_queue_reqs": 0,
|
| 2266 |
+
"max_queue_reqs": 0,
|
| 2267 |
+
"queue_fraction": 0.0,
|
| 2268 |
+
"underfilled": true,
|
| 2269 |
+
"warmup_timed_out": false,
|
| 2270 |
+
"warmup_duration": 8.576,
|
| 2271 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 2272 |
+
"timeout_reason": "",
|
| 2273 |
+
"capacity_limited": false,
|
| 2274 |
+
"hardware_summary": {
|
| 2275 |
+
"samples": 9,
|
| 2276 |
+
"duration_seconds": 19.299,
|
| 2277 |
+
"gpu_count": 4,
|
| 2278 |
+
"cpu_util_avg_pct": 11.51,
|
| 2279 |
+
"cpu_temp_max_c": 76.75,
|
| 2280 |
+
"gpu_util_avg_pct": 100.0,
|
| 2281 |
+
"gpu_util_max_pct": 100.0,
|
| 2282 |
+
"mem_util_avg_pct": 32.06,
|
| 2283 |
+
"mem_util_max_pct": 45.0,
|
| 2284 |
+
"temp_avg_c": 69.14,
|
| 2285 |
+
"temp_max_c": 84.0,
|
| 2286 |
+
"power_total_avg_w": 1161.68,
|
| 2287 |
+
"power_total_max_w": 1174.21,
|
| 2288 |
+
"power_limit_total_w": 1200.0,
|
| 2289 |
+
"vram_used_avg_mb": 384778.0,
|
| 2290 |
+
"vram_used_max_mb": 384778.0,
|
| 2291 |
+
"vram_total_mb": 391548.0,
|
| 2292 |
+
"vram_used_avg_pct": 98.27,
|
| 2293 |
+
"vram_used_max_pct": 98.27,
|
| 2294 |
+
"pcie_rx_avg_mb_s": 35105.33,
|
| 2295 |
+
"pcie_rx_max_mb_s": 74341.0,
|
| 2296 |
+
"pcie_tx_avg_mb_s": 34509.67,
|
| 2297 |
+
"pcie_tx_max_mb_s": 70985.0
|
| 2298 |
+
}
|
| 2299 |
+
}
|
| 2300 |
+
],
|
| 2301 |
+
"summary_table": {
|
| 2302 |
+
"0": {
|
| 2303 |
+
"1": 191.1473049181911,
|
| 2304 |
+
"2": 257.76658819245483,
|
| 2305 |
+
"4": 305.62106309879454
|
| 2306 |
+
},
|
| 2307 |
+
"8192": {
|
| 2308 |
+
"1": 165.15431387835127,
|
| 2309 |
+
"2": 183.53550127953065,
|
| 2310 |
+
"4": 214.8878284941506
|
| 2311 |
+
},
|
| 2312 |
+
"32768": {
|
| 2313 |
+
"1": 162.35299164317425,
|
| 2314 |
+
"2": 186.39233603677485,
|
| 2315 |
+
"4": 212.98394100240475
|
| 2316 |
+
}
|
| 2317 |
+
},
|
| 2318 |
+
"burst_results": [],
|
| 2319 |
+
"burst_summary_table": {},
|
| 2320 |
+
"methodology": {
|
| 2321 |
+
"prefill": {
|
| 2322 |
+
"name": "Prefill",
|
| 2323 |
+
"present": false,
|
| 2324 |
+
"mode": "skipped",
|
| 2325 |
+
"formula": "prompt_tokens / TTFT",
|
| 2326 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 2327 |
+
},
|
| 2328 |
+
"sustained_decode": {
|
| 2329 |
+
"name": "Sustained Decode",
|
| 2330 |
+
"present": true,
|
| 2331 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 2332 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 2333 |
+
},
|
| 2334 |
+
"burst_e2e_decode": {
|
| 2335 |
+
"name": "Burst / E2E Decode",
|
| 2336 |
+
"present": false,
|
| 2337 |
+
"status": "not run; use --run-burst",
|
| 2338 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 2339 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 2340 |
+
}
|
| 2341 |
+
}
|
| 2342 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap512.log
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0', '8k', '32k'] │
|
| 16 |
+
│ Duration: 20.0s per decode test | Max tokens: 512 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill: skipped | Sustained decode: 9 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
|
| 23 |
+
7,337,984 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: skipped
|
| 26 |
+
Calibrating padding text (run=mkkehtszimwn, up to 32k)...
|
| 27 |
+
8k: 50,558 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,152 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
Token targeting: /tokenize exact
|
| 30 |
+
Done.
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
llm-decode-bench v0.4.29
|
| 35 |
+
╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
|
| 36 |
+
│ Sustained Decode │
|
| 37 |
+
│ Steady-state decode throughput after the engine has admitted the requested │
|
| 38 |
+
│ concurrency and passed warmup. Use this as the main tuning/regression signal │
|
| 39 |
+
│ for kernels, NCCL, DCP, MTP, and scheduler changes. │
|
| 40 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 41 |
+
Aggregate tok/s + TTFT/ITL
|
| 42 |
+
╭────────────┬─────────────┬──────────────────┬───────────────────╮
|
| 43 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 44 |
+
├────────────┼─────────────┼──────────────────┼───────────────────┤
|
| 45 |
+
│ 0 │ 191.1 80/5 │ 257.8 123/8 │ ∅ (4/4)* 182/13 │
|
| 46 |
+
│ 8k │ 165.2 629/5 │ 183.5 (2/2) 1k/8 │ ∅ (4/4)* 1k/15 │
|
| 47 |
+
│ 32k │ 162.4 630/5 │ 186.4 (2/2) 1k/8 │ 213.0 (4/4) 1k/15 │
|
| 48 |
+
╰────────────┴─────────────┴──────────────────┴───────────────────╯
|
| 49 |
+
Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
|
| 50 |
+
(continuous completion_tokens when the server supports it). Prometheus is kept
|
| 51 |
+
as validation/scheduler data.
|
| 52 |
+
Aggregate source(s): openai_continuous_usage
|
| 53 |
+
∅ = skipped/hidden because the cell does not fit in KV cache; exact deficit is
|
| 54 |
+
kept in JSON timeout_reason
|
| 55 |
+
(X/Y) = avg running / requested concurrency from Prometheus; * =
|
| 56 |
+
capacity-limited or warmup timed out
|
| 57 |
+
Per-Request tok/s
|
| 58 |
+
╭────────────┬───────┬────────────┬────────────╮
|
| 59 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 60 |
+
├────────────┼───────┼──────���─────┼────────────┤
|
| 61 |
+
│ 0 │ 191.1 │ 128.9 │ ∅ (4/4)* │
|
| 62 |
+
│ 8k │ 165.2 │ 91.8 (2/2) │ ∅ (4/4)* │
|
| 63 |
+
│ 32k │ 162.4 │ 93.2 (2/2) │ 53.2 (4/4) │
|
| 64 |
+
╰────────────┴───────┴────────────┴────────────╯
|
| 65 |
+
Client request latency: p50 / p90 ms
|
| 66 |
+
╭────────────┬───────────┬───────────┬────────────╮
|
| 67 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 68 |
+
├────────────┼───────────┼───────────┼────────────┤
|
| 69 |
+
│ 0 │ 2.6k/2.8k │ 4k/4.3k │ 6.8k/7.1k │
|
| 70 |
+
│ 8k │ 3.2k/3.2k │ 5.4k/5.7k │ 9.9k/10.6k │
|
| 71 |
+
│ 32k │ 3.2k/3.3k │ 5.4k/5.5k │ 9.6k/11.2k │
|
| 72 |
+
╰────────────┴───────────┴───────────┴────────────╯
|
| 73 |
+
Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
|
| 74 |
+
coordinate. ITL is computed from observed generated tokens, including streams
|
| 75 |
+
stopped at the measurement boundary; a missing ITL means no stream produced at
|
| 76 |
+
least two measured output tokens. Per-request tok/s and request latency are
|
| 77 |
+
shown in separate per-cell matrices. Completion/sample counts and full
|
| 78 |
+
request-level distributions remain in JSON under request_samples.
|
| 79 |
+
Sustained mode: client latency metrics explain request UX variance; aggregate
|
| 80 |
+
tok/s remains the primary throughput signal.
|
| 81 |
+
ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
|
| 82 |
+
Hardware Summary
|
| 83 |
+
╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
|
| 84 |
+
│ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
|
| 85 |
+
├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
|
| 86 |
+
│ 0 │ │ sust… │ 99/100% │ 43% │ 1153/1… │ 83C │ 76C │ 98… │ 8389/8246 │
|
| 87 |
+
│ … │ │ sust… │ 99/100% │ 40% │ 1154/1… │ 84C │ 76C │ 98… │ 11118/9898 │
|
| 88 |
+
│ … │ │ sust… │ 96/100% │ 39% │ 1152/1… │ 84C │ 77C │ 98… │ 12769/12689 │
|
| 89 |
+
│ 0 │ │ sust… │ 100/100% │ 41% │ 1175/1… │ 84C │ 76C │ 98… │ 11425/11473 │
|
| 90 |
+
│ 0 │ │ sust… │ 100/100% │ 36% │ 1172/1… │ 84C │ 77C │ 98… │ 8124/8457 │
|
| 91 |
+
│ … │ │ sust… │ 100/100% │ 38% │ 1164/1… │ 84C │ 77C │ 98… │ 37320/33538 │
|
| 92 |
+
│ … │ │ sust… │ 100/100% │ 31% │ 1160/1… │ 84C │ 77C │ 98… │ 9737/7864 │
|
| 93 |
+
│ … │ │ sust… │ 100/100% │ 38% │ 1160/1… │ 84C │ 77C │ 98… │ 18466/18393 │
|
| 94 |
+
│ … │ │ sust… │ 100/100% │ 32% │ 1162/1… │ 84C │ 77C │ 98… │ 35105/34510 │
|
| 95 |
+
╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
|
| 96 |
+
╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
|
| 97 |
+
│ avg 1,097 W | max 1,179 W | limit 1,200 W | over 4m 29s | 113 samples │
|
| 98 |
+
╰───────────────────────────────────────────────────────────────────────╯
|
| 99 |
+
Hardware summary is sampled from nvidia-smi during the measured part of each
|
| 100 |
+
cell. Whole-run GPU power is the sampled sum of GPU power draw across the
|
| 101 |
+
complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
|
| 102 |
+
a coarse live diagnostic, not a per-kernel NCCL profiler.
|
| 103 |
+
|
| 104 |
+
╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
|
| 105 |
+
│ Burst / E2E Decode │
|
| 106 |
+
│ Not run. Re-run with --run-burst to append a finite client-facing request │
|
| 107 |
+
│ burst after Sustained Decode. This is intentionally disabled by default │
|
| 108 |
+
│ because it adds another full decode matrix. │
|
| 109 |
+
╰───────���──────────────────────────────────────────────────────────────────────╯
|
| 110 |
+
|
| 111 |
+
╭────────────────────────────── Primary Summary ───────────────────────────────╮
|
| 112 |
+
│ Primary matrices repeated last so the important numbers are visible without │
|
| 113 |
+
│ scrolling back through diagnostics. │
|
| 114 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 115 |
+
Aggregate decode tok/s
|
| 116 |
+
╭────────────┬───────┬─────────────┬─────────────╮
|
| 117 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 118 |
+
├────────────┼───────┼─────────────┼─────────────┤
|
| 119 |
+
│ 0 │ 191.1 │ 257.8 │ ∅ (4/4)* │
|
| 120 |
+
│ 8k │ 165.2 │ 183.5 (2/2) │ ∅ (4/4)* │
|
| 121 |
+
│ 32k │ 162.4 │ 186.4 (2/2) │ 213.0 (4/4) │
|
| 122 |
+
╰────────────┴───────┴─────────────┴─────────────╯
|
| 123 |
+
|
| 124 |
+
Results saved to
|
| 125 |
+
<campaign>/candidate-speed-wi
|
| 126 |
+
ndow-01/results-01/decode-warp-quant/rep-2/decode-cap512.json
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-command.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0,8k,32k",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--skip-prefill",
|
| 25 |
+
"--cell-warmup-timeout-seconds",
|
| 26 |
+
"180"
|
| 27 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "c97b1e6c81137e30eeaae4e137a8605c633dbe6eaa2b5037e53dc6f63dbeae19"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json
ADDED
|
@@ -0,0 +1,1394 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:33:25.169970",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "skipped",
|
| 15 |
+
"standalone_prefill": false,
|
| 16 |
+
"prefill_only": false,
|
| 17 |
+
"skip_prefill": true,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 32768,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0,
|
| 42 |
+
8192,
|
| 43 |
+
32768
|
| 44 |
+
],
|
| 45 |
+
"startup_diagnostics_available": true,
|
| 46 |
+
"nvidia_p2p_override_effective": true,
|
| 47 |
+
"p2pmark_status": "not_run",
|
| 48 |
+
"amd_fabric_status": "not_run"
|
| 49 |
+
},
|
| 50 |
+
"startup_diagnostics": {
|
| 51 |
+
"version": "0.4.29",
|
| 52 |
+
"server_url": "http://127.0.0.1:8001",
|
| 53 |
+
"hostname": "<host>",
|
| 54 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 55 |
+
"env": {},
|
| 56 |
+
"args": {
|
| 57 |
+
"concurrency": "1,2,4",
|
| 58 |
+
"contexts": "0,8k,32k",
|
| 59 |
+
"max_tokens": 8192,
|
| 60 |
+
"duration": 20.0,
|
| 61 |
+
"request_count": 0,
|
| 62 |
+
"run_burst": false,
|
| 63 |
+
"standalone_prefill": false,
|
| 64 |
+
"prefill_only": false,
|
| 65 |
+
"skip_prefill": true,
|
| 66 |
+
"prefill_contexts": "8k,64k,128k",
|
| 67 |
+
"prefill_metric": "client",
|
| 68 |
+
"dcp_size": 0,
|
| 69 |
+
"kv_budget": 0
|
| 70 |
+
},
|
| 71 |
+
"nvidia_p2p_override": {
|
| 72 |
+
"effective": true,
|
| 73 |
+
"configured": true,
|
| 74 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 75 |
+
"params_available": true,
|
| 76 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 77 |
+
"modprobe_available": true,
|
| 78 |
+
"runtime": {
|
| 79 |
+
"ForceP2P": "0x11",
|
| 80 |
+
"RMForceP2PType": "1",
|
| 81 |
+
"RMPcieP2PType": "2",
|
| 82 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 83 |
+
"EnableResizableBar": "1",
|
| 84 |
+
"DmaRemapPeerMmio": "1"
|
| 85 |
+
},
|
| 86 |
+
"expected": {
|
| 87 |
+
"ForceP2P": "0x11",
|
| 88 |
+
"RMForceP2PType": "1",
|
| 89 |
+
"RMPcieP2PType": "2",
|
| 90 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 91 |
+
"EnableResizableBar": "1"
|
| 92 |
+
},
|
| 93 |
+
"missing": [],
|
| 94 |
+
"mismatched": {},
|
| 95 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 96 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 97 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 98 |
+
},
|
| 99 |
+
"p2pmark": {
|
| 100 |
+
"status": "not_run"
|
| 101 |
+
},
|
| 102 |
+
"amd_fabric": {
|
| 103 |
+
"status": "not_run"
|
| 104 |
+
},
|
| 105 |
+
"nvidia_smi_query": {
|
| 106 |
+
"cmd": [
|
| 107 |
+
"nvidia-smi",
|
| 108 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 109 |
+
"--format=csv,noheader,nounits"
|
| 110 |
+
],
|
| 111 |
+
"returncode": 0,
|
| 112 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 113 |
+
"stderr": ""
|
| 114 |
+
},
|
| 115 |
+
"nvidia_smi_topo": {
|
| 116 |
+
"cmd": [
|
| 117 |
+
"nvidia-smi",
|
| 118 |
+
"topo",
|
| 119 |
+
"-m"
|
| 120 |
+
],
|
| 121 |
+
"returncode": 0,
|
| 122 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 123 |
+
"stderr": ""
|
| 124 |
+
}
|
| 125 |
+
},
|
| 126 |
+
"nvidia_p2p_override": {
|
| 127 |
+
"effective": true,
|
| 128 |
+
"configured": true,
|
| 129 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 130 |
+
"params_available": true,
|
| 131 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 132 |
+
"modprobe_available": true,
|
| 133 |
+
"runtime": {
|
| 134 |
+
"ForceP2P": "0x11",
|
| 135 |
+
"RMForceP2PType": "1",
|
| 136 |
+
"RMPcieP2PType": "2",
|
| 137 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 138 |
+
"EnableResizableBar": "1",
|
| 139 |
+
"DmaRemapPeerMmio": "1"
|
| 140 |
+
},
|
| 141 |
+
"expected": {
|
| 142 |
+
"ForceP2P": "0x11",
|
| 143 |
+
"RMForceP2PType": "1",
|
| 144 |
+
"RMPcieP2PType": "2",
|
| 145 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 146 |
+
"EnableResizableBar": "1"
|
| 147 |
+
},
|
| 148 |
+
"missing": [],
|
| 149 |
+
"mismatched": {},
|
| 150 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 151 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 152 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 153 |
+
},
|
| 154 |
+
"p2pmark": {
|
| 155 |
+
"status": "not_run"
|
| 156 |
+
},
|
| 157 |
+
"amd_fabric": {
|
| 158 |
+
"status": "not_run"
|
| 159 |
+
},
|
| 160 |
+
"hardware_run_summary": {
|
| 161 |
+
"samples": 113,
|
| 162 |
+
"duration_seconds": 269.983,
|
| 163 |
+
"gpu_count": 4,
|
| 164 |
+
"cpu_util_avg_pct": 11.42,
|
| 165 |
+
"cpu_temp_max_c": 77.38,
|
| 166 |
+
"gpu_util_avg_pct": 89.04,
|
| 167 |
+
"gpu_util_max_pct": 100.0,
|
| 168 |
+
"mem_util_avg_pct": 34.99,
|
| 169 |
+
"mem_util_max_pct": 56.0,
|
| 170 |
+
"temp_avg_c": 67.89,
|
| 171 |
+
"temp_max_c": 85.0,
|
| 172 |
+
"power_total_avg_w": 1105.19,
|
| 173 |
+
"power_total_max_w": 1178.45,
|
| 174 |
+
"power_limit_total_w": 1200.0,
|
| 175 |
+
"vram_used_avg_mb": 384778.0,
|
| 176 |
+
"vram_used_max_mb": 384778.0,
|
| 177 |
+
"vram_total_mb": 391548.0,
|
| 178 |
+
"vram_used_avg_pct": 98.27,
|
| 179 |
+
"vram_used_max_pct": 98.27,
|
| 180 |
+
"pcie_rx_avg_mb_s": 12268.5,
|
| 181 |
+
"pcie_rx_max_mb_s": 68298.0,
|
| 182 |
+
"pcie_tx_avg_mb_s": 12130.98,
|
| 183 |
+
"pcie_tx_max_mb_s": 63483.0
|
| 184 |
+
},
|
| 185 |
+
"event_log": [
|
| 186 |
+
"02:28:53 benchmark start engine=vllm",
|
| 187 |
+
"02:28:53 startup server=http://127.0.0.1:8001 model=glm53-flash-trellismx-p8-k45",
|
| 188 |
+
"02:28:53 startup decode concurrency=1,2,4 contexts=0,8k,32k",
|
| 189 |
+
"02:28:53 startup NVIDIA P2P override: enabled: runtime NVIDIA P2P override matches expected RegistryDwords",
|
| 190 |
+
"02:28:53 startup engine vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f models=['glm53-flash-trellismx-p8-k45']",
|
| 191 |
+
"02:28:53 startup KV cache budget from vLLM metrics: 29,351,936 tokens (3583 blocks x 2048; local 7,337,984 \u00d7 CP 4; CP source: local process)",
|
| 192 |
+
"02:28:53 startup model context length: 1,000,000 tokens",
|
| 193 |
+
"02:28:53 startup prefill tests: skipped",
|
| 194 |
+
"02:28:53 startup calibrating padding text run=lijekedoguqy up_to=32k",
|
| 195 |
+
"02:28:53 startup context 8k: 50,540 chars (8,192 prompt tokens via /tokenize)",
|
| 196 |
+
"02:28:53 startup context 32k: 205,139 chars (32,768 prompt tokens via /tokenize)",
|
| 197 |
+
"02:28:53 startup token targeting: /tokenize exact",
|
| 198 |
+
"02:28:53 startup startup preparation done",
|
| 199 |
+
"02:28:53 hardware monitor interval=2s",
|
| 200 |
+
"02:28:53 decode warmup start",
|
| 201 |
+
"02:28:53 decode warmup start C=1 ctx=32k 3s",
|
| 202 |
+
"02:28:53 cell start C=1 ctx=32k",
|
| 203 |
+
"02:29:01 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 204 |
+
"02:29:04 cell done C=1 ctx=32k 184.6 tok/s",
|
| 205 |
+
"02:29:04 decode warmup done C=1 ctx=32k",
|
| 206 |
+
"02:29:06 cell start C=1 ctx=0",
|
| 207 |
+
"02:29:12 ready C=1 ctx=0 running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 208 |
+
"02:29:32 cell done C=1 ctx=0 175.2 tok/s",
|
| 209 |
+
"02:29:34 cell start C=1 ctx=8k",
|
| 210 |
+
"02:29:40 ready C=1 ctx=8k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 211 |
+
"02:30:00 cell done C=1 ctx=8k 174.2 tok/s",
|
| 212 |
+
"02:30:02 cell start C=1 ctx=32k",
|
| 213 |
+
"02:30:11 ready C=1 ctx=32k running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 214 |
+
"02:30:31 cell done C=1 ctx=32k 179.3 tok/s",
|
| 215 |
+
"02:30:33 cell start C=2 ctx=0",
|
| 216 |
+
"02:30:39 ready C=2 ctx=0 running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 217 |
+
"02:30:59 cell done C=2 ctx=0 236.4 tok/s",
|
| 218 |
+
"02:31:01 cell start C=4 ctx=0",
|
| 219 |
+
"02:31:06 ready C=4 ctx=0 running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 220 |
+
"02:31:26 cell done C=4 ctx=0 296.7 tok/s",
|
| 221 |
+
"02:31:28 cell start C=2 ctx=8k",
|
| 222 |
+
"02:31:34 ready C=2 ctx=8k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 223 |
+
"02:31:54 cell done C=2 ctx=8k 246.8 tok/s",
|
| 224 |
+
"02:31:56 cell start C=4 ctx=8k",
|
| 225 |
+
"02:32:04 ready C=4 ctx=8k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 226 |
+
"02:32:24 cell done C=4 ctx=8k 291.4 tok/s",
|
| 227 |
+
"02:32:26 cell start C=2 ctx=32k",
|
| 228 |
+
"02:32:32 ready C=2 ctx=32k running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 229 |
+
"02:32:52 cell done C=2 ctx=32k 239.9 tok/s",
|
| 230 |
+
"02:32:54 cell start C=4 ctx=32k",
|
| 231 |
+
"02:33:03 ready C=4 ctx=32k running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 232 |
+
"02:33:23 cell done C=4 ctx=32k 302.2 tok/s"
|
| 233 |
+
],
|
| 234 |
+
"prefill": {},
|
| 235 |
+
"results": [
|
| 236 |
+
{
|
| 237 |
+
"concurrency": 1,
|
| 238 |
+
"context_tokens": 0,
|
| 239 |
+
"benchmark_mode": "duration",
|
| 240 |
+
"request_count_target": 0,
|
| 241 |
+
"warmup_request_count": 0,
|
| 242 |
+
"measurement_seconds": 19.996529,
|
| 243 |
+
"measurement_wall_seconds": 20.000654,
|
| 244 |
+
"client_output_tokens": 3503,
|
| 245 |
+
"server_output_tokens": 3503,
|
| 246 |
+
"aggregate_source": "openai_continuous_usage",
|
| 247 |
+
"aggregate_tps": 175.1804006140207,
|
| 248 |
+
"per_request_avg_tps": 175.1804006140207,
|
| 249 |
+
"ttft_avg": 0.06969588715583086,
|
| 250 |
+
"ttft_p50": 0.06969588715583086,
|
| 251 |
+
"ttft_p90": 0.06969588715583086,
|
| 252 |
+
"ttft_p99": 0.06969588715583086,
|
| 253 |
+
"time_to_second_token_avg": 0.013065006816759706,
|
| 254 |
+
"time_to_second_token_p50": 0.013065006816759706,
|
| 255 |
+
"time_to_second_token_p90": 0.013065006816759706,
|
| 256 |
+
"time_to_second_token_p99": 0.013065006816759706,
|
| 257 |
+
"request_latency_avg": 0.0,
|
| 258 |
+
"request_latency_p50": 0.0,
|
| 259 |
+
"request_latency_p90": 0.0,
|
| 260 |
+
"request_latency_p99": 0.0,
|
| 261 |
+
"inter_token_latency_avg": 0.005635859511267837,
|
| 262 |
+
"inter_token_latency_p50": 0.005635859511267837,
|
| 263 |
+
"inter_token_latency_p90": 0.005635859511267837,
|
| 264 |
+
"inter_token_latency_p99": 0.005635859511267837,
|
| 265 |
+
"output_tps_per_user_avg": 177.43522492011888,
|
| 266 |
+
"output_tps_per_user_p50": 177.43522492011888,
|
| 267 |
+
"output_tps_per_user_p90": 177.43522492011888,
|
| 268 |
+
"output_tps_per_user_p99": 177.43522492011888,
|
| 269 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 270 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 271 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 272 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 273 |
+
"chunk_inter_token_latency_avg": 0.01424891621259546,
|
| 274 |
+
"chunk_inter_token_latency_p50": 0.01424891621259546,
|
| 275 |
+
"chunk_inter_token_latency_p90": 0.01424891621259546,
|
| 276 |
+
"chunk_inter_token_latency_p99": 0.01424891621259546,
|
| 277 |
+
"input_seq_len_avg": 78.0,
|
| 278 |
+
"output_seq_len_avg": 4519.0,
|
| 279 |
+
"output_seq_len_p50": 4519.0,
|
| 280 |
+
"output_seq_len_p90": 4519.0,
|
| 281 |
+
"output_seq_len_p99": 4519.0,
|
| 282 |
+
"request_count": 1,
|
| 283 |
+
"completed_request_count": 0,
|
| 284 |
+
"request_samples": [
|
| 285 |
+
{
|
| 286 |
+
"ttft": 0.06969588715583086,
|
| 287 |
+
"time_to_second_token": 0.013065006816759706,
|
| 288 |
+
"latency": 0.0,
|
| 289 |
+
"inter_token_latency_avg": 0.005635859511267837,
|
| 290 |
+
"chunk_inter_token_latency_avg": 0.01424891621259546,
|
| 291 |
+
"input_tokens": 78,
|
| 292 |
+
"output_tokens": 4519,
|
| 293 |
+
"output_tps_per_user": 177.43522492011888,
|
| 294 |
+
"e2e_output_tps_per_user": 0.0,
|
| 295 |
+
"completed": false
|
| 296 |
+
}
|
| 297 |
+
],
|
| 298 |
+
"total_tokens": 3503,
|
| 299 |
+
"wall_time": 25.547412615967914,
|
| 300 |
+
"num_completed": 1,
|
| 301 |
+
"num_errors": 0,
|
| 302 |
+
"server_gen_throughput": 175.1028092847603,
|
| 303 |
+
"server_utilization": 0.005862646566164198,
|
| 304 |
+
"server_spec_accept_rate": 0.4835680751173709,
|
| 305 |
+
"server_spec_accept_length": 0.0,
|
| 306 |
+
"avg_running_reqs": 1,
|
| 307 |
+
"max_running_reqs": 1,
|
| 308 |
+
"effective_concurrency": 1,
|
| 309 |
+
"avg_queue_reqs": 0,
|
| 310 |
+
"max_queue_reqs": 0,
|
| 311 |
+
"queue_fraction": 0.0,
|
| 312 |
+
"underfilled": false,
|
| 313 |
+
"warmup_timed_out": false,
|
| 314 |
+
"warmup_duration": 5.536,
|
| 315 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 316 |
+
"timeout_reason": "",
|
| 317 |
+
"capacity_limited": false,
|
| 318 |
+
"hardware_summary": {
|
| 319 |
+
"samples": 8,
|
| 320 |
+
"duration_seconds": 16.909,
|
| 321 |
+
"gpu_count": 4,
|
| 322 |
+
"cpu_util_avg_pct": 11.56,
|
| 323 |
+
"cpu_temp_max_c": 75.88,
|
| 324 |
+
"gpu_util_avg_pct": 99.0,
|
| 325 |
+
"gpu_util_max_pct": 99.0,
|
| 326 |
+
"mem_util_avg_pct": 44.09,
|
| 327 |
+
"mem_util_max_pct": 56.0,
|
| 328 |
+
"temp_avg_c": 67.31,
|
| 329 |
+
"temp_max_c": 82.0,
|
| 330 |
+
"power_total_avg_w": 1153.01,
|
| 331 |
+
"power_total_max_w": 1155.24,
|
| 332 |
+
"power_limit_total_w": 1200.0,
|
| 333 |
+
"vram_used_avg_mb": 384778.0,
|
| 334 |
+
"vram_used_max_mb": 384778.0,
|
| 335 |
+
"vram_total_mb": 391548.0,
|
| 336 |
+
"vram_used_avg_pct": 98.27,
|
| 337 |
+
"vram_used_max_pct": 98.27,
|
| 338 |
+
"pcie_rx_avg_mb_s": 8331.0,
|
| 339 |
+
"pcie_rx_max_mb_s": 8554.0,
|
| 340 |
+
"pcie_tx_avg_mb_s": 8194.0,
|
| 341 |
+
"pcie_tx_max_mb_s": 8520.0
|
| 342 |
+
}
|
| 343 |
+
},
|
| 344 |
+
{
|
| 345 |
+
"concurrency": 1,
|
| 346 |
+
"context_tokens": 8192,
|
| 347 |
+
"benchmark_mode": "duration",
|
| 348 |
+
"request_count_target": 0,
|
| 349 |
+
"warmup_request_count": 0,
|
| 350 |
+
"measurement_seconds": 19.992808,
|
| 351 |
+
"measurement_wall_seconds": 20.000923,
|
| 352 |
+
"client_output_tokens": 3482,
|
| 353 |
+
"server_output_tokens": 3482,
|
| 354 |
+
"aggregate_source": "openai_continuous_usage",
|
| 355 |
+
"aggregate_tps": 174.16263081468745,
|
| 356 |
+
"per_request_avg_tps": 174.16263081468745,
|
| 357 |
+
"ttft_avg": 0.5914782441686839,
|
| 358 |
+
"ttft_p50": 0.5914782441686839,
|
| 359 |
+
"ttft_p90": 0.5914782441686839,
|
| 360 |
+
"ttft_p99": 0.5914782441686839,
|
| 361 |
+
"time_to_second_token_avg": 0.01696636783890426,
|
| 362 |
+
"time_to_second_token_p50": 0.01696636783890426,
|
| 363 |
+
"time_to_second_token_p90": 0.01696636783890426,
|
| 364 |
+
"time_to_second_token_p99": 0.01696636783890426,
|
| 365 |
+
"request_latency_avg": 0.0,
|
| 366 |
+
"request_latency_p50": 0.0,
|
| 367 |
+
"request_latency_p90": 0.0,
|
| 368 |
+
"request_latency_p99": 0.0,
|
| 369 |
+
"inter_token_latency_avg": 0.005593080112173436,
|
| 370 |
+
"inter_token_latency_p50": 0.005593080112173436,
|
| 371 |
+
"inter_token_latency_p90": 0.005593080112173436,
|
| 372 |
+
"inter_token_latency_p99": 0.005593080112173436,
|
| 373 |
+
"output_tps_per_user_avg": 178.79236126503582,
|
| 374 |
+
"output_tps_per_user_p50": 178.79236126503582,
|
| 375 |
+
"output_tps_per_user_p90": 178.79236126503582,
|
| 376 |
+
"output_tps_per_user_p99": 178.79236126503582,
|
| 377 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 378 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 379 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 380 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 381 |
+
"chunk_inter_token_latency_avg": 0.014262687604284941,
|
| 382 |
+
"chunk_inter_token_latency_p50": 0.014262687604284941,
|
| 383 |
+
"chunk_inter_token_latency_p90": 0.014262687604284941,
|
| 384 |
+
"chunk_inter_token_latency_p99": 0.014262687604284941,
|
| 385 |
+
"input_seq_len_avg": 8192.0,
|
| 386 |
+
"output_seq_len_avg": 4280.0,
|
| 387 |
+
"output_seq_len_p50": 4280.0,
|
| 388 |
+
"output_seq_len_p90": 4280.0,
|
| 389 |
+
"output_seq_len_p99": 4280.0,
|
| 390 |
+
"request_count": 1,
|
| 391 |
+
"completed_request_count": 0,
|
| 392 |
+
"request_samples": [
|
| 393 |
+
{
|
| 394 |
+
"ttft": 0.5914782441686839,
|
| 395 |
+
"time_to_second_token": 0.01696636783890426,
|
| 396 |
+
"latency": 0.0,
|
| 397 |
+
"inter_token_latency_avg": 0.005593080112173436,
|
| 398 |
+
"chunk_inter_token_latency_avg": 0.014262687604284941,
|
| 399 |
+
"input_tokens": 8192,
|
| 400 |
+
"output_tokens": 4280,
|
| 401 |
+
"output_tps_per_user": 178.79236126503582,
|
| 402 |
+
"e2e_output_tps_per_user": 0.0,
|
| 403 |
+
"completed": false
|
| 404 |
+
}
|
| 405 |
+
],
|
| 406 |
+
"total_tokens": 3482,
|
| 407 |
+
"wall_time": 26.0623870131094,
|
| 408 |
+
"num_completed": 1,
|
| 409 |
+
"num_errors": 0,
|
| 410 |
+
"server_gen_throughput": 174.05206214713238,
|
| 411 |
+
"server_utilization": 0.006141820212172022,
|
| 412 |
+
"server_spec_accept_rate": 0.5047619047619047,
|
| 413 |
+
"server_spec_accept_length": 0.0,
|
| 414 |
+
"avg_running_reqs": 1,
|
| 415 |
+
"max_running_reqs": 1,
|
| 416 |
+
"effective_concurrency": 1,
|
| 417 |
+
"avg_queue_reqs": 0,
|
| 418 |
+
"max_queue_reqs": 0,
|
| 419 |
+
"queue_fraction": 0.0,
|
| 420 |
+
"underfilled": false,
|
| 421 |
+
"warmup_timed_out": false,
|
| 422 |
+
"warmup_duration": 6.055,
|
| 423 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 424 |
+
"timeout_reason": "",
|
| 425 |
+
"capacity_limited": false,
|
| 426 |
+
"hardware_summary": {
|
| 427 |
+
"samples": 8,
|
| 428 |
+
"duration_seconds": 16.885,
|
| 429 |
+
"gpu_count": 4,
|
| 430 |
+
"cpu_util_avg_pct": 11.53,
|
| 431 |
+
"cpu_temp_max_c": 76.62,
|
| 432 |
+
"gpu_util_avg_pct": 99.0,
|
| 433 |
+
"gpu_util_max_pct": 99.0,
|
| 434 |
+
"mem_util_avg_pct": 43.56,
|
| 435 |
+
"mem_util_max_pct": 55.0,
|
| 436 |
+
"temp_avg_c": 67.78,
|
| 437 |
+
"temp_max_c": 83.0,
|
| 438 |
+
"power_total_avg_w": 1153.83,
|
| 439 |
+
"power_total_max_w": 1155.28,
|
| 440 |
+
"power_limit_total_w": 1200.0,
|
| 441 |
+
"vram_used_avg_mb": 384778.0,
|
| 442 |
+
"vram_used_max_mb": 384778.0,
|
| 443 |
+
"vram_total_mb": 391548.0,
|
| 444 |
+
"vram_used_avg_pct": 98.27,
|
| 445 |
+
"vram_used_max_pct": 98.27,
|
| 446 |
+
"pcie_rx_avg_mb_s": 8357.0,
|
| 447 |
+
"pcie_rx_max_mb_s": 8584.0,
|
| 448 |
+
"pcie_tx_avg_mb_s": 8151.12,
|
| 449 |
+
"pcie_tx_max_mb_s": 8441.0
|
| 450 |
+
}
|
| 451 |
+
},
|
| 452 |
+
{
|
| 453 |
+
"concurrency": 1,
|
| 454 |
+
"context_tokens": 32768,
|
| 455 |
+
"benchmark_mode": "duration",
|
| 456 |
+
"request_count_target": 0,
|
| 457 |
+
"warmup_request_count": 0,
|
| 458 |
+
"measurement_seconds": 19.987041,
|
| 459 |
+
"measurement_wall_seconds": 20.00121,
|
| 460 |
+
"client_output_tokens": 3584,
|
| 461 |
+
"server_output_tokens": 3586,
|
| 462 |
+
"aggregate_source": "openai_continuous_usage",
|
| 463 |
+
"aggregate_tps": 179.316188111569,
|
| 464 |
+
"per_request_avg_tps": 179.316188111569,
|
| 465 |
+
"ttft_avg": 0.6033541450742632,
|
| 466 |
+
"ttft_p50": 0.6033541450742632,
|
| 467 |
+
"ttft_p90": 0.6033541450742632,
|
| 468 |
+
"ttft_p99": 0.6033541450742632,
|
| 469 |
+
"time_to_second_token_avg": 0.012713581090793014,
|
| 470 |
+
"time_to_second_token_p50": 0.012713581090793014,
|
| 471 |
+
"time_to_second_token_p90": 0.012713581090793014,
|
| 472 |
+
"time_to_second_token_p99": 0.012713581090793014,
|
| 473 |
+
"request_latency_avg": 0.0,
|
| 474 |
+
"request_latency_p50": 0.0,
|
| 475 |
+
"request_latency_p90": 0.0,
|
| 476 |
+
"request_latency_p99": 0.0,
|
| 477 |
+
"inter_token_latency_avg": 0.005507921420528074,
|
| 478 |
+
"inter_token_latency_p50": 0.005507921420528074,
|
| 479 |
+
"inter_token_latency_p90": 0.005507921420528074,
|
| 480 |
+
"inter_token_latency_p99": 0.005507921420528074,
|
| 481 |
+
"output_tps_per_user_avg": 181.55669328777836,
|
| 482 |
+
"output_tps_per_user_p50": 181.55669328777836,
|
| 483 |
+
"output_tps_per_user_p90": 181.55669328777836,
|
| 484 |
+
"output_tps_per_user_p99": 181.55669328777836,
|
| 485 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 486 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 487 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 488 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 489 |
+
"chunk_inter_token_latency_avg": 0.014389527561933152,
|
| 490 |
+
"chunk_inter_token_latency_p50": 0.014389527561933152,
|
| 491 |
+
"chunk_inter_token_latency_p90": 0.014389527561933152,
|
| 492 |
+
"chunk_inter_token_latency_p99": 0.014389527561933152,
|
| 493 |
+
"input_seq_len_avg": 32768.0,
|
| 494 |
+
"output_seq_len_avg": 4343.0,
|
| 495 |
+
"output_seq_len_p50": 4343.0,
|
| 496 |
+
"output_seq_len_p90": 4343.0,
|
| 497 |
+
"output_seq_len_p99": 4343.0,
|
| 498 |
+
"request_count": 1,
|
| 499 |
+
"completed_request_count": 0,
|
| 500 |
+
"request_samples": [
|
| 501 |
+
{
|
| 502 |
+
"ttft": 0.6033541450742632,
|
| 503 |
+
"time_to_second_token": 0.012713581090793014,
|
| 504 |
+
"latency": 0.0,
|
| 505 |
+
"inter_token_latency_avg": 0.005507921420528074,
|
| 506 |
+
"chunk_inter_token_latency_avg": 0.014389527561933152,
|
| 507 |
+
"input_tokens": 32768,
|
| 508 |
+
"output_tokens": 4343,
|
| 509 |
+
"output_tps_per_user": 181.55669328777836,
|
| 510 |
+
"e2e_output_tps_per_user": 0.0,
|
| 511 |
+
"completed": false
|
| 512 |
+
}
|
| 513 |
+
],
|
| 514 |
+
"total_tokens": 3584,
|
| 515 |
+
"wall_time": 29.107660971814767,
|
| 516 |
+
"num_completed": 1,
|
| 517 |
+
"num_errors": 0,
|
| 518 |
+
"server_gen_throughput": 179.23767853872448,
|
| 519 |
+
"server_utilization": 0.006979341150195384,
|
| 520 |
+
"server_spec_accept_rate": 0.46190476190476193,
|
| 521 |
+
"server_spec_accept_length": 0.0,
|
| 522 |
+
"avg_running_reqs": 1,
|
| 523 |
+
"max_running_reqs": 1,
|
| 524 |
+
"effective_concurrency": 1,
|
| 525 |
+
"avg_queue_reqs": 0,
|
| 526 |
+
"max_queue_reqs": 0,
|
| 527 |
+
"queue_fraction": 0.0,
|
| 528 |
+
"underfilled": false,
|
| 529 |
+
"warmup_timed_out": false,
|
| 530 |
+
"warmup_duration": 9.101,
|
| 531 |
+
"ready_reason": "running_reqs=1/1, queue_reqs=0, active_streams=1/1, stable=3.0s",
|
| 532 |
+
"timeout_reason": "",
|
| 533 |
+
"capacity_limited": false,
|
| 534 |
+
"hardware_summary": {
|
| 535 |
+
"samples": 8,
|
| 536 |
+
"duration_seconds": 16.891,
|
| 537 |
+
"gpu_count": 4,
|
| 538 |
+
"cpu_util_avg_pct": 11.56,
|
| 539 |
+
"cpu_temp_max_c": 75.88,
|
| 540 |
+
"gpu_util_avg_pct": 99.0,
|
| 541 |
+
"gpu_util_max_pct": 99.0,
|
| 542 |
+
"mem_util_avg_pct": 44.0,
|
| 543 |
+
"mem_util_max_pct": 56.0,
|
| 544 |
+
"temp_avg_c": 68.16,
|
| 545 |
+
"temp_max_c": 83.0,
|
| 546 |
+
"power_total_avg_w": 1155.58,
|
| 547 |
+
"power_total_max_w": 1156.94,
|
| 548 |
+
"power_limit_total_w": 1200.0,
|
| 549 |
+
"vram_used_avg_mb": 384778.0,
|
| 550 |
+
"vram_used_max_mb": 384778.0,
|
| 551 |
+
"vram_total_mb": 391548.0,
|
| 552 |
+
"vram_used_avg_pct": 98.27,
|
| 553 |
+
"vram_used_max_pct": 98.27,
|
| 554 |
+
"pcie_rx_avg_mb_s": 8340.12,
|
| 555 |
+
"pcie_rx_max_mb_s": 8569.0,
|
| 556 |
+
"pcie_tx_avg_mb_s": 8143.62,
|
| 557 |
+
"pcie_tx_max_mb_s": 8438.0
|
| 558 |
+
}
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"concurrency": 2,
|
| 562 |
+
"context_tokens": 0,
|
| 563 |
+
"benchmark_mode": "duration",
|
| 564 |
+
"request_count_target": 0,
|
| 565 |
+
"warmup_request_count": 0,
|
| 566 |
+
"measurement_seconds": 20.00082,
|
| 567 |
+
"measurement_wall_seconds": 20.000837,
|
| 568 |
+
"client_output_tokens": 4728,
|
| 569 |
+
"server_output_tokens": 4728,
|
| 570 |
+
"aggregate_source": "openai_continuous_usage",
|
| 571 |
+
"aggregate_tps": 236.39031276129717,
|
| 572 |
+
"per_request_avg_tps": 118.19515638064858,
|
| 573 |
+
"ttft_avg": 0.11200506216846406,
|
| 574 |
+
"ttft_p50": 0.11200506216846406,
|
| 575 |
+
"ttft_p90": 0.1434264814015478,
|
| 576 |
+
"ttft_p99": 0.15049630072899162,
|
| 577 |
+
"time_to_second_token_avg": 0.01528907532338053,
|
| 578 |
+
"time_to_second_token_p50": 0.01528907532338053,
|
| 579 |
+
"time_to_second_token_p90": 0.017035022121854128,
|
| 580 |
+
"time_to_second_token_p99": 0.017427860151510686,
|
| 581 |
+
"request_latency_avg": 0.0,
|
| 582 |
+
"request_latency_p50": 0.0,
|
| 583 |
+
"request_latency_p90": 0.0,
|
| 584 |
+
"request_latency_p99": 0.0,
|
| 585 |
+
"inter_token_latency_avg": 0.00843379404097247,
|
| 586 |
+
"inter_token_latency_p50": 0.00843379404097247,
|
| 587 |
+
"inter_token_latency_p90": 0.008573998497973905,
|
| 588 |
+
"inter_token_latency_p99": 0.008605544500799228,
|
| 589 |
+
"output_tps_per_user_avg": 118.62182033895986,
|
| 590 |
+
"output_tps_per_user_p50": 118.62182033895986,
|
| 591 |
+
"output_tps_per_user_p90": 120.59380445765525,
|
| 592 |
+
"output_tps_per_user_p99": 121.03750088436172,
|
| 593 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 594 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 595 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 596 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 597 |
+
"chunk_inter_token_latency_avg": 0.020858210011846075,
|
| 598 |
+
"chunk_inter_token_latency_p50": 0.020858210011846075,
|
| 599 |
+
"chunk_inter_token_latency_p90": 0.02088407432588644,
|
| 600 |
+
"chunk_inter_token_latency_p99": 0.02088989379654552,
|
| 601 |
+
"input_seq_len_avg": 78.0,
|
| 602 |
+
"output_seq_len_avg": 3017.0,
|
| 603 |
+
"output_seq_len_p50": 3017.0,
|
| 604 |
+
"output_seq_len_p90": 3063.4,
|
| 605 |
+
"output_seq_len_p99": 3073.84,
|
| 606 |
+
"request_count": 2,
|
| 607 |
+
"completed_request_count": 0,
|
| 608 |
+
"request_samples": [
|
| 609 |
+
{
|
| 610 |
+
"ttft": 0.07272828812710941,
|
| 611 |
+
"time_to_second_token": 0.013106641825288534,
|
| 612 |
+
"latency": 0.0,
|
| 613 |
+
"inter_token_latency_avg": 0.008609049612224263,
|
| 614 |
+
"chunk_inter_token_latency_avg": 0.02089054040439653,
|
| 615 |
+
"input_tokens": 78,
|
| 616 |
+
"output_tokens": 2959,
|
| 617 |
+
"output_tps_per_user": 116.15684019059063,
|
| 618 |
+
"e2e_output_tps_per_user": 0.0,
|
| 619 |
+
"completed": false
|
| 620 |
+
},
|
| 621 |
+
{
|
| 622 |
+
"ttft": 0.15128183620981872,
|
| 623 |
+
"time_to_second_token": 0.017471508821472526,
|
| 624 |
+
"latency": 0.0,
|
| 625 |
+
"inter_token_latency_avg": 0.008258538469720678,
|
| 626 |
+
"chunk_inter_token_latency_avg": 0.020825879619295624,
|
| 627 |
+
"input_tokens": 78,
|
| 628 |
+
"output_tokens": 3075,
|
| 629 |
+
"output_tps_per_user": 121.0868004873291,
|
| 630 |
+
"e2e_output_tps_per_user": 0.0,
|
| 631 |
+
"completed": false
|
| 632 |
+
}
|
| 633 |
+
],
|
| 634 |
+
"total_tokens": 4728,
|
| 635 |
+
"wall_time": 25.560176193946972,
|
| 636 |
+
"num_completed": 2,
|
| 637 |
+
"num_errors": 0,
|
| 638 |
+
"server_gen_throughput": 236.33660483401786,
|
| 639 |
+
"server_utilization": 0.011725293132328285,
|
| 640 |
+
"server_spec_accept_rate": 0.3958333333333333,
|
| 641 |
+
"server_spec_accept_length": 0.0,
|
| 642 |
+
"avg_running_reqs": 2,
|
| 643 |
+
"max_running_reqs": 2,
|
| 644 |
+
"effective_concurrency": 2,
|
| 645 |
+
"avg_queue_reqs": 0,
|
| 646 |
+
"max_queue_reqs": 0,
|
| 647 |
+
"queue_fraction": 0.0,
|
| 648 |
+
"underfilled": false,
|
| 649 |
+
"warmup_timed_out": false,
|
| 650 |
+
"warmup_duration": 5.538,
|
| 651 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 652 |
+
"timeout_reason": "",
|
| 653 |
+
"capacity_limited": false,
|
| 654 |
+
"hardware_summary": {
|
| 655 |
+
"samples": 8,
|
| 656 |
+
"duration_seconds": 16.86,
|
| 657 |
+
"gpu_count": 4,
|
| 658 |
+
"cpu_util_avg_pct": 11.55,
|
| 659 |
+
"cpu_temp_max_c": 77.12,
|
| 660 |
+
"gpu_util_avg_pct": 100.0,
|
| 661 |
+
"gpu_util_max_pct": 100.0,
|
| 662 |
+
"mem_util_avg_pct": 40.34,
|
| 663 |
+
"mem_util_max_pct": 50.0,
|
| 664 |
+
"temp_avg_c": 68.5,
|
| 665 |
+
"temp_max_c": 84.0,
|
| 666 |
+
"power_total_avg_w": 1176.31,
|
| 667 |
+
"power_total_max_w": 1178.12,
|
| 668 |
+
"power_limit_total_w": 1200.0,
|
| 669 |
+
"vram_used_avg_mb": 384778.0,
|
| 670 |
+
"vram_used_max_mb": 384778.0,
|
| 671 |
+
"vram_total_mb": 391548.0,
|
| 672 |
+
"vram_used_avg_pct": 98.27,
|
| 673 |
+
"vram_used_max_pct": 98.27,
|
| 674 |
+
"pcie_rx_avg_mb_s": 11179.25,
|
| 675 |
+
"pcie_rx_max_mb_s": 11225.0,
|
| 676 |
+
"pcie_tx_avg_mb_s": 11087.75,
|
| 677 |
+
"pcie_tx_max_mb_s": 11179.0
|
| 678 |
+
}
|
| 679 |
+
},
|
| 680 |
+
{
|
| 681 |
+
"concurrency": 4,
|
| 682 |
+
"context_tokens": 0,
|
| 683 |
+
"benchmark_mode": "duration",
|
| 684 |
+
"request_count_target": 0,
|
| 685 |
+
"warmup_request_count": 0,
|
| 686 |
+
"measurement_seconds": 19.975466,
|
| 687 |
+
"measurement_wall_seconds": 20.000669,
|
| 688 |
+
"client_output_tokens": 5926,
|
| 689 |
+
"server_output_tokens": 5926,
|
| 690 |
+
"aggregate_source": "openai_continuous_usage",
|
| 691 |
+
"aggregate_tps": 296.6639108412111,
|
| 692 |
+
"per_request_avg_tps": 74.16597771030277,
|
| 693 |
+
"ttft_avg": 0.1473790240706876,
|
| 694 |
+
"ttft_p50": 0.17257179808802903,
|
| 695 |
+
"ttft_p90": 0.17268561611417682,
|
| 696 |
+
"ttft_p99": 0.17268778499448673,
|
| 697 |
+
"time_to_second_token_avg": 0.025300663721282035,
|
| 698 |
+
"time_to_second_token_p50": 0.02924731746315956,
|
| 699 |
+
"time_to_second_token_p90": 0.029298453708179295,
|
| 700 |
+
"time_to_second_token_p99": 0.029313364170957357,
|
| 701 |
+
"request_latency_avg": 0.0,
|
| 702 |
+
"request_latency_p50": 0.0,
|
| 703 |
+
"request_latency_p90": 0.0,
|
| 704 |
+
"request_latency_p99": 0.0,
|
| 705 |
+
"inter_token_latency_avg": 0.013187405590978741,
|
| 706 |
+
"inter_token_latency_p50": 0.01321511895063593,
|
| 707 |
+
"inter_token_latency_p90": 0.01351998806064279,
|
| 708 |
+
"inter_token_latency_p99": 0.01357511053376785,
|
| 709 |
+
"output_tps_per_user_avg": 75.87497211232,
|
| 710 |
+
"output_tps_per_user_p50": 75.68227161840176,
|
| 711 |
+
"output_tps_per_user_p90": 77.93597879757434,
|
| 712 |
+
"output_tps_per_user_p99": 78.44750412533332,
|
| 713 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 714 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 715 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 716 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 717 |
+
"chunk_inter_token_latency_avg": 0.034238496269222964,
|
| 718 |
+
"chunk_inter_token_latency_p50": 0.03419213106811758,
|
| 719 |
+
"chunk_inter_token_latency_p90": 0.03441501889490805,
|
| 720 |
+
"chunk_inter_token_latency_p99": 0.0344653942706813,
|
| 721 |
+
"input_seq_len_avg": 78.0,
|
| 722 |
+
"output_seq_len_avg": 1925.25,
|
| 723 |
+
"output_seq_len_p50": 1918.5,
|
| 724 |
+
"output_seq_len_p90": 1975.6,
|
| 725 |
+
"output_seq_len_p99": 1988.56,
|
| 726 |
+
"request_count": 4,
|
| 727 |
+
"completed_request_count": 0,
|
| 728 |
+
"request_samples": [
|
| 729 |
+
{
|
| 730 |
+
"ttft": 0.0716844741255045,
|
| 731 |
+
"time_to_second_token": 0.01339299906976521,
|
| 732 |
+
"latency": 0.0,
|
| 733 |
+
"inter_token_latency_avg": 0.013581235253003969,
|
| 734 |
+
"chunk_inter_token_latency_avg": 0.03409873140600058,
|
| 735 |
+
"input_tokens": 78,
|
| 736 |
+
"output_tokens": 1874,
|
| 737 |
+
"output_tps_per_user": 73.63100493961437,
|
| 738 |
+
"e2e_output_tps_per_user": 0.0,
|
| 739 |
+
"completed": false
|
| 740 |
+
},
|
| 741 |
+
{
|
| 742 |
+
"ttft": 0.17267999309115112,
|
| 743 |
+
"time_to_second_token": 0.029315020889043808,
|
| 744 |
+
"latency": 0.0,
|
| 745 |
+
"inter_token_latency_avg": 0.013053159956138491,
|
| 746 |
+
"chunk_inter_token_latency_avg": 0.034284416068829246,
|
| 747 |
+
"input_tokens": 78,
|
| 748 |
+
"output_tokens": 1942,
|
| 749 |
+
"output_tps_per_user": 76.60980202190285,
|
| 750 |
+
"e2e_output_tps_per_user": 0.0,
|
| 751 |
+
"completed": false
|
| 752 |
+
},
|
| 753 |
+
{
|
| 754 |
+
"ttft": 0.17268802598118782,
|
| 755 |
+
"time_to_second_token": 0.029259796952828765,
|
| 756 |
+
"latency": 0.0,
|
| 757 |
+
"inter_token_latency_avg": 0.012738149209639133,
|
| 758 |
+
"chunk_inter_token_latency_avg": 0.034470991534656104,
|
| 759 |
+
"input_tokens": 78,
|
| 760 |
+
"output_tokens": 1990,
|
| 761 |
+
"output_tps_per_user": 78.50434027286211,
|
| 762 |
+
"e2e_output_tps_per_user": 0.0,
|
| 763 |
+
"completed": false
|
| 764 |
+
},
|
| 765 |
+
{
|
| 766 |
+
"ttft": 0.17246360308490694,
|
| 767 |
+
"time_to_second_token": 0.029234837973490357,
|
| 768 |
+
"latency": 0.0,
|
| 769 |
+
"inter_token_latency_avg": 0.01337707794513337,
|
| 770 |
+
"chunk_inter_token_latency_avg": 0.034099846067405924,
|
| 771 |
+
"input_tokens": 78,
|
| 772 |
+
"output_tokens": 1895,
|
| 773 |
+
"output_tps_per_user": 74.75474121490065,
|
| 774 |
+
"e2e_output_tps_per_user": 0.0,
|
| 775 |
+
"completed": false
|
| 776 |
+
}
|
| 777 |
+
],
|
| 778 |
+
"total_tokens": 5926,
|
| 779 |
+
"wall_time": 25.54523406806402,
|
| 780 |
+
"num_completed": 4,
|
| 781 |
+
"num_errors": 0,
|
| 782 |
+
"server_gen_throughput": 296.1937254544746,
|
| 783 |
+
"server_utilization": 0.02345058626465657,
|
| 784 |
+
"server_spec_accept_rate": 0.5028735632183908,
|
| 785 |
+
"server_spec_accept_length": 0.0,
|
| 786 |
+
"avg_running_reqs": 4,
|
| 787 |
+
"max_running_reqs": 4,
|
| 788 |
+
"effective_concurrency": 4,
|
| 789 |
+
"avg_queue_reqs": 0,
|
| 790 |
+
"max_queue_reqs": 0,
|
| 791 |
+
"queue_fraction": 0.0,
|
| 792 |
+
"underfilled": false,
|
| 793 |
+
"warmup_timed_out": false,
|
| 794 |
+
"warmup_duration": 5.534,
|
| 795 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 796 |
+
"timeout_reason": "",
|
| 797 |
+
"capacity_limited": false,
|
| 798 |
+
"hardware_summary": {
|
| 799 |
+
"samples": 8,
|
| 800 |
+
"duration_seconds": 16.831,
|
| 801 |
+
"gpu_count": 4,
|
| 802 |
+
"cpu_util_avg_pct": 11.49,
|
| 803 |
+
"cpu_temp_max_c": 77.38,
|
| 804 |
+
"gpu_util_avg_pct": 100.0,
|
| 805 |
+
"gpu_util_max_pct": 100.0,
|
| 806 |
+
"mem_util_avg_pct": 35.0,
|
| 807 |
+
"mem_util_max_pct": 43.0,
|
| 808 |
+
"temp_avg_c": 69.0,
|
| 809 |
+
"temp_max_c": 84.0,
|
| 810 |
+
"power_total_avg_w": 1172.77,
|
| 811 |
+
"power_total_max_w": 1173.84,
|
| 812 |
+
"power_limit_total_w": 1200.0,
|
| 813 |
+
"vram_used_avg_mb": 384778.0,
|
| 814 |
+
"vram_used_max_mb": 384778.0,
|
| 815 |
+
"vram_total_mb": 391548.0,
|
| 816 |
+
"vram_used_avg_pct": 98.27,
|
| 817 |
+
"vram_used_max_pct": 98.27,
|
| 818 |
+
"pcie_rx_avg_mb_s": 8004.38,
|
| 819 |
+
"pcie_rx_max_mb_s": 8287.0,
|
| 820 |
+
"pcie_tx_avg_mb_s": 7682.5,
|
| 821 |
+
"pcie_tx_max_mb_s": 7820.0
|
| 822 |
+
}
|
| 823 |
+
},
|
| 824 |
+
{
|
| 825 |
+
"concurrency": 2,
|
| 826 |
+
"context_tokens": 8192,
|
| 827 |
+
"benchmark_mode": "duration",
|
| 828 |
+
"request_count_target": 0,
|
| 829 |
+
"warmup_request_count": 0,
|
| 830 |
+
"measurement_seconds": 19.987602,
|
| 831 |
+
"measurement_wall_seconds": 20.001775,
|
| 832 |
+
"client_output_tokens": 4932,
|
| 833 |
+
"server_output_tokens": 4932,
|
| 834 |
+
"aggregate_source": "openai_continuous_usage",
|
| 835 |
+
"aggregate_tps": 246.75295838278893,
|
| 836 |
+
"per_request_avg_tps": 123.37647919139447,
|
| 837 |
+
"ttft_avg": 0.9467372725484893,
|
| 838 |
+
"ttft_p50": 0.9467372725484893,
|
| 839 |
+
"ttft_p90": 1.2301197802415118,
|
| 840 |
+
"ttft_p99": 1.293880844472442,
|
| 841 |
+
"time_to_second_token_avg": 0.02229496242944151,
|
| 842 |
+
"time_to_second_token_p50": 0.02229496242944151,
|
| 843 |
+
"time_to_second_token_p90": 0.02584285002667457,
|
| 844 |
+
"time_to_second_token_p99": 0.026641124736052006,
|
| 845 |
+
"request_latency_avg": 0.0,
|
| 846 |
+
"request_latency_p50": 0.0,
|
| 847 |
+
"request_latency_p90": 0.0,
|
| 848 |
+
"request_latency_p99": 0.0,
|
| 849 |
+
"inter_token_latency_avg": 0.008086084075571213,
|
| 850 |
+
"inter_token_latency_p50": 0.008086084075571213,
|
| 851 |
+
"inter_token_latency_p90": 0.008137498278031636,
|
| 852 |
+
"inter_token_latency_p99": 0.008149066473585232,
|
| 853 |
+
"output_tps_per_user_avg": 123.67706846442331,
|
| 854 |
+
"output_tps_per_user_p50": 123.67706846442331,
|
| 855 |
+
"output_tps_per_user_p90": 124.46345131405901,
|
| 856 |
+
"output_tps_per_user_p99": 124.64038745522704,
|
| 857 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 858 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 859 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 860 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 861 |
+
"chunk_inter_token_latency_avg": 0.021277465417892594,
|
| 862 |
+
"chunk_inter_token_latency_p50": 0.021277465417892594,
|
| 863 |
+
"chunk_inter_token_latency_p90": 0.021564048664052024,
|
| 864 |
+
"chunk_inter_token_latency_p99": 0.021628529894437896,
|
| 865 |
+
"input_seq_len_avg": 8192.0,
|
| 866 |
+
"output_seq_len_avg": 2917.0,
|
| 867 |
+
"output_seq_len_p50": 2917.0,
|
| 868 |
+
"output_seq_len_p90": 2970.6,
|
| 869 |
+
"output_seq_len_p99": 2982.66,
|
| 870 |
+
"request_count": 2,
|
| 871 |
+
"completed_request_count": 0,
|
| 872 |
+
"request_samples": [
|
| 873 |
+
{
|
| 874 |
+
"ttft": 0.5925091379322112,
|
| 875 |
+
"time_to_second_token": 0.01786010293290019,
|
| 876 |
+
"latency": 0.0,
|
| 877 |
+
"inter_token_latency_avg": 0.008021816322495684,
|
| 878 |
+
"chunk_inter_token_latency_avg": 0.021635694475591882,
|
| 879 |
+
"input_tokens": 8192,
|
| 880 |
+
"output_tokens": 2984,
|
| 881 |
+
"output_tps_per_user": 124.66004702646794,
|
| 882 |
+
"e2e_output_tps_per_user": 0.0,
|
| 883 |
+
"completed": false
|
| 884 |
+
},
|
| 885 |
+
{
|
| 886 |
+
"ttft": 1.3009654071647674,
|
| 887 |
+
"time_to_second_token": 0.026729821925982833,
|
| 888 |
+
"latency": 0.0,
|
| 889 |
+
"inter_token_latency_avg": 0.008150351828646742,
|
| 890 |
+
"chunk_inter_token_latency_avg": 0.020919236360193307,
|
| 891 |
+
"input_tokens": 8192,
|
| 892 |
+
"output_tokens": 2850,
|
| 893 |
+
"output_tps_per_user": 122.6940899023787,
|
| 894 |
+
"e2e_output_tps_per_user": 0.0,
|
| 895 |
+
"completed": false
|
| 896 |
+
}
|
| 897 |
+
],
|
| 898 |
+
"total_tokens": 4932,
|
| 899 |
+
"wall_time": 25.56168243405409,
|
| 900 |
+
"num_completed": 2,
|
| 901 |
+
"num_errors": 0,
|
| 902 |
+
"server_gen_throughput": 246.50722758411473,
|
| 903 |
+
"server_utilization": 0.012283640424343933,
|
| 904 |
+
"server_spec_accept_rate": 0.5763888888888888,
|
| 905 |
+
"server_spec_accept_length": 0.0,
|
| 906 |
+
"avg_running_reqs": 2,
|
| 907 |
+
"max_running_reqs": 2,
|
| 908 |
+
"effective_concurrency": 2,
|
| 909 |
+
"avg_queue_reqs": 0,
|
| 910 |
+
"max_queue_reqs": 0,
|
| 911 |
+
"queue_fraction": 0.0,
|
| 912 |
+
"underfilled": false,
|
| 913 |
+
"warmup_timed_out": false,
|
| 914 |
+
"warmup_duration": 5.552,
|
| 915 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 916 |
+
"timeout_reason": "",
|
| 917 |
+
"capacity_limited": false,
|
| 918 |
+
"hardware_summary": {
|
| 919 |
+
"samples": 8,
|
| 920 |
+
"duration_seconds": 16.901,
|
| 921 |
+
"gpu_count": 4,
|
| 922 |
+
"cpu_util_avg_pct": 11.55,
|
| 923 |
+
"cpu_temp_max_c": 77.25,
|
| 924 |
+
"gpu_util_avg_pct": 100.0,
|
| 925 |
+
"gpu_util_max_pct": 100.0,
|
| 926 |
+
"mem_util_avg_pct": 40.5,
|
| 927 |
+
"mem_util_max_pct": 50.0,
|
| 928 |
+
"temp_avg_c": 68.62,
|
| 929 |
+
"temp_max_c": 84.0,
|
| 930 |
+
"power_total_avg_w": 1177.05,
|
| 931 |
+
"power_total_max_w": 1177.91,
|
| 932 |
+
"power_limit_total_w": 1200.0,
|
| 933 |
+
"vram_used_avg_mb": 384778.0,
|
| 934 |
+
"vram_used_max_mb": 384778.0,
|
| 935 |
+
"vram_total_mb": 391548.0,
|
| 936 |
+
"vram_used_avg_pct": 98.27,
|
| 937 |
+
"vram_used_max_pct": 98.27,
|
| 938 |
+
"pcie_rx_avg_mb_s": 11224.88,
|
| 939 |
+
"pcie_rx_max_mb_s": 11341.0,
|
| 940 |
+
"pcie_tx_avg_mb_s": 11103.0,
|
| 941 |
+
"pcie_tx_max_mb_s": 11202.0
|
| 942 |
+
}
|
| 943 |
+
},
|
| 944 |
+
{
|
| 945 |
+
"concurrency": 4,
|
| 946 |
+
"context_tokens": 8192,
|
| 947 |
+
"benchmark_mode": "duration",
|
| 948 |
+
"request_count_target": 0,
|
| 949 |
+
"warmup_request_count": 0,
|
| 950 |
+
"measurement_seconds": 19.99956,
|
| 951 |
+
"measurement_wall_seconds": 20.000651,
|
| 952 |
+
"client_output_tokens": 5828,
|
| 953 |
+
"server_output_tokens": 5828,
|
| 954 |
+
"aggregate_source": "openai_continuous_usage",
|
| 955 |
+
"aggregate_tps": 291.40641309659526,
|
| 956 |
+
"per_request_avg_tps": 72.85160327414881,
|
| 957 |
+
"ttft_avg": 2.3201283884700388,
|
| 958 |
+
"ttft_p50": 2.213119035004638,
|
| 959 |
+
"ttft_p90": 3.644294320815243,
|
| 960 |
+
"ttft_p99": 4.196257012102286,
|
| 961 |
+
"time_to_second_token_avg": 0.034855086996685714,
|
| 962 |
+
"time_to_second_token_p50": 0.040691820555366576,
|
| 963 |
+
"time_to_second_token_p90": 0.041562757641077044,
|
| 964 |
+
"time_to_second_token_p99": 0.0415682226838544,
|
| 965 |
+
"request_latency_avg": 0.0,
|
| 966 |
+
"request_latency_p50": 0.0,
|
| 967 |
+
"request_latency_p90": 0.0,
|
| 968 |
+
"request_latency_p99": 0.0,
|
| 969 |
+
"inter_token_latency_avg": 0.013789638304529176,
|
| 970 |
+
"inter_token_latency_p50": 0.013799817255296618,
|
| 971 |
+
"inter_token_latency_p90": 0.013947021664650167,
|
| 972 |
+
"inter_token_latency_p99": 0.013999457133027752,
|
| 973 |
+
"output_tps_per_user_avg": 72.52803020492948,
|
| 974 |
+
"output_tps_per_user_p50": 72.46477597135552,
|
| 975 |
+
"output_tps_per_user_p90": 73.40383210307152,
|
| 976 |
+
"output_tps_per_user_p99": 73.74323179562586,
|
| 977 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 978 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 979 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 980 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 981 |
+
"chunk_inter_token_latency_avg": 0.0354126605281524,
|
| 982 |
+
"chunk_inter_token_latency_p50": 0.03551991527773096,
|
| 983 |
+
"chunk_inter_token_latency_p90": 0.035908490963787884,
|
| 984 |
+
"chunk_inter_token_latency_p99": 0.03602955673091524,
|
| 985 |
+
"input_seq_len_avg": 8192.0,
|
| 986 |
+
"output_seq_len_avg": 1830.25,
|
| 987 |
+
"output_seq_len_p50": 1837.5,
|
| 988 |
+
"output_seq_len_p90": 1899.9,
|
| 989 |
+
"output_seq_len_p99": 1923.3899999999999,
|
| 990 |
+
"request_count": 4,
|
| 991 |
+
"completed_request_count": 0,
|
| 992 |
+
"request_samples": [
|
| 993 |
+
{
|
| 994 |
+
"ttft": 0.5966892838478088,
|
| 995 |
+
"time_to_second_token": 0.01646787696518004,
|
| 996 |
+
"latency": 0.0,
|
| 997 |
+
"inter_token_latency_avg": 0.014005283296180816,
|
| 998 |
+
"chunk_inter_token_latency_avg": 0.03604300848281828,
|
| 999 |
+
"input_tokens": 8192,
|
| 1000 |
+
"output_tokens": 1926,
|
| 1001 |
+
"output_tps_per_user": 71.40162600443048,
|
| 1002 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1003 |
+
"completed": false
|
| 1004 |
+
},
|
| 1005 |
+
{
|
| 1006 |
+
"ttft": 2.2132799359969795,
|
| 1007 |
+
"time_to_second_token": 0.04156882991082966,
|
| 1008 |
+
"latency": 0.0,
|
| 1009 |
+
"inter_token_latency_avg": 0.013811077857745319,
|
| 1010 |
+
"chunk_inter_token_latency_avg": 0.03544521380274498,
|
| 1011 |
+
"input_tokens": 8192,
|
| 1012 |
+
"output_tokens": 1836,
|
| 1013 |
+
"output_tps_per_user": 72.40564496848414,
|
| 1014 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1015 |
+
"completed": false
|
| 1016 |
+
},
|
| 1017 |
+
{
|
| 1018 |
+
"ttft": 2.212958134012297,
|
| 1019 |
+
"time_to_second_token": 0.04154858901165426,
|
| 1020 |
+
"latency": 0.0,
|
| 1021 |
+
"inter_token_latency_avg": 0.013788556652847917,
|
| 1022 |
+
"chunk_inter_token_latency_avg": 0.035594616752716954,
|
| 1023 |
+
"input_tokens": 8192,
|
| 1024 |
+
"output_tokens": 1839,
|
| 1025 |
+
"output_tps_per_user": 72.52390697422692,
|
| 1026 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1027 |
+
"completed": false
|
| 1028 |
+
},
|
| 1029 |
+
{
|
| 1030 |
+
"ttft": 4.25758620002307,
|
| 1031 |
+
"time_to_second_token": 0.039835052099078894,
|
| 1032 |
+
"latency": 0.0,
|
| 1033 |
+
"inter_token_latency_avg": 0.013553635411342652,
|
| 1034 |
+
"chunk_inter_token_latency_avg": 0.034567803074329405,
|
| 1035 |
+
"input_tokens": 8192,
|
| 1036 |
+
"output_tokens": 1720,
|
| 1037 |
+
"output_tps_per_user": 73.78094287257635,
|
| 1038 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1039 |
+
"completed": false
|
| 1040 |
+
}
|
| 1041 |
+
],
|
| 1042 |
+
"total_tokens": 5828,
|
| 1043 |
+
"wall_time": 28.611056566005573,
|
| 1044 |
+
"num_completed": 4,
|
| 1045 |
+
"num_errors": 0,
|
| 1046 |
+
"server_gen_throughput": 291.3185246321431,
|
| 1047 |
+
"server_utilization": 0.024567280848687867,
|
| 1048 |
+
"server_spec_accept_rate": 0.4511494252873563,
|
| 1049 |
+
"server_spec_accept_length": 0.0,
|
| 1050 |
+
"avg_running_reqs": 4,
|
| 1051 |
+
"max_running_reqs": 4,
|
| 1052 |
+
"effective_concurrency": 4,
|
| 1053 |
+
"avg_queue_reqs": 0,
|
| 1054 |
+
"max_queue_reqs": 0,
|
| 1055 |
+
"queue_fraction": 0.0,
|
| 1056 |
+
"underfilled": false,
|
| 1057 |
+
"warmup_timed_out": false,
|
| 1058 |
+
"warmup_duration": 8.576,
|
| 1059 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1060 |
+
"timeout_reason": "",
|
| 1061 |
+
"capacity_limited": false,
|
| 1062 |
+
"hardware_summary": {
|
| 1063 |
+
"samples": 8,
|
| 1064 |
+
"duration_seconds": 16.82,
|
| 1065 |
+
"gpu_count": 4,
|
| 1066 |
+
"cpu_util_avg_pct": 11.51,
|
| 1067 |
+
"cpu_temp_max_c": 76.88,
|
| 1068 |
+
"gpu_util_avg_pct": 100.0,
|
| 1069 |
+
"gpu_util_max_pct": 100.0,
|
| 1070 |
+
"mem_util_avg_pct": 34.94,
|
| 1071 |
+
"mem_util_max_pct": 43.0,
|
| 1072 |
+
"temp_avg_c": 69.09,
|
| 1073 |
+
"temp_max_c": 84.0,
|
| 1074 |
+
"power_total_avg_w": 1172.67,
|
| 1075 |
+
"power_total_max_w": 1173.43,
|
| 1076 |
+
"power_limit_total_w": 1200.0,
|
| 1077 |
+
"vram_used_avg_mb": 384778.0,
|
| 1078 |
+
"vram_used_max_mb": 384778.0,
|
| 1079 |
+
"vram_total_mb": 391548.0,
|
| 1080 |
+
"vram_used_avg_pct": 98.27,
|
| 1081 |
+
"vram_used_max_pct": 98.27,
|
| 1082 |
+
"pcie_rx_avg_mb_s": 7728.12,
|
| 1083 |
+
"pcie_rx_max_mb_s": 7936.0,
|
| 1084 |
+
"pcie_tx_avg_mb_s": 7832.62,
|
| 1085 |
+
"pcie_tx_max_mb_s": 8312.0
|
| 1086 |
+
}
|
| 1087 |
+
},
|
| 1088 |
+
{
|
| 1089 |
+
"concurrency": 2,
|
| 1090 |
+
"context_tokens": 32768,
|
| 1091 |
+
"benchmark_mode": "duration",
|
| 1092 |
+
"request_count_target": 0,
|
| 1093 |
+
"warmup_request_count": 0,
|
| 1094 |
+
"measurement_seconds": 19.985443,
|
| 1095 |
+
"measurement_wall_seconds": 20.000566,
|
| 1096 |
+
"client_output_tokens": 4794,
|
| 1097 |
+
"server_output_tokens": 4794,
|
| 1098 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1099 |
+
"aggregate_tps": 239.87458866483988,
|
| 1100 |
+
"per_request_avg_tps": 119.93729433241994,
|
| 1101 |
+
"ttft_avg": 0.9611447914503515,
|
| 1102 |
+
"ttft_p50": 0.9611447914503515,
|
| 1103 |
+
"ttft_p90": 1.2436655103228986,
|
| 1104 |
+
"ttft_p99": 1.3072326720692218,
|
| 1105 |
+
"time_to_second_token_avg": 0.014196591568179429,
|
| 1106 |
+
"time_to_second_token_p50": 0.014196591568179429,
|
| 1107 |
+
"time_to_second_token_p90": 0.017423492693342268,
|
| 1108 |
+
"time_to_second_token_p99": 0.018149545446503906,
|
| 1109 |
+
"request_latency_avg": 0.0,
|
| 1110 |
+
"request_latency_p50": 0.0,
|
| 1111 |
+
"request_latency_p90": 0.0,
|
| 1112 |
+
"request_latency_p99": 0.0,
|
| 1113 |
+
"inter_token_latency_avg": 0.00824110461840657,
|
| 1114 |
+
"inter_token_latency_p50": 0.00824110461840657,
|
| 1115 |
+
"inter_token_latency_p90": 0.008593042280594048,
|
| 1116 |
+
"inter_token_latency_p99": 0.008672228254586231,
|
| 1117 |
+
"output_tps_per_user_avg": 121.68972103528682,
|
| 1118 |
+
"output_tps_per_user_p50": 121.68972103528682,
|
| 1119 |
+
"output_tps_per_user_p90": 126.88649961248755,
|
| 1120 |
+
"output_tps_per_user_p99": 128.0557747923577,
|
| 1121 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1122 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1123 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1124 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1125 |
+
"chunk_inter_token_latency_avg": 0.021374294281339835,
|
| 1126 |
+
"chunk_inter_token_latency_p50": 0.021374294281339835,
|
| 1127 |
+
"chunk_inter_token_latency_p90": 0.02164637385802789,
|
| 1128 |
+
"chunk_inter_token_latency_p99": 0.0217075917627827,
|
| 1129 |
+
"input_seq_len_avg": 32768.0,
|
| 1130 |
+
"output_seq_len_avg": 2865.0,
|
| 1131 |
+
"output_seq_len_p50": 2865.0,
|
| 1132 |
+
"output_seq_len_p90": 2953.0,
|
| 1133 |
+
"output_seq_len_p99": 2972.8,
|
| 1134 |
+
"request_count": 2,
|
| 1135 |
+
"completed_request_count": 0,
|
| 1136 |
+
"request_samples": [
|
| 1137 |
+
{
|
| 1138 |
+
"ttft": 0.6079938928596675,
|
| 1139 |
+
"time_to_second_token": 0.010162965161725879,
|
| 1140 |
+
"latency": 0.0,
|
| 1141 |
+
"inter_token_latency_avg": 0.008681026696140919,
|
| 1142 |
+
"chunk_inter_token_latency_avg": 0.0217143937521999,
|
| 1143 |
+
"input_tokens": 32768,
|
| 1144 |
+
"output_tokens": 2755,
|
| 1145 |
+
"output_tps_per_user": 115.1937478137859,
|
| 1146 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1147 |
+
"completed": false
|
| 1148 |
+
},
|
| 1149 |
+
{
|
| 1150 |
+
"ttft": 1.3142956900410354,
|
| 1151 |
+
"time_to_second_token": 0.01823021797463298,
|
| 1152 |
+
"latency": 0.0,
|
| 1153 |
+
"inter_token_latency_avg": 0.007801182540672222,
|
| 1154 |
+
"chunk_inter_token_latency_avg": 0.021034194810479773,
|
| 1155 |
+
"input_tokens": 32768,
|
| 1156 |
+
"output_tokens": 2975,
|
| 1157 |
+
"output_tps_per_user": 128.18569425678774,
|
| 1158 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1159 |
+
"completed": false
|
| 1160 |
+
}
|
| 1161 |
+
],
|
| 1162 |
+
"total_tokens": 4794,
|
| 1163 |
+
"wall_time": 25.558490577852353,
|
| 1164 |
+
"num_completed": 2,
|
| 1165 |
+
"num_errors": 0,
|
| 1166 |
+
"server_gen_throughput": 239.6320204448516,
|
| 1167 |
+
"server_utilization": 0.013121161362367406,
|
| 1168 |
+
"server_spec_accept_rate": 0.5555555555555556,
|
| 1169 |
+
"server_spec_accept_length": 0.0,
|
| 1170 |
+
"avg_running_reqs": 2,
|
| 1171 |
+
"max_running_reqs": 2,
|
| 1172 |
+
"effective_concurrency": 2,
|
| 1173 |
+
"avg_queue_reqs": 0,
|
| 1174 |
+
"max_queue_reqs": 0,
|
| 1175 |
+
"queue_fraction": 0.0,
|
| 1176 |
+
"underfilled": false,
|
| 1177 |
+
"warmup_timed_out": false,
|
| 1178 |
+
"warmup_duration": 5.551,
|
| 1179 |
+
"ready_reason": "running_reqs=2/2, queue_reqs=0, active_streams=2/2, stable=3.0s",
|
| 1180 |
+
"timeout_reason": "",
|
| 1181 |
+
"capacity_limited": false,
|
| 1182 |
+
"hardware_summary": {
|
| 1183 |
+
"samples": 9,
|
| 1184 |
+
"duration_seconds": 19.321,
|
| 1185 |
+
"gpu_count": 4,
|
| 1186 |
+
"cpu_util_avg_pct": 11.53,
|
| 1187 |
+
"cpu_temp_max_c": 76.88,
|
| 1188 |
+
"gpu_util_avg_pct": 100.0,
|
| 1189 |
+
"gpu_util_max_pct": 100.0,
|
| 1190 |
+
"mem_util_avg_pct": 40.97,
|
| 1191 |
+
"mem_util_max_pct": 50.0,
|
| 1192 |
+
"temp_avg_c": 68.61,
|
| 1193 |
+
"temp_max_c": 84.0,
|
| 1194 |
+
"power_total_avg_w": 1177.52,
|
| 1195 |
+
"power_total_max_w": 1178.4,
|
| 1196 |
+
"power_limit_total_w": 1200.0,
|
| 1197 |
+
"vram_used_avg_mb": 384778.0,
|
| 1198 |
+
"vram_used_max_mb": 384778.0,
|
| 1199 |
+
"vram_total_mb": 391548.0,
|
| 1200 |
+
"vram_used_avg_pct": 98.27,
|
| 1201 |
+
"vram_used_max_pct": 98.27,
|
| 1202 |
+
"pcie_rx_avg_mb_s": 11127.78,
|
| 1203 |
+
"pcie_rx_max_mb_s": 11295.0,
|
| 1204 |
+
"pcie_tx_avg_mb_s": 10971.33,
|
| 1205 |
+
"pcie_tx_max_mb_s": 11178.0
|
| 1206 |
+
}
|
| 1207 |
+
},
|
| 1208 |
+
{
|
| 1209 |
+
"concurrency": 4,
|
| 1210 |
+
"context_tokens": 32768,
|
| 1211 |
+
"benchmark_mode": "duration",
|
| 1212 |
+
"request_count_target": 0,
|
| 1213 |
+
"warmup_request_count": 0,
|
| 1214 |
+
"measurement_seconds": 19.994025,
|
| 1215 |
+
"measurement_wall_seconds": 20.000158,
|
| 1216 |
+
"client_output_tokens": 6042,
|
| 1217 |
+
"server_output_tokens": 6042,
|
| 1218 |
+
"aggregate_source": "openai_continuous_usage",
|
| 1219 |
+
"aggregate_tps": 302.19028317254777,
|
| 1220 |
+
"per_request_avg_tps": 75.54757079313694,
|
| 1221 |
+
"ttft_avg": 2.305535151215736,
|
| 1222 |
+
"ttft_p50": 2.2079672454856336,
|
| 1223 |
+
"ttft_p90": 3.5996002558851616,
|
| 1224 |
+
"ttft_p99": 4.136311722563113,
|
| 1225 |
+
"time_to_second_token_avg": 0.026320659555494785,
|
| 1226 |
+
"time_to_second_token_p50": 0.03037197550293058,
|
| 1227 |
+
"time_to_second_token_p90": 0.03217388866469264,
|
| 1228 |
+
"time_to_second_token_p99": 0.032832831228151914,
|
| 1229 |
+
"request_latency_avg": 0.0,
|
| 1230 |
+
"request_latency_p50": 0.0,
|
| 1231 |
+
"request_latency_p90": 0.0,
|
| 1232 |
+
"request_latency_p99": 0.0,
|
| 1233 |
+
"inter_token_latency_avg": 0.013299206291348701,
|
| 1234 |
+
"inter_token_latency_p50": 0.013299051093053463,
|
| 1235 |
+
"inter_token_latency_p90": 0.013574522517023369,
|
| 1236 |
+
"inter_token_latency_p99": 0.01365233632768284,
|
| 1237 |
+
"output_tps_per_user_avg": 75.22142959253767,
|
| 1238 |
+
"output_tps_per_user_p50": 75.19564602998912,
|
| 1239 |
+
"output_tps_per_user_p90": 76.78903624677658,
|
| 1240 |
+
"output_tps_per_user_p99": 77.24282693804788,
|
| 1241 |
+
"e2e_output_tps_per_user_avg": 0.0,
|
| 1242 |
+
"e2e_output_tps_per_user_p50": 0.0,
|
| 1243 |
+
"e2e_output_tps_per_user_p90": 0.0,
|
| 1244 |
+
"e2e_output_tps_per_user_p99": 0.0,
|
| 1245 |
+
"chunk_inter_token_latency_avg": 0.035561150767445995,
|
| 1246 |
+
"chunk_inter_token_latency_p50": 0.03564201678101365,
|
| 1247 |
+
"chunk_inter_token_latency_p90": 0.036069497146972274,
|
| 1248 |
+
"chunk_inter_token_latency_p99": 0.0361956293509813,
|
| 1249 |
+
"input_seq_len_avg": 32768.0,
|
| 1250 |
+
"output_seq_len_avg": 1899.0,
|
| 1251 |
+
"output_seq_len_p50": 1876.0,
|
| 1252 |
+
"output_seq_len_p90": 1995.4,
|
| 1253 |
+
"output_seq_len_p99": 2033.74,
|
| 1254 |
+
"request_count": 4,
|
| 1255 |
+
"completed_request_count": 0,
|
| 1256 |
+
"request_samples": [
|
| 1257 |
+
{
|
| 1258 |
+
"ttft": 0.6102597839199007,
|
| 1259 |
+
"time_to_second_token": 0.011632640147581697,
|
| 1260 |
+
"latency": 0.0,
|
| 1261 |
+
"inter_token_latency_avg": 0.013225319178200705,
|
| 1262 |
+
"chunk_inter_token_latency_avg": 0.03620964404031564,
|
| 1263 |
+
"input_tokens": 32768,
|
| 1264 |
+
"output_tokens": 2038,
|
| 1265 |
+
"output_tps_per_user": 75.61254186199908,
|
| 1266 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1267 |
+
"completed": false
|
| 1268 |
+
},
|
| 1269 |
+
{
|
| 1270 |
+
"ttft": 2.2081260830163956,
|
| 1271 |
+
"time_to_second_token": 0.030465519055724144,
|
| 1272 |
+
"latency": 0.0,
|
| 1273 |
+
"inter_token_latency_avg": 0.013372783007906223,
|
| 1274 |
+
"chunk_inter_token_latency_avg": 0.03574248772917108,
|
| 1275 |
+
"input_tokens": 32768,
|
| 1276 |
+
"output_tokens": 1896,
|
| 1277 |
+
"output_tps_per_user": 74.77875019797916,
|
| 1278 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1279 |
+
"completed": false
|
| 1280 |
+
},
|
| 1281 |
+
{
|
| 1282 |
+
"ttft": 2.2078084079548717,
|
| 1283 |
+
"time_to_second_token": 0.03027843195013702,
|
| 1284 |
+
"latency": 0.0,
|
| 1285 |
+
"inter_token_latency_avg": 0.013660982306645003,
|
| 1286 |
+
"chunk_inter_token_latency_avg": 0.035541545832856215,
|
| 1287 |
+
"input_tokens": 32768,
|
| 1288 |
+
"output_tokens": 1856,
|
| 1289 |
+
"output_tps_per_user": 73.20117818420553,
|
| 1290 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1291 |
+
"completed": false
|
| 1292 |
+
},
|
| 1293 |
+
{
|
| 1294 |
+
"ttft": 4.195946329971775,
|
| 1295 |
+
"time_to_second_token": 0.03290604706853628,
|
| 1296 |
+
"latency": 0.0,
|
| 1297 |
+
"inter_token_latency_avg": 0.012937740672642875,
|
| 1298 |
+
"chunk_inter_token_latency_avg": 0.03475092546744106,
|
| 1299 |
+
"input_tokens": 32768,
|
| 1300 |
+
"output_tokens": 1806,
|
| 1301 |
+
"output_tps_per_user": 77.29324812596693,
|
| 1302 |
+
"e2e_output_tps_per_user": 0.0,
|
| 1303 |
+
"completed": false
|
| 1304 |
+
}
|
| 1305 |
+
],
|
| 1306 |
+
"total_tokens": 6042,
|
| 1307 |
+
"wall_time": 28.607491098809987,
|
| 1308 |
+
"num_completed": 4,
|
| 1309 |
+
"num_errors": 0,
|
| 1310 |
+
"server_gen_throughput": 302.0169777555693,
|
| 1311 |
+
"server_utilization": 0.02540480178671134,
|
| 1312 |
+
"server_spec_accept_rate": 0.5201149425287356,
|
| 1313 |
+
"server_spec_accept_length": 0.0,
|
| 1314 |
+
"avg_running_reqs": 4,
|
| 1315 |
+
"max_running_reqs": 4,
|
| 1316 |
+
"effective_concurrency": 4,
|
| 1317 |
+
"avg_queue_reqs": 0,
|
| 1318 |
+
"max_queue_reqs": 0,
|
| 1319 |
+
"queue_fraction": 0.0,
|
| 1320 |
+
"underfilled": false,
|
| 1321 |
+
"warmup_timed_out": false,
|
| 1322 |
+
"warmup_duration": 8.578,
|
| 1323 |
+
"ready_reason": "running_reqs=4/4, queue_reqs=0, active_streams=4/4, stable=3.0s",
|
| 1324 |
+
"timeout_reason": "",
|
| 1325 |
+
"capacity_limited": false,
|
| 1326 |
+
"hardware_summary": {
|
| 1327 |
+
"samples": 8,
|
| 1328 |
+
"duration_seconds": 16.85,
|
| 1329 |
+
"gpu_count": 4,
|
| 1330 |
+
"cpu_util_avg_pct": 11.53,
|
| 1331 |
+
"cpu_temp_max_c": 76.88,
|
| 1332 |
+
"gpu_util_avg_pct": 100.0,
|
| 1333 |
+
"gpu_util_max_pct": 100.0,
|
| 1334 |
+
"mem_util_avg_pct": 35.12,
|
| 1335 |
+
"mem_util_max_pct": 44.0,
|
| 1336 |
+
"temp_avg_c": 69.22,
|
| 1337 |
+
"temp_max_c": 84.0,
|
| 1338 |
+
"power_total_avg_w": 1172.46,
|
| 1339 |
+
"power_total_max_w": 1173.31,
|
| 1340 |
+
"power_limit_total_w": 1200.0,
|
| 1341 |
+
"vram_used_avg_mb": 384778.0,
|
| 1342 |
+
"vram_used_max_mb": 384778.0,
|
| 1343 |
+
"vram_total_mb": 391548.0,
|
| 1344 |
+
"vram_used_avg_pct": 98.27,
|
| 1345 |
+
"vram_used_max_pct": 98.27,
|
| 1346 |
+
"pcie_rx_avg_mb_s": 7907.88,
|
| 1347 |
+
"pcie_rx_max_mb_s": 8173.0,
|
| 1348 |
+
"pcie_tx_avg_mb_s": 7679.75,
|
| 1349 |
+
"pcie_tx_max_mb_s": 8296.0
|
| 1350 |
+
}
|
| 1351 |
+
}
|
| 1352 |
+
],
|
| 1353 |
+
"summary_table": {
|
| 1354 |
+
"0": {
|
| 1355 |
+
"1": 175.1804006140207,
|
| 1356 |
+
"2": 236.39031276129717,
|
| 1357 |
+
"4": 296.6639108412111
|
| 1358 |
+
},
|
| 1359 |
+
"8192": {
|
| 1360 |
+
"1": 174.16263081468745,
|
| 1361 |
+
"2": 246.75295838278893,
|
| 1362 |
+
"4": 291.40641309659526
|
| 1363 |
+
},
|
| 1364 |
+
"32768": {
|
| 1365 |
+
"1": 179.316188111569,
|
| 1366 |
+
"2": 239.87458866483988,
|
| 1367 |
+
"4": 302.19028317254777
|
| 1368 |
+
}
|
| 1369 |
+
},
|
| 1370 |
+
"burst_results": [],
|
| 1371 |
+
"burst_summary_table": {},
|
| 1372 |
+
"methodology": {
|
| 1373 |
+
"prefill": {
|
| 1374 |
+
"name": "Prefill",
|
| 1375 |
+
"present": false,
|
| 1376 |
+
"mode": "skipped",
|
| 1377 |
+
"formula": "prompt_tokens / TTFT",
|
| 1378 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 1379 |
+
},
|
| 1380 |
+
"sustained_decode": {
|
| 1381 |
+
"name": "Sustained Decode",
|
| 1382 |
+
"present": true,
|
| 1383 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 1384 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 1385 |
+
},
|
| 1386 |
+
"burst_e2e_decode": {
|
| 1387 |
+
"name": "Burst / E2E Decode",
|
| 1388 |
+
"present": false,
|
| 1389 |
+
"status": "not run; use --run-burst",
|
| 1390 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 1391 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 1392 |
+
}
|
| 1393 |
+
}
|
| 1394 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/decode-cap8192.log
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
New version available: v0.6.2 (current: v0.4.29)
|
| 3 |
+
Upgrade and restart? [Y/n]: Skipping update.
|
| 4 |
+
|
| 5 |
+
╭──────────────────────────── NVIDIA P2P Override ─────────────────────────────╮
|
| 6 |
+
│ Effective: yes │
|
| 7 |
+
│ Configured file: yes (/etc/modprobe.d/nvidia-p2p-override.conf) │
|
| 8 |
+
│ Runtime: ForceP2P=0x11; RMForceP2PType=1; RMPcieP2PType=2; │
|
| 9 |
+
│ GrdmaPciTopoCheckOverride=1; EnableResizableBar=1; DmaRemapPeerMmio=1 │
|
| 10 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 11 |
+
╭─────────────────────────────── Configuration ────────────────────────────────╮
|
| 12 |
+
│ LLM Inference Benchmark │
|
| 13 |
+
│ Model: glm53-flash-trellismx-p8-k45 @ 127.0.0.1:8001 │
|
| 14 |
+
│ Decode concurrency: [1, 2, 4] │
|
| 15 |
+
│ Decode contexts: ['0', '8k', '32k'] │
|
| 16 |
+
│ Duration: 20.0s per decode test | Max tokens: 8192 │
|
| 17 |
+
│ Pre-decode warmup: C=1 max-runnable context for 3s │
|
| 18 |
+
│ Prefill: skipped | Sustained decode: 9 cells │
|
| 19 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 20 |
+
Engine: vLLM 0.26.1rc0+glm53.flash.nvfp4.luke.clean.r1.vllme75bcfd.b12x58a046f
|
| 21 |
+
Models: ['glm53-flash-trellismx-p8-k45']
|
| 22 |
+
KV cache budget (vLLM metrics): 29,351,936 tokens (3583 blocks × 2048; local
|
| 23 |
+
7,337,984 × CP 4; CP source: local process)
|
| 24 |
+
Model context length: 1,000,000 tokens
|
| 25 |
+
Prefill tests: skipped
|
| 26 |
+
Calibrating padding text (run=lijekedoguqy, up to 32k)...
|
| 27 |
+
8k: 50,540 chars (8,192 prompt tokens via /tokenize)
|
| 28 |
+
32k: 205,139 chars (32,768 prompt tokens via /tokenize)
|
| 29 |
+
Token targeting: /tokenize exact
|
| 30 |
+
Done.
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
llm-decode-bench v0.4.29
|
| 35 |
+
╭────────────────────────────────── Phase 2 ───────────────────────────────────╮
|
| 36 |
+
│ Sustained Decode │
|
| 37 |
+
│ Steady-state decode throughput after the engine has admitted the requested │
|
| 38 |
+
│ concurrency and passed warmup. Use this as the main tuning/regression signal │
|
| 39 |
+
│ for kernels, NCCL, DCP, MTP, and scheduler changes. │
|
| 40 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 41 |
+
Aggregate tok/s + TTFT/ITL
|
| 42 |
+
╭────────────┬─────────────┬─────────────┬──────────────╮
|
| 43 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 44 |
+
├────────────┼─────────────┼─────────────┼──────────────┤
|
| 45 |
+
│ 0 │ 175.2 70/6 │ 236.4 112/8 │ 296.7 173/13 │
|
| 46 |
+
│ 8k │ 174.2 591/6 │ 246.8 947/8 │ 291.4 2k/14 │
|
| 47 |
+
│ 32k │ 179.3 603/6 │ 239.9 961/8 │ 302.2 2k/13 │
|
| 48 |
+
╰────────────┴─────────────┴─────────────┴──────────────╯
|
| 49 |
+
Sustained Decode: aggregate tok/s uses OpenAI stream usage by default
|
| 50 |
+
(continuous completion_tokens when the server supports it). Prometheus is kept
|
| 51 |
+
as validation/scheduler data.
|
| 52 |
+
Aggregate source(s): openai_continuous_usage
|
| 53 |
+
Per-Request tok/s
|
| 54 |
+
╭────────────┬───────┬───────┬──────╮
|
| 55 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 56 |
+
├────────────┼───────┼───────┼──────┤
|
| 57 |
+
│ 0 │ 175.2 │ 118.2 │ 74.2 │
|
| 58 |
+
│ 8k │ 174.2 │ 123.4 │ 72.9 │
|
| 59 |
+
│ 32k │ 179.3 │ 119.9 │ 75.5 │
|
| 60 |
+
╰────────────┴───────┴───────┴──────╯
|
| 61 |
+
Client request latency: p50 /
|
| 62 |
+
p90 ms
|
| 63 |
+
╭────────────┬─────┬─────┬─────╮
|
| 64 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 65 |
+
├────────────┼─────┼─────┼─────┤
|
| 66 |
+
│ 0 │ —/— │ —/— │ —/— │
|
| 67 |
+
│ 8k │ —/— │ —/— │ —/— │
|
| 68 |
+
│ 32k │ —/— │ —/— │ —/— │
|
| 69 |
+
╰────────────┴─────┴─────┴─────╯
|
| 70 |
+
Aggregate cells show dim detail as TTFT ms / ITL ms for the same ctx/conc
|
| 71 |
+
coordinate. ITL is computed from observed generated tokens, including streams
|
| 72 |
+
stopped at the measurement boundary; a missing ITL means no stream produced at
|
| 73 |
+
least two measured output tokens. Per-request tok/s and request latency are
|
| 74 |
+
shown in separate per-cell matrices. Completion/sample counts and full
|
| 75 |
+
request-level distributions remain in JSON under request_samples.
|
| 76 |
+
Sustained mode: client latency metrics explain request UX variance; aggregate
|
| 77 |
+
tok/s remains the primary throughput signal.
|
| 78 |
+
ITL=(last_token_time-first_token_time)/(output_tokens-1), user tok/s=1/ITL.
|
| 79 |
+
Hardware Summary
|
| 80 |
+
╭───┬─┬───────┬───────────┬───────┬─────────┬─────┬──────┬─────┬───────────────╮
|
| 81 |
+
│ … │ │ mode │ GPU avg/… │ Mem … │ W avg/… │ T … │ CPU… │ VR… │ PCIe rx/tx a… │
|
| 82 |
+
├───┼─┼───────┼───────────┼───────┼─────────┼─────┼──────┼─────┼───────────────┤
|
| 83 |
+
│ 0 │ │ sust… │ 99/99% │ 44% │ 1153/1… │ 82C │ 76C │ 98… │ 8331/8194 │
|
| 84 |
+
│ … │ │ sust… │ 99/99% │ 44% │ 1154/1… │ 83C │ 77C │ 98… │ 8357/8151 │
|
| 85 |
+
│ … │ │ sust… │ 99/99% │ 44% │ 1156/1… │ 83C │ 76C │ 98… │ 8340/8144 │
|
| 86 |
+
│ 0 │ │ sust… │ 100/100% │ 40% │ 1176/1… │ 84C │ 77C │ 98… │ 11179/11088 │
|
| 87 |
+
│ 0 │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 77C │ 98… │ 8004/7682 │
|
| 88 |
+
│ … │ │ sust… │ 100/100% │ 40% │ 1177/1… │ 84C │ 77C │ 98… │ 11225/11103 │
|
| 89 |
+
│ … │ │ sust… │ 100/100% │ 35% │ 1173/1… │ 84C │ 77C │ 98… │ 7728/7833 │
|
| 90 |
+
│ … │ │ sust… │ 100/100% │ 41% │ 1178/1… │ 84C │ 77C │ 98… │ 11128/10971 │
|
| 91 |
+
│ … │ │ sust… │ 100/100% │ 35% │ 1172/1… │ 84C │ 77C │ 98… │ 7908/7680 │
|
| 92 |
+
╰───┴─┴───────┴───────────┴───────┴─────────┴─────┴──────┴─────┴───────────────╯
|
| 93 |
+
╭───────────────────────── Whole-run GPU Power ─────────────────────────╮
|
| 94 |
+
│ avg 1,105 W | max 1,178 W | limit 1,200 W | over 4m 29s | 113 samples │
|
| 95 |
+
╰───────────────────────────────────────────────────────────────────────╯
|
| 96 |
+
Hardware summary is sampled from nvidia-smi during the measured part of each
|
| 97 |
+
cell. Whole-run GPU power is the sampled sum of GPU power draw across the
|
| 98 |
+
complete benchmark run, not wall-outlet system power. PCIe rx/tx is MB/s and is
|
| 99 |
+
a coarse live diagnostic, not a per-kernel NCCL profiler.
|
| 100 |
+
|
| 101 |
+
╭────────────────────────────────── Phase 3 ───────────────────────────────────╮
|
| 102 |
+
│ Burst / E2E Decode │
|
| 103 |
+
│ Not run. Re-run with --run-burst to append a finite client-facing request │
|
| 104 |
+
│ burst after Sustained Decode. This is intentionally disabled by default │
|
| 105 |
+
│ because it adds another full decode matrix. │
|
| 106 |
+
╰──────────────────────────────────────────────────────────────────────────────╯
|
| 107 |
+
|
| 108 |
+
╭────────────────────────────── Primary Summary ───────────────────────────────╮
|
| 109 |
+
│ Primary matrices repeated last so the important numbers are visible without │
|
| 110 |
+
│ scrolling back through diagnostics. │
|
| 111 |
+
╰─────────────────────────────────────��────────────────────────────────────────╯
|
| 112 |
+
Aggregate decode tok/s
|
| 113 |
+
╭────────────┬───────┬───────┬───────╮
|
| 114 |
+
│ ctx \ conc │ 1 │ 2 │ 4 │
|
| 115 |
+
├────────────┼───────┼───────┼───────┤
|
| 116 |
+
│ 0 │ 175.2 │ 236.4 │ 296.7 │
|
| 117 |
+
│ 8k │ 174.2 │ 246.8 │ 291.4 │
|
| 118 |
+
│ 32k │ 179.3 │ 239.9 │ 302.2 │
|
| 119 |
+
╰────────────┴───────┴───────┴───────╯
|
| 120 |
+
|
| 121 |
+
Results saved to
|
| 122 |
+
<campaign>/candidate-speed-wi
|
| 123 |
+
ndow-01/results-01/decode-warp-quant/rep-2/decode-cap8192.json
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-command.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
"/usr/bin/python3",
|
| 3 |
+
"<workspace>/trellismx-performance-audit-20260908/llm_decode_bench.py",
|
| 4 |
+
"--host",
|
| 5 |
+
"127.0.0.1",
|
| 6 |
+
"--port",
|
| 7 |
+
"8001",
|
| 8 |
+
"--model",
|
| 9 |
+
"glm53-flash-trellismx-p8-k45",
|
| 10 |
+
"--duration",
|
| 11 |
+
"20",
|
| 12 |
+
"--max-tokens",
|
| 13 |
+
"8192",
|
| 14 |
+
"--token-targeting",
|
| 15 |
+
"exact",
|
| 16 |
+
"--display-mode",
|
| 17 |
+
"plain",
|
| 18 |
+
"--output",
|
| 19 |
+
"<campaign>/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json",
|
| 20 |
+
"--contexts",
|
| 21 |
+
"0",
|
| 22 |
+
"--concurrency",
|
| 23 |
+
"1,2,4",
|
| 24 |
+
"--prefill-only",
|
| 25 |
+
"--prefill-contexts",
|
| 26 |
+
"8k,32k,64k,128k",
|
| 27 |
+
"--prefill-duration",
|
| 28 |
+
"20",
|
| 29 |
+
"--cell-warmup-timeout-seconds",
|
| 30 |
+
"180"
|
| 31 |
+
]
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill-receipt.json
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"exit_code": 0,
|
| 3 |
+
"result_exists": true,
|
| 4 |
+
"sha256": "d0342998106c3361f6c09187c2c66554d2169d2910240fa71372ec70b061bdbb"
|
| 5 |
+
}
|
results/speed-20260909/evidence/candidate-speed-window-01/results-01/decode-warp-quant/rep-2/prefill.json
ADDED
|
@@ -0,0 +1,396 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"version": "0.4.29",
|
| 4 |
+
"engine": "vllm",
|
| 5 |
+
"model": "glm53-flash-trellismx-p8-k45",
|
| 6 |
+
"server": "127.0.0.1:8001",
|
| 7 |
+
"timestamp": "2026-09-09T02:28:52.054869",
|
| 8 |
+
"decode_mode": "duration",
|
| 9 |
+
"primary_decode_layer": "sustained_decode",
|
| 10 |
+
"duration_per_test": 20.0,
|
| 11 |
+
"request_count": 0,
|
| 12 |
+
"warmup_request_count": 0,
|
| 13 |
+
"run_burst": false,
|
| 14 |
+
"prefill_mode": "standalone_cold",
|
| 15 |
+
"standalone_prefill": true,
|
| 16 |
+
"prefill_only": true,
|
| 17 |
+
"skip_prefill": false,
|
| 18 |
+
"burst_e2e_status": "not_run_use_--run-burst",
|
| 19 |
+
"burst_request_count": 0,
|
| 20 |
+
"burst_warmup_request_count": 0,
|
| 21 |
+
"burst_requests_per_concurrency": 5,
|
| 22 |
+
"decode_warmup_seconds": 3.0,
|
| 23 |
+
"decode_warmup_context": 0,
|
| 24 |
+
"decode_warmup_concurrency": 1,
|
| 25 |
+
"cell_warmup_timeout_seconds": 180.0,
|
| 26 |
+
"cell_warmup_timeout_policy": "<=32k:60s,64k:120s,>=128k:180s when override is 0",
|
| 27 |
+
"show_capacity_limited_values": false,
|
| 28 |
+
"max_tokens": 8192,
|
| 29 |
+
"temperature": null,
|
| 30 |
+
"ignore_eos": true,
|
| 31 |
+
"max_total_tokens": 29351936,
|
| 32 |
+
"dcp_size": 0,
|
| 33 |
+
"metrics_available": true,
|
| 34 |
+
"metrics_warning": "",
|
| 35 |
+
"concurrency_levels": [
|
| 36 |
+
1,
|
| 37 |
+
2,
|
| 38 |
+
4
|
| 39 |
+
],
|
| 40 |
+
"context_lengths": [
|
| 41 |
+
0
|
| 42 |
+
],
|
| 43 |
+
"startup_diagnostics_available": true,
|
| 44 |
+
"nvidia_p2p_override_effective": true,
|
| 45 |
+
"p2pmark_status": "not_run",
|
| 46 |
+
"amd_fabric_status": "not_run"
|
| 47 |
+
},
|
| 48 |
+
"startup_diagnostics": {
|
| 49 |
+
"version": "0.4.29",
|
| 50 |
+
"server_url": "http://127.0.0.1:8001",
|
| 51 |
+
"hostname": "<host>",
|
| 52 |
+
"uname": "Linux <host> 6.18.7-76061807-generic #202601231045~1769703228~24.04~cb87b5b SMP PREEMPT_DYNAMIC Thu J x86_64 x86_64 x86_64 GNU/Linux",
|
| 53 |
+
"env": {},
|
| 54 |
+
"args": {
|
| 55 |
+
"concurrency": "1,2,4",
|
| 56 |
+
"contexts": "0",
|
| 57 |
+
"max_tokens": 8192,
|
| 58 |
+
"duration": 20.0,
|
| 59 |
+
"request_count": 0,
|
| 60 |
+
"run_burst": false,
|
| 61 |
+
"standalone_prefill": true,
|
| 62 |
+
"prefill_only": true,
|
| 63 |
+
"skip_prefill": false,
|
| 64 |
+
"prefill_contexts": "8k,32k,64k,128k",
|
| 65 |
+
"prefill_metric": "client",
|
| 66 |
+
"dcp_size": 0,
|
| 67 |
+
"kv_budget": 0
|
| 68 |
+
},
|
| 69 |
+
"nvidia_p2p_override": {
|
| 70 |
+
"effective": true,
|
| 71 |
+
"configured": true,
|
| 72 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 73 |
+
"params_available": true,
|
| 74 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 75 |
+
"modprobe_available": true,
|
| 76 |
+
"runtime": {
|
| 77 |
+
"ForceP2P": "0x11",
|
| 78 |
+
"RMForceP2PType": "1",
|
| 79 |
+
"RMPcieP2PType": "2",
|
| 80 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 81 |
+
"EnableResizableBar": "1",
|
| 82 |
+
"DmaRemapPeerMmio": "1"
|
| 83 |
+
},
|
| 84 |
+
"expected": {
|
| 85 |
+
"ForceP2P": "0x11",
|
| 86 |
+
"RMForceP2PType": "1",
|
| 87 |
+
"RMPcieP2PType": "2",
|
| 88 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 89 |
+
"EnableResizableBar": "1"
|
| 90 |
+
},
|
| 91 |
+
"missing": [],
|
| 92 |
+
"mismatched": {},
|
| 93 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 94 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 95 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 96 |
+
},
|
| 97 |
+
"p2pmark": {
|
| 98 |
+
"status": "not_run"
|
| 99 |
+
},
|
| 100 |
+
"amd_fabric": {
|
| 101 |
+
"status": "not_run"
|
| 102 |
+
},
|
| 103 |
+
"nvidia_smi_query": {
|
| 104 |
+
"cmd": [
|
| 105 |
+
"nvidia-smi",
|
| 106 |
+
"--query-gpu=index,name,driver_version,pci.bus_id,pcie.link.gen.current,pcie.link.width.current,power.limit",
|
| 107 |
+
"--format=csv,noheader,nounits"
|
| 108 |
+
],
|
| 109 |
+
"returncode": 0,
|
| 110 |
+
"stdout": "0, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:01:00.0, 5, 16, 300.00\n1, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:21:00.0, 5, 16, 300.00\n2, NVIDIA RTX PRO 6000 Blackwell Max-Q Workstation Edition, 610.57.04, 00000000:81:00.0, 5, 16, 300.00\n3, NVIDIA RTX PRO 6000 Blackwell Workstation Edition, 610.57.04, 00000000:C1:00.0, 5, 16, 300.00",
|
| 111 |
+
"stderr": ""
|
| 112 |
+
},
|
| 113 |
+
"nvidia_smi_topo": {
|
| 114 |
+
"cmd": [
|
| 115 |
+
"nvidia-smi",
|
| 116 |
+
"topo",
|
| 117 |
+
"-m"
|
| 118 |
+
],
|
| 119 |
+
"returncode": 0,
|
| 120 |
+
"stdout": "\u001b[4mGPU0\tGPU1\tGPU2\tGPU3\tCPU Affinity\tNUMA Affinity\tGPU NUMA ID\u001b[0m\nGPU0\t X \tNODE\tNODE\tNODE\t0-47\t0\t\tN/A\nGPU1\tNODE\t X \tNODE\tNODE\t0-47\t0\t\tN/A\nGPU2\tNODE\tNODE\t X \tNODE\t0-47\t0\t\tN/A\nGPU3\tNODE\tNODE\tNODE\t X \t0-47\t0\t\tN/A\n\nLegend:\n\n X = Self\n SYS = Connection traversing PCIe as well as the SMP interconnect between NUMA nodes (e.g., QPI/UPI)\n NODE = Connection traversing PCIe as well as the interconnect between PCIe Host Bridges within a NUMA node\n PHB = Connection traversing PCIe as well as a PCIe Host Bridge (typically the CPU)\n PXB = Connection traversing multiple PCIe bridges (without traversing the PCIe Host Bridge)\n PIX = Connection traversing at most a single PCIe bridge\n NV# = Connection traversing a bonded set of # NVLinks",
|
| 121 |
+
"stderr": ""
|
| 122 |
+
}
|
| 123 |
+
},
|
| 124 |
+
"nvidia_p2p_override": {
|
| 125 |
+
"effective": true,
|
| 126 |
+
"configured": true,
|
| 127 |
+
"params_path": "/proc/driver/nvidia/params",
|
| 128 |
+
"params_available": true,
|
| 129 |
+
"modprobe_path": "/etc/modprobe.d/nvidia-p2p-override.conf",
|
| 130 |
+
"modprobe_available": true,
|
| 131 |
+
"runtime": {
|
| 132 |
+
"ForceP2P": "0x11",
|
| 133 |
+
"RMForceP2PType": "1",
|
| 134 |
+
"RMPcieP2PType": "2",
|
| 135 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 136 |
+
"EnableResizableBar": "1",
|
| 137 |
+
"DmaRemapPeerMmio": "1"
|
| 138 |
+
},
|
| 139 |
+
"expected": {
|
| 140 |
+
"ForceP2P": "0x11",
|
| 141 |
+
"RMForceP2PType": "1",
|
| 142 |
+
"RMPcieP2PType": "2",
|
| 143 |
+
"GrdmaPciTopoCheckOverride": "1",
|
| 144 |
+
"EnableResizableBar": "1"
|
| 145 |
+
},
|
| 146 |
+
"missing": [],
|
| 147 |
+
"mismatched": {},
|
| 148 |
+
"registry_dwords": "ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1",
|
| 149 |
+
"suggested_modprobe_line": "options nvidia NVreg_RegistryDwords=\"ForceP2P=0x11;RMForceP2PType=1;RMPcieP2PType=2;GrdmaPciTopoCheckOverride=1;EnableResizableBar=1\"",
|
| 150 |
+
"suggested_reload": "stop GPU workloads, then reload NVIDIA modules or reboot; the modprobe file alone is not enough until the nvidia module is reloaded"
|
| 151 |
+
},
|
| 152 |
+
"p2pmark": {
|
| 153 |
+
"status": "not_run"
|
| 154 |
+
},
|
| 155 |
+
"amd_fabric": {
|
| 156 |
+
"status": "not_run"
|
| 157 |
+
},
|
| 158 |
+
"hardware_run_summary": {
|
| 159 |
+
"samples": 48,
|
| 160 |
+
"duration_seconds": 113.123,
|
| 161 |
+
"gpu_count": 4,
|
| 162 |
+
"cpu_util_avg_pct": 10.91,
|
| 163 |
+
"cpu_temp_max_c": 77.25,
|
| 164 |
+
"gpu_util_avg_pct": 88.84,
|
| 165 |
+
"gpu_util_max_pct": 100.0,
|
| 166 |
+
"mem_util_avg_pct": 19.54,
|
| 167 |
+
"mem_util_max_pct": 34.0,
|
| 168 |
+
"temp_avg_c": 67.36,
|
| 169 |
+
"temp_max_c": 84.0,
|
| 170 |
+
"power_total_avg_w": 1049.68,
|
| 171 |
+
"power_total_max_w": 1145.97,
|
| 172 |
+
"power_limit_total_w": 1200.0,
|
| 173 |
+
"vram_used_avg_mb": 384778.0,
|
| 174 |
+
"vram_used_max_mb": 384778.0,
|
| 175 |
+
"vram_total_mb": 391548.0,
|
| 176 |
+
"vram_used_avg_pct": 98.27,
|
| 177 |
+
"vram_used_max_pct": 98.27,
|
| 178 |
+
"pcie_rx_avg_mb_s": 39825.88,
|
| 179 |
+
"pcie_rx_max_mb_s": 55605.0,
|
| 180 |
+
"pcie_tx_avg_mb_s": 40385.83,
|
| 181 |
+
"pcie_tx_max_mb_s": 53259.0
|
| 182 |
+
},
|
| 183 |
+
"event_log": [],
|
| 184 |
+
"prefill": {
|
| 185 |
+
"8192": {
|
| 186 |
+
"ttft_seconds": 1.155,
|
| 187 |
+
"prefill_seconds": 1.155,
|
| 188 |
+
"tok_per_sec": 7097.0,
|
| 189 |
+
"client_ttft_seconds": 1.155,
|
| 190 |
+
"client_tok_per_sec": 7097.0,
|
| 191 |
+
"prompt_tokens": 8194,
|
| 192 |
+
"samples": 14,
|
| 193 |
+
"method": "client",
|
| 194 |
+
"server_validation": {
|
| 195 |
+
"method": "",
|
| 196 |
+
"tok_per_sec": 0.0,
|
| 197 |
+
"prefill_seconds": 0.0,
|
| 198 |
+
"prompt_tokens": 0,
|
| 199 |
+
"request_prompt_tokens": 0,
|
| 200 |
+
"cached_tokens": 0,
|
| 201 |
+
"token_source": "",
|
| 202 |
+
"samples": 0,
|
| 203 |
+
"invalid_reason": ""
|
| 204 |
+
},
|
| 205 |
+
"hardware_summary": {
|
| 206 |
+
"samples": 9,
|
| 207 |
+
"duration_seconds": 19.191,
|
| 208 |
+
"gpu_count": 4,
|
| 209 |
+
"cpu_util_avg_pct": 11.01,
|
| 210 |
+
"cpu_temp_max_c": 75.75,
|
| 211 |
+
"gpu_util_avg_pct": 76.69,
|
| 212 |
+
"gpu_util_max_pct": 100.0,
|
| 213 |
+
"mem_util_avg_pct": 18.22,
|
| 214 |
+
"mem_util_max_pct": 34.0,
|
| 215 |
+
"temp_avg_c": 65.86,
|
| 216 |
+
"temp_max_c": 80.0,
|
| 217 |
+
"power_total_avg_w": 987.33,
|
| 218 |
+
"power_total_max_w": 1139.37,
|
| 219 |
+
"power_limit_total_w": 1200.0,
|
| 220 |
+
"vram_used_avg_mb": 384778.0,
|
| 221 |
+
"vram_used_max_mb": 384778.0,
|
| 222 |
+
"vram_total_mb": 391548.0,
|
| 223 |
+
"vram_used_avg_pct": 98.27,
|
| 224 |
+
"vram_used_max_pct": 98.27,
|
| 225 |
+
"pcie_rx_avg_mb_s": 29085.56,
|
| 226 |
+
"pcie_rx_max_mb_s": 53572.0,
|
| 227 |
+
"pcie_tx_avg_mb_s": 29491.11,
|
| 228 |
+
"pcie_tx_max_mb_s": 53259.0
|
| 229 |
+
}
|
| 230 |
+
},
|
| 231 |
+
"32768": {
|
| 232 |
+
"ttft_seconds": 4.454,
|
| 233 |
+
"prefill_seconds": 4.454,
|
| 234 |
+
"tok_per_sec": 7357.0,
|
| 235 |
+
"client_ttft_seconds": 4.454,
|
| 236 |
+
"client_tok_per_sec": 7357.0,
|
| 237 |
+
"prompt_tokens": 32770,
|
| 238 |
+
"samples": 5,
|
| 239 |
+
"method": "client",
|
| 240 |
+
"server_validation": {
|
| 241 |
+
"method": "",
|
| 242 |
+
"tok_per_sec": 0.0,
|
| 243 |
+
"prefill_seconds": 0.0,
|
| 244 |
+
"prompt_tokens": 0,
|
| 245 |
+
"request_prompt_tokens": 0,
|
| 246 |
+
"cached_tokens": 0,
|
| 247 |
+
"token_source": "",
|
| 248 |
+
"samples": 0,
|
| 249 |
+
"invalid_reason": ""
|
| 250 |
+
},
|
| 251 |
+
"hardware_summary": {
|
| 252 |
+
"samples": 10,
|
| 253 |
+
"duration_seconds": 21.644,
|
| 254 |
+
"gpu_count": 4,
|
| 255 |
+
"cpu_util_avg_pct": 11.16,
|
| 256 |
+
"cpu_temp_max_c": 76.38,
|
| 257 |
+
"gpu_util_avg_pct": 90.4,
|
| 258 |
+
"gpu_util_max_pct": 100.0,
|
| 259 |
+
"mem_util_avg_pct": 19.95,
|
| 260 |
+
"mem_util_max_pct": 30.0,
|
| 261 |
+
"temp_avg_c": 67.0,
|
| 262 |
+
"temp_max_c": 83.0,
|
| 263 |
+
"power_total_avg_w": 1056.58,
|
| 264 |
+
"power_total_max_w": 1144.59,
|
| 265 |
+
"power_limit_total_w": 1200.0,
|
| 266 |
+
"vram_used_avg_mb": 384778.0,
|
| 267 |
+
"vram_used_max_mb": 384778.0,
|
| 268 |
+
"vram_total_mb": 391548.0,
|
| 269 |
+
"vram_used_avg_pct": 98.27,
|
| 270 |
+
"vram_used_max_pct": 98.27,
|
| 271 |
+
"pcie_rx_avg_mb_s": 45582.6,
|
| 272 |
+
"pcie_rx_max_mb_s": 51598.0,
|
| 273 |
+
"pcie_tx_avg_mb_s": 44600.2,
|
| 274 |
+
"pcie_tx_max_mb_s": 48299.0
|
| 275 |
+
}
|
| 276 |
+
},
|
| 277 |
+
"65536": {
|
| 278 |
+
"ttft_seconds": 8.891,
|
| 279 |
+
"prefill_seconds": 8.891,
|
| 280 |
+
"tok_per_sec": 7371.0,
|
| 281 |
+
"client_ttft_seconds": 8.891,
|
| 282 |
+
"client_tok_per_sec": 7371.0,
|
| 283 |
+
"prompt_tokens": 65538,
|
| 284 |
+
"samples": 3,
|
| 285 |
+
"method": "client",
|
| 286 |
+
"server_validation": {
|
| 287 |
+
"method": "",
|
| 288 |
+
"tok_per_sec": 0.0,
|
| 289 |
+
"prefill_seconds": 0.0,
|
| 290 |
+
"prompt_tokens": 0,
|
| 291 |
+
"request_prompt_tokens": 0,
|
| 292 |
+
"cached_tokens": 0,
|
| 293 |
+
"token_source": "",
|
| 294 |
+
"samples": 0,
|
| 295 |
+
"invalid_reason": ""
|
| 296 |
+
},
|
| 297 |
+
"hardware_summary": {
|
| 298 |
+
"samples": 12,
|
| 299 |
+
"duration_seconds": 26.544,
|
| 300 |
+
"gpu_count": 4,
|
| 301 |
+
"cpu_util_avg_pct": 11.23,
|
| 302 |
+
"cpu_temp_max_c": 76.75,
|
| 303 |
+
"gpu_util_avg_pct": 97.67,
|
| 304 |
+
"gpu_util_max_pct": 100.0,
|
| 305 |
+
"mem_util_avg_pct": 21.02,
|
| 306 |
+
"mem_util_max_pct": 29.0,
|
| 307 |
+
"temp_avg_c": 68.04,
|
| 308 |
+
"temp_max_c": 83.0,
|
| 309 |
+
"power_total_avg_w": 1074.72,
|
| 310 |
+
"power_total_max_w": 1145.97,
|
| 311 |
+
"power_limit_total_w": 1200.0,
|
| 312 |
+
"vram_used_avg_mb": 384778.0,
|
| 313 |
+
"vram_used_max_mb": 384778.0,
|
| 314 |
+
"vram_total_mb": 391548.0,
|
| 315 |
+
"vram_used_avg_pct": 98.27,
|
| 316 |
+
"vram_used_max_pct": 98.27,
|
| 317 |
+
"pcie_rx_avg_mb_s": 42799.33,
|
| 318 |
+
"pcie_rx_max_mb_s": 54328.0,
|
| 319 |
+
"pcie_tx_avg_mb_s": 43858.08,
|
| 320 |
+
"pcie_tx_max_mb_s": 53032.0
|
| 321 |
+
}
|
| 322 |
+
},
|
| 323 |
+
"131072": {
|
| 324 |
+
"ttft_seconds": 17.957,
|
| 325 |
+
"prefill_seconds": 17.957,
|
| 326 |
+
"tok_per_sec": 7299.0,
|
| 327 |
+
"client_ttft_seconds": 17.957,
|
| 328 |
+
"client_tok_per_sec": 7299.0,
|
| 329 |
+
"prompt_tokens": 131073,
|
| 330 |
+
"samples": 2,
|
| 331 |
+
"method": "client",
|
| 332 |
+
"server_validation": {
|
| 333 |
+
"method": "",
|
| 334 |
+
"tok_per_sec": 0.0,
|
| 335 |
+
"prefill_seconds": 0.0,
|
| 336 |
+
"prompt_tokens": 0,
|
| 337 |
+
"request_prompt_tokens": 0,
|
| 338 |
+
"cached_tokens": 0,
|
| 339 |
+
"token_source": "",
|
| 340 |
+
"samples": 0,
|
| 341 |
+
"invalid_reason": ""
|
| 342 |
+
},
|
| 343 |
+
"hardware_summary": {
|
| 344 |
+
"samples": 15,
|
| 345 |
+
"duration_seconds": 33.726,
|
| 346 |
+
"gpu_count": 4,
|
| 347 |
+
"cpu_util_avg_pct": 11.25,
|
| 348 |
+
"cpu_temp_max_c": 77.25,
|
| 349 |
+
"gpu_util_avg_pct": 99.87,
|
| 350 |
+
"gpu_util_max_pct": 100.0,
|
| 351 |
+
"mem_util_avg_pct": 21.47,
|
| 352 |
+
"mem_util_max_pct": 29.0,
|
| 353 |
+
"temp_avg_c": 68.65,
|
| 354 |
+
"temp_max_c": 84.0,
|
| 355 |
+
"power_total_avg_w": 1123.91,
|
| 356 |
+
"power_total_max_w": 1145.94,
|
| 357 |
+
"power_limit_total_w": 1200.0,
|
| 358 |
+
"vram_used_avg_mb": 384778.0,
|
| 359 |
+
"vram_used_max_mb": 384778.0,
|
| 360 |
+
"vram_total_mb": 391548.0,
|
| 361 |
+
"vram_used_avg_pct": 98.27,
|
| 362 |
+
"vram_used_max_pct": 98.27,
|
| 363 |
+
"pcie_rx_avg_mb_s": 43837.13,
|
| 364 |
+
"pcie_rx_max_mb_s": 55605.0,
|
| 365 |
+
"pcie_tx_avg_mb_s": 44351.67,
|
| 366 |
+
"pcie_tx_max_mb_s": 52140.0
|
| 367 |
+
}
|
| 368 |
+
}
|
| 369 |
+
},
|
| 370 |
+
"results": [],
|
| 371 |
+
"summary_table": {},
|
| 372 |
+
"burst_results": [],
|
| 373 |
+
"burst_summary_table": {},
|
| 374 |
+
"methodology": {
|
| 375 |
+
"prefill": {
|
| 376 |
+
"name": "Prefill",
|
| 377 |
+
"present": true,
|
| 378 |
+
"mode": "standalone_cold",
|
| 379 |
+
"formula": "prompt_tokens / TTFT",
|
| 380 |
+
"notes": "Default mode records the required decode scout request for each non-zero decode context, so normal runs do not pay for a separate prefill phase. Standalone mode repeats cold-prefill samples. Prometheus prefill counters, when available and uncontaminated, are stored as validation."
|
| 381 |
+
},
|
| 382 |
+
"sustained_decode": {
|
| 383 |
+
"name": "Sustained Decode",
|
| 384 |
+
"present": false,
|
| 385 |
+
"formula": "OpenAI stream usage completion_tokens per measured window; client chunk fallback only when continuous usage is unavailable",
|
| 386 |
+
"notes": "Duration-based steady-state cell after warmup. This is the main tuning/regression signal for kernels, NCCL, DCP, MTP, and scheduling. Prometheus metrics are stored as validation and scheduler state, not the default headline."
|
| 387 |
+
},
|
| 388 |
+
"burst_e2e_decode": {
|
| 389 |
+
"name": "Burst / E2E Decode",
|
| 390 |
+
"present": false,
|
| 391 |
+
"status": "not run; use --run-burst",
|
| 392 |
+
"formula": "sum(completion_tokens) / profiling_wall_time",
|
| 393 |
+
"notes": "Finite client-facing request burst using OpenAI stream usage. It includes request admission, scheduling, prefill/cache behavior, and completion."
|
| 394 |
+
}
|
| 395 |
+
}
|
| 396 |
+
}
|