ProCreations commited on
Commit
1961af5
·
verified ·
1 Parent(s): d205471

Release calibrated Image2.1 NVFP4 transformer with dynamic scaling and BF16 rank correction, native SM120 runtime, quality evidence and real-time demo

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +49 -0
  2. LICENSE +55 -0
  3. Notice +4 -0
  4. README.md +86 -0
  5. acceleration.py +91 -0
  6. comparisons/evaluation-dynamic/00.jpg +3 -0
  7. comparisons/evaluation-dynamic/01.jpg +3 -0
  8. comparisons/evaluation-dynamic/02.jpg +3 -0
  9. comparisons/evaluation-dynamic/03.jpg +3 -0
  10. comparisons/evaluation-dynamic/04.jpg +3 -0
  11. comparisons/evaluation-dynamic/05.jpg +3 -0
  12. comparisons/evaluation-dynamic/06.jpg +3 -0
  13. comparisons/evaluation-dynamic/07.jpg +3 -0
  14. comparisons/evaluation-dynamic/08.jpg +3 -0
  15. comparisons/evaluation-dynamic/09.jpg +3 -0
  16. comparisons/evaluation-dynamic/10.jpg +0 -0
  17. comparisons/evaluation-dynamic/11.jpg +3 -0
  18. comparisons/evaluation-dynamic/12.jpg +3 -0
  19. comparisons/evaluation-dynamic/13.jpg +0 -0
  20. comparisons/evaluation-dynamic/14.jpg +3 -0
  21. comparisons/evaluation-dynamic/15.jpg +3 -0
  22. comparisons/evaluation-dynamic/edit-0.jpg +3 -0
  23. comparisons/evaluation-dynamic/edit-1.jpg +3 -0
  24. comparisons/heldout-nvfp4/00.jpg +3 -0
  25. comparisons/heldout-nvfp4/01.jpg +0 -0
  26. comparisons/heldout-nvfp4/02.jpg +3 -0
  27. comparisons/heldout-nvfp4/03.jpg +3 -0
  28. comparisons/heldout-nvfp4/04.jpg +3 -0
  29. comparisons/heldout-nvfp4/05.jpg +3 -0
  30. comparisons/heldout-nvfp4/06.jpg +0 -0
  31. comparisons/heldout-nvfp4/07.jpg +3 -0
  32. demo/capture_receipt.json +0 -0
  33. demo/realtime-30s.mp4 +3 -0
  34. dynamic_scale.py +29 -0
  35. fp8_runtime.py +104 -0
  36. generate.py +54 -0
  37. install.sh +12 -0
  38. manifest.json +510 -0
  39. nvfp4_runtime.py +81 -0
  40. reports/benchmark.json +36 -0
  41. reports/calibration_manifest.json +881 -0
  42. reports/calibration_search.json +0 -0
  43. reports/correction-probe.json +122 -0
  44. reports/cudagraph-trial.json +10 -0
  45. reports/earlier-bf16-benchmark.json +41 -0
  46. reports/environment.json +95 -0
  47. reports/evaluation-environment.json +9 -0
  48. reports/fresh-fp8-benchmark.json +31 -0
  49. reports/heldout-manifest.json +69 -0
  50. reports/kernel_evidence.json +81 -0
.gitattributes CHANGED
@@ -33,3 +33,52 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ comparisons/evaluation-dynamic/00.jpg filter=lfs diff=lfs merge=lfs -text
37
+ comparisons/evaluation-dynamic/01.jpg filter=lfs diff=lfs merge=lfs -text
38
+ comparisons/evaluation-dynamic/02.jpg filter=lfs diff=lfs merge=lfs -text
39
+ comparisons/evaluation-dynamic/03.jpg filter=lfs diff=lfs merge=lfs -text
40
+ comparisons/evaluation-dynamic/04.jpg filter=lfs diff=lfs merge=lfs -text
41
+ comparisons/evaluation-dynamic/05.jpg filter=lfs diff=lfs merge=lfs -text
42
+ comparisons/evaluation-dynamic/06.jpg filter=lfs diff=lfs merge=lfs -text
43
+ comparisons/evaluation-dynamic/07.jpg filter=lfs diff=lfs merge=lfs -text
44
+ comparisons/evaluation-dynamic/08.jpg filter=lfs diff=lfs merge=lfs -text
45
+ comparisons/evaluation-dynamic/09.jpg filter=lfs diff=lfs merge=lfs -text
46
+ comparisons/evaluation-dynamic/11.jpg filter=lfs diff=lfs merge=lfs -text
47
+ comparisons/evaluation-dynamic/12.jpg filter=lfs diff=lfs merge=lfs -text
48
+ comparisons/evaluation-dynamic/14.jpg filter=lfs diff=lfs merge=lfs -text
49
+ comparisons/evaluation-dynamic/15.jpg filter=lfs diff=lfs merge=lfs -text
50
+ comparisons/evaluation-dynamic/edit-0.jpg filter=lfs diff=lfs merge=lfs -text
51
+ comparisons/evaluation-dynamic/edit-1.jpg filter=lfs diff=lfs merge=lfs -text
52
+ comparisons/heldout-nvfp4/00.jpg filter=lfs diff=lfs merge=lfs -text
53
+ comparisons/heldout-nvfp4/02.jpg filter=lfs diff=lfs merge=lfs -text
54
+ comparisons/heldout-nvfp4/03.jpg filter=lfs diff=lfs merge=lfs -text
55
+ comparisons/heldout-nvfp4/04.jpg filter=lfs diff=lfs merge=lfs -text
56
+ comparisons/heldout-nvfp4/05.jpg filter=lfs diff=lfs merge=lfs -text
57
+ comparisons/heldout-nvfp4/07.jpg filter=lfs diff=lfs merge=lfs -text
58
+ demo/realtime-30s.mp4 filter=lfs diff=lfs merge=lfs -text
59
+ samples/heldout/00.png filter=lfs diff=lfs merge=lfs -text
60
+ samples/heldout/01.png filter=lfs diff=lfs merge=lfs -text
61
+ samples/heldout/02.png filter=lfs diff=lfs merge=lfs -text
62
+ samples/heldout/03.png filter=lfs diff=lfs merge=lfs -text
63
+ samples/heldout/04.png filter=lfs diff=lfs merge=lfs -text
64
+ samples/heldout/05.png filter=lfs diff=lfs merge=lfs -text
65
+ samples/heldout/06.png filter=lfs diff=lfs merge=lfs -text
66
+ samples/heldout/07.png filter=lfs diff=lfs merge=lfs -text
67
+ samples/validation/00.png filter=lfs diff=lfs merge=lfs -text
68
+ samples/validation/01.png filter=lfs diff=lfs merge=lfs -text
69
+ samples/validation/02.png filter=lfs diff=lfs merge=lfs -text
70
+ samples/validation/03.png filter=lfs diff=lfs merge=lfs -text
71
+ samples/validation/04.png filter=lfs diff=lfs merge=lfs -text
72
+ samples/validation/05.png filter=lfs diff=lfs merge=lfs -text
73
+ samples/validation/06.png filter=lfs diff=lfs merge=lfs -text
74
+ samples/validation/07.png filter=lfs diff=lfs merge=lfs -text
75
+ samples/validation/08.png filter=lfs diff=lfs merge=lfs -text
76
+ samples/validation/09.png filter=lfs diff=lfs merge=lfs -text
77
+ samples/validation/10.png filter=lfs diff=lfs merge=lfs -text
78
+ samples/validation/11.png filter=lfs diff=lfs merge=lfs -text
79
+ samples/validation/12.png filter=lfs diff=lfs merge=lfs -text
80
+ samples/validation/13.png filter=lfs diff=lfs merge=lfs -text
81
+ samples/validation/14.png filter=lfs diff=lfs merge=lfs -text
82
+ samples/validation/15.png filter=lfs diff=lfs merge=lfs -text
83
+ samples/validation/edit-0.png filter=lfs diff=lfs merge=lfs -text
84
+ samples/validation/edit-1.png filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
21
+
22
+ 3. Redistribution
23
+ Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+ c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
33
+
34
+ 5. Intellectual Property
35
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
36
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
37
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
38
+
39
+ 6. Disclaimer of Warranty and Limitation of Liability
40
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
41
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
42
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
43
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
44
+
45
+ 7. Survival and Termination.
46
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
47
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
48
+
49
+ 8. Governing Law and Jurisdiction.
50
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
51
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
52
+
53
+ 9. Other Terms and Conditions.
54
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
55
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
Notice ADDED
@@ -0,0 +1,4 @@
 
 
 
 
 
1
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
2
+
3
+ Built with Qwen.
4
+ ProCreations modified the original BF16 transformer through calibrated NVFP4 quantization with BF16 low-rank corrections and dynamic activation scaling on 2026-09-20. The custom runtime and calibration/evaluation scripts were added by ProCreations. Non-commercial research and evaluation under the included upstream license.
README.md ADDED
@@ -0,0 +1,86 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: qwen-research
4
+ license_link: LICENSE
5
+ base_model: Qwen/Qwen-Image-2.1
6
+ library_name: diffusers
7
+ pipeline_tag: text-to-image
8
+ tags:
9
+ - nvfp4
10
+ - svdquant
11
+ - blackwell
12
+ - sm120
13
+ - image-editing
14
+ ---
15
+ # Image 2.1 Calibrated NVFP4
16
+
17
+ Built with Qwen. A calibrated NVFP4 transformer derived from `Qwen/Qwen-Image-2.1` at revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`.
18
+
19
+ **4.87 GB transformer; native SM120 W4A4 Tensor Core kernels; all 40 denoising steps.** Encoder, VAE, attention, conditioning and small projections retain BF16. Each of 224 large projections uses NVFP4 plus a BF16 rank-128 correction, accumulated together in FP32. This is a custom Diffusers/FlashInfer format and requires the included loader.
20
+
21
+ ## Measured speed
22
+
23
+ RTX PRO 6000 Blackwell Workstation 96 GB, batch 1, CFG 1, prefix KV cache. Mean CUDA-synchronized wall time includes text encoding, all denoising and VAE decoding. Loading, one full warmup per resolution, compilation and PNG writing are excluded. NVFP4 uses five measured 1024 runs and three 2048 runs; fresh FP8 controls use two per resolution in the same clean environment.
24
+
25
+ | Resolution | This NVFP4 | Fresh optimized FP8 control | Speedup |
26
+ |---|---:|---:|---:|
27
+ | 1024 × 1024 | 4.589 s/image | 5.910 s/image | 1.29× |
28
+ | 2048 × 2048 | 32.658 s/image | 37.214 s/image | 1.14× |
29
+
30
+ Earlier BF16 measurements on this workstation with the same generation settings were 9.792 and 55.169 seconds respectively. These are historical controls from the same project, not newly timed in this release. First-use loading and compilation cost extra; full warmup times are in [benchmark.json](reports/benchmark.json). Performance depends on resolution and prompt. No steps are skipped, no approximate feature reuse is used, and attention is not quantized.
31
+
32
+ The [30-second real-time video](demo/realtime-30s.mp4) has no text, overlays or audio. It begins with one completed warmup image, then displays actual newly completed generations with all waits preserved. The [capture log](demo/capture_receipt.json) records every display update and frame timestamp.
33
+
34
+ ## Quality and calibration
35
+
36
+ Calibration uses 64 original BF16 trajectories: 56 text-to-image and eight edits, with 1024, 2048 and alternate aspect ratios. Each projection has 1,536 sampled activation rows across six denoising timesteps; 384 stratified fit rows and 384 disjoint diagnostic rows. Five smoothing exponents and two weight-scale methods are searched using native FP4 kernels. Weight-scale refinement searches 15 per-block scale factors, weighted by activation second moments. Rank-128 SVD corrections absorb dominant weight components. Original BF16 weights are the source; this is not a requantization of FP8.
37
+
38
+ Activations use a **fresh actual tensor-wide amax on every call**, followed by dynamic E4M3 scales per 16 values. Sparse calibration alone missed rare large activations; static activation ranges caused clipping and visible texture degradation, so that candidate was rejected. The final runtime recalibrates and evaluates with dynamic scaling. BF16 correction weights are rescaled consistently when the activation global scale changes.
39
+
40
+ The 18-case development validation set includes 16 generation prompts and two edits. Against BF16, mean LPIPS(AlexNet,512px) is **0.122970**, SSIM **0.897363**, and full-latent cosine **0.976903**. Eight additional prompts were frozen after candidate selection; their mean LPIPS is **0.124012** and SSIM **0.871713**. Full images, paired comparisons and individual measurements are included.
41
+
42
+ **This is lossy quantization, not a zero-quality-loss guarantee.** All 26 pairs were visually reviewed. The corrected candidate preserves readable primary English/Chinese text, transparency, image editing, fine felt/fur textures and plausible hands in these tests. Same-seed composition, poses, decorative marks, geometry and fine detail can change; the largest validation differences are the pottery and train scenes. The prior calibrated FP8 release is closer to BF16 numerically (original 18-case mean LPIPS 0.03447). Fidelity metrics do not establish broad human preference or guarantee every prompt.
43
+
44
+ ## Native FP4 evidence
45
+
46
+ An actual denoising-step profile records **224 SM120 block-scaled `f4E2M1FN` GEMM launches**, 32 native BF16 FlashAttention launches and 40 transformer calls per 40-step generation. Packed E2M1 values and E4M3 scale layouts were independently decoded and checked against FP32 matrix products. See [kernel evidence](reports/kernel_evidence.json) and [numerical verification](reports/native-math-verification.json).
47
+
48
+ The engine uses [FlashInfer NVFP4 SVDQuant](https://docs.flashinfer.ai/generated/flashinfer.gemm.mm_nvfp4_svdquant.html), pinned to commit `975f90583d9ac8896db14cf0f26e99a853c2f136`. PyPI 0.6.18 lacked the SM120 fusion shown in the online documentation during development, so use the pinned source revision. Generic Transformers FP8/FP4 loading does not select this custom runtime.
49
+
50
+ ## Install and generate
51
+
52
+ Tested on Linux x86_64, Python 3.12, Torch 2.14.0+cu130, CUDA toolkit 13.3, driver 615.71.09, compute capability 12.0. Other hardware is unverified. Download this repository, then run `bash install.sh` from its directory. The script installs pinned dependencies and the pinned FlashInfer source; a recent NVIDIA driver and CUDA toolkit must already be available.
53
+
54
+ ```bash
55
+ .venv/bin/python generate.py --prompt 'A kingfisher on a mossy branch, detailed feathers, no text' --width 1024 --height 1024 --steps 40 --warmup --output kingfisher.png
56
+ ```
57
+
58
+ `--base /path/to/local/original-model` reuses a local upstream checkpoint. Otherwise the loader downloads the pinned upstream components. `--image input.png` enables editing. `--prompts-json prompts.json` accepts a JSON list and reuses the loaded pipeline. `--eager` disables block compilation. Warmup and loading costs are reported separately; without warmup the reported generation time includes any compilation on that call.
59
+
60
+ ```python
61
+ from nvfp4_runtime import load_pipeline
62
+ from acceleration import accelerate_pipeline
63
+ pipe = accelerate_pipeline(load_pipeline('Qwen/Qwen-Image-2.1', './transformer'))
64
+ image = pipe(prompt='A detailed watercolor garden', width=1024, height=1024, num_inference_steps=40).images[0]
65
+ image.save('garden.png')
66
+ ```
67
+
68
+ Do not cast the loaded quantized transformer with `.to(dtype=...)`; packed values and FP32 scales must retain their stored types. The normal loader handles device placement. The checkpoint is not a generic Transformers, ComfyUI, TensorRT or Nunchaku checkpoint.
69
+
70
+ ## Reproduce calibration
71
+
72
+ Download the exact BF16 upstream revision to a local directory first. The generation runtime environment also supports calibration. From this release directory:
73
+
74
+ ```bash
75
+ export IMAGE21_BASE=/absolute/path/to/original-model
76
+ export IMAGE21_WORK_DIR="$PWD/rebuild"
77
+ export PYTHONPATH="$PWD"
78
+ .venv/bin/python source/collect_calibration.py
79
+ .venv/bin/python source/quantize_dynamic.py
80
+ ```
81
+
82
+ The first command collects the published 64-case calibration manifest; the second reconstructs the NVFP4 transformer under `rebuild/release/transformer`. Original fitting activations are retained on the build workstation and can be regenerated with these scripts. Evaluation additionally needs LPIPS, scikit-image and SciPy; video capture uses imageio-ffmpeg. [Environment versions](reports/environment.json), source, calibration search, manifests and timing records are included. Earlier static-range, mixed-precision, ridge-correction, kernel-autotuning and CUDA-graph experiments were rejected or provided no useful gain; they are not enabled in this runtime.
83
+
84
+ ## License and modification notice
85
+
86
+ Built with Qwen. ProCreations modified the original transformer through calibrated NVFP4 quantization and supplied this custom inference runtime on 2026-09-20. This derivative is distributed under the included [Qwen Research License](LICENSE) and [Notice](Notice), for non-commercial research/evaluation. Preserve those files and the upstream terms when redistributing. This is an independent derivative, not an official Qwen release.
acceleration.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Full-compute acceleration for Image 2.1 Calibrated NVFP4. Built with Qwen.
2
+
3
+ Keeps every denoising step, BF16 attention, calibrated native FP4 GEMMs with BF16 rank correction and
4
+ FP32 accumulation/scales. No approximate residual cache or attention quantization.
5
+ Only cached-prefix decode blocks compile; prefill keeps upstream behavior.
6
+ """
7
+ import types
8
+ import torch
9
+ from diffusers.models.transformers.transformer_qwenimage21 import (
10
+ QwenImage21AttnProcessor, QwenImage21TransformerBlock,
11
+ )
12
+
13
+ _ORIGINAL_BLOCK_FORWARD = QwenImage21TransformerBlock.forward
14
+
15
+
16
+ def _real_rope(x, frequencies):
17
+ paired = x.float().unflatten(-1, (-1, 2))
18
+ cosine = frequencies.real[None, :, None, :]
19
+ sine = frequencies.imag[None, :, None, :]
20
+ return torch.stack((
21
+ paired[..., 0] * cosine - paired[..., 1] * sine,
22
+ paired[..., 0] * sine + paired[..., 1] * cosine,
23
+ ), dim=-1).flatten(-2).to(x.dtype)
24
+
25
+
26
+ class NativeAttentionProcessor(QwenImage21AttnProcessor):
27
+ def __call__(self, attn, hidden_states, attention_mask=None, rotary_emb=None,
28
+ layer_cache=None, kv_cache_mode=None, cache_write_slice=None,
29
+ segments=None, key_valid=None):
30
+ if kv_cache_mode != 'cached' or attention_mask is not None:
31
+ return super().__call__(attn, hidden_states, attention_mask, rotary_emb,
32
+ layer_cache, kv_cache_mode, cache_write_slice,
33
+ segments, key_valid)
34
+ query = attn.to_q(hidden_states).unflatten(-1, (attn.heads, -1))
35
+ key = attn.to_k(hidden_states).unflatten(-1, (attn.heads, -1))
36
+ value = attn.to_v(hidden_states).unflatten(-1, (attn.heads, -1))
37
+ query = attn.norm_q(query)
38
+ key = attn.norm_k(key)
39
+ if rotary_emb is not None:
40
+ query = _real_rope(query, rotary_emb)
41
+ key = _real_rope(key, rotary_emb)
42
+ cached_key, cached_value = layer_cache.get()
43
+ key = torch.cat((cached_key, key), dim=1)
44
+ value = torch.cat((cached_value, value), dim=1)
45
+ output = torch.nn.functional.scaled_dot_product_attention(
46
+ query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2)
47
+ ).transpose(1, 2)
48
+ return attn.to_out[1](attn.to_out[0](output.flatten(2, 3)))
49
+
50
+
51
+ def _decode_block(self, hidden_states, modulation, rotary_emb=None,
52
+ attention_mask=None, target_token_mask=None, layer_cache=None,
53
+ kv_cache_mode=None, cache_write_slice=None, segments=None,
54
+ key_valid=None):
55
+ # Cached decode contains only target image tokens. The final t=0 modulation
56
+ # row belongs to the prefix, which was already evaluated during prefill.
57
+ if kv_cache_mode == 'cached':
58
+ modulation = modulation[:-1]
59
+ target_token_mask = None
60
+ return _ORIGINAL_BLOCK_FORWARD(
61
+ self, hidden_states, modulation, rotary_emb, attention_mask,
62
+ target_token_mask, layer_cache, kv_cache_mode, cache_write_slice,
63
+ segments, key_valid,
64
+ )
65
+
66
+
67
+ def _dispatch_block(self, **kwargs):
68
+ if kwargs.get('kv_cache_mode') == 'cached':
69
+ return self._image21_compiled(**kwargs)
70
+ return _ORIGINAL_BLOCK_FORWARD(self, **kwargs)
71
+
72
+
73
+ def accelerate_pipeline(pipe):
74
+ """Enable once after loading. Initial compilation is excluded from warm timings.
75
+
76
+ Dynamic sequence lengths reduce recompilation across prompts/resolutions.
77
+ emulate_precision_casts preserves the upstream intermediate BF16 rounding
78
+ boundaries in fused code; GPU reduction ordering can still differ.
79
+ """
80
+ if getattr(pipe, '_image21_accelerated', False):
81
+ return pipe
82
+ torch._dynamo.config.recompile_limit = max(torch._dynamo.config.recompile_limit, 64)
83
+ for block in pipe.transformer.transformer_blocks:
84
+ block.attn.set_processor(NativeAttentionProcessor())
85
+ block._image21_compiled = torch.compile(
86
+ types.MethodType(_decode_block, block), fullgraph=True, dynamic=True,
87
+ options={'emulate_precision_casts': True},
88
+ )
89
+ block.forward = types.MethodType(_dispatch_block, block)
90
+ pipe._image21_accelerated = True
91
+ return pipe
comparisons/evaluation-dynamic/00.jpg ADDED

Git LFS Details

  • SHA256: 801c01744c1a1c5d33b5e06b9503e92cf9df1bff52e530e65bb062713f442162
  • Pointer size: 131 Bytes
  • Size of remote file: 176 kB
comparisons/evaluation-dynamic/01.jpg ADDED

Git LFS Details

  • SHA256: b0423321243e0f5b210ce6e32c4565f993294aa8fd89583699a3caebeded99a1
  • Pointer size: 131 Bytes
  • Size of remote file: 180 kB
comparisons/evaluation-dynamic/02.jpg ADDED

Git LFS Details

  • SHA256: e519b8431f20a90c6c619b91e7966dd7c638973f2b33b012ea3aabb991a738d8
  • Pointer size: 131 Bytes
  • Size of remote file: 180 kB
comparisons/evaluation-dynamic/03.jpg ADDED

Git LFS Details

  • SHA256: 3a0138e1eb0eb79abe01ff768be4eccba5c2114488f840b8575c16b02af7b169
  • Pointer size: 131 Bytes
  • Size of remote file: 146 kB
comparisons/evaluation-dynamic/04.jpg ADDED

Git LFS Details

  • SHA256: 51e0a532f9f4de9d65d14fd4edc543cf87b5bc35ea36e4697e42f2e3997afa52
  • Pointer size: 131 Bytes
  • Size of remote file: 183 kB
comparisons/evaluation-dynamic/05.jpg ADDED

Git LFS Details

  • SHA256: 81fd3d157a7175fee7b108206fc5bdb44f6542cf06f42e7ee19e2ca1d79d3636
  • Pointer size: 131 Bytes
  • Size of remote file: 310 kB
comparisons/evaluation-dynamic/06.jpg ADDED

Git LFS Details

  • SHA256: 626595f20c55ef0cbad9ac793180dc8f7bf84da82ec29c60d8b13244a9c64f40
  • Pointer size: 131 Bytes
  • Size of remote file: 105 kB
comparisons/evaluation-dynamic/07.jpg ADDED

Git LFS Details

  • SHA256: 96259410bc3679ffa69cfef803b38af74974068467de97160defe7893b29930e
  • Pointer size: 131 Bytes
  • Size of remote file: 124 kB
comparisons/evaluation-dynamic/08.jpg ADDED

Git LFS Details

  • SHA256: dad753aef0935797375233f3c6d0b64c4360dfc45c4227c3d3cd72c58e64b8f3
  • Pointer size: 131 Bytes
  • Size of remote file: 113 kB
comparisons/evaluation-dynamic/09.jpg ADDED

Git LFS Details

  • SHA256: 68a84e802c40234eb830362be6e9405c7c4f765a0ac12061864d2e8c6dd3b746
  • Pointer size: 131 Bytes
  • Size of remote file: 102 kB
comparisons/evaluation-dynamic/10.jpg ADDED
comparisons/evaluation-dynamic/11.jpg ADDED

Git LFS Details

  • SHA256: a57d36f2e4bd995fa540234dce491f209fa4524700b1ebdb0cbb8966bbbc5778
  • Pointer size: 131 Bytes
  • Size of remote file: 260 kB
comparisons/evaluation-dynamic/12.jpg ADDED

Git LFS Details

  • SHA256: 9b35d184b7efa4ff9a3bc7abef9e4cf50d6ed99bf6967724312038e52f8e2672
  • Pointer size: 131 Bytes
  • Size of remote file: 277 kB
comparisons/evaluation-dynamic/13.jpg ADDED
comparisons/evaluation-dynamic/14.jpg ADDED

Git LFS Details

  • SHA256: ccdb2b7222304c5049c324750433dfa0f4862dfc3833864e90030a1a0e03d1a7
  • Pointer size: 131 Bytes
  • Size of remote file: 190 kB
comparisons/evaluation-dynamic/15.jpg ADDED

Git LFS Details

  • SHA256: e3586ffca8d68cebde081ef42c4e7fab7fd5c8cbe1b2531a54ee8474a07bc7a0
  • Pointer size: 131 Bytes
  • Size of remote file: 204 kB
comparisons/evaluation-dynamic/edit-0.jpg ADDED

Git LFS Details

  • SHA256: 5747ce5f80fd014383a89ad06d0b31209612bbbe2981e5082b086819c8db720f
  • Pointer size: 131 Bytes
  • Size of remote file: 240 kB
comparisons/evaluation-dynamic/edit-1.jpg ADDED

Git LFS Details

  • SHA256: e93f685092a7f8729a135df81d8fc91d2f7f446e6857d03d36055e4926c820b2
  • Pointer size: 131 Bytes
  • Size of remote file: 140 kB
comparisons/heldout-nvfp4/00.jpg ADDED

Git LFS Details

  • SHA256: fef99d1fed7bec5ca673bf94aae5f0f534db77d474e45f04a8c6b52798b20801
  • Pointer size: 131 Bytes
  • Size of remote file: 144 kB
comparisons/heldout-nvfp4/01.jpg ADDED
comparisons/heldout-nvfp4/02.jpg ADDED

Git LFS Details

  • SHA256: 16c8ab16fe20d594d46ad396ed4c415690f2c3082ac4287b080806a435d82fb3
  • Pointer size: 131 Bytes
  • Size of remote file: 137 kB
comparisons/heldout-nvfp4/03.jpg ADDED

Git LFS Details

  • SHA256: 907a7a68a36afc83d506442e84a538e0bab1af77582fb76b067809685df15e20
  • Pointer size: 131 Bytes
  • Size of remote file: 105 kB
comparisons/heldout-nvfp4/04.jpg ADDED

Git LFS Details

  • SHA256: 7abc30394201e4f7e1456f06df38ab6b492ffbf3a869ab5d9decc70231312667
  • Pointer size: 131 Bytes
  • Size of remote file: 142 kB
comparisons/heldout-nvfp4/05.jpg ADDED

Git LFS Details

  • SHA256: db68b9fa5e5ce3453fbe74d569b051274f300106d6c3aecd60996418f52ed86b
  • Pointer size: 131 Bytes
  • Size of remote file: 204 kB
comparisons/heldout-nvfp4/06.jpg ADDED
comparisons/heldout-nvfp4/07.jpg ADDED

Git LFS Details

  • SHA256: 9f625acf906a08a69543723a8af7254ecd4e0837101c152fb2140d894ea41371
  • Pointer size: 131 Bytes
  • Size of remote file: 135 kB
demo/capture_receipt.json ADDED
The diff for this file is too large to render. See raw diff
 
demo/realtime-30s.mp4 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7585d0c1e7618d2fb710ebf08f70541d36d88ac49152bfcbd4f10d093ae9ca94
3
+ size 2301594
dynamic_scale.py ADDED
@@ -0,0 +1,29 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Dynamic tensor-wide NVFP4 range, with FP32 reductions and correction rescaling."""
2
+ import torch,triton
3
+ import triton.language as tl
4
+ @triton.jit
5
+ def _partials(X,P,T,COUNT:tl.constexpr,K:tl.constexpr,B:tl.constexpr):
6
+ i=tl.program_id(0)*B+tl.arange(0,B)
7
+ x=tl.load(X+i,i<COUNT,0).to(tl.float32)
8
+ p=tl.load(P+i%K).to(tl.float32)
9
+ tl.store(T+tl.program_id(0),tl.max(tl.abs(x*p),0))
10
+ @triton.jit
11
+ def _finish(T,OLDG,OLDA,G,A,R,N:tl.constexpr,B:tl.constexpr):
12
+ i=tl.arange(0,B);v=tl.load(T+i,i<N,0)
13
+ g=2688./tl.maximum(tl.max(v,0),1.e-12)
14
+ oldg=tl.load(OLDG);olda=tl.load(OLDA)
15
+ tl.store(G,g);tl.store(A,olda*oldg/g);tl.store(R,g/oldg)
16
+ @triton.jit
17
+ def _upscale(U,R,V,N:tl.constexpr,B:tl.constexpr):
18
+ i=tl.program_id(0)*B+tl.arange(0,B)
19
+ u=tl.load(U+i,i<N,0).to(tl.float32);r=tl.load(R)
20
+ tl.store(V+i,u*r,i<N)
21
+ def scale(x,pre,gx,alpha,up):
22
+ count=x.numel();blocks=triton.cdiv(count,16384)
23
+ partial=torch.empty(blocks,device=x.device,dtype=torch.float32)
24
+ g=torch.empty_like(gx);a=torch.empty_like(alpha);r=torch.empty_like(gx)
25
+ _partials[(blocks,)](x,pre,partial,count,x.shape[-1],16384,num_warps=8)
26
+ _finish[(1,)](partial,gx,alpha,g,a,r,blocks,triton.next_power_of_2(blocks),num_warps=8)
27
+ u=torch.empty_like(up)
28
+ _upscale[(triton.cdiv(up.numel(),1024),)](up,r,u,up.numel(),1024)
29
+ return g,a,u
fp8_runtime.py ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Calibrated W8A8 E4M3 inference using native CUTLASS SM120 scaled GEMMs.
2
+
3
+ Built with Qwen. Quantization modifications by ProCreations, 2026.
4
+ The upstream model is subject to the included Qwen Research License.
5
+ """
6
+ import json
7
+ from pathlib import Path
8
+ import torch
9
+ from torch import nn
10
+ import triton
11
+ import triton.language as tl
12
+ from safetensors.torch import load_file
13
+
14
+
15
+ @triton.jit
16
+ def _quantize_rows(X, SMOOTH, Q, SCALE, K: tl.constexpr, BLOCK: tl.constexpr):
17
+ row = tl.program_id(0)
18
+ cols = tl.arange(0, BLOCK)
19
+ x = tl.load(X + row * K + cols, cols < K, 0).to(tl.float32)
20
+ s = tl.load(SMOOTH + cols, cols < K, 1).to(tl.float32)
21
+ z = x / s
22
+ scale = tl.maximum(tl.max(tl.abs(z), 0) / 448.0, 1.e-12)
23
+ tl.store(Q + row * K + cols, z / scale, cols < K)
24
+ tl.store(SCALE + row, scale)
25
+
26
+
27
+ def quantize_activation(x, smooth):
28
+ x = x.reshape(-1, x.shape[-1]).contiguous()
29
+ q = torch.empty_like(x, dtype=torch.float8_e4m3fn)
30
+ scale = torch.empty((x.shape[0], 1), device=x.device, dtype=torch.float32)
31
+ _quantize_rows[(x.shape[0],)](x, smooth, q, scale, x.shape[1],
32
+ triton.next_power_of_2(x.shape[1]), num_warps=8)
33
+ return q, scale
34
+
35
+
36
+ class CalibratedFP8Linear(nn.Module):
37
+ def __init__(self, weight, scale, smooth):
38
+ super().__init__()
39
+ self.register_buffer("weight", weight)
40
+ self.register_buffer("weight_scale", scale.float())
41
+ self.register_buffer("smooth", smooth.float())
42
+ self.in_features = weight.shape[1]
43
+ self.out_features = weight.shape[0]
44
+ self.bias = None
45
+
46
+ def forward(self, x):
47
+ shape = x.shape[:-1]
48
+ q, scale = quantize_activation(x, self.smooth)
49
+ y = torch._scaled_mm(q, self.weight.t(), scale_a=scale,
50
+ scale_b=self.weight_scale.t(),
51
+ out_dtype=torch.bfloat16, use_fast_accum=False)
52
+ return y.reshape(*shape, self.out_features)
53
+
54
+
55
+ def eligible_modules(model):
56
+ return {n:m for n,m in model.named_modules()
57
+ if isinstance(m, nn.Linear) and n.startswith("transformer_blocks.")
58
+ and m.bias is None}
59
+
60
+
61
+ def replace_module(root, name, replacement):
62
+ parent_name, leaf = name.rsplit(".", 1)
63
+ setattr(root.get_submodule(parent_name), leaf, replacement)
64
+
65
+
66
+ def load_fp8_transformer(directory, device="cuda"):
67
+ from diffusers import QwenImage21Transformer2DModel
68
+ directory = Path(directory)
69
+ config = json.loads((directory / "config.json").read_text())
70
+ meta = json.loads((directory / "quantization_config.json").read_text())
71
+ # Construct without allocating an intermediate BF16 model.
72
+ with torch.device("meta"):
73
+ model = QwenImage21Transformer2DModel.from_config(config)
74
+ for n, spec in meta["quantized_modules"].items():
75
+ out_f, in_f = spec["shape"]
76
+ replace_module(model, n, CalibratedFP8Linear(
77
+ torch.empty((out_f,in_f),dtype=torch.float8_e4m3fn),
78
+ torch.empty((out_f,1),dtype=torch.float32),
79
+ torch.empty(in_f,dtype=torch.float32)))
80
+ state = {}
81
+ for path in sorted(directory.glob("model-*.safetensors")):
82
+ state.update(load_file(str(path), device="cpu"))
83
+ model.load_state_dict(state, strict=True, assign=True)
84
+ # The original rotary module has nonpersistent buffers / Python tensor lists.
85
+ # Reconstruct these by creating a lightweight ordinary position module.
86
+ from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21Rope
87
+ model.pos_embed = QwenImage21Rope(theta=10000, axes_dim=list(model.config.axes_dims_rope))
88
+ from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21TemporalTimesteps
89
+ model.time_text_embed.time_proj = QwenImage21TemporalTimesteps(timestep_dim=256)
90
+ return model.eval().to(device=device)
91
+
92
+
93
+ def load_pipeline(base, quant=None):
94
+ from diffusers import QwenImage21Pipeline
95
+ args = {}
96
+ if not Path(base).exists():
97
+ args["revision"] = "b3179ad355be050328e483a9dfdd9e60cd62adfa"
98
+ if quant:
99
+ args["transformer"] = load_fp8_transformer(quant, device="cuda")
100
+ pipe = QwenImage21Pipeline.from_pretrained(base, torch_dtype=torch.bfloat16, **args)
101
+ # Do not call pipe.to(dtype=...) after FP8 loading; scales are FP32.
102
+ pipe.to(device="cuda")
103
+ pipe.set_progress_bar_config(disable=True)
104
+ return pipe
generate.py ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Generate with Image 2.1 Calibrated NVFP4. Built with Qwen."""
2
+ import argparse, json, time
3
+ from pathlib import Path
4
+ import torch
5
+ from nvfp4_runtime import load_pipeline
6
+
7
+ @torch.inference_mode()
8
+ def main():
9
+ ap=argparse.ArgumentParser()
10
+ inputs=ap.add_mutually_exclusive_group(required=True)
11
+ inputs.add_argument('--prompt')
12
+ inputs.add_argument('--prompts-json', help='JSON list of prompt strings; one loaded pipeline serves the whole list')
13
+ ap.add_argument('--base',default='Qwen/Qwen-Image-2.1')
14
+ ap.add_argument('--quant',default=str(Path(__file__).parent/'transformer'))
15
+ ap.add_argument('--width',type=int,default=2048)
16
+ ap.add_argument('--height',type=int,default=2048)
17
+ ap.add_argument('--steps',type=int,default=40)
18
+ ap.add_argument('--seed',type=int,default=42)
19
+ ap.add_argument('--output',default='output.png')
20
+ ap.add_argument('--image')
21
+ ap.add_argument('--eager', action='store_true', help='Use the original uncompiled runtime')
22
+ ap.add_argument('--warmup', action='store_true', help='Run a full untimed warmup; report its cost separately')
23
+ args=ap.parse_args()
24
+ prompts=json.loads(Path(args.prompts_json).read_text()) if args.prompts_json else [args.prompt]
25
+ if not isinstance(prompts,list) or not prompts or not all(isinstance(p,str) and p for p in prompts):
26
+ ap.error('prompts-json must contain a nonempty list of prompt strings')
27
+ start=time.perf_counter();pipe=load_pipeline(args.base,args.quant)
28
+ if not args.eager:
29
+ from acceleration import accelerate_pipeline
30
+ accelerate_pipeline(pipe)
31
+ torch.cuda.synchronize();load_seconds=time.perf_counter()-start
32
+ kw={}
33
+ if args.image:
34
+ from PIL import Image
35
+ kw['image']=Image.open(args.image)
36
+ def run(prompt,seed):
37
+ return pipe(prompt=prompt,width=args.width,height=args.height,num_inference_steps=args.steps,
38
+ generator=torch.Generator('cuda').manual_seed(seed),**kw).images[0]
39
+ warmup_seconds=0
40
+ if args.warmup:
41
+ start=time.perf_counter();run(prompts[0],args.seed);torch.cuda.synchronize()
42
+ warmup_seconds=time.perf_counter()-start
43
+ path=Path(args.output);path.parent.mkdir(parents=True,exist_ok=True)
44
+ for i,prompt in enumerate(prompts):
45
+ torch.cuda.synchronize();start=time.perf_counter();result=run(prompt,args.seed+i)
46
+ torch.cuda.synchronize();seconds=time.perf_counter()-start
47
+ destination=path if len(prompts)==1 else path.with_name(f'{path.stem}-{i:03d}{path.suffix or ".png"}')
48
+ result.save(destination)
49
+ print(json.dumps({'seconds':seconds,'output':str(destination),'width':result.width,'height':result.height,
50
+ 'steps':args.steps,'eager':args.eager,'load_seconds':load_seconds,
51
+ 'warmup_seconds':warmup_seconds,
52
+ 'timing_note':'Generation includes any compilation on this call; excludes model load and PNG writing.'}),flush=True)
53
+
54
+ if __name__=='__main__':main()
install.sh ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env bash
2
+ # Run from the downloaded release directory on Linux x86_64 / SM120.
3
+ set -euo pipefail
4
+ python3.12 -m venv .venv
5
+ .venv/bin/python -m pip install --upgrade pip
6
+ .venv/bin/python -m pip install torch==2.14.0 torchvision==0.29.0 --index-url https://download.pytorch.org/whl/cu130
7
+ .venv/bin/python -m pip install -r requirements.txt
8
+ git clone https://github.com/flashinfer-ai/flashinfer.git flashinfer-src
9
+ git -C flashinfer-src checkout 975f90583d9ac8896db14cf0f26e99a853c2f136
10
+ git -C flashinfer-src submodule update --init --depth 1 3rdparty/cutlass 3rdparty/cccl 3rdparty/spdlog
11
+ BUILD_NVEP=0 .venv/bin/python -m pip install --no-deps ./flashinfer-src
12
+ .venv/bin/python -m pip check
manifest.json ADDED
@@ -0,0 +1,510 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "files": [
3
+ {
4
+ "path": "LICENSE",
5
+ "bytes": 7831,
6
+ "sha256": "8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d"
7
+ },
8
+ {
9
+ "path": "Notice",
10
+ "bytes": 491,
11
+ "sha256": "c517ea86da1c4b8f3f659944f5e7b5a66128fe024af5d64655b6a1c2ade6ae65"
12
+ },
13
+ {
14
+ "path": "README.md",
15
+ "bytes": 8107,
16
+ "sha256": "d849e2126e39ff5d1586de3a17995afc2358b47ade44dda7dd1bdb15f91cea8f"
17
+ },
18
+ {
19
+ "path": "acceleration.py",
20
+ "bytes": 4157,
21
+ "sha256": "a5a43f98da091fb81fed8f493a736f68820d4cd1b5dc2f30e8a3f8f65659b3c8"
22
+ },
23
+ {
24
+ "path": "comparisons/evaluation-dynamic/00.jpg",
25
+ "bytes": 176056,
26
+ "sha256": "801c01744c1a1c5d33b5e06b9503e92cf9df1bff52e530e65bb062713f442162"
27
+ },
28
+ {
29
+ "path": "comparisons/evaluation-dynamic/01.jpg",
30
+ "bytes": 179746,
31
+ "sha256": "b0423321243e0f5b210ce6e32c4565f993294aa8fd89583699a3caebeded99a1"
32
+ },
33
+ {
34
+ "path": "comparisons/evaluation-dynamic/02.jpg",
35
+ "bytes": 180435,
36
+ "sha256": "e519b8431f20a90c6c619b91e7966dd7c638973f2b33b012ea3aabb991a738d8"
37
+ },
38
+ {
39
+ "path": "comparisons/evaluation-dynamic/03.jpg",
40
+ "bytes": 146351,
41
+ "sha256": "3a0138e1eb0eb79abe01ff768be4eccba5c2114488f840b8575c16b02af7b169"
42
+ },
43
+ {
44
+ "path": "comparisons/evaluation-dynamic/04.jpg",
45
+ "bytes": 182735,
46
+ "sha256": "51e0a532f9f4de9d65d14fd4edc543cf87b5bc35ea36e4697e42f2e3997afa52"
47
+ },
48
+ {
49
+ "path": "comparisons/evaluation-dynamic/05.jpg",
50
+ "bytes": 309805,
51
+ "sha256": "81fd3d157a7175fee7b108206fc5bdb44f6542cf06f42e7ee19e2ca1d79d3636"
52
+ },
53
+ {
54
+ "path": "comparisons/evaluation-dynamic/06.jpg",
55
+ "bytes": 104835,
56
+ "sha256": "626595f20c55ef0cbad9ac793180dc8f7bf84da82ec29c60d8b13244a9c64f40"
57
+ },
58
+ {
59
+ "path": "comparisons/evaluation-dynamic/07.jpg",
60
+ "bytes": 124166,
61
+ "sha256": "96259410bc3679ffa69cfef803b38af74974068467de97160defe7893b29930e"
62
+ },
63
+ {
64
+ "path": "comparisons/evaluation-dynamic/08.jpg",
65
+ "bytes": 113262,
66
+ "sha256": "dad753aef0935797375233f3c6d0b64c4360dfc45c4227c3d3cd72c58e64b8f3"
67
+ },
68
+ {
69
+ "path": "comparisons/evaluation-dynamic/09.jpg",
70
+ "bytes": 101923,
71
+ "sha256": "68a84e802c40234eb830362be6e9405c7c4f765a0ac12061864d2e8c6dd3b746"
72
+ },
73
+ {
74
+ "path": "comparisons/evaluation-dynamic/10.jpg",
75
+ "bytes": 60148,
76
+ "sha256": "434f2e92839951456722954e64c66f35a287a75b4416d1aad33a6362c89c15f6"
77
+ },
78
+ {
79
+ "path": "comparisons/evaluation-dynamic/11.jpg",
80
+ "bytes": 260028,
81
+ "sha256": "a57d36f2e4bd995fa540234dce491f209fa4524700b1ebdb0cbb8966bbbc5778"
82
+ },
83
+ {
84
+ "path": "comparisons/evaluation-dynamic/12.jpg",
85
+ "bytes": 276643,
86
+ "sha256": "9b35d184b7efa4ff9a3bc7abef9e4cf50d6ed99bf6967724312038e52f8e2672"
87
+ },
88
+ {
89
+ "path": "comparisons/evaluation-dynamic/13.jpg",
90
+ "bytes": 86912,
91
+ "sha256": "84f4bbe544f92623675f448e01710fdf8e799dbc8ed2af4674c1c6c94b00eb22"
92
+ },
93
+ {
94
+ "path": "comparisons/evaluation-dynamic/14.jpg",
95
+ "bytes": 189569,
96
+ "sha256": "ccdb2b7222304c5049c324750433dfa0f4862dfc3833864e90030a1a0e03d1a7"
97
+ },
98
+ {
99
+ "path": "comparisons/evaluation-dynamic/15.jpg",
100
+ "bytes": 203642,
101
+ "sha256": "e3586ffca8d68cebde081ef42c4e7fab7fd5c8cbe1b2531a54ee8474a07bc7a0"
102
+ },
103
+ {
104
+ "path": "comparisons/evaluation-dynamic/edit-0.jpg",
105
+ "bytes": 239945,
106
+ "sha256": "5747ce5f80fd014383a89ad06d0b31209612bbbe2981e5082b086819c8db720f"
107
+ },
108
+ {
109
+ "path": "comparisons/evaluation-dynamic/edit-1.jpg",
110
+ "bytes": 140013,
111
+ "sha256": "e93f685092a7f8729a135df81d8fc91d2f7f446e6857d03d36055e4926c820b2"
112
+ },
113
+ {
114
+ "path": "comparisons/heldout-nvfp4/00.jpg",
115
+ "bytes": 144117,
116
+ "sha256": "fef99d1fed7bec5ca673bf94aae5f0f534db77d474e45f04a8c6b52798b20801"
117
+ },
118
+ {
119
+ "path": "comparisons/heldout-nvfp4/01.jpg",
120
+ "bytes": 95024,
121
+ "sha256": "bf3c15a3f9acaaaac7deb73d89501a73a5bfe6d3dd402107deff0bef5218651f"
122
+ },
123
+ {
124
+ "path": "comparisons/heldout-nvfp4/02.jpg",
125
+ "bytes": 137218,
126
+ "sha256": "16c8ab16fe20d594d46ad396ed4c415690f2c3082ac4287b080806a435d82fb3"
127
+ },
128
+ {
129
+ "path": "comparisons/heldout-nvfp4/03.jpg",
130
+ "bytes": 105496,
131
+ "sha256": "907a7a68a36afc83d506442e84a538e0bab1af77582fb76b067809685df15e20"
132
+ },
133
+ {
134
+ "path": "comparisons/heldout-nvfp4/04.jpg",
135
+ "bytes": 142102,
136
+ "sha256": "7abc30394201e4f7e1456f06df38ab6b492ffbf3a869ab5d9decc70231312667"
137
+ },
138
+ {
139
+ "path": "comparisons/heldout-nvfp4/05.jpg",
140
+ "bytes": 203606,
141
+ "sha256": "db68b9fa5e5ce3453fbe74d569b051274f300106d6c3aecd60996418f52ed86b"
142
+ },
143
+ {
144
+ "path": "comparisons/heldout-nvfp4/06.jpg",
145
+ "bytes": 95263,
146
+ "sha256": "194e2d5bafe9ef00fae0f62f891125169c4b79f379bb582b72af82eae51a351b"
147
+ },
148
+ {
149
+ "path": "comparisons/heldout-nvfp4/07.jpg",
150
+ "bytes": 134634,
151
+ "sha256": "9f625acf906a08a69543723a8af7254ecd4e0837101c152fb2140d894ea41371"
152
+ },
153
+ {
154
+ "path": "demo/capture_receipt.json",
155
+ "bytes": 131978,
156
+ "sha256": "b9e7e82391fd38f7536bc48a1ab1f0cc711a715a04b9e634679bdc4da89ab17c"
157
+ },
158
+ {
159
+ "path": "demo/realtime-30s.mp4",
160
+ "bytes": 2301594,
161
+ "sha256": "7585d0c1e7618d2fb710ebf08f70541d36d88ac49152bfcbd4f10d093ae9ca94"
162
+ },
163
+ {
164
+ "path": "dynamic_scale.py",
165
+ "bytes": 1288,
166
+ "sha256": "d908cd50c73cbdf40133775460fe599278fd6d2e7d10648782798d6dfbbbbd7b"
167
+ },
168
+ {
169
+ "path": "fp8_runtime.py",
170
+ "bytes": 4376,
171
+ "sha256": "5892e8b5963f66fee67c16dd3996e05b4b64297e570f3115630a19b224d69413"
172
+ },
173
+ {
174
+ "path": "generate.py",
175
+ "bytes": 3007,
176
+ "sha256": "b383aa9a93d7980ba1b2b74eb6bd7f1accb9c1651c4e468c319fd48608ac42ea"
177
+ },
178
+ {
179
+ "path": "install.sh",
180
+ "bytes": 697,
181
+ "sha256": "8fff3b147208b3651fcc71d7b09ded4674769e4321d840a7871a412298d0be7a"
182
+ },
183
+ {
184
+ "path": "nvfp4_runtime.py",
185
+ "bytes": 3972,
186
+ "sha256": "86fbf7003b755f824b35bf2370feab2f360c79ee4683e4f79267fb7db1e307d7"
187
+ },
188
+ {
189
+ "path": "reports/benchmark.json",
190
+ "bytes": 1158,
191
+ "sha256": "c0ef7fbedaf7b976dab4fc623130fe8e09838b1b0f7147d27d9b367524172c98"
192
+ },
193
+ {
194
+ "path": "reports/calibration_manifest.json",
195
+ "bytes": 29336,
196
+ "sha256": "e067532fc1e6779e22cac2c2beb2bbf4840cd4252aba0da142c5b37f5f97024a"
197
+ },
198
+ {
199
+ "path": "reports/calibration_search.json",
200
+ "bytes": 378392,
201
+ "sha256": "8b71f8e6d43303f3341b4c08e10652dad8c1585f31bf0d0a77dee9a55db8381b"
202
+ },
203
+ {
204
+ "path": "reports/correction-probe.json",
205
+ "bytes": 2835,
206
+ "sha256": "5a3f2d003e8c7c5ab928260b56b12add38bc94d7f7e25ab00fb3b3fddcafcdce"
207
+ },
208
+ {
209
+ "path": "reports/cudagraph-trial.json",
210
+ "bytes": 218,
211
+ "sha256": "9e16d5dc46a1a580ea2dac0e7eec8655c18a8a313870a160e6b091b56a96b4d2"
212
+ },
213
+ {
214
+ "path": "reports/earlier-bf16-benchmark.json",
215
+ "bytes": 1031,
216
+ "sha256": "640be194397d85b94022b5a960141c92e3f65de30d98d2586bc466b5df8beb7c"
217
+ },
218
+ {
219
+ "path": "reports/environment.json",
220
+ "bytes": 2884,
221
+ "sha256": "c375ec1b13547ee0ee2ff0f9a67a9de1d2fd8c47aaa92f3f6de31731c5994af7"
222
+ },
223
+ {
224
+ "path": "reports/evaluation-environment.json",
225
+ "bytes": 161,
226
+ "sha256": "727e74ca45de746f0efa9f78dec96ac8ebc925b92dfddaf1e7e4da812b9cd583"
227
+ },
228
+ {
229
+ "path": "reports/fresh-fp8-benchmark.json",
230
+ "bytes": 913,
231
+ "sha256": "4194b2a3b646711daa393474b9bb911900d65cf2d792fe224e5ab619ff5470fb"
232
+ },
233
+ {
234
+ "path": "reports/heldout-manifest.json",
235
+ "bytes": 2446,
236
+ "sha256": "ec71f7211480de9bc445c9c44649c5811ff20b51b568c4132915bea2ac2c7f91"
237
+ },
238
+ {
239
+ "path": "reports/kernel_evidence.json",
240
+ "bytes": 24824,
241
+ "sha256": "1b622cd32d8e9c6524a27421a3d077d1a0df2c5512f4ff1914e336178e189e2b"
242
+ },
243
+ {
244
+ "path": "reports/native-math-verification.json",
245
+ "bytes": 765,
246
+ "sha256": "3ed24479ca50f9dabe5bfaa57931a6711d4a47c4eb9b493db854d35bc0495be7"
247
+ },
248
+ {
249
+ "path": "reports/qa-review.json",
250
+ "bytes": 817,
251
+ "sha256": "2fc1a5a431daf87ea5e3e3ff2bdcc85239dc2d08ce1ecc18b903815846d8deb3"
252
+ },
253
+ {
254
+ "path": "reports/quality_metrics-bf16.json",
255
+ "bytes": 5502,
256
+ "sha256": "bb644a0dd2aef533c0d9896c52e5a0fcaf376fb4acb433474c62a0367f9e126c"
257
+ },
258
+ {
259
+ "path": "reports/quality_metrics-fp8.json",
260
+ "bytes": 5501,
261
+ "sha256": "99ba88af5444d308e278e94a207ee3c4443faff1e8f321d4e0f304270faa8678"
262
+ },
263
+ {
264
+ "path": "reports/quality_metrics-heldout-bf16.json",
265
+ "bytes": 2880,
266
+ "sha256": "849cfd94a16b602554644bbae1b569111527be909f0004bc16e64e07a107f5cc"
267
+ },
268
+ {
269
+ "path": "reports/quantization_summary.json",
270
+ "bytes": 715,
271
+ "sha256": "ece079fec55b6864a777541a5d959ace269f2a366963890cea957d87b9291c91"
272
+ },
273
+ {
274
+ "path": "reports/release-cli-smoke.json",
275
+ "bytes": 319,
276
+ "sha256": "a0c67c3e7f108714c9be3dd33e5bc66dc8896a792ed900fd45e4766d75cb74ac"
277
+ },
278
+ {
279
+ "path": "reports/tuning-results.json",
280
+ "bytes": 1035,
281
+ "sha256": "bc3e79a8cc7cc99759e58cd9640f1b3eb9c39cbdbd96b44dfc8d269435d8c0e2"
282
+ },
283
+ {
284
+ "path": "reports/video-verification.json",
285
+ "bytes": 1981,
286
+ "sha256": "45a15cb8d97bd795cf7e3bb62143d5725feb9a362d75cb5d28f9907320898cb3"
287
+ },
288
+ {
289
+ "path": "requirements.txt",
290
+ "bytes": 488,
291
+ "sha256": "8687c948b6aeeaf3e20a013f3073f51c4235a5459767afb487191699075369cc"
292
+ },
293
+ {
294
+ "path": "samples/heldout/00.png",
295
+ "bytes": 1470245,
296
+ "sha256": "2ecc7ba864f3065c98c0d256fd92b74c0b068cc391f66ed0c548a3143d2191c1"
297
+ },
298
+ {
299
+ "path": "samples/heldout/01.png",
300
+ "bytes": 1144635,
301
+ "sha256": "298853daf5abb029ba7d715b052392c872301aa934e6482c08f0ea6a00757000"
302
+ },
303
+ {
304
+ "path": "samples/heldout/02.png",
305
+ "bytes": 1593770,
306
+ "sha256": "20312fb53fc29ad2c92ddfb3b2b2583dccb0b42678077dcfc4fa6613fbf35bf6"
307
+ },
308
+ {
309
+ "path": "samples/heldout/03.png",
310
+ "bytes": 1128454,
311
+ "sha256": "86148823326a6a8426f8317a73a2437c6eed1e665b9be9962140a065d7710694"
312
+ },
313
+ {
314
+ "path": "samples/heldout/04.png",
315
+ "bytes": 1395649,
316
+ "sha256": "c0c0d4d5492f443c301925508d034b88bc6102ae197e9f79e6bc646cccc299bc"
317
+ },
318
+ {
319
+ "path": "samples/heldout/05.png",
320
+ "bytes": 7047766,
321
+ "sha256": "63d8d2f130ddc3ff0f8946f0a7ffcfe1302a981507f2e94dc4cb0d82e4a81154"
322
+ },
323
+ {
324
+ "path": "samples/heldout/06.png",
325
+ "bytes": 800963,
326
+ "sha256": "ed89a263c649b8d0b260aef08eb2a7727b9557d4e0937a6d8ef992c9f8bba3c1"
327
+ },
328
+ {
329
+ "path": "samples/heldout/07.png",
330
+ "bytes": 1670613,
331
+ "sha256": "9536f8a08d250dc7d9c6b273e011239e20a6b38effdee3d1c20ac672f5d33bbf"
332
+ },
333
+ {
334
+ "path": "samples/validation/00.png",
335
+ "bytes": 7025357,
336
+ "sha256": "c3f5df94d3b2d9c2958f4c79042ae3ceca55c8dc1a22f3737f589119be8dd6ef"
337
+ },
338
+ {
339
+ "path": "samples/validation/01.png",
340
+ "bytes": 1829665,
341
+ "sha256": "1f50d577a9d22baff9034ddc2cb73bd47ac835e85cf43f47965146f8d223be00"
342
+ },
343
+ {
344
+ "path": "samples/validation/02.png",
345
+ "bytes": 1692847,
346
+ "sha256": "7d746a9909df831af3b3b6ff2de4ed76aefbdd5d9d78098d689ac0648732ff60"
347
+ },
348
+ {
349
+ "path": "samples/validation/03.png",
350
+ "bytes": 1648591,
351
+ "sha256": "b17cf4fd884cfe6a9c04357e170766a34c6029dcf6b1e81eae9f3fec500e5adb"
352
+ },
353
+ {
354
+ "path": "samples/validation/04.png",
355
+ "bytes": 6420433,
356
+ "sha256": "5be7d5c89e014c10907613f6872385d189e4a1352c4bab892130bb8745660e85"
357
+ },
358
+ {
359
+ "path": "samples/validation/05.png",
360
+ "bytes": 2458939,
361
+ "sha256": "317896f85a6606473ef1f7e18ae448829eb73a6bbeadda477d31ea092ca301e5"
362
+ },
363
+ {
364
+ "path": "samples/validation/06.png",
365
+ "bytes": 1341617,
366
+ "sha256": "1e5c2554f35b7934b0a8b27f2d85388a2316dc4307fdbd978debd22a1edeff93"
367
+ },
368
+ {
369
+ "path": "samples/validation/07.png",
370
+ "bytes": 1497363,
371
+ "sha256": "7dc3aec9e5bde92189dc9888658b410f44ee9a377d62f8f1be8c7b7c5b6d6418"
372
+ },
373
+ {
374
+ "path": "samples/validation/08.png",
375
+ "bytes": 3343177,
376
+ "sha256": "73f41a041094d8d800b37d0649a1a898fd8ebeaa0cfff3cfabcb98bab43de9c6"
377
+ },
378
+ {
379
+ "path": "samples/validation/09.png",
380
+ "bytes": 1183289,
381
+ "sha256": "7a1ba8598fed1f8909ae14d2e59bfc966a8a62d79933c35075abf9b4548ad3f1"
382
+ },
383
+ {
384
+ "path": "samples/validation/10.png",
385
+ "bytes": 552063,
386
+ "sha256": "1d571a6cc024720f63a966cc8a3105d72b3f933c7569978686b11ce3734fc3a3"
387
+ },
388
+ {
389
+ "path": "samples/validation/11.png",
390
+ "bytes": 2181491,
391
+ "sha256": "9cb287b21ca746a11fa9aebc2667f984e81fc8dfe95ffc5e980e68c68f3521ee"
392
+ },
393
+ {
394
+ "path": "samples/validation/12.png",
395
+ "bytes": 3164193,
396
+ "sha256": "96f8f77f8062b940f2300470c96b6c62663d988978a9d6cd153288e476b9f45b"
397
+ },
398
+ {
399
+ "path": "samples/validation/13.png",
400
+ "bytes": 1426920,
401
+ "sha256": "a66963ce46511705e0c7ad521068814fa6ae80d6dc555388c9ac860aad47cd1b"
402
+ },
403
+ {
404
+ "path": "samples/validation/14.png",
405
+ "bytes": 1772462,
406
+ "sha256": "74ac4c13e7adcf2bbcf595fa11699790bb233727fcd655cf3bfb0b8e3757a493"
407
+ },
408
+ {
409
+ "path": "samples/validation/15.png",
410
+ "bytes": 2000815,
411
+ "sha256": "2a3614b3f7c39e04c6c9c9840adfc5b60aea8823b437376421b784841618078a"
412
+ },
413
+ {
414
+ "path": "samples/validation/edit-0.png",
415
+ "bytes": 2070198,
416
+ "sha256": "c7f1845565b14c01e651e3b0561ec8b50b2461818754c009421d56d9d29d2e35"
417
+ },
418
+ {
419
+ "path": "samples/validation/edit-1.png",
420
+ "bytes": 1575594,
421
+ "sha256": "d6784ec61e127873fc3deed239b0c258887d0e99069d9df8323f4431e4245ad6"
422
+ },
423
+ {
424
+ "path": "source/benchmark.py",
425
+ "bytes": 3838,
426
+ "sha256": "ea305538d5bd04efb6163960ff729ab95a3e1a347dbc2ee356ddda23a24c19de"
427
+ },
428
+ {
429
+ "path": "source/collect_calibration.py",
430
+ "bytes": 3259,
431
+ "sha256": "41a518464980d92dab2dab9a7281a103a7baac765466f96b879319b83f24609f"
432
+ },
433
+ {
434
+ "path": "source/dynamic_scale.py",
435
+ "bytes": 1288,
436
+ "sha256": "d908cd50c73cbdf40133775460fe599278fd6d2e7d10648782798d6dfbbbbd7b"
437
+ },
438
+ {
439
+ "path": "source/evaluate.py",
440
+ "bytes": 2361,
441
+ "sha256": "d499fe31851e3898f72ff30d41b244e7a59697b77cf72231464c1311c1ac89c1"
442
+ },
443
+ {
444
+ "path": "source/heldout.py",
445
+ "bytes": 2872,
446
+ "sha256": "075ee0c597d32a24c5eef42cecef0f2adca404009b86162c9c2940e45a5de4c6"
447
+ },
448
+ {
449
+ "path": "source/probe_dynamic.py",
450
+ "bytes": 3484,
451
+ "sha256": "581b458d0bc3dd9e0cc4bdc212d7b46382094467feb2f8de24cb51eb5dc05045"
452
+ },
453
+ {
454
+ "path": "source/prompts.py",
455
+ "bytes": 9671,
456
+ "sha256": "626db2851c9d6266a9f397e73a5ffd04d9a8389c011860e925c0396ccf2a274f"
457
+ },
458
+ {
459
+ "path": "source/quality_metrics.py",
460
+ "bytes": 3325,
461
+ "sha256": "2d59b3f9e596dd111f0864ef7ea48111d66f942ec9d0e5e87f849db0cf566d9f"
462
+ },
463
+ {
464
+ "path": "source/quantize_dynamic.py",
465
+ "bytes": 4971,
466
+ "sha256": "7c98c0c2637e08274c4998c7379d72ff4736fbce4a961249c9a13136784807cf"
467
+ },
468
+ {
469
+ "path": "source/verify_native.py",
470
+ "bytes": 1969,
471
+ "sha256": "1e7743563bf25b437971fc8ef98a1e1211e95f22b96198b93b8194fe6d5dc95b"
472
+ },
473
+ {
474
+ "path": "source/verify_video.py",
475
+ "bytes": 2562,
476
+ "sha256": "afe6a1a2b059b015f245d6ad7de6903440415f056f44020cbb9e736ee3fb7439"
477
+ },
478
+ {
479
+ "path": "source/video.py",
480
+ "bytes": 3931,
481
+ "sha256": "5448b3118fd5d4dcd2a6f50afc5294cf694c6cc6e9e62177afea2de38da52f09"
482
+ },
483
+ {
484
+ "path": "source/weight_quant.py",
485
+ "bytes": 1819,
486
+ "sha256": "74eb0257e09065ceec838d5ec966f9038a2f4001736ba833270680dcd70faa71"
487
+ },
488
+ {
489
+ "path": "transformer/config.json",
490
+ "bytes": 370,
491
+ "sha256": "56ae3281c4e6c2d1aa3658252d187488071815fd79bef15808bb0205fc1a2241"
492
+ },
493
+ {
494
+ "path": "transformer/model-00001-of-00002.safetensors",
495
+ "bytes": 2977392992,
496
+ "sha256": "9fa418cbe2610486997bf5f251db26f98e1f2fcb53d291992a63e16660268afb"
497
+ },
498
+ {
499
+ "path": "transformer/model-00002-of-00002.safetensors",
500
+ "bytes": 1893707392,
501
+ "sha256": "a820829d3b2d89a09c99bda3f32b1340b37779ad3072d22e7ec6232f71ed83da"
502
+ },
503
+ {
504
+ "path": "transformer/quantization_config.json",
505
+ "bytes": 42776,
506
+ "sha256": "87565b07b201f5171baf85a4f109302ea8414c5e0c8910d5afd0a630907fdb16"
507
+ }
508
+ ],
509
+ "total_bytes": 4937691362
510
+ }
nvfp4_runtime.py ADDED
@@ -0,0 +1,81 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Calibrated NVFP4 W4A4 with BF16 rank-128 correction on SM120.
2
+
3
+ Built with Qwen. Quantization modifications by ProCreations, 2026.
4
+ Requires the pinned FlashInfer source revision shipped in requirements.
5
+ """
6
+ import json
7
+ from pathlib import Path
8
+ import torch
9
+ from torch import nn
10
+ from safetensors.torch import load_file
11
+
12
+ @torch.library.custom_op('image21_nvfp4::linear',mutates_args=())
13
+ def _linear(x:torch.Tensor,weight:torch.Tensor,sf:torch.Tensor,pre:torch.Tensor,
14
+ gx:torch.Tensor,alpha:torch.Tensor,down:torch.Tensor,up:torch.Tensor)->torch.Tensor:
15
+ from flashinfer.gemm import nvfp4_quantize_smooth,mm_nvfp4_svdquant
16
+ from dynamic_scale import scale
17
+ x=x.contiguous()
18
+ gx,alpha,up=scale(x,pre,gx,alpha,up)
19
+ q,sc=nvfp4_quantize_smooth(x,pre,gx,backend='cute-dsl')
20
+ d=torch.mm(x,down)
21
+ return mm_nvfp4_svdquant(q,weight,sc,sf,alpha,d,up,backend='cute-dsl')
22
+
23
+ @_linear.register_fake
24
+ def _linear_fake(x,weight,sf,pre,gx,alpha,down,up):
25
+ return x.new_empty((x.shape[0],weight.shape[0]),dtype=torch.bfloat16)
26
+
27
+ class CalibratedNVFP4Linear(nn.Module):
28
+ def __init__(self,state):
29
+ super().__init__()
30
+ for key in ['weight','sf','pre','gx','alpha','down','up']:
31
+ self.register_buffer(key,state[key])
32
+ self.out_features,self.in_features=self.weight.shape[0],self.weight.shape[1]*2
33
+ self.bias=None
34
+ def forward(self,x):
35
+ shape=x.shape[:-1]
36
+ return _linear(x.reshape(-1,self.in_features),self.weight,self.sf,self.pre,
37
+ self.gx,self.alpha,self.down,self.up).reshape(*shape,self.out_features)
38
+
39
+ def replace_module(model,name,replacement):
40
+ parent,leaf=name.rsplit('.',1)
41
+ setattr(model.get_submodule(parent),leaf,replacement)
42
+
43
+ def load_nvfp4_transformer(directory,device='cuda'):
44
+ from diffusers import QwenImage21Transformer2DModel
45
+ directory=Path(directory)
46
+ config=json.loads((directory/'config.json').read_text())
47
+ meta=json.loads((directory/'quantization_config.json').read_text())
48
+ with torch.device('meta'):
49
+ model=QwenImage21Transformer2DModel.from_config(config)
50
+ for name,spec in meta['quantized_modules'].items():
51
+ n,k=spec['shape'];r=spec['rank']
52
+ state={
53
+ 'weight':torch.empty((n,k//2),dtype=torch.uint8),
54
+ 'sf':torch.empty(n*k//16,dtype=torch.uint8),
55
+ 'pre':torch.empty(k,dtype=torch.bfloat16),
56
+ 'gx':torch.empty(1,dtype=torch.float32),
57
+ 'alpha':torch.empty(1,dtype=torch.float32),
58
+ 'down':torch.empty((k,r),dtype=torch.bfloat16),
59
+ 'up':torch.empty((n,r),dtype=torch.bfloat16),
60
+ }
61
+ replace_module(model,name,CalibratedNVFP4Linear(state))
62
+ for name,spec in meta.get('fp8_modules',{}).items():
63
+ from fp8_runtime import CalibratedFP8Linear
64
+ n,k=spec['shape']
65
+ replace_module(model,name,CalibratedFP8Linear(
66
+ torch.empty((n,k),dtype=torch.float8_e4m3fn),
67
+ torch.empty((n,1),dtype=torch.float32),torch.empty(k,dtype=torch.float32)))
68
+ state={}
69
+ for p in sorted(directory.glob('model-*.safetensors')):state.update(load_file(str(p)))
70
+ model.load_state_dict(state,strict=True,assign=True)
71
+ from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21Rope,QwenImage21TemporalTimesteps
72
+ model.pos_embed=QwenImage21Rope(theta=10000,axes_dim=list(model.config.axes_dims_rope))
73
+ model.time_text_embed.time_proj=QwenImage21TemporalTimesteps(timestep_dim=256)
74
+ return model.eval().to(device=device)
75
+
76
+ def load_pipeline(base,quant):
77
+ from diffusers import QwenImage21Pipeline
78
+ kwargs={} if Path(base).exists() else {'revision':'b3179ad355be050328e483a9dfdd9e60cd62adfa'}
79
+ p=QwenImage21Pipeline.from_pretrained(base,transformer=load_nvfp4_transformer(quant),torch_dtype=torch.bfloat16,**kwargs)
80
+ p.to(device='cuda');p.set_progress_bar_config(disable=True)
81
+ return p
reports/benchmark.json ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "load_seconds": 2.297096138005145,
3
+ "torch": "2.14.0+cu130",
4
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
5
+ "nvfp4_linears": 224,
6
+ "steps": 40,
7
+ "cfg": 1,
8
+ "bf16_rank": 128,
9
+ "attention_dtype": "bfloat16",
10
+ "approximate_cache": false,
11
+ "timing": {
12
+ "1024": {
13
+ "seconds": [
14
+ 4.547408219019417,
15
+ 4.571850906999316,
16
+ 4.592371508013457,
17
+ 4.610020697989967,
18
+ 4.624509044981096
19
+ ],
20
+ "mean": 4.58923207540065,
21
+ "warmup_seconds": 8.535866206977516,
22
+ "peak_gb": 30.255306752
23
+ },
24
+ "2048": {
25
+ "seconds": [
26
+ 32.58121982298326,
27
+ 32.67262674000813,
28
+ 32.71958359400742
29
+ ],
30
+ "mean": 32.657810052332934,
31
+ "warmup_seconds": 35.41050831298344,
32
+ "peak_gb": 51.385951232
33
+ }
34
+ },
35
+ "protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Prefix KV cache enabled. All large projections use either native NVFP4 with BF16 rank128 correction or explicitly listed calibrated FP8 safety layers. Compiled mode emulates intermediate precision casts."
36
+ }
reports/calibration_manifest.json ADDED
@@ -0,0 +1,881 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source_revision": "b3179ad355be050328e483a9dfdd9e60cd62adfa",
3
+ "rows_per_layer": 1536,
4
+ "sampled_steps": [
5
+ 0,
6
+ 3,
7
+ 9,
8
+ 19,
9
+ 29,
10
+ 39
11
+ ],
12
+ "modules": [
13
+ "transformer_blocks.0.attn.to_q",
14
+ "transformer_blocks.0.attn.to_k",
15
+ "transformer_blocks.0.attn.to_v",
16
+ "transformer_blocks.0.attn.to_out.0",
17
+ "transformer_blocks.0.img_mlp.proj",
18
+ "transformer_blocks.0.img_mlp.out",
19
+ "transformer_blocks.0.img_mlp.gate_layer",
20
+ "transformer_blocks.1.attn.to_q",
21
+ "transformer_blocks.1.attn.to_k",
22
+ "transformer_blocks.1.attn.to_v",
23
+ "transformer_blocks.1.attn.to_out.0",
24
+ "transformer_blocks.1.img_mlp.proj",
25
+ "transformer_blocks.1.img_mlp.out",
26
+ "transformer_blocks.1.img_mlp.gate_layer",
27
+ "transformer_blocks.2.attn.to_q",
28
+ "transformer_blocks.2.attn.to_k",
29
+ "transformer_blocks.2.attn.to_v",
30
+ "transformer_blocks.2.attn.to_out.0",
31
+ "transformer_blocks.2.img_mlp.proj",
32
+ "transformer_blocks.2.img_mlp.out",
33
+ "transformer_blocks.2.img_mlp.gate_layer",
34
+ "transformer_blocks.3.attn.to_q",
35
+ "transformer_blocks.3.attn.to_k",
36
+ "transformer_blocks.3.attn.to_v",
37
+ "transformer_blocks.3.attn.to_out.0",
38
+ "transformer_blocks.3.img_mlp.proj",
39
+ "transformer_blocks.3.img_mlp.out",
40
+ "transformer_blocks.3.img_mlp.gate_layer",
41
+ "transformer_blocks.4.attn.to_q",
42
+ "transformer_blocks.4.attn.to_k",
43
+ "transformer_blocks.4.attn.to_v",
44
+ "transformer_blocks.4.attn.to_out.0",
45
+ "transformer_blocks.4.img_mlp.proj",
46
+ "transformer_blocks.4.img_mlp.out",
47
+ "transformer_blocks.4.img_mlp.gate_layer",
48
+ "transformer_blocks.5.attn.to_q",
49
+ "transformer_blocks.5.attn.to_k",
50
+ "transformer_blocks.5.attn.to_v",
51
+ "transformer_blocks.5.attn.to_out.0",
52
+ "transformer_blocks.5.img_mlp.proj",
53
+ "transformer_blocks.5.img_mlp.out",
54
+ "transformer_blocks.5.img_mlp.gate_layer",
55
+ "transformer_blocks.6.attn.to_q",
56
+ "transformer_blocks.6.attn.to_k",
57
+ "transformer_blocks.6.attn.to_v",
58
+ "transformer_blocks.6.attn.to_out.0",
59
+ "transformer_blocks.6.img_mlp.proj",
60
+ "transformer_blocks.6.img_mlp.out",
61
+ "transformer_blocks.6.img_mlp.gate_layer",
62
+ "transformer_blocks.7.attn.to_q",
63
+ "transformer_blocks.7.attn.to_k",
64
+ "transformer_blocks.7.attn.to_v",
65
+ "transformer_blocks.7.attn.to_out.0",
66
+ "transformer_blocks.7.img_mlp.proj",
67
+ "transformer_blocks.7.img_mlp.out",
68
+ "transformer_blocks.7.img_mlp.gate_layer",
69
+ "transformer_blocks.8.attn.to_q",
70
+ "transformer_blocks.8.attn.to_k",
71
+ "transformer_blocks.8.attn.to_v",
72
+ "transformer_blocks.8.attn.to_out.0",
73
+ "transformer_blocks.8.img_mlp.proj",
74
+ "transformer_blocks.8.img_mlp.out",
75
+ "transformer_blocks.8.img_mlp.gate_layer",
76
+ "transformer_blocks.9.attn.to_q",
77
+ "transformer_blocks.9.attn.to_k",
78
+ "transformer_blocks.9.attn.to_v",
79
+ "transformer_blocks.9.attn.to_out.0",
80
+ "transformer_blocks.9.img_mlp.proj",
81
+ "transformer_blocks.9.img_mlp.out",
82
+ "transformer_blocks.9.img_mlp.gate_layer",
83
+ "transformer_blocks.10.attn.to_q",
84
+ "transformer_blocks.10.attn.to_k",
85
+ "transformer_blocks.10.attn.to_v",
86
+ "transformer_blocks.10.attn.to_out.0",
87
+ "transformer_blocks.10.img_mlp.proj",
88
+ "transformer_blocks.10.img_mlp.out",
89
+ "transformer_blocks.10.img_mlp.gate_layer",
90
+ "transformer_blocks.11.attn.to_q",
91
+ "transformer_blocks.11.attn.to_k",
92
+ "transformer_blocks.11.attn.to_v",
93
+ "transformer_blocks.11.attn.to_out.0",
94
+ "transformer_blocks.11.img_mlp.proj",
95
+ "transformer_blocks.11.img_mlp.out",
96
+ "transformer_blocks.11.img_mlp.gate_layer",
97
+ "transformer_blocks.12.attn.to_q",
98
+ "transformer_blocks.12.attn.to_k",
99
+ "transformer_blocks.12.attn.to_v",
100
+ "transformer_blocks.12.attn.to_out.0",
101
+ "transformer_blocks.12.img_mlp.proj",
102
+ "transformer_blocks.12.img_mlp.out",
103
+ "transformer_blocks.12.img_mlp.gate_layer",
104
+ "transformer_blocks.13.attn.to_q",
105
+ "transformer_blocks.13.attn.to_k",
106
+ "transformer_blocks.13.attn.to_v",
107
+ "transformer_blocks.13.attn.to_out.0",
108
+ "transformer_blocks.13.img_mlp.proj",
109
+ "transformer_blocks.13.img_mlp.out",
110
+ "transformer_blocks.13.img_mlp.gate_layer",
111
+ "transformer_blocks.14.attn.to_q",
112
+ "transformer_blocks.14.attn.to_k",
113
+ "transformer_blocks.14.attn.to_v",
114
+ "transformer_blocks.14.attn.to_out.0",
115
+ "transformer_blocks.14.img_mlp.proj",
116
+ "transformer_blocks.14.img_mlp.out",
117
+ "transformer_blocks.14.img_mlp.gate_layer",
118
+ "transformer_blocks.15.attn.to_q",
119
+ "transformer_blocks.15.attn.to_k",
120
+ "transformer_blocks.15.attn.to_v",
121
+ "transformer_blocks.15.attn.to_out.0",
122
+ "transformer_blocks.15.img_mlp.proj",
123
+ "transformer_blocks.15.img_mlp.out",
124
+ "transformer_blocks.15.img_mlp.gate_layer",
125
+ "transformer_blocks.16.attn.to_q",
126
+ "transformer_blocks.16.attn.to_k",
127
+ "transformer_blocks.16.attn.to_v",
128
+ "transformer_blocks.16.attn.to_out.0",
129
+ "transformer_blocks.16.img_mlp.proj",
130
+ "transformer_blocks.16.img_mlp.out",
131
+ "transformer_blocks.16.img_mlp.gate_layer",
132
+ "transformer_blocks.17.attn.to_q",
133
+ "transformer_blocks.17.attn.to_k",
134
+ "transformer_blocks.17.attn.to_v",
135
+ "transformer_blocks.17.attn.to_out.0",
136
+ "transformer_blocks.17.img_mlp.proj",
137
+ "transformer_blocks.17.img_mlp.out",
138
+ "transformer_blocks.17.img_mlp.gate_layer",
139
+ "transformer_blocks.18.attn.to_q",
140
+ "transformer_blocks.18.attn.to_k",
141
+ "transformer_blocks.18.attn.to_v",
142
+ "transformer_blocks.18.attn.to_out.0",
143
+ "transformer_blocks.18.img_mlp.proj",
144
+ "transformer_blocks.18.img_mlp.out",
145
+ "transformer_blocks.18.img_mlp.gate_layer",
146
+ "transformer_blocks.19.attn.to_q",
147
+ "transformer_blocks.19.attn.to_k",
148
+ "transformer_blocks.19.attn.to_v",
149
+ "transformer_blocks.19.attn.to_out.0",
150
+ "transformer_blocks.19.img_mlp.proj",
151
+ "transformer_blocks.19.img_mlp.out",
152
+ "transformer_blocks.19.img_mlp.gate_layer",
153
+ "transformer_blocks.20.attn.to_q",
154
+ "transformer_blocks.20.attn.to_k",
155
+ "transformer_blocks.20.attn.to_v",
156
+ "transformer_blocks.20.attn.to_out.0",
157
+ "transformer_blocks.20.img_mlp.proj",
158
+ "transformer_blocks.20.img_mlp.out",
159
+ "transformer_blocks.20.img_mlp.gate_layer",
160
+ "transformer_blocks.21.attn.to_q",
161
+ "transformer_blocks.21.attn.to_k",
162
+ "transformer_blocks.21.attn.to_v",
163
+ "transformer_blocks.21.attn.to_out.0",
164
+ "transformer_blocks.21.img_mlp.proj",
165
+ "transformer_blocks.21.img_mlp.out",
166
+ "transformer_blocks.21.img_mlp.gate_layer",
167
+ "transformer_blocks.22.attn.to_q",
168
+ "transformer_blocks.22.attn.to_k",
169
+ "transformer_blocks.22.attn.to_v",
170
+ "transformer_blocks.22.attn.to_out.0",
171
+ "transformer_blocks.22.img_mlp.proj",
172
+ "transformer_blocks.22.img_mlp.out",
173
+ "transformer_blocks.22.img_mlp.gate_layer",
174
+ "transformer_blocks.23.attn.to_q",
175
+ "transformer_blocks.23.attn.to_k",
176
+ "transformer_blocks.23.attn.to_v",
177
+ "transformer_blocks.23.attn.to_out.0",
178
+ "transformer_blocks.23.img_mlp.proj",
179
+ "transformer_blocks.23.img_mlp.out",
180
+ "transformer_blocks.23.img_mlp.gate_layer",
181
+ "transformer_blocks.24.attn.to_q",
182
+ "transformer_blocks.24.attn.to_k",
183
+ "transformer_blocks.24.attn.to_v",
184
+ "transformer_blocks.24.attn.to_out.0",
185
+ "transformer_blocks.24.img_mlp.proj",
186
+ "transformer_blocks.24.img_mlp.out",
187
+ "transformer_blocks.24.img_mlp.gate_layer",
188
+ "transformer_blocks.25.attn.to_q",
189
+ "transformer_blocks.25.attn.to_k",
190
+ "transformer_blocks.25.attn.to_v",
191
+ "transformer_blocks.25.attn.to_out.0",
192
+ "transformer_blocks.25.img_mlp.proj",
193
+ "transformer_blocks.25.img_mlp.out",
194
+ "transformer_blocks.25.img_mlp.gate_layer",
195
+ "transformer_blocks.26.attn.to_q",
196
+ "transformer_blocks.26.attn.to_k",
197
+ "transformer_blocks.26.attn.to_v",
198
+ "transformer_blocks.26.attn.to_out.0",
199
+ "transformer_blocks.26.img_mlp.proj",
200
+ "transformer_blocks.26.img_mlp.out",
201
+ "transformer_blocks.26.img_mlp.gate_layer",
202
+ "transformer_blocks.27.attn.to_q",
203
+ "transformer_blocks.27.attn.to_k",
204
+ "transformer_blocks.27.attn.to_v",
205
+ "transformer_blocks.27.attn.to_out.0",
206
+ "transformer_blocks.27.img_mlp.proj",
207
+ "transformer_blocks.27.img_mlp.out",
208
+ "transformer_blocks.27.img_mlp.gate_layer",
209
+ "transformer_blocks.28.attn.to_q",
210
+ "transformer_blocks.28.attn.to_k",
211
+ "transformer_blocks.28.attn.to_v",
212
+ "transformer_blocks.28.attn.to_out.0",
213
+ "transformer_blocks.28.img_mlp.proj",
214
+ "transformer_blocks.28.img_mlp.out",
215
+ "transformer_blocks.28.img_mlp.gate_layer",
216
+ "transformer_blocks.29.attn.to_q",
217
+ "transformer_blocks.29.attn.to_k",
218
+ "transformer_blocks.29.attn.to_v",
219
+ "transformer_blocks.29.attn.to_out.0",
220
+ "transformer_blocks.29.img_mlp.proj",
221
+ "transformer_blocks.29.img_mlp.out",
222
+ "transformer_blocks.29.img_mlp.gate_layer",
223
+ "transformer_blocks.30.attn.to_q",
224
+ "transformer_blocks.30.attn.to_k",
225
+ "transformer_blocks.30.attn.to_v",
226
+ "transformer_blocks.30.attn.to_out.0",
227
+ "transformer_blocks.30.img_mlp.proj",
228
+ "transformer_blocks.30.img_mlp.out",
229
+ "transformer_blocks.30.img_mlp.gate_layer",
230
+ "transformer_blocks.31.attn.to_q",
231
+ "transformer_blocks.31.attn.to_k",
232
+ "transformer_blocks.31.attn.to_v",
233
+ "transformer_blocks.31.attn.to_out.0",
234
+ "transformer_blocks.31.img_mlp.proj",
235
+ "transformer_blocks.31.img_mlp.out",
236
+ "transformer_blocks.31.img_mlp.gate_layer"
237
+ ],
238
+ "records": [
239
+ {
240
+ "index": 0,
241
+ "prompt": "A documentary photograph of an elderly fisherman repairing a blue net on a wooden pier, overcast daylight, realistic hands and skin.",
242
+ "seed": 10000,
243
+ "width": 2048,
244
+ "height": 2048,
245
+ "steps": 40,
246
+ "seconds_with_collection": 55.55586172902258,
247
+ "editing": false
248
+ },
249
+ {
250
+ "index": 1,
251
+ "prompt": "A studio portrait of a woman with curly black hair wearing a green wool sweater, soft side lighting, natural skin texture.",
252
+ "seed": 10001,
253
+ "width": 1024,
254
+ "height": 1024,
255
+ "steps": 40,
256
+ "seconds_with_collection": 9.916995454987045,
257
+ "editing": false
258
+ },
259
+ {
260
+ "index": 2,
261
+ "prompt": "A red panda resting on a mossy branch in a misty bamboo forest, wildlife photography, intricate fur.",
262
+ "seed": 10002,
263
+ "width": 1024,
264
+ "height": 1024,
265
+ "steps": 40,
266
+ "seconds_with_collection": 9.919777019997127,
267
+ "editing": false
268
+ },
269
+ {
270
+ "index": 3,
271
+ "prompt": "An aerial photograph of turquoise river channels winding through a dark volcanic plain at sunrise.",
272
+ "seed": 10003,
273
+ "width": 1024,
274
+ "height": 1024,
275
+ "steps": 40,
276
+ "seconds_with_collection": 9.944503634003922,
277
+ "editing": false
278
+ },
279
+ {
280
+ "index": 4,
281
+ "prompt": "A macro photograph of dew on a purple iris, shallow depth of field, intricate translucent petals.",
282
+ "seed": 10004,
283
+ "width": 2048,
284
+ "height": 2048,
285
+ "steps": 40,
286
+ "seconds_with_collection": 55.75184188003186,
287
+ "editing": false
288
+ },
289
+ {
290
+ "index": 5,
291
+ "prompt": "An architectural photograph of a quiet concrete library with tall windows and warm wooden furniture.",
292
+ "seed": 10005,
293
+ "width": 1024,
294
+ "height": 1024,
295
+ "steps": 40,
296
+ "seconds_with_collection": 10.009491505974438,
297
+ "editing": false
298
+ },
299
+ {
300
+ "index": 6,
301
+ "prompt": "A still life of three ceramic bowls, a linen cloth, two pears and a glass of water in window light.",
302
+ "seed": 10006,
303
+ "width": 1024,
304
+ "height": 1024,
305
+ "steps": 40,
306
+ "seconds_with_collection": 9.987017144041602,
307
+ "editing": false
308
+ },
309
+ {
310
+ "index": 7,
311
+ "prompt": "A cinematic photograph of a wet narrow street in Kyoto at dusk, umbrellas and reflected shop lights.",
312
+ "seed": 10007,
313
+ "width": 1024,
314
+ "height": 1024,
315
+ "steps": 40,
316
+ "seconds_with_collection": 9.978878663969226,
317
+ "editing": false
318
+ },
319
+ {
320
+ "index": 8,
321
+ "prompt": "A wide desert landscape with rippling dunes and a lone acacia tree beneath a star-filled sky.",
322
+ "seed": 10008,
323
+ "width": 2048,
324
+ "height": 2048,
325
+ "steps": 40,
326
+ "seconds_with_collection": 55.9559129459667,
327
+ "editing": false
328
+ },
329
+ {
330
+ "index": 9,
331
+ "prompt": "An underwater photograph of a green sea turtle above a colorful coral reef, sunlight shafts.",
332
+ "seed": 10009,
333
+ "width": 1024,
334
+ "height": 1024,
335
+ "steps": 40,
336
+ "seconds_with_collection": 9.959987735957839,
337
+ "editing": false
338
+ },
339
+ {
340
+ "index": 10,
341
+ "prompt": "A watercolor illustration of a small cottage surrounded by wildflowers, delicate washes on textured paper.",
342
+ "seed": 10010,
343
+ "width": 1024,
344
+ "height": 1024,
345
+ "steps": 40,
346
+ "seconds_with_collection": 10.016518865944818,
347
+ "editing": false
348
+ },
349
+ {
350
+ "index": 11,
351
+ "prompt": "A detailed oil painting of a stormy ocean and a distant lighthouse, dramatic layered clouds.",
352
+ "seed": 10011,
353
+ "width": 1024,
354
+ "height": 1024,
355
+ "steps": 40,
356
+ "seconds_with_collection": 10.014705530018546,
357
+ "editing": false
358
+ },
359
+ {
360
+ "index": 12,
361
+ "prompt": "An isometric clay render of a miniature bakery with croissants and copper baking trays, pastel colors.",
362
+ "seed": 10012,
363
+ "width": 2048,
364
+ "height": 2048,
365
+ "steps": 40,
366
+ "seconds_with_collection": 55.98055850400124,
367
+ "editing": false
368
+ },
369
+ {
370
+ "index": 13,
371
+ "prompt": "A black and white ink drawing of an old oak tree with twisted roots, intricate crosshatching.",
372
+ "seed": 10013,
373
+ "width": 1024,
374
+ "height": 1024,
375
+ "steps": 40,
376
+ "seconds_with_collection": 9.911684828985017,
377
+ "editing": false
378
+ },
379
+ {
380
+ "index": 14,
381
+ "prompt": "A friendly orange robot tending a greenhouse full of tropical plants, polished 3D animation style.",
382
+ "seed": 10014,
383
+ "width": 1024,
384
+ "height": 1024,
385
+ "steps": 40,
386
+ "seconds_with_collection": 10.013557444966864,
387
+ "editing": false
388
+ },
389
+ {
390
+ "index": 15,
391
+ "prompt": "A crisp product photograph of a translucent blue perfume bottle on pale stone, precise reflections.",
392
+ "seed": 10015,
393
+ "width": 1024,
394
+ "height": 1024,
395
+ "steps": 40,
396
+ "seconds_with_collection": 10.004387759021483,
397
+ "editing": false
398
+ },
399
+ {
400
+ "index": 16,
401
+ "prompt": "A close-up of a mechanical wristwatch with brushed steel and visible gears, luxury product lighting.",
402
+ "seed": 10016,
403
+ "width": 2048,
404
+ "height": 2048,
405
+ "steps": 40,
406
+ "seconds_with_collection": 56.012295930995606,
407
+ "editing": false
408
+ },
409
+ {
410
+ "index": 17,
411
+ "prompt": "A photograph of a chef pulling a pizza from a brick oven, flour dust and warm firelight.",
412
+ "seed": 10017,
413
+ "width": 1024,
414
+ "height": 1024,
415
+ "steps": 40,
416
+ "seconds_with_collection": 10.00824662600644,
417
+ "editing": false
418
+ },
419
+ {
420
+ "index": 18,
421
+ "prompt": "A handmade paper collage of mountains, a winding river and a yellow sun, visible cut edges.",
422
+ "seed": 10018,
423
+ "width": 1024,
424
+ "height": 1024,
425
+ "steps": 40,
426
+ "seconds_with_collection": 9.993421372026205,
427
+ "editing": false
428
+ },
429
+ {
430
+ "index": 19,
431
+ "prompt": "A charcoal portrait of an elderly violinist, expressive eyes and detailed instrument strings.",
432
+ "seed": 10019,
433
+ "width": 1024,
434
+ "height": 1024,
435
+ "steps": 40,
436
+ "seconds_with_collection": 9.94143287598854,
437
+ "editing": false
438
+ },
439
+ {
440
+ "index": 20,
441
+ "prompt": "A ceramic teapot shaped like a sleeping cat, clean product photography on a cream background.",
442
+ "seed": 10020,
443
+ "width": 2048,
444
+ "height": 2048,
445
+ "steps": 40,
446
+ "seconds_with_collection": 55.824732757988386,
447
+ "editing": false
448
+ },
449
+ {
450
+ "index": 21,
451
+ "prompt": "A cozy reading nook in an attic with a round window, books and a sleeping dog, morning light.",
452
+ "seed": 10021,
453
+ "width": 1024,
454
+ "height": 1024,
455
+ "steps": 40,
456
+ "seconds_with_collection": 10.017206886026543,
457
+ "editing": false
458
+ },
459
+ {
460
+ "index": 22,
461
+ "prompt": "A vibrant science fiction city built inside a giant glass dome on a rocky moon, cinematic wide shot.",
462
+ "seed": 10022,
463
+ "width": 1024,
464
+ "height": 1024,
465
+ "steps": 40,
466
+ "seconds_with_collection": 9.905753094004467,
467
+ "editing": false
468
+ },
469
+ {
470
+ "index": 23,
471
+ "prompt": "A botanical illustration of a sunflower showing leaves, roots and flower head, clean ivory paper.",
472
+ "seed": 10023,
473
+ "width": 1024,
474
+ "height": 1024,
475
+ "steps": 40,
476
+ "seconds_with_collection": 9.950438202009536,
477
+ "editing": false
478
+ },
479
+ {
480
+ "index": 24,
481
+ "prompt": "A photograph of two dancers in flowing red and white costumes on a dark stage, frozen movement.",
482
+ "seed": 10024,
483
+ "width": 2048,
484
+ "height": 2048,
485
+ "steps": 40,
486
+ "seconds_with_collection": 55.87080749304732,
487
+ "editing": false
488
+ },
489
+ {
490
+ "index": 25,
491
+ "prompt": "A bowl of ramen with noodles, mushrooms, eggs and scallions, overhead food photography.",
492
+ "seed": 10025,
493
+ "width": 1024,
494
+ "height": 1024,
495
+ "steps": 40,
496
+ "seconds_with_collection": 10.003536691016052,
497
+ "editing": false
498
+ },
499
+ {
500
+ "index": 26,
501
+ "prompt": "A snowy alpine village reflected in a still lake, blue hour, high detail.",
502
+ "seed": 10026,
503
+ "width": 1024,
504
+ "height": 1024,
505
+ "steps": 40,
506
+ "seconds_with_collection": 9.961158728983719,
507
+ "editing": false
508
+ },
509
+ {
510
+ "index": 27,
511
+ "prompt": "A glass sculpture of a hummingbird with rainbow refractions, black studio background.",
512
+ "seed": 10027,
513
+ "width": 1024,
514
+ "height": 1024,
515
+ "steps": 40,
516
+ "seconds_with_collection": 9.926164573989809,
517
+ "editing": false
518
+ },
519
+ {
520
+ "index": 28,
521
+ "prompt": "A poster with the exact large words \"SPRING GARDEN\" and small text \"OPEN SATURDAY\", flowers and green borders.",
522
+ "seed": 10028,
523
+ "width": 2048,
524
+ "height": 2048,
525
+ "steps": 40,
526
+ "seconds_with_collection": 55.95907588303089,
527
+ "editing": false
528
+ },
529
+ {
530
+ "index": 29,
531
+ "prompt": "A neatly designed cafe menu with the headings \"COFFEE\", \"TEA\" and \"PASTRIES\", black typography on cream paper.",
532
+ "seed": 10029,
533
+ "width": 1024,
534
+ "height": 1024,
535
+ "steps": 40,
536
+ "seconds_with_collection": 9.982229411019944,
537
+ "editing": false
538
+ },
539
+ {
540
+ "index": 30,
541
+ "prompt": "一张精美的海报,标题为“山水之间”,画面是清晨薄雾中的青山和湖泊,优雅的中文排版。",
542
+ "seed": 10030,
543
+ "width": 1024,
544
+ "height": 1024,
545
+ "steps": 40,
546
+ "seconds_with_collection": 9.996177694993094,
547
+ "editing": false
548
+ },
549
+ {
550
+ "index": 31,
551
+ "prompt": "一只橘猫坐在木窗边,窗外下着细雨,室内有温暖的灯光,真实摄影风格。",
552
+ "seed": 10031,
553
+ "width": 1024,
554
+ "height": 1024,
555
+ "steps": 40,
556
+ "seconds_with_collection": 10.037701113033108,
557
+ "editing": false
558
+ },
559
+ {
560
+ "index": 32,
561
+ "prompt": "Une photographie réaliste de lavande dans la campagne française, une petite maison en pierre au loin.",
562
+ "seed": 10032,
563
+ "width": 2048,
564
+ "height": 2048,
565
+ "steps": 40,
566
+ "seconds_with_collection": 55.914367380028125,
567
+ "editing": false
568
+ },
569
+ {
570
+ "index": 33,
571
+ "prompt": "Un mercado de frutas al amanecer, naranjas, limones y flores, fotografía documental con colores naturales.",
572
+ "seed": 10033,
573
+ "width": 1024,
574
+ "height": 1024,
575
+ "steps": 40,
576
+ "seconds_with_collection": 10.030677798960824,
577
+ "editing": false
578
+ },
579
+ {
580
+ "index": 34,
581
+ "prompt": "日本の静かな庭園、石灯籠と池、秋の赤いもみじ、柔らかい朝の光、写真。",
582
+ "seed": 10034,
583
+ "width": 1024,
584
+ "height": 1024,
585
+ "steps": 40,
586
+ "seconds_with_collection": 9.955376817961223,
587
+ "editing": false
588
+ },
589
+ {
590
+ "index": 35,
591
+ "prompt": "صورة فوتوغرافية لقارب خشبي على بحيرة هادئة عند شروق الشمس، جبال في الخلفية.",
592
+ "seed": 10035,
593
+ "width": 1024,
594
+ "height": 1024,
595
+ "steps": 40,
596
+ "seconds_with_collection": 10.00634750595782,
597
+ "editing": false
598
+ },
599
+ {
600
+ "index": 36,
601
+ "prompt": "A group photograph of four friends seated at a picnic table, diverse appearances, natural candid expressions.",
602
+ "seed": 10036,
603
+ "width": 2048,
604
+ "height": 2048,
605
+ "steps": 40,
606
+ "seconds_with_collection": 56.10474174999399,
607
+ "editing": false
608
+ },
609
+ {
610
+ "index": 37,
611
+ "prompt": "A close photograph of a hand holding a delicate seashell, realistic fingers and softly blurred beach.",
612
+ "seed": 10037,
613
+ "width": 1024,
614
+ "height": 1024,
615
+ "steps": 40,
616
+ "seconds_with_collection": 10.005634397966787,
617
+ "editing": false
618
+ },
619
+ {
620
+ "index": 38,
621
+ "prompt": "Three distinct colored cubes: a red cube left, a green cube center and a blue cube right, clean studio photograph.",
622
+ "seed": 10038,
623
+ "width": 1024,
624
+ "height": 1024,
625
+ "steps": 40,
626
+ "seconds_with_collection": 9.967702204012312,
627
+ "editing": false
628
+ },
629
+ {
630
+ "index": 39,
631
+ "prompt": "A tiny astronaut standing on an enormous open book, imaginative photorealistic diorama, soft light.",
632
+ "seed": 10039,
633
+ "width": 1024,
634
+ "height": 1024,
635
+ "steps": 40,
636
+ "seconds_with_collection": 10.009398482972756,
637
+ "editing": false
638
+ },
639
+ {
640
+ "index": 40,
641
+ "prompt": "This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent.",
642
+ "seed": 10040,
643
+ "width": 2048,
644
+ "height": 2048,
645
+ "steps": 40,
646
+ "seconds_with_collection": 55.86863625398837,
647
+ "editing": false
648
+ },
649
+ {
650
+ "index": 41,
651
+ "prompt": "This is an RGBA image with transparency. A realistic pink rose with green leaves. The image has alpha channel and the background is transparent.",
652
+ "seed": 10041,
653
+ "width": 1024,
654
+ "height": 1024,
655
+ "steps": 40,
656
+ "seconds_with_collection": 9.862337956030387,
657
+ "editing": false
658
+ },
659
+ {
660
+ "index": 42,
661
+ "prompt": "This is an RGBA image with transparency. A polished golden compass seen from above. The image has alpha channel and the background is transparent.",
662
+ "seed": 10042,
663
+ "width": 1024,
664
+ "height": 1024,
665
+ "steps": 40,
666
+ "seconds_with_collection": 9.913998158997856,
667
+ "editing": false
668
+ },
669
+ {
670
+ "index": 43,
671
+ "prompt": "This is an RGBA image with transparency. A fluffy white puppy sitting with its paws visible. The image has alpha channel and the background is transparent.",
672
+ "seed": 10043,
673
+ "width": 1024,
674
+ "height": 1024,
675
+ "steps": 40,
676
+ "seconds_with_collection": 9.922774217964616,
677
+ "editing": false
678
+ },
679
+ {
680
+ "index": 44,
681
+ "prompt": "A sunlit photograph of a waterfall pouring through a lush rocky ravine, long exposure water.",
682
+ "seed": 10044,
683
+ "width": 2048,
684
+ "height": 2048,
685
+ "steps": 40,
686
+ "seconds_with_collection": 56.01795354997739,
687
+ "editing": false
688
+ },
689
+ {
690
+ "index": 45,
691
+ "prompt": "A close-up of iridescent soap bubbles against dark velvet, fine colorful interference patterns.",
692
+ "seed": 10045,
693
+ "width": 1024,
694
+ "height": 1024,
695
+ "steps": 40,
696
+ "seconds_with_collection": 10.013371845008805,
697
+ "editing": false
698
+ },
699
+ {
700
+ "index": 46,
701
+ "prompt": "A handwoven basket filled with apples, walnuts and autumn leaves, rustic still life photograph.",
702
+ "seed": 10046,
703
+ "width": 1024,
704
+ "height": 1024,
705
+ "steps": 40,
706
+ "seconds_with_collection": 10.028369715961162,
707
+ "editing": false
708
+ },
709
+ {
710
+ "index": 47,
711
+ "prompt": "A minimalist geometric illustration with overlapping indigo circles and coral triangles on cream.",
712
+ "seed": 10047,
713
+ "width": 1024,
714
+ "height": 1024,
715
+ "steps": 40,
716
+ "seconds_with_collection": 9.945206430973485,
717
+ "editing": false
718
+ },
719
+ {
720
+ "index": 48,
721
+ "prompt": "A double exposure portrait combining a human silhouette with a pine forest, subtle photographic artwork.",
722
+ "seed": 10048,
723
+ "width": 2048,
724
+ "height": 2048,
725
+ "steps": 40,
726
+ "seconds_with_collection": 56.25246442400385,
727
+ "editing": false
728
+ },
729
+ {
730
+ "index": 49,
731
+ "prompt": "An elegant silver spaceship orbiting a blue planet, realistic cinematic lighting and star field.",
732
+ "seed": 10049,
733
+ "width": 1024,
734
+ "height": 1024,
735
+ "steps": 40,
736
+ "seconds_with_collection": 9.991588920995127,
737
+ "editing": false
738
+ },
739
+ {
740
+ "index": 50,
741
+ "prompt": "A photorealistic brown horse galloping across a green meadow, detailed muscles and flowing mane.",
742
+ "seed": 10050,
743
+ "width": 1024,
744
+ "height": 1024,
745
+ "steps": 40,
746
+ "seconds_with_collection": 10.032783773029223,
747
+ "editing": false
748
+ },
749
+ {
750
+ "index": 51,
751
+ "prompt": "A richly detailed mosaic of tropical fish made from small colored glass tiles, museum lighting.",
752
+ "seed": 10051,
753
+ "width": 1024,
754
+ "height": 1024,
755
+ "steps": 40,
756
+ "seconds_with_collection": 9.938231437001377,
757
+ "editing": false
758
+ },
759
+ {
760
+ "index": 52,
761
+ "prompt": "A long horizontal panorama of a rocky seashore under dramatic sunset clouds, realistic photograph.",
762
+ "seed": 10052,
763
+ "width": 1536,
764
+ "height": 864,
765
+ "steps": 40,
766
+ "seconds_with_collection": 13.22676891402807,
767
+ "editing": false
768
+ },
769
+ {
770
+ "index": 53,
771
+ "prompt": "A vertical photograph looking up through a spiral staircase with repeating white railings.",
772
+ "seed": 10053,
773
+ "width": 864,
774
+ "height": 1536,
775
+ "steps": 40,
776
+ "seconds_with_collection": 13.226088834984694,
777
+ "editing": false
778
+ },
779
+ {
780
+ "index": 54,
781
+ "prompt": "A warm sepia photograph of an antique bicycle leaning against a brick wall, climbing roses.",
782
+ "seed": 10054,
783
+ "width": 1024,
784
+ "height": 1024,
785
+ "steps": 40,
786
+ "seconds_with_collection": 9.979204855044372,
787
+ "editing": false
788
+ },
789
+ {
790
+ "index": 55,
791
+ "prompt": "A quiet winter forest of birch trees and soft falling snow, monochromatic fine art photography.",
792
+ "seed": 10055,
793
+ "width": 1024,
794
+ "height": 1024,
795
+ "steps": 40,
796
+ "seconds_with_collection": 9.959185037005227,
797
+ "editing": false
798
+ },
799
+ {
800
+ "index": 56,
801
+ "prompt": "Change the scene to warm sunset lighting while preserving the main subject.",
802
+ "seed": 10056,
803
+ "width": 2048,
804
+ "height": 2048,
805
+ "steps": 40,
806
+ "seconds_with_collection": 61.76318650902249,
807
+ "editing": true
808
+ },
809
+ {
810
+ "index": 57,
811
+ "prompt": "Make the background a lush green garden while preserving the main subject.",
812
+ "seed": 10057,
813
+ "width": 1024,
814
+ "height": 1024,
815
+ "steps": 40,
816
+ "seconds_with_collection": 11.710096635040827,
817
+ "editing": true
818
+ },
819
+ {
820
+ "index": 58,
821
+ "prompt": "Convert this image to a delicate watercolor painting, preserving its composition.",
822
+ "seed": 10058,
823
+ "width": 1024,
824
+ "height": 1024,
825
+ "steps": 40,
826
+ "seconds_with_collection": 11.703723802987952,
827
+ "editing": true
828
+ },
829
+ {
830
+ "index": 59,
831
+ "prompt": "Add gently falling snow and a winter atmosphere while preserving the main subject.",
832
+ "seed": 10059,
833
+ "width": 1024,
834
+ "height": 1024,
835
+ "steps": 40,
836
+ "seconds_with_collection": 11.623913866991643,
837
+ "editing": true
838
+ },
839
+ {
840
+ "index": 60,
841
+ "prompt": "Change the overall color palette to cool blue and silver, preserving details.",
842
+ "seed": 10060,
843
+ "width": 2048,
844
+ "height": 2048,
845
+ "steps": 40,
846
+ "seconds_with_collection": 60.85325455095153,
847
+ "editing": true
848
+ },
849
+ {
850
+ "index": 61,
851
+ "prompt": "Convert the image to a detailed pencil drawing on white paper.",
852
+ "seed": 10061,
853
+ "width": 1024,
854
+ "height": 1024,
855
+ "steps": 40,
856
+ "seconds_with_collection": 11.702045002020895,
857
+ "editing": true
858
+ },
859
+ {
860
+ "index": 62,
861
+ "prompt": "Place a small red flower in the foreground and preserve the rest of the scene.",
862
+ "seed": 10062,
863
+ "width": 1024,
864
+ "height": 1024,
865
+ "steps": 40,
866
+ "seconds_with_collection": 11.716953598021064,
867
+ "editing": true
868
+ },
869
+ {
870
+ "index": 63,
871
+ "prompt": "Give the scene soft cinematic moonlight while preserving its composition.",
872
+ "seed": 10063,
873
+ "width": 1024,
874
+ "height": 1024,
875
+ "steps": 40,
876
+ "seconds_with_collection": 11.728258827002719,
877
+ "editing": true
878
+ }
879
+ ],
880
+ "collection": "BF16 teacher trajectories, 4 token rows at each sampled timestep"
881
+ }
reports/calibration_search.json ADDED
The diff for this file is too large to render. See raw diff
 
reports/correction-probe.json ADDED
@@ -0,0 +1,122 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "name": "transformer_blocks.0.attn.to_q",
4
+ "reg": "original",
5
+ "fit": 0.012696055695414543,
6
+ "check": 0.0126740587875247
7
+ },
8
+ {
9
+ "name": "transformer_blocks.0.attn.to_q",
10
+ "reg": 0.01,
11
+ "fit": 0.008233755826950073,
12
+ "check": 0.012077884748578072
13
+ },
14
+ {
15
+ "name": "transformer_blocks.0.attn.to_q",
16
+ "reg": 0.1,
17
+ "fit": 0.008387493900954723,
18
+ "check": 0.011893227696418762
19
+ },
20
+ {
21
+ "name": "transformer_blocks.0.attn.to_q",
22
+ "reg": 1.0,
23
+ "fit": 0.00852095615118742,
24
+ "check": 0.01182559598237276
25
+ },
26
+ {
27
+ "name": "transformer_blocks.8.attn.to_v",
28
+ "reg": "original",
29
+ "fit": 0.08148191124200821,
30
+ "check": 0.08202973008155823
31
+ },
32
+ {
33
+ "name": "transformer_blocks.8.attn.to_v",
34
+ "reg": 0.01,
35
+ "fit": 0.0604797899723053,
36
+ "check": 0.0942867323756218
37
+ },
38
+ {
39
+ "name": "transformer_blocks.8.attn.to_v",
40
+ "reg": 0.1,
41
+ "fit": 0.06049658730626106,
42
+ "check": 0.09427530318498611
43
+ },
44
+ {
45
+ "name": "transformer_blocks.8.attn.to_v",
46
+ "reg": 1.0,
47
+ "fit": 0.06077614054083824,
48
+ "check": 0.094297856092453
49
+ },
50
+ {
51
+ "name": "transformer_blocks.16.img_mlp.proj",
52
+ "reg": "original",
53
+ "fit": 0.05309119075536728,
54
+ "check": 0.05289224535226822
55
+ },
56
+ {
57
+ "name": "transformer_blocks.16.img_mlp.proj",
58
+ "reg": 0.01,
59
+ "fit": 0.040202125906944275,
60
+ "check": 0.06128205358982086
61
+ },
62
+ {
63
+ "name": "transformer_blocks.16.img_mlp.proj",
64
+ "reg": 0.1,
65
+ "fit": 0.04021118953824043,
66
+ "check": 0.06127419322729111
67
+ },
68
+ {
69
+ "name": "transformer_blocks.16.img_mlp.proj",
70
+ "reg": 1.0,
71
+ "fit": 0.040364813059568405,
72
+ "check": 0.061265550553798676
73
+ },
74
+ {
75
+ "name": "transformer_blocks.16.img_mlp.gate_layer",
76
+ "reg": "original",
77
+ "fit": 0.05491224303841591,
78
+ "check": 0.05476883798837662
79
+ },
80
+ {
81
+ "name": "transformer_blocks.16.img_mlp.gate_layer",
82
+ "reg": 0.01,
83
+ "fit": 0.04161699116230011,
84
+ "check": 0.0634395033121109
85
+ },
86
+ {
87
+ "name": "transformer_blocks.16.img_mlp.gate_layer",
88
+ "reg": 0.1,
89
+ "fit": 0.0416269451379776,
90
+ "check": 0.06343034654855728
91
+ },
92
+ {
93
+ "name": "transformer_blocks.16.img_mlp.gate_layer",
94
+ "reg": 1.0,
95
+ "fit": 0.04178332909941673,
96
+ "check": 0.06342266499996185
97
+ },
98
+ {
99
+ "name": "transformer_blocks.24.attn.to_out.0",
100
+ "reg": "original",
101
+ "fit": 0.06296917796134949,
102
+ "check": 0.06366930156946182
103
+ },
104
+ {
105
+ "name": "transformer_blocks.24.attn.to_out.0",
106
+ "reg": 0.01,
107
+ "fit": 0.04285139590501785,
108
+ "check": 0.071813203394413
109
+ },
110
+ {
111
+ "name": "transformer_blocks.24.attn.to_out.0",
112
+ "reg": 0.1,
113
+ "fit": 0.04286669194698334,
114
+ "check": 0.0718073919415474
115
+ },
116
+ {
117
+ "name": "transformer_blocks.24.attn.to_out.0",
118
+ "reg": 1.0,
119
+ "fit": 0.04305445775389671,
120
+ "check": 0.0718240886926651
121
+ }
122
+ ]
reports/cudagraph-trial.json ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "baseline_latent_only_seconds": 4.459490508015733,
3
+ "warmup": 12.883338742016349,
4
+ "candidate_latent_only_seconds": [
5
+ 4.45742104400415,
6
+ 4.47657939302735,
7
+ 4.488048463012092
8
+ ],
9
+ "latent_nrmse": 0.0
10
+ }
reports/earlier-bf16-benchmark.json ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "torch": "2.14.0+cu130",
3
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
4
+ "capability": [
5
+ 12,
6
+ 0
7
+ ],
8
+ "fp8_linears": 0,
9
+ "transformer_dtypes": {
10
+ "torch.bfloat16": 297
11
+ },
12
+ "resident_gb": 32.457959936,
13
+ "precision": "bf16",
14
+ "timing": {
15
+ "1024": {
16
+ "seconds": [
17
+ 9.784915568016004,
18
+ 9.794134593976196,
19
+ 9.79820777400164,
20
+ 9.79243299702648,
21
+ 9.791788258997258
22
+ ],
23
+ "mean": 9.792295838403515,
24
+ "median": 9.79243299702648,
25
+ "min": 9.784915568016004,
26
+ "max": 9.79820777400164
27
+ },
28
+ "2048": {
29
+ "seconds": [
30
+ 55.14897009101696,
31
+ 55.18838875496294,
32
+ 55.170718070003204
33
+ ],
34
+ "mean": 55.16935897199437,
35
+ "median": 55.170718070003204,
36
+ "min": 55.14897009101696,
37
+ "max": 55.18838875496294
38
+ }
39
+ },
40
+ "protocol": "Batch1;40steps;CFG1;prefixKVcache;fully GPU resident;warmup excluded;CUDA synchronized;includes text encoding, denoising and VAE decode;excludes load and PNG write."
41
+ }
reports/environment.json ADDED
@@ -0,0 +1,95 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "python": "3.12.3 (main, Aug 31 2026, 10:18:26) [GCC 13.3.0]",
3
+ "platform": "Linux-7.0.0-31-generic-x86_64-with-glibc2.39",
4
+ "packages": {
5
+ "Jinja2": "3.1.6",
6
+ "MarkupSafe": "3.0.3",
7
+ "PyYAML": "6.0.3",
8
+ "Pygments": "2.21.0",
9
+ "accelerate": "1.15.0",
10
+ "annotated-doc": "0.0.5",
11
+ "anyio": "4.15.1",
12
+ "apache-tvm-ffi": "0.1.14.post0",
13
+ "certifi": "2026.7.22",
14
+ "charset-normalizer": "3.5.1",
15
+ "click": "8.5.0",
16
+ "cuda-bindings": "13.4.2",
17
+ "cuda-core": "1.2.0",
18
+ "cuda-pathfinder": "1.6.0",
19
+ "cuda-python": "13.4.1",
20
+ "cuda-tile": "1.6.0",
21
+ "cuda-toolkit": "13.0.3.0",
22
+ "diffusers": "0.41.0.dev0",
23
+ "einops": "0.8.2",
24
+ "filelock": "3.32.3",
25
+ "flashinfer-python": "0.7.0",
26
+ "fsspec": "2026.7.0",
27
+ "h11": "0.16.0",
28
+ "hf-xet": "1.6.0",
29
+ "httpcore": "1.0.9",
30
+ "httpx": "0.28.1",
31
+ "huggingface_hub": "1.32.0",
32
+ "idna": "3.20",
33
+ "imageio-ffmpeg": "0.6.0",
34
+ "importlib_metadata": "9.0.1",
35
+ "markdown-it-py": "4.2.0",
36
+ "mdurl": "0.1.2",
37
+ "mpmath": "1.3.0",
38
+ "nccl-extensions": "0.1.0",
39
+ "nccl4py": "0.5.0",
40
+ "networkx": "3.6.1",
41
+ "ninja": "1.13.2",
42
+ "numpy": "2.5.2",
43
+ "nvidia-cublas": "13.1.1.3",
44
+ "nvidia-cuda-cupti": "13.0.85",
45
+ "nvidia-cuda-nvdisasm": "13.4.92",
46
+ "nvidia-cuda-nvrtc": "13.0.88",
47
+ "nvidia-cuda-runtime": "13.0.96",
48
+ "nvidia-cudnn-cu13": "9.24.0.43",
49
+ "nvidia-cudnn-frontend": "1.29.0",
50
+ "nvidia-cufft": "12.0.0.61",
51
+ "nvidia-cufile": "1.15.1.6",
52
+ "nvidia-curand": "10.4.0.35",
53
+ "nvidia-cusolver": "12.0.4.66",
54
+ "nvidia-cusparse": "12.6.3.3",
55
+ "nvidia-cusparselt-cu13": "0.8.1",
56
+ "nvidia-cutlass-dsl": "4.7.1",
57
+ "nvidia-cutlass-dsl-libs-base": "4.7.1",
58
+ "nvidia-cutlass-dsl-libs-core": "4.7.1",
59
+ "nvidia-cutlass-dsl-libs-cu12": "4.7.1",
60
+ "nvidia-cutlass-dsl-libs-cu13": "4.7.1",
61
+ "nvidia-ml-py": "13.610.43",
62
+ "nvidia-nccl-cu13": "2.30.7",
63
+ "nvidia-nvjitlink": "13.3.33",
64
+ "nvidia-nvshmem-cu13": "3.4.5",
65
+ "nvidia-nvtx": "13.0.85",
66
+ "packaging": "26.3",
67
+ "pillow": "12.3.0",
68
+ "protobuf": "7.36.2",
69
+ "psutil": "7.2.2",
70
+ "regex": "2026.9.10",
71
+ "requests": "2.34.2",
72
+ "rich": "15.0.0",
73
+ "safetensors": "0.8.0",
74
+ "sentencepiece": "0.2.2",
75
+ "setuptools": "78.1.0",
76
+ "shellingham": "1.5.4",
77
+ "sympy": "1.14.0",
78
+ "tabulate": "0.10.0",
79
+ "tokenizers": "0.23.2",
80
+ "torch": "2.14.0+cu130",
81
+ "torchvision": "0.29.0+cu130",
82
+ "tqdm": "4.70.1",
83
+ "transformers": "5.17.0",
84
+ "triton": "3.8.0",
85
+ "typer": "0.27.2",
86
+ "typing_extensions": "4.16.0",
87
+ "urllib3": "2.8.0",
88
+ "zipp": "4.1.0"
89
+ },
90
+ "flashinfer_commit": "975f90583d9ac8896db14cf0f26e99a853c2f136",
91
+ "diffusers_commit": "80c7ed262aeffbeb43ef13ae04baeb9b84515a69",
92
+ "gpu": "RTX PRO 6000 Blackwell Workstation Edition",
93
+ "driver": "615.71.09",
94
+ "compute_capability": "12.0"
95
+ }
reports/evaluation-environment.json ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "torch": "2.14.0",
3
+ "torchvision": "0.29.0",
4
+ "lpips": "0.1.4",
5
+ "scikit-image": "0.26.0",
6
+ "scipy": "1.18.1",
7
+ "numpy": "2.5.3",
8
+ "Pillow": "12.3.0"
9
+ }
reports/fresh-fp8-benchmark.json ADDED
@@ -0,0 +1,31 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "load_seconds": 2.801166548044421,
3
+ "torch": "2.14.0+cu130",
4
+ "gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
5
+ "fp8_linears": 224,
6
+ "steps": 40,
7
+ "cfg": 1,
8
+ "extra_quantization": false,
9
+ "approximate_cache": false,
10
+ "timing": {
11
+ "1024": {
12
+ "seconds": [
13
+ 5.894081406004261,
14
+ 5.925464586995076
15
+ ],
16
+ "mean": 5.909772996499669,
17
+ "warmup_seconds": 10.963575308967847,
18
+ "peak_gb": 32.590829568
19
+ },
20
+ "2048": {
21
+ "seconds": [
22
+ 37.19529012899147,
23
+ 37.23217404395109
24
+ ],
25
+ "mean": 37.21373208647128,
26
+ "warmup_seconds": 40.54065499798162,
27
+ "peak_gb": 53.722432512
28
+ }
29
+ },
30
+ "protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
31
+ }
reports/heldout-manifest.json ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "frozen_after_candidate_selection": true,
3
+ "cases": [
4
+ {
5
+ "index": 0,
6
+ "prompt": "A transparent glass teapot filled with amber tea on a slate table beside sliced dragon fruit, soft window light, crisp reflections, studio photograph, no text.",
7
+ "width": 1024,
8
+ "height": 1024,
9
+ "seed": 62000,
10
+ "steps": 40
11
+ },
12
+ {
13
+ "index": 1,
14
+ "prompt": "Three origami cranes arranged in a row on a pale wooden desk: a red crane on the left, a yellow crane in the center, and a blue crane on the right, precise folded paper, no text.",
15
+ "width": 1024,
16
+ "height": 1024,
17
+ "seed": 62001,
18
+ "steps": 40
19
+ },
20
+ {
21
+ "index": 2,
22
+ "prompt": "An elderly pianist playing a black grand piano in a warmly lit room, both hands visible on the keys, realistic fingers, candid documentary photograph, no text.",
23
+ "width": 1024,
24
+ "height": 1024,
25
+ "seed": 62002,
26
+ "steps": 40
27
+ },
28
+ {
29
+ "index": 3,
30
+ "prompt": "A minimalist travel poster with the exact large headline \"SUMMER 2026\", a golden sun above a turquoise sea, elegant bold typography.",
31
+ "width": 1024,
32
+ "height": 1024,
33
+ "seed": 62003,
34
+ "steps": 40
35
+ },
36
+ {
37
+ "index": 4,
38
+ "prompt": "一张精美的中国山水海报,清晰准确的四字标题“山海之间”,远山、碧海和细腻的水墨纹理,优雅留白。",
39
+ "width": 1024,
40
+ "height": 1024,
41
+ "seed": 62004,
42
+ "steps": 40
43
+ },
44
+ {
45
+ "index": 5,
46
+ "prompt": "An intricately engraved brass mechanical dragon sculpture on a dark pedestal, delicate interlocking gears, polished metal highlights, museum product photograph, no text.",
47
+ "width": 2048,
48
+ "height": 2048,
49
+ "seed": 62005,
50
+ "steps": 40
51
+ },
52
+ {
53
+ "index": 6,
54
+ "prompt": "This is an RGBA image with transparency. A charming illustrated red panda holding a small green bamboo leaf, clean outlines, fluffy striped tail. The image has alpha channel and the background is transparent.",
55
+ "width": 1024,
56
+ "height": 1024,
57
+ "seed": 62006,
58
+ "steps": 40
59
+ },
60
+ {
61
+ "index": 7,
62
+ "prompt": "A close-up wildlife photograph of a barn owl on a weathered wooden fence, fine speckled feathers, sharp dark eyes, softly blurred spring meadow in the background, no text.",
63
+ "width": 1024,
64
+ "height": 1024,
65
+ "seed": 62007,
66
+ "steps": 40
67
+ }
68
+ ]
69
+ }
reports/kernel_evidence.json ADDED
@@ -0,0 +1,81 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "nvfp4_sm120_launches": 224,
3
+ "fp8_sm120_launches": 0,
4
+ "native_bf16_flash_attention": 32,
5
+ "all_kernels": {
6
+ "void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8::Params)": 1,
7
+ "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 3,
8
+ "void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
9
+ "void at::native::reduce_kernel<512, 1, at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4> >(at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4>)": 1,
10
+ "void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>)": 2,
11
+ "void at::native::vectorized_elementwise_kernel<4, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
12
+ "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1})": 3,
13
+ "void at::native::vectorized_elementwise_kernel<4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>)": 2,
14
+ "void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8::Params)": 2,
15
+ "void cublasLt::splitKreduce_kernel<32, 16, int, __nv_bfloat16, __nv_bfloat16, float, __nv_bfloat16, false, __nv_bfloat16, __nv_bfloat16, __nv_bfloat16, true, false, false, false>(cublasLt::cublasSplitKParams<float>, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, __nv_bfloat16*, float const*, float const*, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, void*, long, float*, int*, float*, float*, float const*, float const*, float const*, float const*, float const*)": 227,
16
+ "void at::native::vectorized_elementwise_kernel<4, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 1,
17
+ "void at::native::vectorized_elementwise_kernel<2, at::native::FillFunctor<long>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<long>, std::array<char*, 1ul>)": 3,
18
+ "void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1})": 1,
19
+ "void at_cuda_detail::cub::detail::scan::DeviceScanInitKernel<at_cuda_detail::cub::ScanTileState<long, true> >(at_cuda_detail::cub::ScanTileState<long, true>, int)": 3,
20
+ "void at_cuda_detail::cub::detail::scan::DeviceScanKernel<at_cuda_detail::cub::detail::scan::policy_hub<long, long, long, unsigned int, std::plus<long> >::Policy1000, long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int, long, false, at_cuda_detail::cub::NullType>(long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, int, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int)": 3,
21
+ "Memcpy DtoH (Device -> Pinned)": 11,
22
+ "void at::native::vectorized_elementwise_kernel<4, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>, false>(int, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>)": 3,
23
+ "void at::native::reduce_kernel<512, 1, at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4> >(at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4>)": 2,
24
+ "void compute_cuda_kernel<long>(long const*, long const*, long*, long, long)": 3,
25
+ "void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
26
+ "void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>)": 2,
27
+ "void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 2, 128, 1, 16, 8>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
28
+ "void at::native::vectorized_gather_kernel<16, long>(char*, char*, long*, int, long, long, long, long, bool)": 4,
29
+ "void at_cuda_detail::cub::detail::reduce::DeviceReduceKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, unsigned long long, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, int*, unsigned long long, at_cuda_detail::cub::GridEvenShare<unsigned long long>, cuda::std::__4::plus<void>, cuda::std::__4::__identity)": 4,
30
+ "void at_cuda_detail::cub::detail::reduce::DeviceReduceSingleTileKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, int*, int*, int, cuda::std::__4::plus<void>, int, int, cuda::std::__4::__identity>(int*, int*, int, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity)": 4,
31
+ "void at_cuda_detail::cub::detail::scan::DeviceCompactInitKernel<at_cuda_detail::cub::ScanTileState<int, true>, int*>(at_cuda_detail::cub::ScanTileState<int, true>, int, int*)": 4,
32
+ "void at_cuda_detail::cub::detail::select::DeviceSelectSweepKernel<at_cuda_detail::cub::detail::select::policy_hub<long, bool, int, false, (at_cuda_detail::cub::SelectImpl)0>::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, (at_cuda_detail::cub::SelectImpl)0>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, at_cuda_detail::cub::detail::vsmem_t)": 4,
33
+ "void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
34
+ "Memcpy DtoH (Device -> Pageable)": 1,
35
+ "Memcpy HtoD (Pageable -> Device)": 5,
36
+ "Memcpy DtoD (Device -> Device)": 3,
37
+ "void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 3,
38
+ "void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 2, 128, 1, 16, 2>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
39
+ "void (anonymous namespace)::elementwise_kernel_with_index<int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>(int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, function_traits<at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>::result_type*)": 1,
40
+ "void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
41
+ "void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<bool>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<bool>, std::array<char*, 1ul>)": 1,
42
+ "void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
43
+ "void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 1, 128, 1, 8>(at::native::(anonymous namespace)::OpaqueType<2u>*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
44
+ "void at::native::vectorized_elementwise_kernel<4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>, false>(int, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>)": 1,
45
+ "void at::native::vectorized_elementwise_kernel<4, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
46
+ "void at::native::vectorized_elementwise_kernel<4, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
47
+ "void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 2, 128, 1, 16, 4>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
48
+ "void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8::Params)": 1,
49
+ "void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 3,
50
+ "Memset (Device)": 3,
51
+ "void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8::Params)": 2,
52
+ "void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8::Params)": 1,
53
+ "void at::native::vectorized_elementwise_kernel<4, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>, false>(int, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>)": 1,
54
+ "void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 1,
55
+ "void at::native::reduce_kernel<512, 1, at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4> >(at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4>)": 1,
56
+ "triton_red_fused_add_mul_native_layer_norm_slice_split_unsqueeze_0": 32,
57
+ "_partials": 224,
58
+ "_finish": 224,
59
+ "_upscale": 224,
60
+ "kernel_cutlass_kernel_flashinferquantizationkernelsnvfp4_quantizeNVFP4QuantizeSwizzledKernel_object_at__tensorptrbf16gmemalign16o409640961_tensorptri8gmemalign16o204820481_tensorptri8gmem_0": 192,
61
+ "void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x128_32x4_nn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x128_32x4_nn_align8::Params)": 224,
62
+ "kernel_cutlass_kernel_flashinfergemmkernelsdense_blockscaled_gemm_sm120_b12xDenseGemmKernel_object_at__CopyAtom_ThrID10_TVLayoutSrc11638401_TVLayoutDst11638401_Valuetypef4E2M1FN_tensor00o_0": 224,
63
+ "triton_per_fused__to_copy_mean_pow_view_1": 64,
64
+ "triton_poi_fused__to_copy_add_mean_mul_pow_rsqrt_select_sub_unsqueeze_view_2": 64,
65
+ "triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_3": 32,
66
+ "triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_4": 32,
67
+ "triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_5": 32,
68
+ "void pytorch_flash::flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, false, false, cutlass::bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, cutlass::bfloat16_t> >, false, false, false, false, false, true, false, false>(pytorch_flash::Flash_fwd_params)": 32,
69
+ "triton_red_fused_add_linear_mul_native_layer_norm_slice_split_tanh_unsqueeze_view_6": 32,
70
+ "triton_poi_fused_linear_mul_silu_view_7": 32,
71
+ "kernel_cutlass_kernel_flashinferquantizationkernelsnvfp4_quantizeNVFP4QuantizeSwizzledKernel_object_at__tensorptrbf16gmemalign16o12288122881_tensorptri8gmemalign16o614461441_tensorptri8gm_0": 32,
72
+ "triton_poi_fused_add_mul_slice_split_tanh_unsqueeze_view_8": 32,
73
+ "void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1})": 1,
74
+ "void at::native::(anonymous namespace)::vectorized_layer_norm_kernel<c10::BFloat16, float, false>(int, float, c10::BFloat16 const*, c10::BFloat16 const*, c10::BFloat16 const*, float*, float*, c10::BFloat16*)": 1,
75
+ "void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>)": 1,
76
+ "void at::native::vectorized_elementwise_kernel<4, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>, false>(int, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>)": 1,
77
+ "void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8::Params)": 1
78
+ },
79
+ "transformer_calls": 40,
80
+ "full_denoising_steps": 40
81
+ }