Instructions to use ProCreations/Image-2.1-Calibrated-NVFP4 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Diffusers
How to use ProCreations/Image-2.1-Calibrated-NVFP4 with Diffusers:
pip install -U diffusers transformers accelerate
import torch from diffusers import DiffusionPipeline # switch to "mps" for apple devices pipe = DiffusionPipeline.from_pretrained("ProCreations/Image-2.1-Calibrated-NVFP4", dtype=torch.bfloat16, device_map="cuda") prompt = "Astronaut in a jungle, cold color palette, muted colors, detailed, 8k" image = pipe(prompt).images[0] - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- Draw Things
- DiffusionBee
Release calibrated Image2.1 NVFP4 transformer with dynamic scaling and BF16 rank correction, native SM120 runtime, quality evidence and real-time demo
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- .gitattributes +49 -0
- LICENSE +55 -0
- Notice +4 -0
- README.md +86 -0
- acceleration.py +91 -0
- comparisons/evaluation-dynamic/00.jpg +3 -0
- comparisons/evaluation-dynamic/01.jpg +3 -0
- comparisons/evaluation-dynamic/02.jpg +3 -0
- comparisons/evaluation-dynamic/03.jpg +3 -0
- comparisons/evaluation-dynamic/04.jpg +3 -0
- comparisons/evaluation-dynamic/05.jpg +3 -0
- comparisons/evaluation-dynamic/06.jpg +3 -0
- comparisons/evaluation-dynamic/07.jpg +3 -0
- comparisons/evaluation-dynamic/08.jpg +3 -0
- comparisons/evaluation-dynamic/09.jpg +3 -0
- comparisons/evaluation-dynamic/10.jpg +0 -0
- comparisons/evaluation-dynamic/11.jpg +3 -0
- comparisons/evaluation-dynamic/12.jpg +3 -0
- comparisons/evaluation-dynamic/13.jpg +0 -0
- comparisons/evaluation-dynamic/14.jpg +3 -0
- comparisons/evaluation-dynamic/15.jpg +3 -0
- comparisons/evaluation-dynamic/edit-0.jpg +3 -0
- comparisons/evaluation-dynamic/edit-1.jpg +3 -0
- comparisons/heldout-nvfp4/00.jpg +3 -0
- comparisons/heldout-nvfp4/01.jpg +0 -0
- comparisons/heldout-nvfp4/02.jpg +3 -0
- comparisons/heldout-nvfp4/03.jpg +3 -0
- comparisons/heldout-nvfp4/04.jpg +3 -0
- comparisons/heldout-nvfp4/05.jpg +3 -0
- comparisons/heldout-nvfp4/06.jpg +0 -0
- comparisons/heldout-nvfp4/07.jpg +3 -0
- demo/capture_receipt.json +0 -0
- demo/realtime-30s.mp4 +3 -0
- dynamic_scale.py +29 -0
- fp8_runtime.py +104 -0
- generate.py +54 -0
- install.sh +12 -0
- manifest.json +510 -0
- nvfp4_runtime.py +81 -0
- reports/benchmark.json +36 -0
- reports/calibration_manifest.json +881 -0
- reports/calibration_search.json +0 -0
- reports/correction-probe.json +122 -0
- reports/cudagraph-trial.json +10 -0
- reports/earlier-bf16-benchmark.json +41 -0
- reports/environment.json +95 -0
- reports/evaluation-environment.json +9 -0
- reports/fresh-fp8-benchmark.json +31 -0
- reports/heldout-manifest.json +69 -0
- reports/kernel_evidence.json +81 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,52 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
comparisons/evaluation-dynamic/00.jpg filter=lfs diff=lfs merge=lfs -text
|
| 37 |
+
comparisons/evaluation-dynamic/01.jpg filter=lfs diff=lfs merge=lfs -text
|
| 38 |
+
comparisons/evaluation-dynamic/02.jpg filter=lfs diff=lfs merge=lfs -text
|
| 39 |
+
comparisons/evaluation-dynamic/03.jpg filter=lfs diff=lfs merge=lfs -text
|
| 40 |
+
comparisons/evaluation-dynamic/04.jpg filter=lfs diff=lfs merge=lfs -text
|
| 41 |
+
comparisons/evaluation-dynamic/05.jpg filter=lfs diff=lfs merge=lfs -text
|
| 42 |
+
comparisons/evaluation-dynamic/06.jpg filter=lfs diff=lfs merge=lfs -text
|
| 43 |
+
comparisons/evaluation-dynamic/07.jpg filter=lfs diff=lfs merge=lfs -text
|
| 44 |
+
comparisons/evaluation-dynamic/08.jpg filter=lfs diff=lfs merge=lfs -text
|
| 45 |
+
comparisons/evaluation-dynamic/09.jpg filter=lfs diff=lfs merge=lfs -text
|
| 46 |
+
comparisons/evaluation-dynamic/11.jpg filter=lfs diff=lfs merge=lfs -text
|
| 47 |
+
comparisons/evaluation-dynamic/12.jpg filter=lfs diff=lfs merge=lfs -text
|
| 48 |
+
comparisons/evaluation-dynamic/14.jpg filter=lfs diff=lfs merge=lfs -text
|
| 49 |
+
comparisons/evaluation-dynamic/15.jpg filter=lfs diff=lfs merge=lfs -text
|
| 50 |
+
comparisons/evaluation-dynamic/edit-0.jpg filter=lfs diff=lfs merge=lfs -text
|
| 51 |
+
comparisons/evaluation-dynamic/edit-1.jpg filter=lfs diff=lfs merge=lfs -text
|
| 52 |
+
comparisons/heldout-nvfp4/00.jpg filter=lfs diff=lfs merge=lfs -text
|
| 53 |
+
comparisons/heldout-nvfp4/02.jpg filter=lfs diff=lfs merge=lfs -text
|
| 54 |
+
comparisons/heldout-nvfp4/03.jpg filter=lfs diff=lfs merge=lfs -text
|
| 55 |
+
comparisons/heldout-nvfp4/04.jpg filter=lfs diff=lfs merge=lfs -text
|
| 56 |
+
comparisons/heldout-nvfp4/05.jpg filter=lfs diff=lfs merge=lfs -text
|
| 57 |
+
comparisons/heldout-nvfp4/07.jpg filter=lfs diff=lfs merge=lfs -text
|
| 58 |
+
demo/realtime-30s.mp4 filter=lfs diff=lfs merge=lfs -text
|
| 59 |
+
samples/heldout/00.png filter=lfs diff=lfs merge=lfs -text
|
| 60 |
+
samples/heldout/01.png filter=lfs diff=lfs merge=lfs -text
|
| 61 |
+
samples/heldout/02.png filter=lfs diff=lfs merge=lfs -text
|
| 62 |
+
samples/heldout/03.png filter=lfs diff=lfs merge=lfs -text
|
| 63 |
+
samples/heldout/04.png filter=lfs diff=lfs merge=lfs -text
|
| 64 |
+
samples/heldout/05.png filter=lfs diff=lfs merge=lfs -text
|
| 65 |
+
samples/heldout/06.png filter=lfs diff=lfs merge=lfs -text
|
| 66 |
+
samples/heldout/07.png filter=lfs diff=lfs merge=lfs -text
|
| 67 |
+
samples/validation/00.png filter=lfs diff=lfs merge=lfs -text
|
| 68 |
+
samples/validation/01.png filter=lfs diff=lfs merge=lfs -text
|
| 69 |
+
samples/validation/02.png filter=lfs diff=lfs merge=lfs -text
|
| 70 |
+
samples/validation/03.png filter=lfs diff=lfs merge=lfs -text
|
| 71 |
+
samples/validation/04.png filter=lfs diff=lfs merge=lfs -text
|
| 72 |
+
samples/validation/05.png filter=lfs diff=lfs merge=lfs -text
|
| 73 |
+
samples/validation/06.png filter=lfs diff=lfs merge=lfs -text
|
| 74 |
+
samples/validation/07.png filter=lfs diff=lfs merge=lfs -text
|
| 75 |
+
samples/validation/08.png filter=lfs diff=lfs merge=lfs -text
|
| 76 |
+
samples/validation/09.png filter=lfs diff=lfs merge=lfs -text
|
| 77 |
+
samples/validation/10.png filter=lfs diff=lfs merge=lfs -text
|
| 78 |
+
samples/validation/11.png filter=lfs diff=lfs merge=lfs -text
|
| 79 |
+
samples/validation/12.png filter=lfs diff=lfs merge=lfs -text
|
| 80 |
+
samples/validation/13.png filter=lfs diff=lfs merge=lfs -text
|
| 81 |
+
samples/validation/14.png filter=lfs diff=lfs merge=lfs -text
|
| 82 |
+
samples/validation/15.png filter=lfs diff=lfs merge=lfs -text
|
| 83 |
+
samples/validation/edit-0.png filter=lfs diff=lfs merge=lfs -text
|
| 84 |
+
samples/validation/edit-1.png filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Qwen RESEARCH LICENSE AGREEMENT
|
| 2 |
+
|
| 3 |
+
Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
|
| 4 |
+
|
| 5 |
+
By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
|
| 6 |
+
|
| 7 |
+
1. Definitions
|
| 8 |
+
a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
|
| 9 |
+
b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
|
| 10 |
+
c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
|
| 11 |
+
d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
|
| 12 |
+
e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
|
| 13 |
+
f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
|
| 14 |
+
g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
|
| 15 |
+
h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
|
| 16 |
+
i. "Non-Commercial" shall mean for research or evaluation purposes only.
|
| 17 |
+
|
| 18 |
+
2. Grant of Rights
|
| 19 |
+
a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
|
| 20 |
+
b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
|
| 21 |
+
|
| 22 |
+
3. Redistribution
|
| 23 |
+
Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
|
| 24 |
+
a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
|
| 25 |
+
b. You shall cause any modified files to carry prominent notices stating that you changed the files;
|
| 26 |
+
c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
|
| 27 |
+
d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
|
| 28 |
+
|
| 29 |
+
4. Rules of use
|
| 30 |
+
a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
|
| 31 |
+
b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
|
| 32 |
+
c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
|
| 33 |
+
|
| 34 |
+
5. Intellectual Property
|
| 35 |
+
a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
|
| 36 |
+
b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
|
| 37 |
+
c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
|
| 38 |
+
|
| 39 |
+
6. Disclaimer of Warranty and Limitation of Liability
|
| 40 |
+
a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
|
| 41 |
+
b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
|
| 42 |
+
c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
|
| 43 |
+
d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
|
| 44 |
+
|
| 45 |
+
7. Survival and Termination.
|
| 46 |
+
a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
|
| 47 |
+
b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
|
| 48 |
+
|
| 49 |
+
8. Governing Law and Jurisdiction.
|
| 50 |
+
a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
|
| 51 |
+
b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
|
| 52 |
+
|
| 53 |
+
9. Other Terms and Conditions.
|
| 54 |
+
a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
|
| 55 |
+
b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
|
Notice
ADDED
|
@@ -0,0 +1,4 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
|
| 2 |
+
|
| 3 |
+
Built with Qwen.
|
| 4 |
+
ProCreations modified the original BF16 transformer through calibrated NVFP4 quantization with BF16 low-rank corrections and dynamic activation scaling on 2026-09-20. The custom runtime and calibration/evaluation scripts were added by ProCreations. Non-commercial research and evaluation under the included upstream license.
|
README.md
ADDED
|
@@ -0,0 +1,86 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: other
|
| 3 |
+
license_name: qwen-research
|
| 4 |
+
license_link: LICENSE
|
| 5 |
+
base_model: Qwen/Qwen-Image-2.1
|
| 6 |
+
library_name: diffusers
|
| 7 |
+
pipeline_tag: text-to-image
|
| 8 |
+
tags:
|
| 9 |
+
- nvfp4
|
| 10 |
+
- svdquant
|
| 11 |
+
- blackwell
|
| 12 |
+
- sm120
|
| 13 |
+
- image-editing
|
| 14 |
+
---
|
| 15 |
+
# Image 2.1 Calibrated NVFP4
|
| 16 |
+
|
| 17 |
+
Built with Qwen. A calibrated NVFP4 transformer derived from `Qwen/Qwen-Image-2.1` at revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`.
|
| 18 |
+
|
| 19 |
+
**4.87 GB transformer; native SM120 W4A4 Tensor Core kernels; all 40 denoising steps.** Encoder, VAE, attention, conditioning and small projections retain BF16. Each of 224 large projections uses NVFP4 plus a BF16 rank-128 correction, accumulated together in FP32. This is a custom Diffusers/FlashInfer format and requires the included loader.
|
| 20 |
+
|
| 21 |
+
## Measured speed
|
| 22 |
+
|
| 23 |
+
RTX PRO 6000 Blackwell Workstation 96 GB, batch 1, CFG 1, prefix KV cache. Mean CUDA-synchronized wall time includes text encoding, all denoising and VAE decoding. Loading, one full warmup per resolution, compilation and PNG writing are excluded. NVFP4 uses five measured 1024 runs and three 2048 runs; fresh FP8 controls use two per resolution in the same clean environment.
|
| 24 |
+
|
| 25 |
+
| Resolution | This NVFP4 | Fresh optimized FP8 control | Speedup |
|
| 26 |
+
|---|---:|---:|---:|
|
| 27 |
+
| 1024 × 1024 | 4.589 s/image | 5.910 s/image | 1.29× |
|
| 28 |
+
| 2048 × 2048 | 32.658 s/image | 37.214 s/image | 1.14× |
|
| 29 |
+
|
| 30 |
+
Earlier BF16 measurements on this workstation with the same generation settings were 9.792 and 55.169 seconds respectively. These are historical controls from the same project, not newly timed in this release. First-use loading and compilation cost extra; full warmup times are in [benchmark.json](reports/benchmark.json). Performance depends on resolution and prompt. No steps are skipped, no approximate feature reuse is used, and attention is not quantized.
|
| 31 |
+
|
| 32 |
+
The [30-second real-time video](demo/realtime-30s.mp4) has no text, overlays or audio. It begins with one completed warmup image, then displays actual newly completed generations with all waits preserved. The [capture log](demo/capture_receipt.json) records every display update and frame timestamp.
|
| 33 |
+
|
| 34 |
+
## Quality and calibration
|
| 35 |
+
|
| 36 |
+
Calibration uses 64 original BF16 trajectories: 56 text-to-image and eight edits, with 1024, 2048 and alternate aspect ratios. Each projection has 1,536 sampled activation rows across six denoising timesteps; 384 stratified fit rows and 384 disjoint diagnostic rows. Five smoothing exponents and two weight-scale methods are searched using native FP4 kernels. Weight-scale refinement searches 15 per-block scale factors, weighted by activation second moments. Rank-128 SVD corrections absorb dominant weight components. Original BF16 weights are the source; this is not a requantization of FP8.
|
| 37 |
+
|
| 38 |
+
Activations use a **fresh actual tensor-wide amax on every call**, followed by dynamic E4M3 scales per 16 values. Sparse calibration alone missed rare large activations; static activation ranges caused clipping and visible texture degradation, so that candidate was rejected. The final runtime recalibrates and evaluates with dynamic scaling. BF16 correction weights are rescaled consistently when the activation global scale changes.
|
| 39 |
+
|
| 40 |
+
The 18-case development validation set includes 16 generation prompts and two edits. Against BF16, mean LPIPS(AlexNet,512px) is **0.122970**, SSIM **0.897363**, and full-latent cosine **0.976903**. Eight additional prompts were frozen after candidate selection; their mean LPIPS is **0.124012** and SSIM **0.871713**. Full images, paired comparisons and individual measurements are included.
|
| 41 |
+
|
| 42 |
+
**This is lossy quantization, not a zero-quality-loss guarantee.** All 26 pairs were visually reviewed. The corrected candidate preserves readable primary English/Chinese text, transparency, image editing, fine felt/fur textures and plausible hands in these tests. Same-seed composition, poses, decorative marks, geometry and fine detail can change; the largest validation differences are the pottery and train scenes. The prior calibrated FP8 release is closer to BF16 numerically (original 18-case mean LPIPS 0.03447). Fidelity metrics do not establish broad human preference or guarantee every prompt.
|
| 43 |
+
|
| 44 |
+
## Native FP4 evidence
|
| 45 |
+
|
| 46 |
+
An actual denoising-step profile records **224 SM120 block-scaled `f4E2M1FN` GEMM launches**, 32 native BF16 FlashAttention launches and 40 transformer calls per 40-step generation. Packed E2M1 values and E4M3 scale layouts were independently decoded and checked against FP32 matrix products. See [kernel evidence](reports/kernel_evidence.json) and [numerical verification](reports/native-math-verification.json).
|
| 47 |
+
|
| 48 |
+
The engine uses [FlashInfer NVFP4 SVDQuant](https://docs.flashinfer.ai/generated/flashinfer.gemm.mm_nvfp4_svdquant.html), pinned to commit `975f90583d9ac8896db14cf0f26e99a853c2f136`. PyPI 0.6.18 lacked the SM120 fusion shown in the online documentation during development, so use the pinned source revision. Generic Transformers FP8/FP4 loading does not select this custom runtime.
|
| 49 |
+
|
| 50 |
+
## Install and generate
|
| 51 |
+
|
| 52 |
+
Tested on Linux x86_64, Python 3.12, Torch 2.14.0+cu130, CUDA toolkit 13.3, driver 615.71.09, compute capability 12.0. Other hardware is unverified. Download this repository, then run `bash install.sh` from its directory. The script installs pinned dependencies and the pinned FlashInfer source; a recent NVIDIA driver and CUDA toolkit must already be available.
|
| 53 |
+
|
| 54 |
+
```bash
|
| 55 |
+
.venv/bin/python generate.py --prompt 'A kingfisher on a mossy branch, detailed feathers, no text' --width 1024 --height 1024 --steps 40 --warmup --output kingfisher.png
|
| 56 |
+
```
|
| 57 |
+
|
| 58 |
+
`--base /path/to/local/original-model` reuses a local upstream checkpoint. Otherwise the loader downloads the pinned upstream components. `--image input.png` enables editing. `--prompts-json prompts.json` accepts a JSON list and reuses the loaded pipeline. `--eager` disables block compilation. Warmup and loading costs are reported separately; without warmup the reported generation time includes any compilation on that call.
|
| 59 |
+
|
| 60 |
+
```python
|
| 61 |
+
from nvfp4_runtime import load_pipeline
|
| 62 |
+
from acceleration import accelerate_pipeline
|
| 63 |
+
pipe = accelerate_pipeline(load_pipeline('Qwen/Qwen-Image-2.1', './transformer'))
|
| 64 |
+
image = pipe(prompt='A detailed watercolor garden', width=1024, height=1024, num_inference_steps=40).images[0]
|
| 65 |
+
image.save('garden.png')
|
| 66 |
+
```
|
| 67 |
+
|
| 68 |
+
Do not cast the loaded quantized transformer with `.to(dtype=...)`; packed values and FP32 scales must retain their stored types. The normal loader handles device placement. The checkpoint is not a generic Transformers, ComfyUI, TensorRT or Nunchaku checkpoint.
|
| 69 |
+
|
| 70 |
+
## Reproduce calibration
|
| 71 |
+
|
| 72 |
+
Download the exact BF16 upstream revision to a local directory first. The generation runtime environment also supports calibration. From this release directory:
|
| 73 |
+
|
| 74 |
+
```bash
|
| 75 |
+
export IMAGE21_BASE=/absolute/path/to/original-model
|
| 76 |
+
export IMAGE21_WORK_DIR="$PWD/rebuild"
|
| 77 |
+
export PYTHONPATH="$PWD"
|
| 78 |
+
.venv/bin/python source/collect_calibration.py
|
| 79 |
+
.venv/bin/python source/quantize_dynamic.py
|
| 80 |
+
```
|
| 81 |
+
|
| 82 |
+
The first command collects the published 64-case calibration manifest; the second reconstructs the NVFP4 transformer under `rebuild/release/transformer`. Original fitting activations are retained on the build workstation and can be regenerated with these scripts. Evaluation additionally needs LPIPS, scikit-image and SciPy; video capture uses imageio-ffmpeg. [Environment versions](reports/environment.json), source, calibration search, manifests and timing records are included. Earlier static-range, mixed-precision, ridge-correction, kernel-autotuning and CUDA-graph experiments were rejected or provided no useful gain; they are not enabled in this runtime.
|
| 83 |
+
|
| 84 |
+
## License and modification notice
|
| 85 |
+
|
| 86 |
+
Built with Qwen. ProCreations modified the original transformer through calibrated NVFP4 quantization and supplied this custom inference runtime on 2026-09-20. This derivative is distributed under the included [Qwen Research License](LICENSE) and [Notice](Notice), for non-commercial research/evaluation. Preserve those files and the upstream terms when redistributing. This is an independent derivative, not an official Qwen release.
|
acceleration.py
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Full-compute acceleration for Image 2.1 Calibrated NVFP4. Built with Qwen.
|
| 2 |
+
|
| 3 |
+
Keeps every denoising step, BF16 attention, calibrated native FP4 GEMMs with BF16 rank correction and
|
| 4 |
+
FP32 accumulation/scales. No approximate residual cache or attention quantization.
|
| 5 |
+
Only cached-prefix decode blocks compile; prefill keeps upstream behavior.
|
| 6 |
+
"""
|
| 7 |
+
import types
|
| 8 |
+
import torch
|
| 9 |
+
from diffusers.models.transformers.transformer_qwenimage21 import (
|
| 10 |
+
QwenImage21AttnProcessor, QwenImage21TransformerBlock,
|
| 11 |
+
)
|
| 12 |
+
|
| 13 |
+
_ORIGINAL_BLOCK_FORWARD = QwenImage21TransformerBlock.forward
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def _real_rope(x, frequencies):
|
| 17 |
+
paired = x.float().unflatten(-1, (-1, 2))
|
| 18 |
+
cosine = frequencies.real[None, :, None, :]
|
| 19 |
+
sine = frequencies.imag[None, :, None, :]
|
| 20 |
+
return torch.stack((
|
| 21 |
+
paired[..., 0] * cosine - paired[..., 1] * sine,
|
| 22 |
+
paired[..., 0] * sine + paired[..., 1] * cosine,
|
| 23 |
+
), dim=-1).flatten(-2).to(x.dtype)
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
class NativeAttentionProcessor(QwenImage21AttnProcessor):
|
| 27 |
+
def __call__(self, attn, hidden_states, attention_mask=None, rotary_emb=None,
|
| 28 |
+
layer_cache=None, kv_cache_mode=None, cache_write_slice=None,
|
| 29 |
+
segments=None, key_valid=None):
|
| 30 |
+
if kv_cache_mode != 'cached' or attention_mask is not None:
|
| 31 |
+
return super().__call__(attn, hidden_states, attention_mask, rotary_emb,
|
| 32 |
+
layer_cache, kv_cache_mode, cache_write_slice,
|
| 33 |
+
segments, key_valid)
|
| 34 |
+
query = attn.to_q(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 35 |
+
key = attn.to_k(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 36 |
+
value = attn.to_v(hidden_states).unflatten(-1, (attn.heads, -1))
|
| 37 |
+
query = attn.norm_q(query)
|
| 38 |
+
key = attn.norm_k(key)
|
| 39 |
+
if rotary_emb is not None:
|
| 40 |
+
query = _real_rope(query, rotary_emb)
|
| 41 |
+
key = _real_rope(key, rotary_emb)
|
| 42 |
+
cached_key, cached_value = layer_cache.get()
|
| 43 |
+
key = torch.cat((cached_key, key), dim=1)
|
| 44 |
+
value = torch.cat((cached_value, value), dim=1)
|
| 45 |
+
output = torch.nn.functional.scaled_dot_product_attention(
|
| 46 |
+
query.transpose(1, 2), key.transpose(1, 2), value.transpose(1, 2)
|
| 47 |
+
).transpose(1, 2)
|
| 48 |
+
return attn.to_out[1](attn.to_out[0](output.flatten(2, 3)))
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def _decode_block(self, hidden_states, modulation, rotary_emb=None,
|
| 52 |
+
attention_mask=None, target_token_mask=None, layer_cache=None,
|
| 53 |
+
kv_cache_mode=None, cache_write_slice=None, segments=None,
|
| 54 |
+
key_valid=None):
|
| 55 |
+
# Cached decode contains only target image tokens. The final t=0 modulation
|
| 56 |
+
# row belongs to the prefix, which was already evaluated during prefill.
|
| 57 |
+
if kv_cache_mode == 'cached':
|
| 58 |
+
modulation = modulation[:-1]
|
| 59 |
+
target_token_mask = None
|
| 60 |
+
return _ORIGINAL_BLOCK_FORWARD(
|
| 61 |
+
self, hidden_states, modulation, rotary_emb, attention_mask,
|
| 62 |
+
target_token_mask, layer_cache, kv_cache_mode, cache_write_slice,
|
| 63 |
+
segments, key_valid,
|
| 64 |
+
)
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def _dispatch_block(self, **kwargs):
|
| 68 |
+
if kwargs.get('kv_cache_mode') == 'cached':
|
| 69 |
+
return self._image21_compiled(**kwargs)
|
| 70 |
+
return _ORIGINAL_BLOCK_FORWARD(self, **kwargs)
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def accelerate_pipeline(pipe):
|
| 74 |
+
"""Enable once after loading. Initial compilation is excluded from warm timings.
|
| 75 |
+
|
| 76 |
+
Dynamic sequence lengths reduce recompilation across prompts/resolutions.
|
| 77 |
+
emulate_precision_casts preserves the upstream intermediate BF16 rounding
|
| 78 |
+
boundaries in fused code; GPU reduction ordering can still differ.
|
| 79 |
+
"""
|
| 80 |
+
if getattr(pipe, '_image21_accelerated', False):
|
| 81 |
+
return pipe
|
| 82 |
+
torch._dynamo.config.recompile_limit = max(torch._dynamo.config.recompile_limit, 64)
|
| 83 |
+
for block in pipe.transformer.transformer_blocks:
|
| 84 |
+
block.attn.set_processor(NativeAttentionProcessor())
|
| 85 |
+
block._image21_compiled = torch.compile(
|
| 86 |
+
types.MethodType(_decode_block, block), fullgraph=True, dynamic=True,
|
| 87 |
+
options={'emulate_precision_casts': True},
|
| 88 |
+
)
|
| 89 |
+
block.forward = types.MethodType(_dispatch_block, block)
|
| 90 |
+
pipe._image21_accelerated = True
|
| 91 |
+
return pipe
|
comparisons/evaluation-dynamic/00.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/01.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/02.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/03.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/04.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/05.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/06.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/07.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/08.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/09.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/10.jpg
ADDED
|
comparisons/evaluation-dynamic/11.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/12.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/13.jpg
ADDED
|
comparisons/evaluation-dynamic/14.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/15.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/edit-0.jpg
ADDED
|
Git LFS Details
|
comparisons/evaluation-dynamic/edit-1.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/00.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/01.jpg
ADDED
|
comparisons/heldout-nvfp4/02.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/03.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/04.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/05.jpg
ADDED
|
Git LFS Details
|
comparisons/heldout-nvfp4/06.jpg
ADDED
|
comparisons/heldout-nvfp4/07.jpg
ADDED
|
Git LFS Details
|
demo/capture_receipt.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
demo/realtime-30s.mp4
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7585d0c1e7618d2fb710ebf08f70541d36d88ac49152bfcbd4f10d093ae9ca94
|
| 3 |
+
size 2301594
|
dynamic_scale.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Dynamic tensor-wide NVFP4 range, with FP32 reductions and correction rescaling."""
|
| 2 |
+
import torch,triton
|
| 3 |
+
import triton.language as tl
|
| 4 |
+
@triton.jit
|
| 5 |
+
def _partials(X,P,T,COUNT:tl.constexpr,K:tl.constexpr,B:tl.constexpr):
|
| 6 |
+
i=tl.program_id(0)*B+tl.arange(0,B)
|
| 7 |
+
x=tl.load(X+i,i<COUNT,0).to(tl.float32)
|
| 8 |
+
p=tl.load(P+i%K).to(tl.float32)
|
| 9 |
+
tl.store(T+tl.program_id(0),tl.max(tl.abs(x*p),0))
|
| 10 |
+
@triton.jit
|
| 11 |
+
def _finish(T,OLDG,OLDA,G,A,R,N:tl.constexpr,B:tl.constexpr):
|
| 12 |
+
i=tl.arange(0,B);v=tl.load(T+i,i<N,0)
|
| 13 |
+
g=2688./tl.maximum(tl.max(v,0),1.e-12)
|
| 14 |
+
oldg=tl.load(OLDG);olda=tl.load(OLDA)
|
| 15 |
+
tl.store(G,g);tl.store(A,olda*oldg/g);tl.store(R,g/oldg)
|
| 16 |
+
@triton.jit
|
| 17 |
+
def _upscale(U,R,V,N:tl.constexpr,B:tl.constexpr):
|
| 18 |
+
i=tl.program_id(0)*B+tl.arange(0,B)
|
| 19 |
+
u=tl.load(U+i,i<N,0).to(tl.float32);r=tl.load(R)
|
| 20 |
+
tl.store(V+i,u*r,i<N)
|
| 21 |
+
def scale(x,pre,gx,alpha,up):
|
| 22 |
+
count=x.numel();blocks=triton.cdiv(count,16384)
|
| 23 |
+
partial=torch.empty(blocks,device=x.device,dtype=torch.float32)
|
| 24 |
+
g=torch.empty_like(gx);a=torch.empty_like(alpha);r=torch.empty_like(gx)
|
| 25 |
+
_partials[(blocks,)](x,pre,partial,count,x.shape[-1],16384,num_warps=8)
|
| 26 |
+
_finish[(1,)](partial,gx,alpha,g,a,r,blocks,triton.next_power_of_2(blocks),num_warps=8)
|
| 27 |
+
u=torch.empty_like(up)
|
| 28 |
+
_upscale[(triton.cdiv(up.numel(),1024),)](up,r,u,up.numel(),1024)
|
| 29 |
+
return g,a,u
|
fp8_runtime.py
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Calibrated W8A8 E4M3 inference using native CUTLASS SM120 scaled GEMMs.
|
| 2 |
+
|
| 3 |
+
Built with Qwen. Quantization modifications by ProCreations, 2026.
|
| 4 |
+
The upstream model is subject to the included Qwen Research License.
|
| 5 |
+
"""
|
| 6 |
+
import json
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
import torch
|
| 9 |
+
from torch import nn
|
| 10 |
+
import triton
|
| 11 |
+
import triton.language as tl
|
| 12 |
+
from safetensors.torch import load_file
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
@triton.jit
|
| 16 |
+
def _quantize_rows(X, SMOOTH, Q, SCALE, K: tl.constexpr, BLOCK: tl.constexpr):
|
| 17 |
+
row = tl.program_id(0)
|
| 18 |
+
cols = tl.arange(0, BLOCK)
|
| 19 |
+
x = tl.load(X + row * K + cols, cols < K, 0).to(tl.float32)
|
| 20 |
+
s = tl.load(SMOOTH + cols, cols < K, 1).to(tl.float32)
|
| 21 |
+
z = x / s
|
| 22 |
+
scale = tl.maximum(tl.max(tl.abs(z), 0) / 448.0, 1.e-12)
|
| 23 |
+
tl.store(Q + row * K + cols, z / scale, cols < K)
|
| 24 |
+
tl.store(SCALE + row, scale)
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
def quantize_activation(x, smooth):
|
| 28 |
+
x = x.reshape(-1, x.shape[-1]).contiguous()
|
| 29 |
+
q = torch.empty_like(x, dtype=torch.float8_e4m3fn)
|
| 30 |
+
scale = torch.empty((x.shape[0], 1), device=x.device, dtype=torch.float32)
|
| 31 |
+
_quantize_rows[(x.shape[0],)](x, smooth, q, scale, x.shape[1],
|
| 32 |
+
triton.next_power_of_2(x.shape[1]), num_warps=8)
|
| 33 |
+
return q, scale
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
class CalibratedFP8Linear(nn.Module):
|
| 37 |
+
def __init__(self, weight, scale, smooth):
|
| 38 |
+
super().__init__()
|
| 39 |
+
self.register_buffer("weight", weight)
|
| 40 |
+
self.register_buffer("weight_scale", scale.float())
|
| 41 |
+
self.register_buffer("smooth", smooth.float())
|
| 42 |
+
self.in_features = weight.shape[1]
|
| 43 |
+
self.out_features = weight.shape[0]
|
| 44 |
+
self.bias = None
|
| 45 |
+
|
| 46 |
+
def forward(self, x):
|
| 47 |
+
shape = x.shape[:-1]
|
| 48 |
+
q, scale = quantize_activation(x, self.smooth)
|
| 49 |
+
y = torch._scaled_mm(q, self.weight.t(), scale_a=scale,
|
| 50 |
+
scale_b=self.weight_scale.t(),
|
| 51 |
+
out_dtype=torch.bfloat16, use_fast_accum=False)
|
| 52 |
+
return y.reshape(*shape, self.out_features)
|
| 53 |
+
|
| 54 |
+
|
| 55 |
+
def eligible_modules(model):
|
| 56 |
+
return {n:m for n,m in model.named_modules()
|
| 57 |
+
if isinstance(m, nn.Linear) and n.startswith("transformer_blocks.")
|
| 58 |
+
and m.bias is None}
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
def replace_module(root, name, replacement):
|
| 62 |
+
parent_name, leaf = name.rsplit(".", 1)
|
| 63 |
+
setattr(root.get_submodule(parent_name), leaf, replacement)
|
| 64 |
+
|
| 65 |
+
|
| 66 |
+
def load_fp8_transformer(directory, device="cuda"):
|
| 67 |
+
from diffusers import QwenImage21Transformer2DModel
|
| 68 |
+
directory = Path(directory)
|
| 69 |
+
config = json.loads((directory / "config.json").read_text())
|
| 70 |
+
meta = json.loads((directory / "quantization_config.json").read_text())
|
| 71 |
+
# Construct without allocating an intermediate BF16 model.
|
| 72 |
+
with torch.device("meta"):
|
| 73 |
+
model = QwenImage21Transformer2DModel.from_config(config)
|
| 74 |
+
for n, spec in meta["quantized_modules"].items():
|
| 75 |
+
out_f, in_f = spec["shape"]
|
| 76 |
+
replace_module(model, n, CalibratedFP8Linear(
|
| 77 |
+
torch.empty((out_f,in_f),dtype=torch.float8_e4m3fn),
|
| 78 |
+
torch.empty((out_f,1),dtype=torch.float32),
|
| 79 |
+
torch.empty(in_f,dtype=torch.float32)))
|
| 80 |
+
state = {}
|
| 81 |
+
for path in sorted(directory.glob("model-*.safetensors")):
|
| 82 |
+
state.update(load_file(str(path), device="cpu"))
|
| 83 |
+
model.load_state_dict(state, strict=True, assign=True)
|
| 84 |
+
# The original rotary module has nonpersistent buffers / Python tensor lists.
|
| 85 |
+
# Reconstruct these by creating a lightweight ordinary position module.
|
| 86 |
+
from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21Rope
|
| 87 |
+
model.pos_embed = QwenImage21Rope(theta=10000, axes_dim=list(model.config.axes_dims_rope))
|
| 88 |
+
from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21TemporalTimesteps
|
| 89 |
+
model.time_text_embed.time_proj = QwenImage21TemporalTimesteps(timestep_dim=256)
|
| 90 |
+
return model.eval().to(device=device)
|
| 91 |
+
|
| 92 |
+
|
| 93 |
+
def load_pipeline(base, quant=None):
|
| 94 |
+
from diffusers import QwenImage21Pipeline
|
| 95 |
+
args = {}
|
| 96 |
+
if not Path(base).exists():
|
| 97 |
+
args["revision"] = "b3179ad355be050328e483a9dfdd9e60cd62adfa"
|
| 98 |
+
if quant:
|
| 99 |
+
args["transformer"] = load_fp8_transformer(quant, device="cuda")
|
| 100 |
+
pipe = QwenImage21Pipeline.from_pretrained(base, torch_dtype=torch.bfloat16, **args)
|
| 101 |
+
# Do not call pipe.to(dtype=...) after FP8 loading; scales are FP32.
|
| 102 |
+
pipe.to(device="cuda")
|
| 103 |
+
pipe.set_progress_bar_config(disable=True)
|
| 104 |
+
return pipe
|
generate.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Generate with Image 2.1 Calibrated NVFP4. Built with Qwen."""
|
| 2 |
+
import argparse, json, time
|
| 3 |
+
from pathlib import Path
|
| 4 |
+
import torch
|
| 5 |
+
from nvfp4_runtime import load_pipeline
|
| 6 |
+
|
| 7 |
+
@torch.inference_mode()
|
| 8 |
+
def main():
|
| 9 |
+
ap=argparse.ArgumentParser()
|
| 10 |
+
inputs=ap.add_mutually_exclusive_group(required=True)
|
| 11 |
+
inputs.add_argument('--prompt')
|
| 12 |
+
inputs.add_argument('--prompts-json', help='JSON list of prompt strings; one loaded pipeline serves the whole list')
|
| 13 |
+
ap.add_argument('--base',default='Qwen/Qwen-Image-2.1')
|
| 14 |
+
ap.add_argument('--quant',default=str(Path(__file__).parent/'transformer'))
|
| 15 |
+
ap.add_argument('--width',type=int,default=2048)
|
| 16 |
+
ap.add_argument('--height',type=int,default=2048)
|
| 17 |
+
ap.add_argument('--steps',type=int,default=40)
|
| 18 |
+
ap.add_argument('--seed',type=int,default=42)
|
| 19 |
+
ap.add_argument('--output',default='output.png')
|
| 20 |
+
ap.add_argument('--image')
|
| 21 |
+
ap.add_argument('--eager', action='store_true', help='Use the original uncompiled runtime')
|
| 22 |
+
ap.add_argument('--warmup', action='store_true', help='Run a full untimed warmup; report its cost separately')
|
| 23 |
+
args=ap.parse_args()
|
| 24 |
+
prompts=json.loads(Path(args.prompts_json).read_text()) if args.prompts_json else [args.prompt]
|
| 25 |
+
if not isinstance(prompts,list) or not prompts or not all(isinstance(p,str) and p for p in prompts):
|
| 26 |
+
ap.error('prompts-json must contain a nonempty list of prompt strings')
|
| 27 |
+
start=time.perf_counter();pipe=load_pipeline(args.base,args.quant)
|
| 28 |
+
if not args.eager:
|
| 29 |
+
from acceleration import accelerate_pipeline
|
| 30 |
+
accelerate_pipeline(pipe)
|
| 31 |
+
torch.cuda.synchronize();load_seconds=time.perf_counter()-start
|
| 32 |
+
kw={}
|
| 33 |
+
if args.image:
|
| 34 |
+
from PIL import Image
|
| 35 |
+
kw['image']=Image.open(args.image)
|
| 36 |
+
def run(prompt,seed):
|
| 37 |
+
return pipe(prompt=prompt,width=args.width,height=args.height,num_inference_steps=args.steps,
|
| 38 |
+
generator=torch.Generator('cuda').manual_seed(seed),**kw).images[0]
|
| 39 |
+
warmup_seconds=0
|
| 40 |
+
if args.warmup:
|
| 41 |
+
start=time.perf_counter();run(prompts[0],args.seed);torch.cuda.synchronize()
|
| 42 |
+
warmup_seconds=time.perf_counter()-start
|
| 43 |
+
path=Path(args.output);path.parent.mkdir(parents=True,exist_ok=True)
|
| 44 |
+
for i,prompt in enumerate(prompts):
|
| 45 |
+
torch.cuda.synchronize();start=time.perf_counter();result=run(prompt,args.seed+i)
|
| 46 |
+
torch.cuda.synchronize();seconds=time.perf_counter()-start
|
| 47 |
+
destination=path if len(prompts)==1 else path.with_name(f'{path.stem}-{i:03d}{path.suffix or ".png"}')
|
| 48 |
+
result.save(destination)
|
| 49 |
+
print(json.dumps({'seconds':seconds,'output':str(destination),'width':result.width,'height':result.height,
|
| 50 |
+
'steps':args.steps,'eager':args.eager,'load_seconds':load_seconds,
|
| 51 |
+
'warmup_seconds':warmup_seconds,
|
| 52 |
+
'timing_note':'Generation includes any compilation on this call; excludes model load and PNG writing.'}),flush=True)
|
| 53 |
+
|
| 54 |
+
if __name__=='__main__':main()
|
install.sh
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env bash
|
| 2 |
+
# Run from the downloaded release directory on Linux x86_64 / SM120.
|
| 3 |
+
set -euo pipefail
|
| 4 |
+
python3.12 -m venv .venv
|
| 5 |
+
.venv/bin/python -m pip install --upgrade pip
|
| 6 |
+
.venv/bin/python -m pip install torch==2.14.0 torchvision==0.29.0 --index-url https://download.pytorch.org/whl/cu130
|
| 7 |
+
.venv/bin/python -m pip install -r requirements.txt
|
| 8 |
+
git clone https://github.com/flashinfer-ai/flashinfer.git flashinfer-src
|
| 9 |
+
git -C flashinfer-src checkout 975f90583d9ac8896db14cf0f26e99a853c2f136
|
| 10 |
+
git -C flashinfer-src submodule update --init --depth 1 3rdparty/cutlass 3rdparty/cccl 3rdparty/spdlog
|
| 11 |
+
BUILD_NVEP=0 .venv/bin/python -m pip install --no-deps ./flashinfer-src
|
| 12 |
+
.venv/bin/python -m pip check
|
manifest.json
ADDED
|
@@ -0,0 +1,510 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"files": [
|
| 3 |
+
{
|
| 4 |
+
"path": "LICENSE",
|
| 5 |
+
"bytes": 7831,
|
| 6 |
+
"sha256": "8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d"
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
"path": "Notice",
|
| 10 |
+
"bytes": 491,
|
| 11 |
+
"sha256": "c517ea86da1c4b8f3f659944f5e7b5a66128fe024af5d64655b6a1c2ade6ae65"
|
| 12 |
+
},
|
| 13 |
+
{
|
| 14 |
+
"path": "README.md",
|
| 15 |
+
"bytes": 8107,
|
| 16 |
+
"sha256": "d849e2126e39ff5d1586de3a17995afc2358b47ade44dda7dd1bdb15f91cea8f"
|
| 17 |
+
},
|
| 18 |
+
{
|
| 19 |
+
"path": "acceleration.py",
|
| 20 |
+
"bytes": 4157,
|
| 21 |
+
"sha256": "a5a43f98da091fb81fed8f493a736f68820d4cd1b5dc2f30e8a3f8f65659b3c8"
|
| 22 |
+
},
|
| 23 |
+
{
|
| 24 |
+
"path": "comparisons/evaluation-dynamic/00.jpg",
|
| 25 |
+
"bytes": 176056,
|
| 26 |
+
"sha256": "801c01744c1a1c5d33b5e06b9503e92cf9df1bff52e530e65bb062713f442162"
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"path": "comparisons/evaluation-dynamic/01.jpg",
|
| 30 |
+
"bytes": 179746,
|
| 31 |
+
"sha256": "b0423321243e0f5b210ce6e32c4565f993294aa8fd89583699a3caebeded99a1"
|
| 32 |
+
},
|
| 33 |
+
{
|
| 34 |
+
"path": "comparisons/evaluation-dynamic/02.jpg",
|
| 35 |
+
"bytes": 180435,
|
| 36 |
+
"sha256": "e519b8431f20a90c6c619b91e7966dd7c638973f2b33b012ea3aabb991a738d8"
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"path": "comparisons/evaluation-dynamic/03.jpg",
|
| 40 |
+
"bytes": 146351,
|
| 41 |
+
"sha256": "3a0138e1eb0eb79abe01ff768be4eccba5c2114488f840b8575c16b02af7b169"
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"path": "comparisons/evaluation-dynamic/04.jpg",
|
| 45 |
+
"bytes": 182735,
|
| 46 |
+
"sha256": "51e0a532f9f4de9d65d14fd4edc543cf87b5bc35ea36e4697e42f2e3997afa52"
|
| 47 |
+
},
|
| 48 |
+
{
|
| 49 |
+
"path": "comparisons/evaluation-dynamic/05.jpg",
|
| 50 |
+
"bytes": 309805,
|
| 51 |
+
"sha256": "81fd3d157a7175fee7b108206fc5bdb44f6542cf06f42e7ee19e2ca1d79d3636"
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"path": "comparisons/evaluation-dynamic/06.jpg",
|
| 55 |
+
"bytes": 104835,
|
| 56 |
+
"sha256": "626595f20c55ef0cbad9ac793180dc8f7bf84da82ec29c60d8b13244a9c64f40"
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"path": "comparisons/evaluation-dynamic/07.jpg",
|
| 60 |
+
"bytes": 124166,
|
| 61 |
+
"sha256": "96259410bc3679ffa69cfef803b38af74974068467de97160defe7893b29930e"
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"path": "comparisons/evaluation-dynamic/08.jpg",
|
| 65 |
+
"bytes": 113262,
|
| 66 |
+
"sha256": "dad753aef0935797375233f3c6d0b64c4360dfc45c4227c3d3cd72c58e64b8f3"
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"path": "comparisons/evaluation-dynamic/09.jpg",
|
| 70 |
+
"bytes": 101923,
|
| 71 |
+
"sha256": "68a84e802c40234eb830362be6e9405c7c4f765a0ac12061864d2e8c6dd3b746"
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"path": "comparisons/evaluation-dynamic/10.jpg",
|
| 75 |
+
"bytes": 60148,
|
| 76 |
+
"sha256": "434f2e92839951456722954e64c66f35a287a75b4416d1aad33a6362c89c15f6"
|
| 77 |
+
},
|
| 78 |
+
{
|
| 79 |
+
"path": "comparisons/evaluation-dynamic/11.jpg",
|
| 80 |
+
"bytes": 260028,
|
| 81 |
+
"sha256": "a57d36f2e4bd995fa540234dce491f209fa4524700b1ebdb0cbb8966bbbc5778"
|
| 82 |
+
},
|
| 83 |
+
{
|
| 84 |
+
"path": "comparisons/evaluation-dynamic/12.jpg",
|
| 85 |
+
"bytes": 276643,
|
| 86 |
+
"sha256": "9b35d184b7efa4ff9a3bc7abef9e4cf50d6ed99bf6967724312038e52f8e2672"
|
| 87 |
+
},
|
| 88 |
+
{
|
| 89 |
+
"path": "comparisons/evaluation-dynamic/13.jpg",
|
| 90 |
+
"bytes": 86912,
|
| 91 |
+
"sha256": "84f4bbe544f92623675f448e01710fdf8e799dbc8ed2af4674c1c6c94b00eb22"
|
| 92 |
+
},
|
| 93 |
+
{
|
| 94 |
+
"path": "comparisons/evaluation-dynamic/14.jpg",
|
| 95 |
+
"bytes": 189569,
|
| 96 |
+
"sha256": "ccdb2b7222304c5049c324750433dfa0f4862dfc3833864e90030a1a0e03d1a7"
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"path": "comparisons/evaluation-dynamic/15.jpg",
|
| 100 |
+
"bytes": 203642,
|
| 101 |
+
"sha256": "e3586ffca8d68cebde081ef42c4e7fab7fd5c8cbe1b2531a54ee8474a07bc7a0"
|
| 102 |
+
},
|
| 103 |
+
{
|
| 104 |
+
"path": "comparisons/evaluation-dynamic/edit-0.jpg",
|
| 105 |
+
"bytes": 239945,
|
| 106 |
+
"sha256": "5747ce5f80fd014383a89ad06d0b31209612bbbe2981e5082b086819c8db720f"
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"path": "comparisons/evaluation-dynamic/edit-1.jpg",
|
| 110 |
+
"bytes": 140013,
|
| 111 |
+
"sha256": "e93f685092a7f8729a135df81d8fc91d2f7f446e6857d03d36055e4926c820b2"
|
| 112 |
+
},
|
| 113 |
+
{
|
| 114 |
+
"path": "comparisons/heldout-nvfp4/00.jpg",
|
| 115 |
+
"bytes": 144117,
|
| 116 |
+
"sha256": "fef99d1fed7bec5ca673bf94aae5f0f534db77d474e45f04a8c6b52798b20801"
|
| 117 |
+
},
|
| 118 |
+
{
|
| 119 |
+
"path": "comparisons/heldout-nvfp4/01.jpg",
|
| 120 |
+
"bytes": 95024,
|
| 121 |
+
"sha256": "bf3c15a3f9acaaaac7deb73d89501a73a5bfe6d3dd402107deff0bef5218651f"
|
| 122 |
+
},
|
| 123 |
+
{
|
| 124 |
+
"path": "comparisons/heldout-nvfp4/02.jpg",
|
| 125 |
+
"bytes": 137218,
|
| 126 |
+
"sha256": "16c8ab16fe20d594d46ad396ed4c415690f2c3082ac4287b080806a435d82fb3"
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"path": "comparisons/heldout-nvfp4/03.jpg",
|
| 130 |
+
"bytes": 105496,
|
| 131 |
+
"sha256": "907a7a68a36afc83d506442e84a538e0bab1af77582fb76b067809685df15e20"
|
| 132 |
+
},
|
| 133 |
+
{
|
| 134 |
+
"path": "comparisons/heldout-nvfp4/04.jpg",
|
| 135 |
+
"bytes": 142102,
|
| 136 |
+
"sha256": "7abc30394201e4f7e1456f06df38ab6b492ffbf3a869ab5d9decc70231312667"
|
| 137 |
+
},
|
| 138 |
+
{
|
| 139 |
+
"path": "comparisons/heldout-nvfp4/05.jpg",
|
| 140 |
+
"bytes": 203606,
|
| 141 |
+
"sha256": "db68b9fa5e5ce3453fbe74d569b051274f300106d6c3aecd60996418f52ed86b"
|
| 142 |
+
},
|
| 143 |
+
{
|
| 144 |
+
"path": "comparisons/heldout-nvfp4/06.jpg",
|
| 145 |
+
"bytes": 95263,
|
| 146 |
+
"sha256": "194e2d5bafe9ef00fae0f62f891125169c4b79f379bb582b72af82eae51a351b"
|
| 147 |
+
},
|
| 148 |
+
{
|
| 149 |
+
"path": "comparisons/heldout-nvfp4/07.jpg",
|
| 150 |
+
"bytes": 134634,
|
| 151 |
+
"sha256": "9f625acf906a08a69543723a8af7254ecd4e0837101c152fb2140d894ea41371"
|
| 152 |
+
},
|
| 153 |
+
{
|
| 154 |
+
"path": "demo/capture_receipt.json",
|
| 155 |
+
"bytes": 131978,
|
| 156 |
+
"sha256": "b9e7e82391fd38f7536bc48a1ab1f0cc711a715a04b9e634679bdc4da89ab17c"
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"path": "demo/realtime-30s.mp4",
|
| 160 |
+
"bytes": 2301594,
|
| 161 |
+
"sha256": "7585d0c1e7618d2fb710ebf08f70541d36d88ac49152bfcbd4f10d093ae9ca94"
|
| 162 |
+
},
|
| 163 |
+
{
|
| 164 |
+
"path": "dynamic_scale.py",
|
| 165 |
+
"bytes": 1288,
|
| 166 |
+
"sha256": "d908cd50c73cbdf40133775460fe599278fd6d2e7d10648782798d6dfbbbbd7b"
|
| 167 |
+
},
|
| 168 |
+
{
|
| 169 |
+
"path": "fp8_runtime.py",
|
| 170 |
+
"bytes": 4376,
|
| 171 |
+
"sha256": "5892e8b5963f66fee67c16dd3996e05b4b64297e570f3115630a19b224d69413"
|
| 172 |
+
},
|
| 173 |
+
{
|
| 174 |
+
"path": "generate.py",
|
| 175 |
+
"bytes": 3007,
|
| 176 |
+
"sha256": "b383aa9a93d7980ba1b2b74eb6bd7f1accb9c1651c4e468c319fd48608ac42ea"
|
| 177 |
+
},
|
| 178 |
+
{
|
| 179 |
+
"path": "install.sh",
|
| 180 |
+
"bytes": 697,
|
| 181 |
+
"sha256": "8fff3b147208b3651fcc71d7b09ded4674769e4321d840a7871a412298d0be7a"
|
| 182 |
+
},
|
| 183 |
+
{
|
| 184 |
+
"path": "nvfp4_runtime.py",
|
| 185 |
+
"bytes": 3972,
|
| 186 |
+
"sha256": "86fbf7003b755f824b35bf2370feab2f360c79ee4683e4f79267fb7db1e307d7"
|
| 187 |
+
},
|
| 188 |
+
{
|
| 189 |
+
"path": "reports/benchmark.json",
|
| 190 |
+
"bytes": 1158,
|
| 191 |
+
"sha256": "c0ef7fbedaf7b976dab4fc623130fe8e09838b1b0f7147d27d9b367524172c98"
|
| 192 |
+
},
|
| 193 |
+
{
|
| 194 |
+
"path": "reports/calibration_manifest.json",
|
| 195 |
+
"bytes": 29336,
|
| 196 |
+
"sha256": "e067532fc1e6779e22cac2c2beb2bbf4840cd4252aba0da142c5b37f5f97024a"
|
| 197 |
+
},
|
| 198 |
+
{
|
| 199 |
+
"path": "reports/calibration_search.json",
|
| 200 |
+
"bytes": 378392,
|
| 201 |
+
"sha256": "8b71f8e6d43303f3341b4c08e10652dad8c1585f31bf0d0a77dee9a55db8381b"
|
| 202 |
+
},
|
| 203 |
+
{
|
| 204 |
+
"path": "reports/correction-probe.json",
|
| 205 |
+
"bytes": 2835,
|
| 206 |
+
"sha256": "5a3f2d003e8c7c5ab928260b56b12add38bc94d7f7e25ab00fb3b3fddcafcdce"
|
| 207 |
+
},
|
| 208 |
+
{
|
| 209 |
+
"path": "reports/cudagraph-trial.json",
|
| 210 |
+
"bytes": 218,
|
| 211 |
+
"sha256": "9e16d5dc46a1a580ea2dac0e7eec8655c18a8a313870a160e6b091b56a96b4d2"
|
| 212 |
+
},
|
| 213 |
+
{
|
| 214 |
+
"path": "reports/earlier-bf16-benchmark.json",
|
| 215 |
+
"bytes": 1031,
|
| 216 |
+
"sha256": "640be194397d85b94022b5a960141c92e3f65de30d98d2586bc466b5df8beb7c"
|
| 217 |
+
},
|
| 218 |
+
{
|
| 219 |
+
"path": "reports/environment.json",
|
| 220 |
+
"bytes": 2884,
|
| 221 |
+
"sha256": "c375ec1b13547ee0ee2ff0f9a67a9de1d2fd8c47aaa92f3f6de31731c5994af7"
|
| 222 |
+
},
|
| 223 |
+
{
|
| 224 |
+
"path": "reports/evaluation-environment.json",
|
| 225 |
+
"bytes": 161,
|
| 226 |
+
"sha256": "727e74ca45de746f0efa9f78dec96ac8ebc925b92dfddaf1e7e4da812b9cd583"
|
| 227 |
+
},
|
| 228 |
+
{
|
| 229 |
+
"path": "reports/fresh-fp8-benchmark.json",
|
| 230 |
+
"bytes": 913,
|
| 231 |
+
"sha256": "4194b2a3b646711daa393474b9bb911900d65cf2d792fe224e5ab619ff5470fb"
|
| 232 |
+
},
|
| 233 |
+
{
|
| 234 |
+
"path": "reports/heldout-manifest.json",
|
| 235 |
+
"bytes": 2446,
|
| 236 |
+
"sha256": "ec71f7211480de9bc445c9c44649c5811ff20b51b568c4132915bea2ac2c7f91"
|
| 237 |
+
},
|
| 238 |
+
{
|
| 239 |
+
"path": "reports/kernel_evidence.json",
|
| 240 |
+
"bytes": 24824,
|
| 241 |
+
"sha256": "1b622cd32d8e9c6524a27421a3d077d1a0df2c5512f4ff1914e336178e189e2b"
|
| 242 |
+
},
|
| 243 |
+
{
|
| 244 |
+
"path": "reports/native-math-verification.json",
|
| 245 |
+
"bytes": 765,
|
| 246 |
+
"sha256": "3ed24479ca50f9dabe5bfaa57931a6711d4a47c4eb9b493db854d35bc0495be7"
|
| 247 |
+
},
|
| 248 |
+
{
|
| 249 |
+
"path": "reports/qa-review.json",
|
| 250 |
+
"bytes": 817,
|
| 251 |
+
"sha256": "2fc1a5a431daf87ea5e3e3ff2bdcc85239dc2d08ce1ecc18b903815846d8deb3"
|
| 252 |
+
},
|
| 253 |
+
{
|
| 254 |
+
"path": "reports/quality_metrics-bf16.json",
|
| 255 |
+
"bytes": 5502,
|
| 256 |
+
"sha256": "bb644a0dd2aef533c0d9896c52e5a0fcaf376fb4acb433474c62a0367f9e126c"
|
| 257 |
+
},
|
| 258 |
+
{
|
| 259 |
+
"path": "reports/quality_metrics-fp8.json",
|
| 260 |
+
"bytes": 5501,
|
| 261 |
+
"sha256": "99ba88af5444d308e278e94a207ee3c4443faff1e8f321d4e0f304270faa8678"
|
| 262 |
+
},
|
| 263 |
+
{
|
| 264 |
+
"path": "reports/quality_metrics-heldout-bf16.json",
|
| 265 |
+
"bytes": 2880,
|
| 266 |
+
"sha256": "849cfd94a16b602554644bbae1b569111527be909f0004bc16e64e07a107f5cc"
|
| 267 |
+
},
|
| 268 |
+
{
|
| 269 |
+
"path": "reports/quantization_summary.json",
|
| 270 |
+
"bytes": 715,
|
| 271 |
+
"sha256": "ece079fec55b6864a777541a5d959ace269f2a366963890cea957d87b9291c91"
|
| 272 |
+
},
|
| 273 |
+
{
|
| 274 |
+
"path": "reports/release-cli-smoke.json",
|
| 275 |
+
"bytes": 319,
|
| 276 |
+
"sha256": "a0c67c3e7f108714c9be3dd33e5bc66dc8896a792ed900fd45e4766d75cb74ac"
|
| 277 |
+
},
|
| 278 |
+
{
|
| 279 |
+
"path": "reports/tuning-results.json",
|
| 280 |
+
"bytes": 1035,
|
| 281 |
+
"sha256": "bc3e79a8cc7cc99759e58cd9640f1b3eb9c39cbdbd96b44dfc8d269435d8c0e2"
|
| 282 |
+
},
|
| 283 |
+
{
|
| 284 |
+
"path": "reports/video-verification.json",
|
| 285 |
+
"bytes": 1981,
|
| 286 |
+
"sha256": "45a15cb8d97bd795cf7e3bb62143d5725feb9a362d75cb5d28f9907320898cb3"
|
| 287 |
+
},
|
| 288 |
+
{
|
| 289 |
+
"path": "requirements.txt",
|
| 290 |
+
"bytes": 488,
|
| 291 |
+
"sha256": "8687c948b6aeeaf3e20a013f3073f51c4235a5459767afb487191699075369cc"
|
| 292 |
+
},
|
| 293 |
+
{
|
| 294 |
+
"path": "samples/heldout/00.png",
|
| 295 |
+
"bytes": 1470245,
|
| 296 |
+
"sha256": "2ecc7ba864f3065c98c0d256fd92b74c0b068cc391f66ed0c548a3143d2191c1"
|
| 297 |
+
},
|
| 298 |
+
{
|
| 299 |
+
"path": "samples/heldout/01.png",
|
| 300 |
+
"bytes": 1144635,
|
| 301 |
+
"sha256": "298853daf5abb029ba7d715b052392c872301aa934e6482c08f0ea6a00757000"
|
| 302 |
+
},
|
| 303 |
+
{
|
| 304 |
+
"path": "samples/heldout/02.png",
|
| 305 |
+
"bytes": 1593770,
|
| 306 |
+
"sha256": "20312fb53fc29ad2c92ddfb3b2b2583dccb0b42678077dcfc4fa6613fbf35bf6"
|
| 307 |
+
},
|
| 308 |
+
{
|
| 309 |
+
"path": "samples/heldout/03.png",
|
| 310 |
+
"bytes": 1128454,
|
| 311 |
+
"sha256": "86148823326a6a8426f8317a73a2437c6eed1e665b9be9962140a065d7710694"
|
| 312 |
+
},
|
| 313 |
+
{
|
| 314 |
+
"path": "samples/heldout/04.png",
|
| 315 |
+
"bytes": 1395649,
|
| 316 |
+
"sha256": "c0c0d4d5492f443c301925508d034b88bc6102ae197e9f79e6bc646cccc299bc"
|
| 317 |
+
},
|
| 318 |
+
{
|
| 319 |
+
"path": "samples/heldout/05.png",
|
| 320 |
+
"bytes": 7047766,
|
| 321 |
+
"sha256": "63d8d2f130ddc3ff0f8946f0a7ffcfe1302a981507f2e94dc4cb0d82e4a81154"
|
| 322 |
+
},
|
| 323 |
+
{
|
| 324 |
+
"path": "samples/heldout/06.png",
|
| 325 |
+
"bytes": 800963,
|
| 326 |
+
"sha256": "ed89a263c649b8d0b260aef08eb2a7727b9557d4e0937a6d8ef992c9f8bba3c1"
|
| 327 |
+
},
|
| 328 |
+
{
|
| 329 |
+
"path": "samples/heldout/07.png",
|
| 330 |
+
"bytes": 1670613,
|
| 331 |
+
"sha256": "9536f8a08d250dc7d9c6b273e011239e20a6b38effdee3d1c20ac672f5d33bbf"
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"path": "samples/validation/00.png",
|
| 335 |
+
"bytes": 7025357,
|
| 336 |
+
"sha256": "c3f5df94d3b2d9c2958f4c79042ae3ceca55c8dc1a22f3737f589119be8dd6ef"
|
| 337 |
+
},
|
| 338 |
+
{
|
| 339 |
+
"path": "samples/validation/01.png",
|
| 340 |
+
"bytes": 1829665,
|
| 341 |
+
"sha256": "1f50d577a9d22baff9034ddc2cb73bd47ac835e85cf43f47965146f8d223be00"
|
| 342 |
+
},
|
| 343 |
+
{
|
| 344 |
+
"path": "samples/validation/02.png",
|
| 345 |
+
"bytes": 1692847,
|
| 346 |
+
"sha256": "7d746a9909df831af3b3b6ff2de4ed76aefbdd5d9d78098d689ac0648732ff60"
|
| 347 |
+
},
|
| 348 |
+
{
|
| 349 |
+
"path": "samples/validation/03.png",
|
| 350 |
+
"bytes": 1648591,
|
| 351 |
+
"sha256": "b17cf4fd884cfe6a9c04357e170766a34c6029dcf6b1e81eae9f3fec500e5adb"
|
| 352 |
+
},
|
| 353 |
+
{
|
| 354 |
+
"path": "samples/validation/04.png",
|
| 355 |
+
"bytes": 6420433,
|
| 356 |
+
"sha256": "5be7d5c89e014c10907613f6872385d189e4a1352c4bab892130bb8745660e85"
|
| 357 |
+
},
|
| 358 |
+
{
|
| 359 |
+
"path": "samples/validation/05.png",
|
| 360 |
+
"bytes": 2458939,
|
| 361 |
+
"sha256": "317896f85a6606473ef1f7e18ae448829eb73a6bbeadda477d31ea092ca301e5"
|
| 362 |
+
},
|
| 363 |
+
{
|
| 364 |
+
"path": "samples/validation/06.png",
|
| 365 |
+
"bytes": 1341617,
|
| 366 |
+
"sha256": "1e5c2554f35b7934b0a8b27f2d85388a2316dc4307fdbd978debd22a1edeff93"
|
| 367 |
+
},
|
| 368 |
+
{
|
| 369 |
+
"path": "samples/validation/07.png",
|
| 370 |
+
"bytes": 1497363,
|
| 371 |
+
"sha256": "7dc3aec9e5bde92189dc9888658b410f44ee9a377d62f8f1be8c7b7c5b6d6418"
|
| 372 |
+
},
|
| 373 |
+
{
|
| 374 |
+
"path": "samples/validation/08.png",
|
| 375 |
+
"bytes": 3343177,
|
| 376 |
+
"sha256": "73f41a041094d8d800b37d0649a1a898fd8ebeaa0cfff3cfabcb98bab43de9c6"
|
| 377 |
+
},
|
| 378 |
+
{
|
| 379 |
+
"path": "samples/validation/09.png",
|
| 380 |
+
"bytes": 1183289,
|
| 381 |
+
"sha256": "7a1ba8598fed1f8909ae14d2e59bfc966a8a62d79933c35075abf9b4548ad3f1"
|
| 382 |
+
},
|
| 383 |
+
{
|
| 384 |
+
"path": "samples/validation/10.png",
|
| 385 |
+
"bytes": 552063,
|
| 386 |
+
"sha256": "1d571a6cc024720f63a966cc8a3105d72b3f933c7569978686b11ce3734fc3a3"
|
| 387 |
+
},
|
| 388 |
+
{
|
| 389 |
+
"path": "samples/validation/11.png",
|
| 390 |
+
"bytes": 2181491,
|
| 391 |
+
"sha256": "9cb287b21ca746a11fa9aebc2667f984e81fc8dfe95ffc5e980e68c68f3521ee"
|
| 392 |
+
},
|
| 393 |
+
{
|
| 394 |
+
"path": "samples/validation/12.png",
|
| 395 |
+
"bytes": 3164193,
|
| 396 |
+
"sha256": "96f8f77f8062b940f2300470c96b6c62663d988978a9d6cd153288e476b9f45b"
|
| 397 |
+
},
|
| 398 |
+
{
|
| 399 |
+
"path": "samples/validation/13.png",
|
| 400 |
+
"bytes": 1426920,
|
| 401 |
+
"sha256": "a66963ce46511705e0c7ad521068814fa6ae80d6dc555388c9ac860aad47cd1b"
|
| 402 |
+
},
|
| 403 |
+
{
|
| 404 |
+
"path": "samples/validation/14.png",
|
| 405 |
+
"bytes": 1772462,
|
| 406 |
+
"sha256": "74ac4c13e7adcf2bbcf595fa11699790bb233727fcd655cf3bfb0b8e3757a493"
|
| 407 |
+
},
|
| 408 |
+
{
|
| 409 |
+
"path": "samples/validation/15.png",
|
| 410 |
+
"bytes": 2000815,
|
| 411 |
+
"sha256": "2a3614b3f7c39e04c6c9c9840adfc5b60aea8823b437376421b784841618078a"
|
| 412 |
+
},
|
| 413 |
+
{
|
| 414 |
+
"path": "samples/validation/edit-0.png",
|
| 415 |
+
"bytes": 2070198,
|
| 416 |
+
"sha256": "c7f1845565b14c01e651e3b0561ec8b50b2461818754c009421d56d9d29d2e35"
|
| 417 |
+
},
|
| 418 |
+
{
|
| 419 |
+
"path": "samples/validation/edit-1.png",
|
| 420 |
+
"bytes": 1575594,
|
| 421 |
+
"sha256": "d6784ec61e127873fc3deed239b0c258887d0e99069d9df8323f4431e4245ad6"
|
| 422 |
+
},
|
| 423 |
+
{
|
| 424 |
+
"path": "source/benchmark.py",
|
| 425 |
+
"bytes": 3838,
|
| 426 |
+
"sha256": "ea305538d5bd04efb6163960ff729ab95a3e1a347dbc2ee356ddda23a24c19de"
|
| 427 |
+
},
|
| 428 |
+
{
|
| 429 |
+
"path": "source/collect_calibration.py",
|
| 430 |
+
"bytes": 3259,
|
| 431 |
+
"sha256": "41a518464980d92dab2dab9a7281a103a7baac765466f96b879319b83f24609f"
|
| 432 |
+
},
|
| 433 |
+
{
|
| 434 |
+
"path": "source/dynamic_scale.py",
|
| 435 |
+
"bytes": 1288,
|
| 436 |
+
"sha256": "d908cd50c73cbdf40133775460fe599278fd6d2e7d10648782798d6dfbbbbd7b"
|
| 437 |
+
},
|
| 438 |
+
{
|
| 439 |
+
"path": "source/evaluate.py",
|
| 440 |
+
"bytes": 2361,
|
| 441 |
+
"sha256": "d499fe31851e3898f72ff30d41b244e7a59697b77cf72231464c1311c1ac89c1"
|
| 442 |
+
},
|
| 443 |
+
{
|
| 444 |
+
"path": "source/heldout.py",
|
| 445 |
+
"bytes": 2872,
|
| 446 |
+
"sha256": "075ee0c597d32a24c5eef42cecef0f2adca404009b86162c9c2940e45a5de4c6"
|
| 447 |
+
},
|
| 448 |
+
{
|
| 449 |
+
"path": "source/probe_dynamic.py",
|
| 450 |
+
"bytes": 3484,
|
| 451 |
+
"sha256": "581b458d0bc3dd9e0cc4bdc212d7b46382094467feb2f8de24cb51eb5dc05045"
|
| 452 |
+
},
|
| 453 |
+
{
|
| 454 |
+
"path": "source/prompts.py",
|
| 455 |
+
"bytes": 9671,
|
| 456 |
+
"sha256": "626db2851c9d6266a9f397e73a5ffd04d9a8389c011860e925c0396ccf2a274f"
|
| 457 |
+
},
|
| 458 |
+
{
|
| 459 |
+
"path": "source/quality_metrics.py",
|
| 460 |
+
"bytes": 3325,
|
| 461 |
+
"sha256": "2d59b3f9e596dd111f0864ef7ea48111d66f942ec9d0e5e87f849db0cf566d9f"
|
| 462 |
+
},
|
| 463 |
+
{
|
| 464 |
+
"path": "source/quantize_dynamic.py",
|
| 465 |
+
"bytes": 4971,
|
| 466 |
+
"sha256": "7c98c0c2637e08274c4998c7379d72ff4736fbce4a961249c9a13136784807cf"
|
| 467 |
+
},
|
| 468 |
+
{
|
| 469 |
+
"path": "source/verify_native.py",
|
| 470 |
+
"bytes": 1969,
|
| 471 |
+
"sha256": "1e7743563bf25b437971fc8ef98a1e1211e95f22b96198b93b8194fe6d5dc95b"
|
| 472 |
+
},
|
| 473 |
+
{
|
| 474 |
+
"path": "source/verify_video.py",
|
| 475 |
+
"bytes": 2562,
|
| 476 |
+
"sha256": "afe6a1a2b059b015f245d6ad7de6903440415f056f44020cbb9e736ee3fb7439"
|
| 477 |
+
},
|
| 478 |
+
{
|
| 479 |
+
"path": "source/video.py",
|
| 480 |
+
"bytes": 3931,
|
| 481 |
+
"sha256": "5448b3118fd5d4dcd2a6f50afc5294cf694c6cc6e9e62177afea2de38da52f09"
|
| 482 |
+
},
|
| 483 |
+
{
|
| 484 |
+
"path": "source/weight_quant.py",
|
| 485 |
+
"bytes": 1819,
|
| 486 |
+
"sha256": "74eb0257e09065ceec838d5ec966f9038a2f4001736ba833270680dcd70faa71"
|
| 487 |
+
},
|
| 488 |
+
{
|
| 489 |
+
"path": "transformer/config.json",
|
| 490 |
+
"bytes": 370,
|
| 491 |
+
"sha256": "56ae3281c4e6c2d1aa3658252d187488071815fd79bef15808bb0205fc1a2241"
|
| 492 |
+
},
|
| 493 |
+
{
|
| 494 |
+
"path": "transformer/model-00001-of-00002.safetensors",
|
| 495 |
+
"bytes": 2977392992,
|
| 496 |
+
"sha256": "9fa418cbe2610486997bf5f251db26f98e1f2fcb53d291992a63e16660268afb"
|
| 497 |
+
},
|
| 498 |
+
{
|
| 499 |
+
"path": "transformer/model-00002-of-00002.safetensors",
|
| 500 |
+
"bytes": 1893707392,
|
| 501 |
+
"sha256": "a820829d3b2d89a09c99bda3f32b1340b37779ad3072d22e7ec6232f71ed83da"
|
| 502 |
+
},
|
| 503 |
+
{
|
| 504 |
+
"path": "transformer/quantization_config.json",
|
| 505 |
+
"bytes": 42776,
|
| 506 |
+
"sha256": "87565b07b201f5171baf85a4f109302ea8414c5e0c8910d5afd0a630907fdb16"
|
| 507 |
+
}
|
| 508 |
+
],
|
| 509 |
+
"total_bytes": 4937691362
|
| 510 |
+
}
|
nvfp4_runtime.py
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Calibrated NVFP4 W4A4 with BF16 rank-128 correction on SM120.
|
| 2 |
+
|
| 3 |
+
Built with Qwen. Quantization modifications by ProCreations, 2026.
|
| 4 |
+
Requires the pinned FlashInfer source revision shipped in requirements.
|
| 5 |
+
"""
|
| 6 |
+
import json
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
import torch
|
| 9 |
+
from torch import nn
|
| 10 |
+
from safetensors.torch import load_file
|
| 11 |
+
|
| 12 |
+
@torch.library.custom_op('image21_nvfp4::linear',mutates_args=())
|
| 13 |
+
def _linear(x:torch.Tensor,weight:torch.Tensor,sf:torch.Tensor,pre:torch.Tensor,
|
| 14 |
+
gx:torch.Tensor,alpha:torch.Tensor,down:torch.Tensor,up:torch.Tensor)->torch.Tensor:
|
| 15 |
+
from flashinfer.gemm import nvfp4_quantize_smooth,mm_nvfp4_svdquant
|
| 16 |
+
from dynamic_scale import scale
|
| 17 |
+
x=x.contiguous()
|
| 18 |
+
gx,alpha,up=scale(x,pre,gx,alpha,up)
|
| 19 |
+
q,sc=nvfp4_quantize_smooth(x,pre,gx,backend='cute-dsl')
|
| 20 |
+
d=torch.mm(x,down)
|
| 21 |
+
return mm_nvfp4_svdquant(q,weight,sc,sf,alpha,d,up,backend='cute-dsl')
|
| 22 |
+
|
| 23 |
+
@_linear.register_fake
|
| 24 |
+
def _linear_fake(x,weight,sf,pre,gx,alpha,down,up):
|
| 25 |
+
return x.new_empty((x.shape[0],weight.shape[0]),dtype=torch.bfloat16)
|
| 26 |
+
|
| 27 |
+
class CalibratedNVFP4Linear(nn.Module):
|
| 28 |
+
def __init__(self,state):
|
| 29 |
+
super().__init__()
|
| 30 |
+
for key in ['weight','sf','pre','gx','alpha','down','up']:
|
| 31 |
+
self.register_buffer(key,state[key])
|
| 32 |
+
self.out_features,self.in_features=self.weight.shape[0],self.weight.shape[1]*2
|
| 33 |
+
self.bias=None
|
| 34 |
+
def forward(self,x):
|
| 35 |
+
shape=x.shape[:-1]
|
| 36 |
+
return _linear(x.reshape(-1,self.in_features),self.weight,self.sf,self.pre,
|
| 37 |
+
self.gx,self.alpha,self.down,self.up).reshape(*shape,self.out_features)
|
| 38 |
+
|
| 39 |
+
def replace_module(model,name,replacement):
|
| 40 |
+
parent,leaf=name.rsplit('.',1)
|
| 41 |
+
setattr(model.get_submodule(parent),leaf,replacement)
|
| 42 |
+
|
| 43 |
+
def load_nvfp4_transformer(directory,device='cuda'):
|
| 44 |
+
from diffusers import QwenImage21Transformer2DModel
|
| 45 |
+
directory=Path(directory)
|
| 46 |
+
config=json.loads((directory/'config.json').read_text())
|
| 47 |
+
meta=json.loads((directory/'quantization_config.json').read_text())
|
| 48 |
+
with torch.device('meta'):
|
| 49 |
+
model=QwenImage21Transformer2DModel.from_config(config)
|
| 50 |
+
for name,spec in meta['quantized_modules'].items():
|
| 51 |
+
n,k=spec['shape'];r=spec['rank']
|
| 52 |
+
state={
|
| 53 |
+
'weight':torch.empty((n,k//2),dtype=torch.uint8),
|
| 54 |
+
'sf':torch.empty(n*k//16,dtype=torch.uint8),
|
| 55 |
+
'pre':torch.empty(k,dtype=torch.bfloat16),
|
| 56 |
+
'gx':torch.empty(1,dtype=torch.float32),
|
| 57 |
+
'alpha':torch.empty(1,dtype=torch.float32),
|
| 58 |
+
'down':torch.empty((k,r),dtype=torch.bfloat16),
|
| 59 |
+
'up':torch.empty((n,r),dtype=torch.bfloat16),
|
| 60 |
+
}
|
| 61 |
+
replace_module(model,name,CalibratedNVFP4Linear(state))
|
| 62 |
+
for name,spec in meta.get('fp8_modules',{}).items():
|
| 63 |
+
from fp8_runtime import CalibratedFP8Linear
|
| 64 |
+
n,k=spec['shape']
|
| 65 |
+
replace_module(model,name,CalibratedFP8Linear(
|
| 66 |
+
torch.empty((n,k),dtype=torch.float8_e4m3fn),
|
| 67 |
+
torch.empty((n,1),dtype=torch.float32),torch.empty(k,dtype=torch.float32)))
|
| 68 |
+
state={}
|
| 69 |
+
for p in sorted(directory.glob('model-*.safetensors')):state.update(load_file(str(p)))
|
| 70 |
+
model.load_state_dict(state,strict=True,assign=True)
|
| 71 |
+
from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21Rope,QwenImage21TemporalTimesteps
|
| 72 |
+
model.pos_embed=QwenImage21Rope(theta=10000,axes_dim=list(model.config.axes_dims_rope))
|
| 73 |
+
model.time_text_embed.time_proj=QwenImage21TemporalTimesteps(timestep_dim=256)
|
| 74 |
+
return model.eval().to(device=device)
|
| 75 |
+
|
| 76 |
+
def load_pipeline(base,quant):
|
| 77 |
+
from diffusers import QwenImage21Pipeline
|
| 78 |
+
kwargs={} if Path(base).exists() else {'revision':'b3179ad355be050328e483a9dfdd9e60cd62adfa'}
|
| 79 |
+
p=QwenImage21Pipeline.from_pretrained(base,transformer=load_nvfp4_transformer(quant),torch_dtype=torch.bfloat16,**kwargs)
|
| 80 |
+
p.to(device='cuda');p.set_progress_bar_config(disable=True)
|
| 81 |
+
return p
|
reports/benchmark.json
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"load_seconds": 2.297096138005145,
|
| 3 |
+
"torch": "2.14.0+cu130",
|
| 4 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 5 |
+
"nvfp4_linears": 224,
|
| 6 |
+
"steps": 40,
|
| 7 |
+
"cfg": 1,
|
| 8 |
+
"bf16_rank": 128,
|
| 9 |
+
"attention_dtype": "bfloat16",
|
| 10 |
+
"approximate_cache": false,
|
| 11 |
+
"timing": {
|
| 12 |
+
"1024": {
|
| 13 |
+
"seconds": [
|
| 14 |
+
4.547408219019417,
|
| 15 |
+
4.571850906999316,
|
| 16 |
+
4.592371508013457,
|
| 17 |
+
4.610020697989967,
|
| 18 |
+
4.624509044981096
|
| 19 |
+
],
|
| 20 |
+
"mean": 4.58923207540065,
|
| 21 |
+
"warmup_seconds": 8.535866206977516,
|
| 22 |
+
"peak_gb": 30.255306752
|
| 23 |
+
},
|
| 24 |
+
"2048": {
|
| 25 |
+
"seconds": [
|
| 26 |
+
32.58121982298326,
|
| 27 |
+
32.67262674000813,
|
| 28 |
+
32.71958359400742
|
| 29 |
+
],
|
| 30 |
+
"mean": 32.657810052332934,
|
| 31 |
+
"warmup_seconds": 35.41050831298344,
|
| 32 |
+
"peak_gb": 51.385951232
|
| 33 |
+
}
|
| 34 |
+
},
|
| 35 |
+
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Prefix KV cache enabled. All large projections use either native NVFP4 with BF16 rank128 correction or explicitly listed calibrated FP8 safety layers. Compiled mode emulates intermediate precision casts."
|
| 36 |
+
}
|
reports/calibration_manifest.json
ADDED
|
@@ -0,0 +1,881 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"source_revision": "b3179ad355be050328e483a9dfdd9e60cd62adfa",
|
| 3 |
+
"rows_per_layer": 1536,
|
| 4 |
+
"sampled_steps": [
|
| 5 |
+
0,
|
| 6 |
+
3,
|
| 7 |
+
9,
|
| 8 |
+
19,
|
| 9 |
+
29,
|
| 10 |
+
39
|
| 11 |
+
],
|
| 12 |
+
"modules": [
|
| 13 |
+
"transformer_blocks.0.attn.to_q",
|
| 14 |
+
"transformer_blocks.0.attn.to_k",
|
| 15 |
+
"transformer_blocks.0.attn.to_v",
|
| 16 |
+
"transformer_blocks.0.attn.to_out.0",
|
| 17 |
+
"transformer_blocks.0.img_mlp.proj",
|
| 18 |
+
"transformer_blocks.0.img_mlp.out",
|
| 19 |
+
"transformer_blocks.0.img_mlp.gate_layer",
|
| 20 |
+
"transformer_blocks.1.attn.to_q",
|
| 21 |
+
"transformer_blocks.1.attn.to_k",
|
| 22 |
+
"transformer_blocks.1.attn.to_v",
|
| 23 |
+
"transformer_blocks.1.attn.to_out.0",
|
| 24 |
+
"transformer_blocks.1.img_mlp.proj",
|
| 25 |
+
"transformer_blocks.1.img_mlp.out",
|
| 26 |
+
"transformer_blocks.1.img_mlp.gate_layer",
|
| 27 |
+
"transformer_blocks.2.attn.to_q",
|
| 28 |
+
"transformer_blocks.2.attn.to_k",
|
| 29 |
+
"transformer_blocks.2.attn.to_v",
|
| 30 |
+
"transformer_blocks.2.attn.to_out.0",
|
| 31 |
+
"transformer_blocks.2.img_mlp.proj",
|
| 32 |
+
"transformer_blocks.2.img_mlp.out",
|
| 33 |
+
"transformer_blocks.2.img_mlp.gate_layer",
|
| 34 |
+
"transformer_blocks.3.attn.to_q",
|
| 35 |
+
"transformer_blocks.3.attn.to_k",
|
| 36 |
+
"transformer_blocks.3.attn.to_v",
|
| 37 |
+
"transformer_blocks.3.attn.to_out.0",
|
| 38 |
+
"transformer_blocks.3.img_mlp.proj",
|
| 39 |
+
"transformer_blocks.3.img_mlp.out",
|
| 40 |
+
"transformer_blocks.3.img_mlp.gate_layer",
|
| 41 |
+
"transformer_blocks.4.attn.to_q",
|
| 42 |
+
"transformer_blocks.4.attn.to_k",
|
| 43 |
+
"transformer_blocks.4.attn.to_v",
|
| 44 |
+
"transformer_blocks.4.attn.to_out.0",
|
| 45 |
+
"transformer_blocks.4.img_mlp.proj",
|
| 46 |
+
"transformer_blocks.4.img_mlp.out",
|
| 47 |
+
"transformer_blocks.4.img_mlp.gate_layer",
|
| 48 |
+
"transformer_blocks.5.attn.to_q",
|
| 49 |
+
"transformer_blocks.5.attn.to_k",
|
| 50 |
+
"transformer_blocks.5.attn.to_v",
|
| 51 |
+
"transformer_blocks.5.attn.to_out.0",
|
| 52 |
+
"transformer_blocks.5.img_mlp.proj",
|
| 53 |
+
"transformer_blocks.5.img_mlp.out",
|
| 54 |
+
"transformer_blocks.5.img_mlp.gate_layer",
|
| 55 |
+
"transformer_blocks.6.attn.to_q",
|
| 56 |
+
"transformer_blocks.6.attn.to_k",
|
| 57 |
+
"transformer_blocks.6.attn.to_v",
|
| 58 |
+
"transformer_blocks.6.attn.to_out.0",
|
| 59 |
+
"transformer_blocks.6.img_mlp.proj",
|
| 60 |
+
"transformer_blocks.6.img_mlp.out",
|
| 61 |
+
"transformer_blocks.6.img_mlp.gate_layer",
|
| 62 |
+
"transformer_blocks.7.attn.to_q",
|
| 63 |
+
"transformer_blocks.7.attn.to_k",
|
| 64 |
+
"transformer_blocks.7.attn.to_v",
|
| 65 |
+
"transformer_blocks.7.attn.to_out.0",
|
| 66 |
+
"transformer_blocks.7.img_mlp.proj",
|
| 67 |
+
"transformer_blocks.7.img_mlp.out",
|
| 68 |
+
"transformer_blocks.7.img_mlp.gate_layer",
|
| 69 |
+
"transformer_blocks.8.attn.to_q",
|
| 70 |
+
"transformer_blocks.8.attn.to_k",
|
| 71 |
+
"transformer_blocks.8.attn.to_v",
|
| 72 |
+
"transformer_blocks.8.attn.to_out.0",
|
| 73 |
+
"transformer_blocks.8.img_mlp.proj",
|
| 74 |
+
"transformer_blocks.8.img_mlp.out",
|
| 75 |
+
"transformer_blocks.8.img_mlp.gate_layer",
|
| 76 |
+
"transformer_blocks.9.attn.to_q",
|
| 77 |
+
"transformer_blocks.9.attn.to_k",
|
| 78 |
+
"transformer_blocks.9.attn.to_v",
|
| 79 |
+
"transformer_blocks.9.attn.to_out.0",
|
| 80 |
+
"transformer_blocks.9.img_mlp.proj",
|
| 81 |
+
"transformer_blocks.9.img_mlp.out",
|
| 82 |
+
"transformer_blocks.9.img_mlp.gate_layer",
|
| 83 |
+
"transformer_blocks.10.attn.to_q",
|
| 84 |
+
"transformer_blocks.10.attn.to_k",
|
| 85 |
+
"transformer_blocks.10.attn.to_v",
|
| 86 |
+
"transformer_blocks.10.attn.to_out.0",
|
| 87 |
+
"transformer_blocks.10.img_mlp.proj",
|
| 88 |
+
"transformer_blocks.10.img_mlp.out",
|
| 89 |
+
"transformer_blocks.10.img_mlp.gate_layer",
|
| 90 |
+
"transformer_blocks.11.attn.to_q",
|
| 91 |
+
"transformer_blocks.11.attn.to_k",
|
| 92 |
+
"transformer_blocks.11.attn.to_v",
|
| 93 |
+
"transformer_blocks.11.attn.to_out.0",
|
| 94 |
+
"transformer_blocks.11.img_mlp.proj",
|
| 95 |
+
"transformer_blocks.11.img_mlp.out",
|
| 96 |
+
"transformer_blocks.11.img_mlp.gate_layer",
|
| 97 |
+
"transformer_blocks.12.attn.to_q",
|
| 98 |
+
"transformer_blocks.12.attn.to_k",
|
| 99 |
+
"transformer_blocks.12.attn.to_v",
|
| 100 |
+
"transformer_blocks.12.attn.to_out.0",
|
| 101 |
+
"transformer_blocks.12.img_mlp.proj",
|
| 102 |
+
"transformer_blocks.12.img_mlp.out",
|
| 103 |
+
"transformer_blocks.12.img_mlp.gate_layer",
|
| 104 |
+
"transformer_blocks.13.attn.to_q",
|
| 105 |
+
"transformer_blocks.13.attn.to_k",
|
| 106 |
+
"transformer_blocks.13.attn.to_v",
|
| 107 |
+
"transformer_blocks.13.attn.to_out.0",
|
| 108 |
+
"transformer_blocks.13.img_mlp.proj",
|
| 109 |
+
"transformer_blocks.13.img_mlp.out",
|
| 110 |
+
"transformer_blocks.13.img_mlp.gate_layer",
|
| 111 |
+
"transformer_blocks.14.attn.to_q",
|
| 112 |
+
"transformer_blocks.14.attn.to_k",
|
| 113 |
+
"transformer_blocks.14.attn.to_v",
|
| 114 |
+
"transformer_blocks.14.attn.to_out.0",
|
| 115 |
+
"transformer_blocks.14.img_mlp.proj",
|
| 116 |
+
"transformer_blocks.14.img_mlp.out",
|
| 117 |
+
"transformer_blocks.14.img_mlp.gate_layer",
|
| 118 |
+
"transformer_blocks.15.attn.to_q",
|
| 119 |
+
"transformer_blocks.15.attn.to_k",
|
| 120 |
+
"transformer_blocks.15.attn.to_v",
|
| 121 |
+
"transformer_blocks.15.attn.to_out.0",
|
| 122 |
+
"transformer_blocks.15.img_mlp.proj",
|
| 123 |
+
"transformer_blocks.15.img_mlp.out",
|
| 124 |
+
"transformer_blocks.15.img_mlp.gate_layer",
|
| 125 |
+
"transformer_blocks.16.attn.to_q",
|
| 126 |
+
"transformer_blocks.16.attn.to_k",
|
| 127 |
+
"transformer_blocks.16.attn.to_v",
|
| 128 |
+
"transformer_blocks.16.attn.to_out.0",
|
| 129 |
+
"transformer_blocks.16.img_mlp.proj",
|
| 130 |
+
"transformer_blocks.16.img_mlp.out",
|
| 131 |
+
"transformer_blocks.16.img_mlp.gate_layer",
|
| 132 |
+
"transformer_blocks.17.attn.to_q",
|
| 133 |
+
"transformer_blocks.17.attn.to_k",
|
| 134 |
+
"transformer_blocks.17.attn.to_v",
|
| 135 |
+
"transformer_blocks.17.attn.to_out.0",
|
| 136 |
+
"transformer_blocks.17.img_mlp.proj",
|
| 137 |
+
"transformer_blocks.17.img_mlp.out",
|
| 138 |
+
"transformer_blocks.17.img_mlp.gate_layer",
|
| 139 |
+
"transformer_blocks.18.attn.to_q",
|
| 140 |
+
"transformer_blocks.18.attn.to_k",
|
| 141 |
+
"transformer_blocks.18.attn.to_v",
|
| 142 |
+
"transformer_blocks.18.attn.to_out.0",
|
| 143 |
+
"transformer_blocks.18.img_mlp.proj",
|
| 144 |
+
"transformer_blocks.18.img_mlp.out",
|
| 145 |
+
"transformer_blocks.18.img_mlp.gate_layer",
|
| 146 |
+
"transformer_blocks.19.attn.to_q",
|
| 147 |
+
"transformer_blocks.19.attn.to_k",
|
| 148 |
+
"transformer_blocks.19.attn.to_v",
|
| 149 |
+
"transformer_blocks.19.attn.to_out.0",
|
| 150 |
+
"transformer_blocks.19.img_mlp.proj",
|
| 151 |
+
"transformer_blocks.19.img_mlp.out",
|
| 152 |
+
"transformer_blocks.19.img_mlp.gate_layer",
|
| 153 |
+
"transformer_blocks.20.attn.to_q",
|
| 154 |
+
"transformer_blocks.20.attn.to_k",
|
| 155 |
+
"transformer_blocks.20.attn.to_v",
|
| 156 |
+
"transformer_blocks.20.attn.to_out.0",
|
| 157 |
+
"transformer_blocks.20.img_mlp.proj",
|
| 158 |
+
"transformer_blocks.20.img_mlp.out",
|
| 159 |
+
"transformer_blocks.20.img_mlp.gate_layer",
|
| 160 |
+
"transformer_blocks.21.attn.to_q",
|
| 161 |
+
"transformer_blocks.21.attn.to_k",
|
| 162 |
+
"transformer_blocks.21.attn.to_v",
|
| 163 |
+
"transformer_blocks.21.attn.to_out.0",
|
| 164 |
+
"transformer_blocks.21.img_mlp.proj",
|
| 165 |
+
"transformer_blocks.21.img_mlp.out",
|
| 166 |
+
"transformer_blocks.21.img_mlp.gate_layer",
|
| 167 |
+
"transformer_blocks.22.attn.to_q",
|
| 168 |
+
"transformer_blocks.22.attn.to_k",
|
| 169 |
+
"transformer_blocks.22.attn.to_v",
|
| 170 |
+
"transformer_blocks.22.attn.to_out.0",
|
| 171 |
+
"transformer_blocks.22.img_mlp.proj",
|
| 172 |
+
"transformer_blocks.22.img_mlp.out",
|
| 173 |
+
"transformer_blocks.22.img_mlp.gate_layer",
|
| 174 |
+
"transformer_blocks.23.attn.to_q",
|
| 175 |
+
"transformer_blocks.23.attn.to_k",
|
| 176 |
+
"transformer_blocks.23.attn.to_v",
|
| 177 |
+
"transformer_blocks.23.attn.to_out.0",
|
| 178 |
+
"transformer_blocks.23.img_mlp.proj",
|
| 179 |
+
"transformer_blocks.23.img_mlp.out",
|
| 180 |
+
"transformer_blocks.23.img_mlp.gate_layer",
|
| 181 |
+
"transformer_blocks.24.attn.to_q",
|
| 182 |
+
"transformer_blocks.24.attn.to_k",
|
| 183 |
+
"transformer_blocks.24.attn.to_v",
|
| 184 |
+
"transformer_blocks.24.attn.to_out.0",
|
| 185 |
+
"transformer_blocks.24.img_mlp.proj",
|
| 186 |
+
"transformer_blocks.24.img_mlp.out",
|
| 187 |
+
"transformer_blocks.24.img_mlp.gate_layer",
|
| 188 |
+
"transformer_blocks.25.attn.to_q",
|
| 189 |
+
"transformer_blocks.25.attn.to_k",
|
| 190 |
+
"transformer_blocks.25.attn.to_v",
|
| 191 |
+
"transformer_blocks.25.attn.to_out.0",
|
| 192 |
+
"transformer_blocks.25.img_mlp.proj",
|
| 193 |
+
"transformer_blocks.25.img_mlp.out",
|
| 194 |
+
"transformer_blocks.25.img_mlp.gate_layer",
|
| 195 |
+
"transformer_blocks.26.attn.to_q",
|
| 196 |
+
"transformer_blocks.26.attn.to_k",
|
| 197 |
+
"transformer_blocks.26.attn.to_v",
|
| 198 |
+
"transformer_blocks.26.attn.to_out.0",
|
| 199 |
+
"transformer_blocks.26.img_mlp.proj",
|
| 200 |
+
"transformer_blocks.26.img_mlp.out",
|
| 201 |
+
"transformer_blocks.26.img_mlp.gate_layer",
|
| 202 |
+
"transformer_blocks.27.attn.to_q",
|
| 203 |
+
"transformer_blocks.27.attn.to_k",
|
| 204 |
+
"transformer_blocks.27.attn.to_v",
|
| 205 |
+
"transformer_blocks.27.attn.to_out.0",
|
| 206 |
+
"transformer_blocks.27.img_mlp.proj",
|
| 207 |
+
"transformer_blocks.27.img_mlp.out",
|
| 208 |
+
"transformer_blocks.27.img_mlp.gate_layer",
|
| 209 |
+
"transformer_blocks.28.attn.to_q",
|
| 210 |
+
"transformer_blocks.28.attn.to_k",
|
| 211 |
+
"transformer_blocks.28.attn.to_v",
|
| 212 |
+
"transformer_blocks.28.attn.to_out.0",
|
| 213 |
+
"transformer_blocks.28.img_mlp.proj",
|
| 214 |
+
"transformer_blocks.28.img_mlp.out",
|
| 215 |
+
"transformer_blocks.28.img_mlp.gate_layer",
|
| 216 |
+
"transformer_blocks.29.attn.to_q",
|
| 217 |
+
"transformer_blocks.29.attn.to_k",
|
| 218 |
+
"transformer_blocks.29.attn.to_v",
|
| 219 |
+
"transformer_blocks.29.attn.to_out.0",
|
| 220 |
+
"transformer_blocks.29.img_mlp.proj",
|
| 221 |
+
"transformer_blocks.29.img_mlp.out",
|
| 222 |
+
"transformer_blocks.29.img_mlp.gate_layer",
|
| 223 |
+
"transformer_blocks.30.attn.to_q",
|
| 224 |
+
"transformer_blocks.30.attn.to_k",
|
| 225 |
+
"transformer_blocks.30.attn.to_v",
|
| 226 |
+
"transformer_blocks.30.attn.to_out.0",
|
| 227 |
+
"transformer_blocks.30.img_mlp.proj",
|
| 228 |
+
"transformer_blocks.30.img_mlp.out",
|
| 229 |
+
"transformer_blocks.30.img_mlp.gate_layer",
|
| 230 |
+
"transformer_blocks.31.attn.to_q",
|
| 231 |
+
"transformer_blocks.31.attn.to_k",
|
| 232 |
+
"transformer_blocks.31.attn.to_v",
|
| 233 |
+
"transformer_blocks.31.attn.to_out.0",
|
| 234 |
+
"transformer_blocks.31.img_mlp.proj",
|
| 235 |
+
"transformer_blocks.31.img_mlp.out",
|
| 236 |
+
"transformer_blocks.31.img_mlp.gate_layer"
|
| 237 |
+
],
|
| 238 |
+
"records": [
|
| 239 |
+
{
|
| 240 |
+
"index": 0,
|
| 241 |
+
"prompt": "A documentary photograph of an elderly fisherman repairing a blue net on a wooden pier, overcast daylight, realistic hands and skin.",
|
| 242 |
+
"seed": 10000,
|
| 243 |
+
"width": 2048,
|
| 244 |
+
"height": 2048,
|
| 245 |
+
"steps": 40,
|
| 246 |
+
"seconds_with_collection": 55.55586172902258,
|
| 247 |
+
"editing": false
|
| 248 |
+
},
|
| 249 |
+
{
|
| 250 |
+
"index": 1,
|
| 251 |
+
"prompt": "A studio portrait of a woman with curly black hair wearing a green wool sweater, soft side lighting, natural skin texture.",
|
| 252 |
+
"seed": 10001,
|
| 253 |
+
"width": 1024,
|
| 254 |
+
"height": 1024,
|
| 255 |
+
"steps": 40,
|
| 256 |
+
"seconds_with_collection": 9.916995454987045,
|
| 257 |
+
"editing": false
|
| 258 |
+
},
|
| 259 |
+
{
|
| 260 |
+
"index": 2,
|
| 261 |
+
"prompt": "A red panda resting on a mossy branch in a misty bamboo forest, wildlife photography, intricate fur.",
|
| 262 |
+
"seed": 10002,
|
| 263 |
+
"width": 1024,
|
| 264 |
+
"height": 1024,
|
| 265 |
+
"steps": 40,
|
| 266 |
+
"seconds_with_collection": 9.919777019997127,
|
| 267 |
+
"editing": false
|
| 268 |
+
},
|
| 269 |
+
{
|
| 270 |
+
"index": 3,
|
| 271 |
+
"prompt": "An aerial photograph of turquoise river channels winding through a dark volcanic plain at sunrise.",
|
| 272 |
+
"seed": 10003,
|
| 273 |
+
"width": 1024,
|
| 274 |
+
"height": 1024,
|
| 275 |
+
"steps": 40,
|
| 276 |
+
"seconds_with_collection": 9.944503634003922,
|
| 277 |
+
"editing": false
|
| 278 |
+
},
|
| 279 |
+
{
|
| 280 |
+
"index": 4,
|
| 281 |
+
"prompt": "A macro photograph of dew on a purple iris, shallow depth of field, intricate translucent petals.",
|
| 282 |
+
"seed": 10004,
|
| 283 |
+
"width": 2048,
|
| 284 |
+
"height": 2048,
|
| 285 |
+
"steps": 40,
|
| 286 |
+
"seconds_with_collection": 55.75184188003186,
|
| 287 |
+
"editing": false
|
| 288 |
+
},
|
| 289 |
+
{
|
| 290 |
+
"index": 5,
|
| 291 |
+
"prompt": "An architectural photograph of a quiet concrete library with tall windows and warm wooden furniture.",
|
| 292 |
+
"seed": 10005,
|
| 293 |
+
"width": 1024,
|
| 294 |
+
"height": 1024,
|
| 295 |
+
"steps": 40,
|
| 296 |
+
"seconds_with_collection": 10.009491505974438,
|
| 297 |
+
"editing": false
|
| 298 |
+
},
|
| 299 |
+
{
|
| 300 |
+
"index": 6,
|
| 301 |
+
"prompt": "A still life of three ceramic bowls, a linen cloth, two pears and a glass of water in window light.",
|
| 302 |
+
"seed": 10006,
|
| 303 |
+
"width": 1024,
|
| 304 |
+
"height": 1024,
|
| 305 |
+
"steps": 40,
|
| 306 |
+
"seconds_with_collection": 9.987017144041602,
|
| 307 |
+
"editing": false
|
| 308 |
+
},
|
| 309 |
+
{
|
| 310 |
+
"index": 7,
|
| 311 |
+
"prompt": "A cinematic photograph of a wet narrow street in Kyoto at dusk, umbrellas and reflected shop lights.",
|
| 312 |
+
"seed": 10007,
|
| 313 |
+
"width": 1024,
|
| 314 |
+
"height": 1024,
|
| 315 |
+
"steps": 40,
|
| 316 |
+
"seconds_with_collection": 9.978878663969226,
|
| 317 |
+
"editing": false
|
| 318 |
+
},
|
| 319 |
+
{
|
| 320 |
+
"index": 8,
|
| 321 |
+
"prompt": "A wide desert landscape with rippling dunes and a lone acacia tree beneath a star-filled sky.",
|
| 322 |
+
"seed": 10008,
|
| 323 |
+
"width": 2048,
|
| 324 |
+
"height": 2048,
|
| 325 |
+
"steps": 40,
|
| 326 |
+
"seconds_with_collection": 55.9559129459667,
|
| 327 |
+
"editing": false
|
| 328 |
+
},
|
| 329 |
+
{
|
| 330 |
+
"index": 9,
|
| 331 |
+
"prompt": "An underwater photograph of a green sea turtle above a colorful coral reef, sunlight shafts.",
|
| 332 |
+
"seed": 10009,
|
| 333 |
+
"width": 1024,
|
| 334 |
+
"height": 1024,
|
| 335 |
+
"steps": 40,
|
| 336 |
+
"seconds_with_collection": 9.959987735957839,
|
| 337 |
+
"editing": false
|
| 338 |
+
},
|
| 339 |
+
{
|
| 340 |
+
"index": 10,
|
| 341 |
+
"prompt": "A watercolor illustration of a small cottage surrounded by wildflowers, delicate washes on textured paper.",
|
| 342 |
+
"seed": 10010,
|
| 343 |
+
"width": 1024,
|
| 344 |
+
"height": 1024,
|
| 345 |
+
"steps": 40,
|
| 346 |
+
"seconds_with_collection": 10.016518865944818,
|
| 347 |
+
"editing": false
|
| 348 |
+
},
|
| 349 |
+
{
|
| 350 |
+
"index": 11,
|
| 351 |
+
"prompt": "A detailed oil painting of a stormy ocean and a distant lighthouse, dramatic layered clouds.",
|
| 352 |
+
"seed": 10011,
|
| 353 |
+
"width": 1024,
|
| 354 |
+
"height": 1024,
|
| 355 |
+
"steps": 40,
|
| 356 |
+
"seconds_with_collection": 10.014705530018546,
|
| 357 |
+
"editing": false
|
| 358 |
+
},
|
| 359 |
+
{
|
| 360 |
+
"index": 12,
|
| 361 |
+
"prompt": "An isometric clay render of a miniature bakery with croissants and copper baking trays, pastel colors.",
|
| 362 |
+
"seed": 10012,
|
| 363 |
+
"width": 2048,
|
| 364 |
+
"height": 2048,
|
| 365 |
+
"steps": 40,
|
| 366 |
+
"seconds_with_collection": 55.98055850400124,
|
| 367 |
+
"editing": false
|
| 368 |
+
},
|
| 369 |
+
{
|
| 370 |
+
"index": 13,
|
| 371 |
+
"prompt": "A black and white ink drawing of an old oak tree with twisted roots, intricate crosshatching.",
|
| 372 |
+
"seed": 10013,
|
| 373 |
+
"width": 1024,
|
| 374 |
+
"height": 1024,
|
| 375 |
+
"steps": 40,
|
| 376 |
+
"seconds_with_collection": 9.911684828985017,
|
| 377 |
+
"editing": false
|
| 378 |
+
},
|
| 379 |
+
{
|
| 380 |
+
"index": 14,
|
| 381 |
+
"prompt": "A friendly orange robot tending a greenhouse full of tropical plants, polished 3D animation style.",
|
| 382 |
+
"seed": 10014,
|
| 383 |
+
"width": 1024,
|
| 384 |
+
"height": 1024,
|
| 385 |
+
"steps": 40,
|
| 386 |
+
"seconds_with_collection": 10.013557444966864,
|
| 387 |
+
"editing": false
|
| 388 |
+
},
|
| 389 |
+
{
|
| 390 |
+
"index": 15,
|
| 391 |
+
"prompt": "A crisp product photograph of a translucent blue perfume bottle on pale stone, precise reflections.",
|
| 392 |
+
"seed": 10015,
|
| 393 |
+
"width": 1024,
|
| 394 |
+
"height": 1024,
|
| 395 |
+
"steps": 40,
|
| 396 |
+
"seconds_with_collection": 10.004387759021483,
|
| 397 |
+
"editing": false
|
| 398 |
+
},
|
| 399 |
+
{
|
| 400 |
+
"index": 16,
|
| 401 |
+
"prompt": "A close-up of a mechanical wristwatch with brushed steel and visible gears, luxury product lighting.",
|
| 402 |
+
"seed": 10016,
|
| 403 |
+
"width": 2048,
|
| 404 |
+
"height": 2048,
|
| 405 |
+
"steps": 40,
|
| 406 |
+
"seconds_with_collection": 56.012295930995606,
|
| 407 |
+
"editing": false
|
| 408 |
+
},
|
| 409 |
+
{
|
| 410 |
+
"index": 17,
|
| 411 |
+
"prompt": "A photograph of a chef pulling a pizza from a brick oven, flour dust and warm firelight.",
|
| 412 |
+
"seed": 10017,
|
| 413 |
+
"width": 1024,
|
| 414 |
+
"height": 1024,
|
| 415 |
+
"steps": 40,
|
| 416 |
+
"seconds_with_collection": 10.00824662600644,
|
| 417 |
+
"editing": false
|
| 418 |
+
},
|
| 419 |
+
{
|
| 420 |
+
"index": 18,
|
| 421 |
+
"prompt": "A handmade paper collage of mountains, a winding river and a yellow sun, visible cut edges.",
|
| 422 |
+
"seed": 10018,
|
| 423 |
+
"width": 1024,
|
| 424 |
+
"height": 1024,
|
| 425 |
+
"steps": 40,
|
| 426 |
+
"seconds_with_collection": 9.993421372026205,
|
| 427 |
+
"editing": false
|
| 428 |
+
},
|
| 429 |
+
{
|
| 430 |
+
"index": 19,
|
| 431 |
+
"prompt": "A charcoal portrait of an elderly violinist, expressive eyes and detailed instrument strings.",
|
| 432 |
+
"seed": 10019,
|
| 433 |
+
"width": 1024,
|
| 434 |
+
"height": 1024,
|
| 435 |
+
"steps": 40,
|
| 436 |
+
"seconds_with_collection": 9.94143287598854,
|
| 437 |
+
"editing": false
|
| 438 |
+
},
|
| 439 |
+
{
|
| 440 |
+
"index": 20,
|
| 441 |
+
"prompt": "A ceramic teapot shaped like a sleeping cat, clean product photography on a cream background.",
|
| 442 |
+
"seed": 10020,
|
| 443 |
+
"width": 2048,
|
| 444 |
+
"height": 2048,
|
| 445 |
+
"steps": 40,
|
| 446 |
+
"seconds_with_collection": 55.824732757988386,
|
| 447 |
+
"editing": false
|
| 448 |
+
},
|
| 449 |
+
{
|
| 450 |
+
"index": 21,
|
| 451 |
+
"prompt": "A cozy reading nook in an attic with a round window, books and a sleeping dog, morning light.",
|
| 452 |
+
"seed": 10021,
|
| 453 |
+
"width": 1024,
|
| 454 |
+
"height": 1024,
|
| 455 |
+
"steps": 40,
|
| 456 |
+
"seconds_with_collection": 10.017206886026543,
|
| 457 |
+
"editing": false
|
| 458 |
+
},
|
| 459 |
+
{
|
| 460 |
+
"index": 22,
|
| 461 |
+
"prompt": "A vibrant science fiction city built inside a giant glass dome on a rocky moon, cinematic wide shot.",
|
| 462 |
+
"seed": 10022,
|
| 463 |
+
"width": 1024,
|
| 464 |
+
"height": 1024,
|
| 465 |
+
"steps": 40,
|
| 466 |
+
"seconds_with_collection": 9.905753094004467,
|
| 467 |
+
"editing": false
|
| 468 |
+
},
|
| 469 |
+
{
|
| 470 |
+
"index": 23,
|
| 471 |
+
"prompt": "A botanical illustration of a sunflower showing leaves, roots and flower head, clean ivory paper.",
|
| 472 |
+
"seed": 10023,
|
| 473 |
+
"width": 1024,
|
| 474 |
+
"height": 1024,
|
| 475 |
+
"steps": 40,
|
| 476 |
+
"seconds_with_collection": 9.950438202009536,
|
| 477 |
+
"editing": false
|
| 478 |
+
},
|
| 479 |
+
{
|
| 480 |
+
"index": 24,
|
| 481 |
+
"prompt": "A photograph of two dancers in flowing red and white costumes on a dark stage, frozen movement.",
|
| 482 |
+
"seed": 10024,
|
| 483 |
+
"width": 2048,
|
| 484 |
+
"height": 2048,
|
| 485 |
+
"steps": 40,
|
| 486 |
+
"seconds_with_collection": 55.87080749304732,
|
| 487 |
+
"editing": false
|
| 488 |
+
},
|
| 489 |
+
{
|
| 490 |
+
"index": 25,
|
| 491 |
+
"prompt": "A bowl of ramen with noodles, mushrooms, eggs and scallions, overhead food photography.",
|
| 492 |
+
"seed": 10025,
|
| 493 |
+
"width": 1024,
|
| 494 |
+
"height": 1024,
|
| 495 |
+
"steps": 40,
|
| 496 |
+
"seconds_with_collection": 10.003536691016052,
|
| 497 |
+
"editing": false
|
| 498 |
+
},
|
| 499 |
+
{
|
| 500 |
+
"index": 26,
|
| 501 |
+
"prompt": "A snowy alpine village reflected in a still lake, blue hour, high detail.",
|
| 502 |
+
"seed": 10026,
|
| 503 |
+
"width": 1024,
|
| 504 |
+
"height": 1024,
|
| 505 |
+
"steps": 40,
|
| 506 |
+
"seconds_with_collection": 9.961158728983719,
|
| 507 |
+
"editing": false
|
| 508 |
+
},
|
| 509 |
+
{
|
| 510 |
+
"index": 27,
|
| 511 |
+
"prompt": "A glass sculpture of a hummingbird with rainbow refractions, black studio background.",
|
| 512 |
+
"seed": 10027,
|
| 513 |
+
"width": 1024,
|
| 514 |
+
"height": 1024,
|
| 515 |
+
"steps": 40,
|
| 516 |
+
"seconds_with_collection": 9.926164573989809,
|
| 517 |
+
"editing": false
|
| 518 |
+
},
|
| 519 |
+
{
|
| 520 |
+
"index": 28,
|
| 521 |
+
"prompt": "A poster with the exact large words \"SPRING GARDEN\" and small text \"OPEN SATURDAY\", flowers and green borders.",
|
| 522 |
+
"seed": 10028,
|
| 523 |
+
"width": 2048,
|
| 524 |
+
"height": 2048,
|
| 525 |
+
"steps": 40,
|
| 526 |
+
"seconds_with_collection": 55.95907588303089,
|
| 527 |
+
"editing": false
|
| 528 |
+
},
|
| 529 |
+
{
|
| 530 |
+
"index": 29,
|
| 531 |
+
"prompt": "A neatly designed cafe menu with the headings \"COFFEE\", \"TEA\" and \"PASTRIES\", black typography on cream paper.",
|
| 532 |
+
"seed": 10029,
|
| 533 |
+
"width": 1024,
|
| 534 |
+
"height": 1024,
|
| 535 |
+
"steps": 40,
|
| 536 |
+
"seconds_with_collection": 9.982229411019944,
|
| 537 |
+
"editing": false
|
| 538 |
+
},
|
| 539 |
+
{
|
| 540 |
+
"index": 30,
|
| 541 |
+
"prompt": "一张精美的海报,标题为“山水之间”,画面是清晨薄雾中的青山和湖泊,优雅的中文排版。",
|
| 542 |
+
"seed": 10030,
|
| 543 |
+
"width": 1024,
|
| 544 |
+
"height": 1024,
|
| 545 |
+
"steps": 40,
|
| 546 |
+
"seconds_with_collection": 9.996177694993094,
|
| 547 |
+
"editing": false
|
| 548 |
+
},
|
| 549 |
+
{
|
| 550 |
+
"index": 31,
|
| 551 |
+
"prompt": "一只橘猫坐在木窗边,窗外下着细雨,室内有温暖的灯光,真实摄影风格。",
|
| 552 |
+
"seed": 10031,
|
| 553 |
+
"width": 1024,
|
| 554 |
+
"height": 1024,
|
| 555 |
+
"steps": 40,
|
| 556 |
+
"seconds_with_collection": 10.037701113033108,
|
| 557 |
+
"editing": false
|
| 558 |
+
},
|
| 559 |
+
{
|
| 560 |
+
"index": 32,
|
| 561 |
+
"prompt": "Une photographie réaliste de lavande dans la campagne française, une petite maison en pierre au loin.",
|
| 562 |
+
"seed": 10032,
|
| 563 |
+
"width": 2048,
|
| 564 |
+
"height": 2048,
|
| 565 |
+
"steps": 40,
|
| 566 |
+
"seconds_with_collection": 55.914367380028125,
|
| 567 |
+
"editing": false
|
| 568 |
+
},
|
| 569 |
+
{
|
| 570 |
+
"index": 33,
|
| 571 |
+
"prompt": "Un mercado de frutas al amanecer, naranjas, limones y flores, fotografía documental con colores naturales.",
|
| 572 |
+
"seed": 10033,
|
| 573 |
+
"width": 1024,
|
| 574 |
+
"height": 1024,
|
| 575 |
+
"steps": 40,
|
| 576 |
+
"seconds_with_collection": 10.030677798960824,
|
| 577 |
+
"editing": false
|
| 578 |
+
},
|
| 579 |
+
{
|
| 580 |
+
"index": 34,
|
| 581 |
+
"prompt": "日本の静かな庭園、石灯籠と池、秋の赤いもみじ、柔らかい朝の光、写真。",
|
| 582 |
+
"seed": 10034,
|
| 583 |
+
"width": 1024,
|
| 584 |
+
"height": 1024,
|
| 585 |
+
"steps": 40,
|
| 586 |
+
"seconds_with_collection": 9.955376817961223,
|
| 587 |
+
"editing": false
|
| 588 |
+
},
|
| 589 |
+
{
|
| 590 |
+
"index": 35,
|
| 591 |
+
"prompt": "صورة فوتوغرافية لقارب خشبي على بحيرة هادئة عند شروق الشمس، جبال في الخلفية.",
|
| 592 |
+
"seed": 10035,
|
| 593 |
+
"width": 1024,
|
| 594 |
+
"height": 1024,
|
| 595 |
+
"steps": 40,
|
| 596 |
+
"seconds_with_collection": 10.00634750595782,
|
| 597 |
+
"editing": false
|
| 598 |
+
},
|
| 599 |
+
{
|
| 600 |
+
"index": 36,
|
| 601 |
+
"prompt": "A group photograph of four friends seated at a picnic table, diverse appearances, natural candid expressions.",
|
| 602 |
+
"seed": 10036,
|
| 603 |
+
"width": 2048,
|
| 604 |
+
"height": 2048,
|
| 605 |
+
"steps": 40,
|
| 606 |
+
"seconds_with_collection": 56.10474174999399,
|
| 607 |
+
"editing": false
|
| 608 |
+
},
|
| 609 |
+
{
|
| 610 |
+
"index": 37,
|
| 611 |
+
"prompt": "A close photograph of a hand holding a delicate seashell, realistic fingers and softly blurred beach.",
|
| 612 |
+
"seed": 10037,
|
| 613 |
+
"width": 1024,
|
| 614 |
+
"height": 1024,
|
| 615 |
+
"steps": 40,
|
| 616 |
+
"seconds_with_collection": 10.005634397966787,
|
| 617 |
+
"editing": false
|
| 618 |
+
},
|
| 619 |
+
{
|
| 620 |
+
"index": 38,
|
| 621 |
+
"prompt": "Three distinct colored cubes: a red cube left, a green cube center and a blue cube right, clean studio photograph.",
|
| 622 |
+
"seed": 10038,
|
| 623 |
+
"width": 1024,
|
| 624 |
+
"height": 1024,
|
| 625 |
+
"steps": 40,
|
| 626 |
+
"seconds_with_collection": 9.967702204012312,
|
| 627 |
+
"editing": false
|
| 628 |
+
},
|
| 629 |
+
{
|
| 630 |
+
"index": 39,
|
| 631 |
+
"prompt": "A tiny astronaut standing on an enormous open book, imaginative photorealistic diorama, soft light.",
|
| 632 |
+
"seed": 10039,
|
| 633 |
+
"width": 1024,
|
| 634 |
+
"height": 1024,
|
| 635 |
+
"steps": 40,
|
| 636 |
+
"seconds_with_collection": 10.009398482972756,
|
| 637 |
+
"editing": false
|
| 638 |
+
},
|
| 639 |
+
{
|
| 640 |
+
"index": 40,
|
| 641 |
+
"prompt": "This is an RGBA image with transparency. A cute cartoon dragon sticker. The image has alpha channel and the background is transparent.",
|
| 642 |
+
"seed": 10040,
|
| 643 |
+
"width": 2048,
|
| 644 |
+
"height": 2048,
|
| 645 |
+
"steps": 40,
|
| 646 |
+
"seconds_with_collection": 55.86863625398837,
|
| 647 |
+
"editing": false
|
| 648 |
+
},
|
| 649 |
+
{
|
| 650 |
+
"index": 41,
|
| 651 |
+
"prompt": "This is an RGBA image with transparency. A realistic pink rose with green leaves. The image has alpha channel and the background is transparent.",
|
| 652 |
+
"seed": 10041,
|
| 653 |
+
"width": 1024,
|
| 654 |
+
"height": 1024,
|
| 655 |
+
"steps": 40,
|
| 656 |
+
"seconds_with_collection": 9.862337956030387,
|
| 657 |
+
"editing": false
|
| 658 |
+
},
|
| 659 |
+
{
|
| 660 |
+
"index": 42,
|
| 661 |
+
"prompt": "This is an RGBA image with transparency. A polished golden compass seen from above. The image has alpha channel and the background is transparent.",
|
| 662 |
+
"seed": 10042,
|
| 663 |
+
"width": 1024,
|
| 664 |
+
"height": 1024,
|
| 665 |
+
"steps": 40,
|
| 666 |
+
"seconds_with_collection": 9.913998158997856,
|
| 667 |
+
"editing": false
|
| 668 |
+
},
|
| 669 |
+
{
|
| 670 |
+
"index": 43,
|
| 671 |
+
"prompt": "This is an RGBA image with transparency. A fluffy white puppy sitting with its paws visible. The image has alpha channel and the background is transparent.",
|
| 672 |
+
"seed": 10043,
|
| 673 |
+
"width": 1024,
|
| 674 |
+
"height": 1024,
|
| 675 |
+
"steps": 40,
|
| 676 |
+
"seconds_with_collection": 9.922774217964616,
|
| 677 |
+
"editing": false
|
| 678 |
+
},
|
| 679 |
+
{
|
| 680 |
+
"index": 44,
|
| 681 |
+
"prompt": "A sunlit photograph of a waterfall pouring through a lush rocky ravine, long exposure water.",
|
| 682 |
+
"seed": 10044,
|
| 683 |
+
"width": 2048,
|
| 684 |
+
"height": 2048,
|
| 685 |
+
"steps": 40,
|
| 686 |
+
"seconds_with_collection": 56.01795354997739,
|
| 687 |
+
"editing": false
|
| 688 |
+
},
|
| 689 |
+
{
|
| 690 |
+
"index": 45,
|
| 691 |
+
"prompt": "A close-up of iridescent soap bubbles against dark velvet, fine colorful interference patterns.",
|
| 692 |
+
"seed": 10045,
|
| 693 |
+
"width": 1024,
|
| 694 |
+
"height": 1024,
|
| 695 |
+
"steps": 40,
|
| 696 |
+
"seconds_with_collection": 10.013371845008805,
|
| 697 |
+
"editing": false
|
| 698 |
+
},
|
| 699 |
+
{
|
| 700 |
+
"index": 46,
|
| 701 |
+
"prompt": "A handwoven basket filled with apples, walnuts and autumn leaves, rustic still life photograph.",
|
| 702 |
+
"seed": 10046,
|
| 703 |
+
"width": 1024,
|
| 704 |
+
"height": 1024,
|
| 705 |
+
"steps": 40,
|
| 706 |
+
"seconds_with_collection": 10.028369715961162,
|
| 707 |
+
"editing": false
|
| 708 |
+
},
|
| 709 |
+
{
|
| 710 |
+
"index": 47,
|
| 711 |
+
"prompt": "A minimalist geometric illustration with overlapping indigo circles and coral triangles on cream.",
|
| 712 |
+
"seed": 10047,
|
| 713 |
+
"width": 1024,
|
| 714 |
+
"height": 1024,
|
| 715 |
+
"steps": 40,
|
| 716 |
+
"seconds_with_collection": 9.945206430973485,
|
| 717 |
+
"editing": false
|
| 718 |
+
},
|
| 719 |
+
{
|
| 720 |
+
"index": 48,
|
| 721 |
+
"prompt": "A double exposure portrait combining a human silhouette with a pine forest, subtle photographic artwork.",
|
| 722 |
+
"seed": 10048,
|
| 723 |
+
"width": 2048,
|
| 724 |
+
"height": 2048,
|
| 725 |
+
"steps": 40,
|
| 726 |
+
"seconds_with_collection": 56.25246442400385,
|
| 727 |
+
"editing": false
|
| 728 |
+
},
|
| 729 |
+
{
|
| 730 |
+
"index": 49,
|
| 731 |
+
"prompt": "An elegant silver spaceship orbiting a blue planet, realistic cinematic lighting and star field.",
|
| 732 |
+
"seed": 10049,
|
| 733 |
+
"width": 1024,
|
| 734 |
+
"height": 1024,
|
| 735 |
+
"steps": 40,
|
| 736 |
+
"seconds_with_collection": 9.991588920995127,
|
| 737 |
+
"editing": false
|
| 738 |
+
},
|
| 739 |
+
{
|
| 740 |
+
"index": 50,
|
| 741 |
+
"prompt": "A photorealistic brown horse galloping across a green meadow, detailed muscles and flowing mane.",
|
| 742 |
+
"seed": 10050,
|
| 743 |
+
"width": 1024,
|
| 744 |
+
"height": 1024,
|
| 745 |
+
"steps": 40,
|
| 746 |
+
"seconds_with_collection": 10.032783773029223,
|
| 747 |
+
"editing": false
|
| 748 |
+
},
|
| 749 |
+
{
|
| 750 |
+
"index": 51,
|
| 751 |
+
"prompt": "A richly detailed mosaic of tropical fish made from small colored glass tiles, museum lighting.",
|
| 752 |
+
"seed": 10051,
|
| 753 |
+
"width": 1024,
|
| 754 |
+
"height": 1024,
|
| 755 |
+
"steps": 40,
|
| 756 |
+
"seconds_with_collection": 9.938231437001377,
|
| 757 |
+
"editing": false
|
| 758 |
+
},
|
| 759 |
+
{
|
| 760 |
+
"index": 52,
|
| 761 |
+
"prompt": "A long horizontal panorama of a rocky seashore under dramatic sunset clouds, realistic photograph.",
|
| 762 |
+
"seed": 10052,
|
| 763 |
+
"width": 1536,
|
| 764 |
+
"height": 864,
|
| 765 |
+
"steps": 40,
|
| 766 |
+
"seconds_with_collection": 13.22676891402807,
|
| 767 |
+
"editing": false
|
| 768 |
+
},
|
| 769 |
+
{
|
| 770 |
+
"index": 53,
|
| 771 |
+
"prompt": "A vertical photograph looking up through a spiral staircase with repeating white railings.",
|
| 772 |
+
"seed": 10053,
|
| 773 |
+
"width": 864,
|
| 774 |
+
"height": 1536,
|
| 775 |
+
"steps": 40,
|
| 776 |
+
"seconds_with_collection": 13.226088834984694,
|
| 777 |
+
"editing": false
|
| 778 |
+
},
|
| 779 |
+
{
|
| 780 |
+
"index": 54,
|
| 781 |
+
"prompt": "A warm sepia photograph of an antique bicycle leaning against a brick wall, climbing roses.",
|
| 782 |
+
"seed": 10054,
|
| 783 |
+
"width": 1024,
|
| 784 |
+
"height": 1024,
|
| 785 |
+
"steps": 40,
|
| 786 |
+
"seconds_with_collection": 9.979204855044372,
|
| 787 |
+
"editing": false
|
| 788 |
+
},
|
| 789 |
+
{
|
| 790 |
+
"index": 55,
|
| 791 |
+
"prompt": "A quiet winter forest of birch trees and soft falling snow, monochromatic fine art photography.",
|
| 792 |
+
"seed": 10055,
|
| 793 |
+
"width": 1024,
|
| 794 |
+
"height": 1024,
|
| 795 |
+
"steps": 40,
|
| 796 |
+
"seconds_with_collection": 9.959185037005227,
|
| 797 |
+
"editing": false
|
| 798 |
+
},
|
| 799 |
+
{
|
| 800 |
+
"index": 56,
|
| 801 |
+
"prompt": "Change the scene to warm sunset lighting while preserving the main subject.",
|
| 802 |
+
"seed": 10056,
|
| 803 |
+
"width": 2048,
|
| 804 |
+
"height": 2048,
|
| 805 |
+
"steps": 40,
|
| 806 |
+
"seconds_with_collection": 61.76318650902249,
|
| 807 |
+
"editing": true
|
| 808 |
+
},
|
| 809 |
+
{
|
| 810 |
+
"index": 57,
|
| 811 |
+
"prompt": "Make the background a lush green garden while preserving the main subject.",
|
| 812 |
+
"seed": 10057,
|
| 813 |
+
"width": 1024,
|
| 814 |
+
"height": 1024,
|
| 815 |
+
"steps": 40,
|
| 816 |
+
"seconds_with_collection": 11.710096635040827,
|
| 817 |
+
"editing": true
|
| 818 |
+
},
|
| 819 |
+
{
|
| 820 |
+
"index": 58,
|
| 821 |
+
"prompt": "Convert this image to a delicate watercolor painting, preserving its composition.",
|
| 822 |
+
"seed": 10058,
|
| 823 |
+
"width": 1024,
|
| 824 |
+
"height": 1024,
|
| 825 |
+
"steps": 40,
|
| 826 |
+
"seconds_with_collection": 11.703723802987952,
|
| 827 |
+
"editing": true
|
| 828 |
+
},
|
| 829 |
+
{
|
| 830 |
+
"index": 59,
|
| 831 |
+
"prompt": "Add gently falling snow and a winter atmosphere while preserving the main subject.",
|
| 832 |
+
"seed": 10059,
|
| 833 |
+
"width": 1024,
|
| 834 |
+
"height": 1024,
|
| 835 |
+
"steps": 40,
|
| 836 |
+
"seconds_with_collection": 11.623913866991643,
|
| 837 |
+
"editing": true
|
| 838 |
+
},
|
| 839 |
+
{
|
| 840 |
+
"index": 60,
|
| 841 |
+
"prompt": "Change the overall color palette to cool blue and silver, preserving details.",
|
| 842 |
+
"seed": 10060,
|
| 843 |
+
"width": 2048,
|
| 844 |
+
"height": 2048,
|
| 845 |
+
"steps": 40,
|
| 846 |
+
"seconds_with_collection": 60.85325455095153,
|
| 847 |
+
"editing": true
|
| 848 |
+
},
|
| 849 |
+
{
|
| 850 |
+
"index": 61,
|
| 851 |
+
"prompt": "Convert the image to a detailed pencil drawing on white paper.",
|
| 852 |
+
"seed": 10061,
|
| 853 |
+
"width": 1024,
|
| 854 |
+
"height": 1024,
|
| 855 |
+
"steps": 40,
|
| 856 |
+
"seconds_with_collection": 11.702045002020895,
|
| 857 |
+
"editing": true
|
| 858 |
+
},
|
| 859 |
+
{
|
| 860 |
+
"index": 62,
|
| 861 |
+
"prompt": "Place a small red flower in the foreground and preserve the rest of the scene.",
|
| 862 |
+
"seed": 10062,
|
| 863 |
+
"width": 1024,
|
| 864 |
+
"height": 1024,
|
| 865 |
+
"steps": 40,
|
| 866 |
+
"seconds_with_collection": 11.716953598021064,
|
| 867 |
+
"editing": true
|
| 868 |
+
},
|
| 869 |
+
{
|
| 870 |
+
"index": 63,
|
| 871 |
+
"prompt": "Give the scene soft cinematic moonlight while preserving its composition.",
|
| 872 |
+
"seed": 10063,
|
| 873 |
+
"width": 1024,
|
| 874 |
+
"height": 1024,
|
| 875 |
+
"steps": 40,
|
| 876 |
+
"seconds_with_collection": 11.728258827002719,
|
| 877 |
+
"editing": true
|
| 878 |
+
}
|
| 879 |
+
],
|
| 880 |
+
"collection": "BF16 teacher trajectories, 4 token rows at each sampled timestep"
|
| 881 |
+
}
|
reports/calibration_search.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
reports/correction-probe.json
ADDED
|
@@ -0,0 +1,122 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
[
|
| 2 |
+
{
|
| 3 |
+
"name": "transformer_blocks.0.attn.to_q",
|
| 4 |
+
"reg": "original",
|
| 5 |
+
"fit": 0.012696055695414543,
|
| 6 |
+
"check": 0.0126740587875247
|
| 7 |
+
},
|
| 8 |
+
{
|
| 9 |
+
"name": "transformer_blocks.0.attn.to_q",
|
| 10 |
+
"reg": 0.01,
|
| 11 |
+
"fit": 0.008233755826950073,
|
| 12 |
+
"check": 0.012077884748578072
|
| 13 |
+
},
|
| 14 |
+
{
|
| 15 |
+
"name": "transformer_blocks.0.attn.to_q",
|
| 16 |
+
"reg": 0.1,
|
| 17 |
+
"fit": 0.008387493900954723,
|
| 18 |
+
"check": 0.011893227696418762
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"name": "transformer_blocks.0.attn.to_q",
|
| 22 |
+
"reg": 1.0,
|
| 23 |
+
"fit": 0.00852095615118742,
|
| 24 |
+
"check": 0.01182559598237276
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"name": "transformer_blocks.8.attn.to_v",
|
| 28 |
+
"reg": "original",
|
| 29 |
+
"fit": 0.08148191124200821,
|
| 30 |
+
"check": 0.08202973008155823
|
| 31 |
+
},
|
| 32 |
+
{
|
| 33 |
+
"name": "transformer_blocks.8.attn.to_v",
|
| 34 |
+
"reg": 0.01,
|
| 35 |
+
"fit": 0.0604797899723053,
|
| 36 |
+
"check": 0.0942867323756218
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"name": "transformer_blocks.8.attn.to_v",
|
| 40 |
+
"reg": 0.1,
|
| 41 |
+
"fit": 0.06049658730626106,
|
| 42 |
+
"check": 0.09427530318498611
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"name": "transformer_blocks.8.attn.to_v",
|
| 46 |
+
"reg": 1.0,
|
| 47 |
+
"fit": 0.06077614054083824,
|
| 48 |
+
"check": 0.094297856092453
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"name": "transformer_blocks.16.img_mlp.proj",
|
| 52 |
+
"reg": "original",
|
| 53 |
+
"fit": 0.05309119075536728,
|
| 54 |
+
"check": 0.05289224535226822
|
| 55 |
+
},
|
| 56 |
+
{
|
| 57 |
+
"name": "transformer_blocks.16.img_mlp.proj",
|
| 58 |
+
"reg": 0.01,
|
| 59 |
+
"fit": 0.040202125906944275,
|
| 60 |
+
"check": 0.06128205358982086
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"name": "transformer_blocks.16.img_mlp.proj",
|
| 64 |
+
"reg": 0.1,
|
| 65 |
+
"fit": 0.04021118953824043,
|
| 66 |
+
"check": 0.06127419322729111
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"name": "transformer_blocks.16.img_mlp.proj",
|
| 70 |
+
"reg": 1.0,
|
| 71 |
+
"fit": 0.040364813059568405,
|
| 72 |
+
"check": 0.061265550553798676
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"name": "transformer_blocks.16.img_mlp.gate_layer",
|
| 76 |
+
"reg": "original",
|
| 77 |
+
"fit": 0.05491224303841591,
|
| 78 |
+
"check": 0.05476883798837662
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"name": "transformer_blocks.16.img_mlp.gate_layer",
|
| 82 |
+
"reg": 0.01,
|
| 83 |
+
"fit": 0.04161699116230011,
|
| 84 |
+
"check": 0.0634395033121109
|
| 85 |
+
},
|
| 86 |
+
{
|
| 87 |
+
"name": "transformer_blocks.16.img_mlp.gate_layer",
|
| 88 |
+
"reg": 0.1,
|
| 89 |
+
"fit": 0.0416269451379776,
|
| 90 |
+
"check": 0.06343034654855728
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"name": "transformer_blocks.16.img_mlp.gate_layer",
|
| 94 |
+
"reg": 1.0,
|
| 95 |
+
"fit": 0.04178332909941673,
|
| 96 |
+
"check": 0.06342266499996185
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"name": "transformer_blocks.24.attn.to_out.0",
|
| 100 |
+
"reg": "original",
|
| 101 |
+
"fit": 0.06296917796134949,
|
| 102 |
+
"check": 0.06366930156946182
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"name": "transformer_blocks.24.attn.to_out.0",
|
| 106 |
+
"reg": 0.01,
|
| 107 |
+
"fit": 0.04285139590501785,
|
| 108 |
+
"check": 0.071813203394413
|
| 109 |
+
},
|
| 110 |
+
{
|
| 111 |
+
"name": "transformer_blocks.24.attn.to_out.0",
|
| 112 |
+
"reg": 0.1,
|
| 113 |
+
"fit": 0.04286669194698334,
|
| 114 |
+
"check": 0.0718073919415474
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"name": "transformer_blocks.24.attn.to_out.0",
|
| 118 |
+
"reg": 1.0,
|
| 119 |
+
"fit": 0.04305445775389671,
|
| 120 |
+
"check": 0.0718240886926651
|
| 121 |
+
}
|
| 122 |
+
]
|
reports/cudagraph-trial.json
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"baseline_latent_only_seconds": 4.459490508015733,
|
| 3 |
+
"warmup": 12.883338742016349,
|
| 4 |
+
"candidate_latent_only_seconds": [
|
| 5 |
+
4.45742104400415,
|
| 6 |
+
4.47657939302735,
|
| 7 |
+
4.488048463012092
|
| 8 |
+
],
|
| 9 |
+
"latent_nrmse": 0.0
|
| 10 |
+
}
|
reports/earlier-bf16-benchmark.json
ADDED
|
@@ -0,0 +1,41 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"torch": "2.14.0+cu130",
|
| 3 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 4 |
+
"capability": [
|
| 5 |
+
12,
|
| 6 |
+
0
|
| 7 |
+
],
|
| 8 |
+
"fp8_linears": 0,
|
| 9 |
+
"transformer_dtypes": {
|
| 10 |
+
"torch.bfloat16": 297
|
| 11 |
+
},
|
| 12 |
+
"resident_gb": 32.457959936,
|
| 13 |
+
"precision": "bf16",
|
| 14 |
+
"timing": {
|
| 15 |
+
"1024": {
|
| 16 |
+
"seconds": [
|
| 17 |
+
9.784915568016004,
|
| 18 |
+
9.794134593976196,
|
| 19 |
+
9.79820777400164,
|
| 20 |
+
9.79243299702648,
|
| 21 |
+
9.791788258997258
|
| 22 |
+
],
|
| 23 |
+
"mean": 9.792295838403515,
|
| 24 |
+
"median": 9.79243299702648,
|
| 25 |
+
"min": 9.784915568016004,
|
| 26 |
+
"max": 9.79820777400164
|
| 27 |
+
},
|
| 28 |
+
"2048": {
|
| 29 |
+
"seconds": [
|
| 30 |
+
55.14897009101696,
|
| 31 |
+
55.18838875496294,
|
| 32 |
+
55.170718070003204
|
| 33 |
+
],
|
| 34 |
+
"mean": 55.16935897199437,
|
| 35 |
+
"median": 55.170718070003204,
|
| 36 |
+
"min": 55.14897009101696,
|
| 37 |
+
"max": 55.18838875496294
|
| 38 |
+
}
|
| 39 |
+
},
|
| 40 |
+
"protocol": "Batch1;40steps;CFG1;prefixKVcache;fully GPU resident;warmup excluded;CUDA synchronized;includes text encoding, denoising and VAE decode;excludes load and PNG write."
|
| 41 |
+
}
|
reports/environment.json
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"python": "3.12.3 (main, Aug 31 2026, 10:18:26) [GCC 13.3.0]",
|
| 3 |
+
"platform": "Linux-7.0.0-31-generic-x86_64-with-glibc2.39",
|
| 4 |
+
"packages": {
|
| 5 |
+
"Jinja2": "3.1.6",
|
| 6 |
+
"MarkupSafe": "3.0.3",
|
| 7 |
+
"PyYAML": "6.0.3",
|
| 8 |
+
"Pygments": "2.21.0",
|
| 9 |
+
"accelerate": "1.15.0",
|
| 10 |
+
"annotated-doc": "0.0.5",
|
| 11 |
+
"anyio": "4.15.1",
|
| 12 |
+
"apache-tvm-ffi": "0.1.14.post0",
|
| 13 |
+
"certifi": "2026.7.22",
|
| 14 |
+
"charset-normalizer": "3.5.1",
|
| 15 |
+
"click": "8.5.0",
|
| 16 |
+
"cuda-bindings": "13.4.2",
|
| 17 |
+
"cuda-core": "1.2.0",
|
| 18 |
+
"cuda-pathfinder": "1.6.0",
|
| 19 |
+
"cuda-python": "13.4.1",
|
| 20 |
+
"cuda-tile": "1.6.0",
|
| 21 |
+
"cuda-toolkit": "13.0.3.0",
|
| 22 |
+
"diffusers": "0.41.0.dev0",
|
| 23 |
+
"einops": "0.8.2",
|
| 24 |
+
"filelock": "3.32.3",
|
| 25 |
+
"flashinfer-python": "0.7.0",
|
| 26 |
+
"fsspec": "2026.7.0",
|
| 27 |
+
"h11": "0.16.0",
|
| 28 |
+
"hf-xet": "1.6.0",
|
| 29 |
+
"httpcore": "1.0.9",
|
| 30 |
+
"httpx": "0.28.1",
|
| 31 |
+
"huggingface_hub": "1.32.0",
|
| 32 |
+
"idna": "3.20",
|
| 33 |
+
"imageio-ffmpeg": "0.6.0",
|
| 34 |
+
"importlib_metadata": "9.0.1",
|
| 35 |
+
"markdown-it-py": "4.2.0",
|
| 36 |
+
"mdurl": "0.1.2",
|
| 37 |
+
"mpmath": "1.3.0",
|
| 38 |
+
"nccl-extensions": "0.1.0",
|
| 39 |
+
"nccl4py": "0.5.0",
|
| 40 |
+
"networkx": "3.6.1",
|
| 41 |
+
"ninja": "1.13.2",
|
| 42 |
+
"numpy": "2.5.2",
|
| 43 |
+
"nvidia-cublas": "13.1.1.3",
|
| 44 |
+
"nvidia-cuda-cupti": "13.0.85",
|
| 45 |
+
"nvidia-cuda-nvdisasm": "13.4.92",
|
| 46 |
+
"nvidia-cuda-nvrtc": "13.0.88",
|
| 47 |
+
"nvidia-cuda-runtime": "13.0.96",
|
| 48 |
+
"nvidia-cudnn-cu13": "9.24.0.43",
|
| 49 |
+
"nvidia-cudnn-frontend": "1.29.0",
|
| 50 |
+
"nvidia-cufft": "12.0.0.61",
|
| 51 |
+
"nvidia-cufile": "1.15.1.6",
|
| 52 |
+
"nvidia-curand": "10.4.0.35",
|
| 53 |
+
"nvidia-cusolver": "12.0.4.66",
|
| 54 |
+
"nvidia-cusparse": "12.6.3.3",
|
| 55 |
+
"nvidia-cusparselt-cu13": "0.8.1",
|
| 56 |
+
"nvidia-cutlass-dsl": "4.7.1",
|
| 57 |
+
"nvidia-cutlass-dsl-libs-base": "4.7.1",
|
| 58 |
+
"nvidia-cutlass-dsl-libs-core": "4.7.1",
|
| 59 |
+
"nvidia-cutlass-dsl-libs-cu12": "4.7.1",
|
| 60 |
+
"nvidia-cutlass-dsl-libs-cu13": "4.7.1",
|
| 61 |
+
"nvidia-ml-py": "13.610.43",
|
| 62 |
+
"nvidia-nccl-cu13": "2.30.7",
|
| 63 |
+
"nvidia-nvjitlink": "13.3.33",
|
| 64 |
+
"nvidia-nvshmem-cu13": "3.4.5",
|
| 65 |
+
"nvidia-nvtx": "13.0.85",
|
| 66 |
+
"packaging": "26.3",
|
| 67 |
+
"pillow": "12.3.0",
|
| 68 |
+
"protobuf": "7.36.2",
|
| 69 |
+
"psutil": "7.2.2",
|
| 70 |
+
"regex": "2026.9.10",
|
| 71 |
+
"requests": "2.34.2",
|
| 72 |
+
"rich": "15.0.0",
|
| 73 |
+
"safetensors": "0.8.0",
|
| 74 |
+
"sentencepiece": "0.2.2",
|
| 75 |
+
"setuptools": "78.1.0",
|
| 76 |
+
"shellingham": "1.5.4",
|
| 77 |
+
"sympy": "1.14.0",
|
| 78 |
+
"tabulate": "0.10.0",
|
| 79 |
+
"tokenizers": "0.23.2",
|
| 80 |
+
"torch": "2.14.0+cu130",
|
| 81 |
+
"torchvision": "0.29.0+cu130",
|
| 82 |
+
"tqdm": "4.70.1",
|
| 83 |
+
"transformers": "5.17.0",
|
| 84 |
+
"triton": "3.8.0",
|
| 85 |
+
"typer": "0.27.2",
|
| 86 |
+
"typing_extensions": "4.16.0",
|
| 87 |
+
"urllib3": "2.8.0",
|
| 88 |
+
"zipp": "4.1.0"
|
| 89 |
+
},
|
| 90 |
+
"flashinfer_commit": "975f90583d9ac8896db14cf0f26e99a853c2f136",
|
| 91 |
+
"diffusers_commit": "80c7ed262aeffbeb43ef13ae04baeb9b84515a69",
|
| 92 |
+
"gpu": "RTX PRO 6000 Blackwell Workstation Edition",
|
| 93 |
+
"driver": "615.71.09",
|
| 94 |
+
"compute_capability": "12.0"
|
| 95 |
+
}
|
reports/evaluation-environment.json
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"torch": "2.14.0",
|
| 3 |
+
"torchvision": "0.29.0",
|
| 4 |
+
"lpips": "0.1.4",
|
| 5 |
+
"scikit-image": "0.26.0",
|
| 6 |
+
"scipy": "1.18.1",
|
| 7 |
+
"numpy": "2.5.3",
|
| 8 |
+
"Pillow": "12.3.0"
|
| 9 |
+
}
|
reports/fresh-fp8-benchmark.json
ADDED
|
@@ -0,0 +1,31 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"load_seconds": 2.801166548044421,
|
| 3 |
+
"torch": "2.14.0+cu130",
|
| 4 |
+
"gpu": "NVIDIA RTX PRO 6000 Blackwell Workstation Edition",
|
| 5 |
+
"fp8_linears": 224,
|
| 6 |
+
"steps": 40,
|
| 7 |
+
"cfg": 1,
|
| 8 |
+
"extra_quantization": false,
|
| 9 |
+
"approximate_cache": false,
|
| 10 |
+
"timing": {
|
| 11 |
+
"1024": {
|
| 12 |
+
"seconds": [
|
| 13 |
+
5.894081406004261,
|
| 14 |
+
5.925464586995076
|
| 15 |
+
],
|
| 16 |
+
"mean": 5.909772996499669,
|
| 17 |
+
"warmup_seconds": 10.963575308967847,
|
| 18 |
+
"peak_gb": 32.590829568
|
| 19 |
+
},
|
| 20 |
+
"2048": {
|
| 21 |
+
"seconds": [
|
| 22 |
+
37.19529012899147,
|
| 23 |
+
37.23217404395109
|
| 24 |
+
],
|
| 25 |
+
"mean": 37.21373208647128,
|
| 26 |
+
"warmup_seconds": 40.54065499798162,
|
| 27 |
+
"peak_gb": 53.722432512
|
| 28 |
+
}
|
| 29 |
+
},
|
| 30 |
+
"protocol": "CUDA synchronized; batch1; full40steps; includes encoder, denoising and VAE; excludes model load, resolution warmup and file writes. Weights and prefixKVcache unchanged. Compiled mode emulates intermediate precision casts."
|
| 31 |
+
}
|
reports/heldout-manifest.json
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"frozen_after_candidate_selection": true,
|
| 3 |
+
"cases": [
|
| 4 |
+
{
|
| 5 |
+
"index": 0,
|
| 6 |
+
"prompt": "A transparent glass teapot filled with amber tea on a slate table beside sliced dragon fruit, soft window light, crisp reflections, studio photograph, no text.",
|
| 7 |
+
"width": 1024,
|
| 8 |
+
"height": 1024,
|
| 9 |
+
"seed": 62000,
|
| 10 |
+
"steps": 40
|
| 11 |
+
},
|
| 12 |
+
{
|
| 13 |
+
"index": 1,
|
| 14 |
+
"prompt": "Three origami cranes arranged in a row on a pale wooden desk: a red crane on the left, a yellow crane in the center, and a blue crane on the right, precise folded paper, no text.",
|
| 15 |
+
"width": 1024,
|
| 16 |
+
"height": 1024,
|
| 17 |
+
"seed": 62001,
|
| 18 |
+
"steps": 40
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"index": 2,
|
| 22 |
+
"prompt": "An elderly pianist playing a black grand piano in a warmly lit room, both hands visible on the keys, realistic fingers, candid documentary photograph, no text.",
|
| 23 |
+
"width": 1024,
|
| 24 |
+
"height": 1024,
|
| 25 |
+
"seed": 62002,
|
| 26 |
+
"steps": 40
|
| 27 |
+
},
|
| 28 |
+
{
|
| 29 |
+
"index": 3,
|
| 30 |
+
"prompt": "A minimalist travel poster with the exact large headline \"SUMMER 2026\", a golden sun above a turquoise sea, elegant bold typography.",
|
| 31 |
+
"width": 1024,
|
| 32 |
+
"height": 1024,
|
| 33 |
+
"seed": 62003,
|
| 34 |
+
"steps": 40
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"index": 4,
|
| 38 |
+
"prompt": "一张精美的中国山水海报,清晰准确的四字标题“山海之间”,远山、碧海和细腻的水墨纹理,优雅留白。",
|
| 39 |
+
"width": 1024,
|
| 40 |
+
"height": 1024,
|
| 41 |
+
"seed": 62004,
|
| 42 |
+
"steps": 40
|
| 43 |
+
},
|
| 44 |
+
{
|
| 45 |
+
"index": 5,
|
| 46 |
+
"prompt": "An intricately engraved brass mechanical dragon sculpture on a dark pedestal, delicate interlocking gears, polished metal highlights, museum product photograph, no text.",
|
| 47 |
+
"width": 2048,
|
| 48 |
+
"height": 2048,
|
| 49 |
+
"seed": 62005,
|
| 50 |
+
"steps": 40
|
| 51 |
+
},
|
| 52 |
+
{
|
| 53 |
+
"index": 6,
|
| 54 |
+
"prompt": "This is an RGBA image with transparency. A charming illustrated red panda holding a small green bamboo leaf, clean outlines, fluffy striped tail. The image has alpha channel and the background is transparent.",
|
| 55 |
+
"width": 1024,
|
| 56 |
+
"height": 1024,
|
| 57 |
+
"seed": 62006,
|
| 58 |
+
"steps": 40
|
| 59 |
+
},
|
| 60 |
+
{
|
| 61 |
+
"index": 7,
|
| 62 |
+
"prompt": "A close-up wildlife photograph of a barn owl on a weathered wooden fence, fine speckled feathers, sharp dark eyes, softly blurred spring meadow in the background, no text.",
|
| 63 |
+
"width": 1024,
|
| 64 |
+
"height": 1024,
|
| 65 |
+
"seed": 62007,
|
| 66 |
+
"steps": 40
|
| 67 |
+
}
|
| 68 |
+
]
|
| 69 |
+
}
|
reports/kernel_evidence.json
ADDED
|
@@ -0,0 +1,81 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"nvfp4_sm120_launches": 224,
|
| 3 |
+
"fp8_sm120_launches": 0,
|
| 4 |
+
"native_bf16_flash_attention": 32,
|
| 5 |
+
"all_kernels": {
|
| 6 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x256_32x3_tt_align8::Params)": 1,
|
| 7 |
+
"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#7}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 3,
|
| 8 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::pow_tensor_scalar_kernel_impl<float, float>(at::TensorIteratorBase&, float)::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 9 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4> >(at::native::ReduceOp<float, at::native::MeanOps<float, float, float, float>, unsigned int, float, 4, 4>)": 1,
|
| 10 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<float>, std::array<char*, 2ul>)": 2,
|
| 11 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::rsqrt_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 12 |
+
"void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > >(at::TensorIteratorBase&, at::native::BinaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> > const&)::{lambda(int)#1})": 3,
|
| 13 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::bfloat16_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda(float)#1}, std::array<char*, 2ul>)": 2,
|
| 14 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x64_32x6_tn_align8::Params)": 2,
|
| 15 |
+
"void cublasLt::splitKreduce_kernel<32, 16, int, __nv_bfloat16, __nv_bfloat16, float, __nv_bfloat16, false, __nv_bfloat16, __nv_bfloat16, __nv_bfloat16, true, false, false, false>(cublasLt::cublasSplitKParams<float>, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, __nv_bfloat16*, float const*, float const*, __nv_bfloat16 const*, __nv_bfloat16 const*, __nv_bfloat16*, void*, long, float*, int*, float*, float*, float const*, float const*, float const*, float const*, float const*)": 227,
|
| 16 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::GeluCUDAKernelImpl(at::TensorIteratorBase&, at::native::GeluType)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 1,
|
| 17 |
+
"void at::native::vectorized_elementwise_kernel<2, at::native::FillFunctor<long>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<long>, std::array<char*, 1ul>)": 3,
|
| 18 |
+
"void at::native::elementwise_kernel<128, 2, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(bool, unsigned long, unsigned long)#1} const&)::{lambda(int)#1})": 1,
|
| 19 |
+
"void at_cuda_detail::cub::detail::scan::DeviceScanInitKernel<at_cuda_detail::cub::ScanTileState<long, true> >(at_cuda_detail::cub::ScanTileState<long, true>, int)": 3,
|
| 20 |
+
"void at_cuda_detail::cub::detail::scan::DeviceScanKernel<at_cuda_detail::cub::detail::scan::policy_hub<long, long, long, unsigned int, std::plus<long> >::Policy1000, long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int, long, false, at_cuda_detail::cub::NullType>(long const*, long*, at_cuda_detail::cub::ScanTileState<long, true>, int, std::plus<long>, at_cuda_detail::cub::NullType, unsigned int)": 3,
|
| 21 |
+
"Memcpy DtoH (Device -> Pinned)": 11,
|
| 22 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>, false>(int, at::native::compare_scalar_kernel<long>(at::TensorIteratorBase&, at::native::(anonymous namespace)::OpType, long)::{lambda(long)#1}, std::array<char*, 2ul>)": 3,
|
| 23 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4> >(at::native::ReduceOp<bool, at::native::func_wrapper_t<bool, at::native::and_kernel_cuda(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#12}::operator()() const::{lambda(bool, bool)#1}>, unsigned int, bool, 4, 4>)": 2,
|
| 24 |
+
"void compute_cuda_kernel<long>(long const*, long const*, long*, long, long)": 3,
|
| 25 |
+
"void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<1>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
|
| 26 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<c10::BFloat16>, std::array<char*, 1ul>)": 2,
|
| 27 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 2, 128, 1, 16, 8>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 28 |
+
"void at::native::vectorized_gather_kernel<16, long>(char*, char*, long*, int, long, long, long, long, bool)": 4,
|
| 29 |
+
"void at_cuda_detail::cub::detail::reduce::DeviceReduceKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, unsigned long long, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, int*, unsigned long long, at_cuda_detail::cub::GridEvenShare<unsigned long long>, cuda::std::__4::plus<void>, cuda::std::__4::__identity)": 4,
|
| 30 |
+
"void at_cuda_detail::cub::detail::reduce::DeviceReduceSingleTileKernel<at_cuda_detail::cub::detail::reduce::policy_hub<int, unsigned long long, cuda::std::__4::plus<void> >::Policy1000, int*, int*, int, cuda::std::__4::plus<void>, int, int, cuda::std::__4::__identity>(int*, int*, int, cuda::std::__4::plus<void>, int, cuda::std::__4::__identity)": 4,
|
| 31 |
+
"void at_cuda_detail::cub::detail::scan::DeviceCompactInitKernel<at_cuda_detail::cub::ScanTileState<int, true>, int*>(at_cuda_detail::cub::ScanTileState<int, true>, int, int*)": 4,
|
| 32 |
+
"void at_cuda_detail::cub::detail::select::DeviceSelectSweepKernel<at_cuda_detail::cub::detail::select::policy_hub<long, bool, int, false, (at_cuda_detail::cub::SelectImpl)0>::Policy1000, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, (at_cuda_detail::cub::SelectImpl)0>(thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::counting_iterator<long, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::transform_iterator<at::native::(anonymous namespace)::NonZeroOp<bool>, bool const*, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default, thrust::THRUST_300001_SM_750_800_860_900_1000_1200_NS::use_default>, long*, int*, at_cuda_detail::cub::ScanTileState<int, true>, at_cuda_detail::cub::NullType, at_cuda_detail::cub::NullType, int, int, at_cuda_detail::cub::detail::select::streaming_context_t<long, true>, at_cuda_detail::cub::detail::vsmem_t)": 4,
|
| 33 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<2> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
|
| 34 |
+
"Memcpy DtoH (Device -> Pageable)": 1,
|
| 35 |
+
"Memcpy HtoD (Pageable -> Device)": 5,
|
| 36 |
+
"Memcpy DtoD (Device -> Device)": 3,
|
| 37 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<8> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 3,
|
| 38 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 2, 128, 1, 16, 2>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<8u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 39 |
+
"void (anonymous namespace)::elementwise_kernel_with_index<int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>(int, at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, function_traits<at::native::arange_cuda_out(c10::Scalar const&, c10::Scalar const&, c10::Scalar const&, at::Tensor&)::{lambda()#1}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}>::result_type*)": 1,
|
| 40 |
+
"void at::native::_scatter_gather_elementwise_kernel<128, 8, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1}>(int, at::native::_cuda_scatter_gather_internal_kernel<false, at::native::OpaqueType<8>, long>::operator()<at::native::TensorAssign>(at::TensorIterator&, long, long, long, at::native::TensorAssign const&)::{lambda(int)#1})": 1,
|
| 41 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::FillFunctor<bool>, std::array<char*, 1ul>, false>(int, at::native::FillFunctor<bool>, std::array<char*, 1ul>)": 1,
|
| 42 |
+
"void at::native::index_elementwise_kernel<128, 4, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1}>(long, at::native::gpu_index_kernel<at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1}>(at::TensorIteratorBase&, c10::ArrayRef<long>, c10::ArrayRef<long>, at::native::index_put_kernel_impl<at::native::OpaqueType<1> >(at::TensorIterator&, c10::ArrayRef<long>, c10::ArrayRef<long>)::{lambda(char*, char const*, long)#1} const&, bool)::{lambda(int)#1})": 1,
|
| 43 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_alignedK_contig<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 1, 128, 1, 8>(at::native::(anonymous namespace)::OpaqueType<2u>*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<2u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 44 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>, false>(int, at::native::AUnaryFunctor<float, float, float, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 2ul>)": 1,
|
| 45 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::cos_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 46 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>, false>(int, at::native::sin_kernel_cuda(at::TensorIteratorBase&)::{lambda()#2}::operator()() const::{lambda()#2}::operator()() const::{lambda(float)#1}, std::array<char*, 2ul>)": 1,
|
| 47 |
+
"void at::native::(anonymous namespace)::CatArrayBatchedCopy_vectorized<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 2, 128, 1, 16, 4>(char*, at::native::(anonymous namespace)::CatArrInputTensorMetadata<at::native::(anonymous namespace)::OpaqueType<4u>, unsigned int, 128, 1>, at::native::(anonymous namespace)::TensorSizeStride<unsigned int, 4u>, int, unsigned int)": 1,
|
| 48 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x2_tn_align8::Params)": 1,
|
| 49 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>, false>(int, at::native::(anonymous namespace)::silu_kernel(at::TensorIteratorBase&)::{lambda()#1}::operator()() const::{lambda()#6}::operator()() const::{lambda(c10::BFloat16)#1}, std::array<char*, 2ul>)": 3,
|
| 50 |
+
"Memset (Device)": 3,
|
| 51 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_128x1_tn_align8::Params)": 2,
|
| 52 |
+
"void cutlass::Kernel2<cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8>(cutlass_80_wmma_tensorop_bf16_s161616gemm_bf16_32x32_64x1_tn_align8::Params)": 1,
|
| 53 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>, false>(int, at::native::bitwise_not_kernel_cuda(at::TensorIteratorBase&)::{lambda(bool)#1}, std::array<char*, 2ul>)": 1,
|
| 54 |
+
"void at::native::unrolled_elementwise_kernel<at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, 4, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1> >(int, at::native::direct_copy_kernel_cuda(at::TensorIteratorBase&)::{lambda()#3}::operator()() const::{lambda()#4}::operator()() const::{lambda(long)#1}, std::array<char*, 2ul>, TrivialOffsetCalculator<1, unsigned int>, TrivialOffsetCalculator<1, unsigned int>, at::native::memory::LoadWithCast<1>, at::native::memory::StoreWithCast<1>)": 1,
|
| 55 |
+
"void at::native::reduce_kernel<512, 1, at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4> >(at::native::ReduceOp<long, at::native::func_wrapper_t<long, at::native::sum_functor<long, long, long>::operator()(at::TensorIterator&)::{lambda(long, long)#1}>, unsigned int, long, 4, 4>)": 1,
|
| 56 |
+
"triton_red_fused_add_mul_native_layer_norm_slice_split_unsqueeze_0": 32,
|
| 57 |
+
"_partials": 224,
|
| 58 |
+
"_finish": 224,
|
| 59 |
+
"_upscale": 224,
|
| 60 |
+
"kernel_cutlass_kernel_flashinferquantizationkernelsnvfp4_quantizeNVFP4QuantizeSwizzledKernel_object_at__tensorptrbf16gmemalign16o409640961_tensorptri8gmemalign16o204820481_tensorptri8gmem_0": 192,
|
| 61 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x128_32x4_nn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_128x128_32x4_nn_align8::Params)": 224,
|
| 62 |
+
"kernel_cutlass_kernel_flashinfergemmkernelsdense_blockscaled_gemm_sm120_b12xDenseGemmKernel_object_at__CopyAtom_ThrID10_TVLayoutSrc11638401_TVLayoutDst11638401_Valuetypef4E2M1FN_tensor00o_0": 224,
|
| 63 |
+
"triton_per_fused__to_copy_mean_pow_view_1": 64,
|
| 64 |
+
"triton_poi_fused__to_copy_add_mean_mul_pow_rsqrt_select_sub_unsqueeze_view_2": 64,
|
| 65 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_3": 32,
|
| 66 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_4": 32,
|
| 67 |
+
"triton_poi_fused__scaled_dot_product_flash_attention__to_copy_cat_stack_transpose_view_5": 32,
|
| 68 |
+
"void pytorch_flash::flash_fwd_kernel<Flash_fwd_kernel_traits<128, 128, 64, 4, false, false, cutlass::bfloat16_t, Flash_kernel_traits<128, 128, 64, 4, cutlass::bfloat16_t> >, false, false, false, false, false, true, false, false>(pytorch_flash::Flash_fwd_params)": 32,
|
| 69 |
+
"triton_red_fused_add_linear_mul_native_layer_norm_slice_split_tanh_unsqueeze_view_6": 32,
|
| 70 |
+
"triton_poi_fused_linear_mul_silu_view_7": 32,
|
| 71 |
+
"kernel_cutlass_kernel_flashinferquantizationkernelsnvfp4_quantizeNVFP4QuantizeSwizzledKernel_object_at__tensorptrbf16gmemalign16o12288122881_tensorptri8gmemalign16o614461441_tensorptri8gm_0": 32,
|
| 72 |
+
"triton_poi_fused_add_mul_slice_split_tanh_unsqueeze_view_8": 32,
|
| 73 |
+
"void at::native::elementwise_kernel<128, 4, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1}>(int, at::native::gpu_kernel_impl_nocast<at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1}>(at::TensorIteratorBase&, at::native::(anonymous namespace)::where_kernel_impl(at::TensorIterator&)::{lambda()#1}::operator()() const::{lambda()#2}::operator()() const::{lambda(bool, unsigned short, unsigned short)#1} const&)::{lambda(int)#1})": 1,
|
| 74 |
+
"void at::native::(anonymous namespace)::vectorized_layer_norm_kernel<c10::BFloat16, float, false>(int, float, c10::BFloat16 const*, c10::BFloat16 const*, c10::BFloat16 const*, float*, float*, c10::BFloat16*)": 1,
|
| 75 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>, false>(int, at::native::CUDAFunctorOnSelf_add<c10::BFloat16>, std::array<char*, 2ul>)": 1,
|
| 76 |
+
"void at::native::vectorized_elementwise_kernel<4, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>, false>(int, at::native::BinaryFunctor<c10::BFloat16, c10::BFloat16, c10::BFloat16, at::native::binary_internal::MulFunctor<float> >, std::array<char*, 3ul>)": 1,
|
| 77 |
+
"void cutlass::Kernel2<cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8>(cutlass_80_tensorop_bf16_s16816gemm_relu_bf16_64x128_64x3_tn_align8::Params)": 1
|
| 78 |
+
},
|
| 79 |
+
"transformer_calls": 40,
|
| 80 |
+
"full_denoising_steps": 40
|
| 81 |
+
}
|