mesmertech commited on
Commit
051ac76
·
verified ·
1 Parent(s): 2ee3fd6

Publish calibrated rank128 INT4 research model and reproducible runtime

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. .gitattributes +3 -0
  2. LICENSE +55 -0
  3. NOTICE +9 -0
  4. README.md +65 -0
  5. SHA256SUMS +338 -0
  6. THIRD_PARTY_NOTICES.md +10 -0
  7. model_index.json +25 -0
  8. processor/chat_template.jinja +120 -0
  9. processor/processor_config.json +65 -0
  10. processor/tokenizer.json +3 -0
  11. processor/tokenizer_config.json +16 -0
  12. reproduction/Dockerfile +17 -0
  13. reproduction/LICENSE +55 -0
  14. reproduction/NOTICE +9 -0
  15. reproduction/REPORT.md +91 -0
  16. reproduction/THIRD_PARTY_NOTICES.md +10 -0
  17. reproduction/artifacts/qwen21-calibration/activation_stats.safetensors +3 -0
  18. reproduction/artifacts/qwen21-calibration/calibration.json +293 -0
  19. reproduction/environment.freeze.txt +193 -0
  20. reproduction/experiments/fidelity-v3/README.md +82 -0
  21. reproduction/experiments/fidelity-v3/REPORT.md +91 -0
  22. reproduction/experiments/fidelity-v3/calibration-jobs.json +75 -0
  23. reproduction/experiments/fidelity-v3/candidate-remaining.json +326 -0
  24. reproduction/experiments/fidelity-v3/expanded-jobs.json +220 -0
  25. reproduction/experiments/fidelity-v3/expanded-rubric.md +92 -0
  26. reproduction/experiments/fidelity-v3/original-18-jobs.json +164 -0
  27. reproduction/experiments/fidelity-v3/performance.md +30 -0
  28. reproduction/experiments/fidelity-v3/quality-jobs-all.json +382 -0
  29. reproduction/experiments/fidelity-v3/quality-jobs.json +83 -0
  30. reproduction/experiments/fidelity-v3/selection.json +15 -0
  31. reproduction/experiments/fidelity-v3/teacher-priority-bf16.json +26 -0
  32. reproduction/experiments/fidelity-v3/teacher-priority.json +58 -0
  33. reproduction/experiments/fidelity-v3/teacher-remaining.json +358 -0
  34. reproduction/lean_encoder.py +163 -0
  35. reproduction/nunchaku_backend/README.md +54 -0
  36. reproduction/nunchaku_backend/__init__.py +11 -0
  37. reproduction/nunchaku_backend/baseline_candidate.py +265 -0
  38. reproduction/nunchaku_backend/baseline_candidate_test.py +141 -0
  39. reproduction/nunchaku_backend/cache_repeat_diagnostic.py +146 -0
  40. reproduction/nunchaku_backend/checkpoint_io.py +28 -0
  41. reproduction/nunchaku_backend/cleanup_test.py +74 -0
  42. reproduction/nunchaku_backend/collect_v3.py +81 -0
  43. reproduction/nunchaku_backend/compare_iterations_v3.py +79 -0
  44. reproduction/nunchaku_backend/convert_activation_reference.py +102 -0
  45. reproduction/nunchaku_backend/convert_recipe_v3.md +160 -0
  46. reproduction/nunchaku_backend/convert_reference.py +72 -0
  47. reproduction/nunchaku_backend/convert_reference_notes.md +54 -0
  48. reproduction/nunchaku_backend/convert_reference_test.py +59 -0
  49. reproduction/nunchaku_backend/denoiser_probe_v3.py +391 -0
  50. reproduction/nunchaku_backend/diagnose_kernel.py +45 -0
.gitattributes CHANGED
@@ -33,3 +33,6 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ processor/tokenizer.json filter=lfs diff=lfs merge=lfs -text
37
+ reproduction/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png filter=lfs diff=lfs merge=lfs -text
38
+ transformer/manifest.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
21
+
22
+ 3. Redistribution
23
+ Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+ c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
33
+
34
+ 5. Intellectual Property
35
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
36
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
37
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
38
+
39
+ 6. Disclaimer of Warranty and Limitation of Liability
40
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
41
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
42
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
43
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
44
+
45
+ 7. Survival and Termination.
46
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
47
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
48
+
49
+ 8. Governing Law and Jurisdiction.
50
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
51
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
52
+
53
+ 9. Other Terms and Conditions.
54
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
55
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
NOTICE ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ Built with Qwen
2
+
3
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
4
+
5
+ Mesmer Image 21 Nunchaku is a modified, independently calibrated quantization of Qwen Image 2.1. The transformer uses Nunchaku signed INT4 weights and activations with BF16 rank128 residual branches. The text encoder is serialized in bitsandbytes NF4; processor, scheduler and VAE originate from the pinned upstream release. These modifications are by MesmerTech, September 2026, and are not an official Qwen or Nunchaku release.
6
+
7
+ The model and derivatives are for non-commercial research and evaluation under the accompanying LICENSE. Commercial use requires a separate license from Qwen.
8
+
9
+ Runtime dependencies retain their respective licenses. The custom linear runtime uses MIT HAN Lab Nunchaku; the conversion packing adapter uses DeepCompressor. See THIRD_PARTY_NOTICES.md for pinned source and license links.
README.md ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: qwen-research
4
+ license_link: LICENSE
5
+ base_model: Qwen/Qwen-Image-2.1
6
+ pipeline_tag: text-to-image
7
+ tags:
8
+ - image-editing
9
+ - nunchaku
10
+ - svdquant
11
+ - int4
12
+ - research
13
+ ---
14
+ # Mesmer Image 21 Nunchaku
15
+
16
+ **Built with Qwen.** My experimental rank128 INT4 quantization of Qwen Image 2.1, tested for generation and editing on a 16 GB RTX 4070 Ti SUPER. This is an independent implementation, not an official Qwen or Nunchaku release.
17
+
18
+ **Research and evaluation only.** The [Qwen Research License](LICENSE) requires a separate commercial license. See [NOTICE](NOTICE) for modifications and attribution.
19
+
20
+ [View my benchmark and full-size comparisons](https://mesmer.tools/benchmarks/qwen-image-2-1).
21
+
22
+ I compared 21 scenarios at both 25 and 40 steps: 14 generation prompts, six single-image edits and one two-image edit. The selected version improves on my first INT4 attempt, but it is **not lossless**. Faces, hands, object counts and layouts can drift. Repeated runs with the same seed can also differ visibly, including with application caches disabled.
23
+
24
+ | 1024 × 1024, RTX 4070 Ti SUPER | 25 steps | 40 steps |
25
+ |---|---:|---:|
26
+ | Generation, median of 14 | 15.50 s | 22.82 s |
27
+ | Single-reference editing, median of 6 | 19.48 s | 28.43 s |
28
+ | Two-reference editing, one case | 23.76 s | 34.35 s |
29
+
30
+ Maximum observed GPU board memory: 11,994 MiB. These are local eager inference measurements, with CPU offload, built-in prefix KV reuse, and no prompt/reference LRU hits. They exclude cloud queue/startup/upload time. The comparison teacher uses a BF16 transformer **with an NF4 text encoder**, not a fully BF16 pipeline. Forty steps follows the upstream starting recommendation; 25 is a measured speed/quality option, not an equivalent-quality promise.
31
+
32
+ ## Contents
33
+
34
+ - `transformer/`: 224 signed W4A4 block projections using upstream Nunchaku CUDA kernels, plus BF16 rank128 branches and unquantized boundary layers. Packed state is 4,655,177,728 bytes.
35
+ - `text_encoder/`: prequantized bitsandbytes NF4 Qwen3-VL with double quantization and BF16 compute. All 36 decoder layers are retained; the runtime removes only unused output components.
36
+ - `vae/`, `processor/`, `scheduler/`, `model_index.json`: pipeline components.
37
+ - `runtime/`: shared inference server and loader. The custom transformer needs this loader; stock pipeline loading alone will not recognize the packed format.
38
+ - `reproduction/`: calibration/export implementation, configuration and measured report. The obsolete first implementation is not shipped as the active runtime.
39
+
40
+ Base revision: `b3179ad355be050328e483a9dfdd9e60cd62adfa`. Original selected transformer manifest SHA256: `cf69e83a646981d22ee6df8d5239b46a50df25d8eb73c9f0478feae87323e6cb`. The original manifest is retained unchanged, including calibration provenance.
41
+
42
+ ## Run
43
+
44
+ Use Linux, CUDA 12.8, Torch 2.8.0 and the matching Nunchaku 1.2.1 wheel. This release was tested on Ada; do not use this INT4 build on RTX 5090/Blackwell. The pinned serving dependencies and Docker configuration are in `runtime/runpod/`.
45
+
46
+ Download this repository to a local directory, review the runtime code, then:
47
+
48
+ ```bash
49
+ export QWEN_BACKEND=nunchaku
50
+ export QWEN_PREQUANT=/absolute/path/to/model
51
+ export QWEN_NUNCHAKU_CHECKPOINT="$QWEN_PREQUANT/transformer"
52
+ export HF_HUB_OFFLINE=1
53
+ cd "$QWEN_PREQUANT/runtime"
54
+ uvicorn server:app --host 127.0.0.1 --port 8091
55
+ ```
56
+
57
+ ```json
58
+ {"input":{"prompt":"A busy night market after rain. Two friends share an umbrella while a vendor hands them a paper bag. A handwritten sign reads FRESH BREAD.","width":1024,"height":1024,"steps":40,"CFGScale":1,"seed":42,"outputFormat":"PNG"}}
59
+ ```
60
+
61
+ POST this body to `/runsync`. For edits add `referenceImages` with one or two public image URLs or data URIs. Optional `uploadUrl` accepts a presigned PUT URL; otherwise the response includes base64. The server serializes jobs. Current API limits are 256–1024 pixels per dimension in multiples of32, up to60 steps and up to2 references, even though the upstream model has broader capabilities.
62
+
63
+ See `reproduction/REPORT.md` for comparison methodology and limitations. Per-layer calibration minimizes measured projection output error; this is not a reproduction of every official SVDQuant training/calibration choice, nor proof of end-to-end equivalence.
64
+
65
+ For calibration reproduction, use the `reproduction/` directory as the POC workspace mounted at `/poc`, with the upstream BF16 model and a separate writable cache mounted at `/cache`. Copy `reproduction/artifacts/qwen21-calibration` to `/cache/qwen21-calibration`. The calibration edit reference is included at its original relative path. Follow `reproduction/nunchaku_backend/README.md`; randomized SVD/CUDA reductions mean byte-identical regenerated checkpoints are not promised.
SHA256SUMS ADDED
@@ -0,0 +1,338 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ 8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d LICENSE
2
+ 0bed3a6a716e52829e0d78110339d6c028bb4d3893c7221cf9efe2ba485f8bed NOTICE
3
+ b07d6247562cfe168dbc98df3435a3b1ffe85f9a8148d9c0bf20e97cef7be38c README.md
4
+ affa01b1499e11cc0fd209fcc6775e659bcf773959e708e268e27c69be808114 THIRD_PARTY_NOTICES.md
5
+ 11ce832ce35b332259dcaa4bcc370bb391694959ce3f74409a321a6b32919a57 model_index.json
6
+ 3636d0f0bd6bef02654cdffdc447b79cb2cef8ab02cc75267345946291a489e4 processor/chat_template.jinja
7
+ 8993ae056f017c142d0dd91c3950b1bcfb4043e118d2f24436ec8e67272113b8 processor/processor_config.json
8
+ be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506 processor/tokenizer.json
9
+ 09bbc8d235d72ce43f6ab3c1d15e55319f52b672780f977f40441345048c2443 processor/tokenizer_config.json
10
+ 604692671915a90ee81430fd7e1097ec626a05659d9a0f80412f19da50ddcc2d reproduction/Dockerfile
11
+ 8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d reproduction/LICENSE
12
+ 0bed3a6a716e52829e0d78110339d6c028bb4d3893c7221cf9efe2ba485f8bed reproduction/NOTICE
13
+ 7ad48b6b89d0abebde961696b5fd33862d14e7e2d91de3fad74264943102a0bd reproduction/REPORT.md
14
+ affa01b1499e11cc0fd209fcc6775e659bcf773959e708e268e27c69be808114 reproduction/THIRD_PARTY_NOTICES.md
15
+ 7f032d0930881f8352309e10084b3e5e4418ac81fb1c69a9bc7457f5fa731018 reproduction/artifacts/qwen21-calibration/activation_stats.safetensors
16
+ 2efdac15b2ad81492b394604e807921fa4268d7e4d410f25261fe6fb78e0d833 reproduction/artifacts/qwen21-calibration/calibration.json
17
+ 5050e49d34b4dcac35f75fcd25638cc401105ffb7595339a7551e67a4d8abcae reproduction/environment.freeze.txt
18
+ f748f62cb3daf7aae90c01d455889a4a6472716bfe806596f88411695469c998 reproduction/experiments/fidelity-v3/README.md
19
+ 7ad48b6b89d0abebde961696b5fd33862d14e7e2d91de3fad74264943102a0bd reproduction/experiments/fidelity-v3/REPORT.md
20
+ 191c65872a09c37b0672df820b8db7fa3ae639c9bc2207bbb3e437124bf7f68b reproduction/experiments/fidelity-v3/calibration-jobs.json
21
+ 5759b11e2380f888ada234e5b125a1aa6d5f2a5694a35522e0908f50bd95a23b reproduction/experiments/fidelity-v3/candidate-remaining.json
22
+ 4d0f6d5d7c4cc3f83676d1d72826546955c1b15446b8b9e2c5fef7c0395f0dfa reproduction/experiments/fidelity-v3/expanded-jobs.json
23
+ f1fdc8b4d7cda630c480f794f0da53e0caf39e5cf78736f28af9c4fc4e5d9fb5 reproduction/experiments/fidelity-v3/expanded-rubric.md
24
+ a3a1ea160e2f2c0fdbb731b6f21d7d388176abc626ca2f6600024255c746951f reproduction/experiments/fidelity-v3/original-18-jobs.json
25
+ ba9ae69d9b203bc5e5c1c998e726ca697da49fddd869ebea423b235fe5cd1441 reproduction/experiments/fidelity-v3/performance.md
26
+ 8993e784dfc3dca10f508bc4f5658bb900b5b15305082dac1d019975624e0e7a reproduction/experiments/fidelity-v3/quality-jobs-all.json
27
+ 03492ea5750ca7ae9b62d2e8e0dc03ab266af846e078a83363ded3f15dbcb32a reproduction/experiments/fidelity-v3/quality-jobs.json
28
+ f416d01bbb950380ec181d6a9431dc8c106270deb7fc38625deb967e5dbe09ee reproduction/experiments/fidelity-v3/selection.json
29
+ f5571496730809e5d3edfabf087f9dfc6efcd1f90fd538a83fa50aded1ecce09 reproduction/experiments/fidelity-v3/teacher-priority-bf16.json
30
+ b584a8116aaa3584d0268704fa0df28d4db632f3de42b27694ad372218e312b7 reproduction/experiments/fidelity-v3/teacher-priority.json
31
+ 7f52529a6eed5bca8d2f327cff6617e4121e2d80cce4f86d6346482de51c8fd7 reproduction/experiments/fidelity-v3/teacher-remaining.json
32
+ 2787ad3ea400a2df3f96b45643cc8c480976e787bb17d2e9e58b54e7872195d8 reproduction/lean_encoder.py
33
+ a94c85703a3ae47eea0ca7edacfc4d9dfe5b96235a9cf8a4a61df69056251630 reproduction/nunchaku_backend/README.md
34
+ 888e726eec3fbf637073e7861a546b2c11cff6a00b5cad60cbe905b3a2782069 reproduction/nunchaku_backend/__init__.py
35
+ bd864b154571b93825f4bd260e66f745e6d1d0fd1767ca8918f759b8127f20fc reproduction/nunchaku_backend/baseline_candidate.py
36
+ 36242e9d2b06d1c77542b030061def860ec8263458d4e57cfa1495f6df51c7f7 reproduction/nunchaku_backend/baseline_candidate_test.py
37
+ d5c0409c0804a66c99549d16ea741e36f2e2fb029ea3662c2592bc54916926a8 reproduction/nunchaku_backend/cache_repeat_diagnostic.py
38
+ dcbc86bf6f6b2cd20317a9aa26c30b6986fed6aab1b231c507f3c88d44fee939 reproduction/nunchaku_backend/checkpoint_io.py
39
+ 5680d0a4cd7d941d69e0df0e436d8e6794c8331ad31c9d61fd95ac1cdfa67bfa reproduction/nunchaku_backend/cleanup_test.py
40
+ 81f1a59b36b15834a1ba81d1ecad627bc23b19fda96f26d2ae4cbf80bd71d765 reproduction/nunchaku_backend/collect_v3.py
41
+ 2f6bdeb29564fbfcf46162df876ab56fbd913b31db8884345285caae3d3caf2b reproduction/nunchaku_backend/compare_iterations_v3.py
42
+ e170904ae03cdf47551c763d182376218e02b28b47be47e524b18877614ebed1 reproduction/nunchaku_backend/convert_activation_reference.py
43
+ 765bd94af8fd02957b5b1b9cd18b7c4432eb8011c3e376fa6dec52380c76ef69 reproduction/nunchaku_backend/convert_recipe_v3.md
44
+ 37e5a426e20416d516a8ba4a1e7ebd55c5366fea042467961852951a3c9fad5e reproduction/nunchaku_backend/convert_reference.py
45
+ 3f0b095b08c3d9022559e9c1dc6d9412b42b444e31897e437040f2c3347b8051 reproduction/nunchaku_backend/convert_reference_notes.md
46
+ f6d5ec714ad98077e0bea90cc0af3e08c417e6a0ef4ea135863939294d3be046 reproduction/nunchaku_backend/convert_reference_test.py
47
+ bac0f1bc6eb5feeae974154363fc2ad978852045b7182165e9c3b8297b7aa7fc reproduction/nunchaku_backend/denoiser_probe_v3.py
48
+ 84bd1d8b1993062686a7406f994e2cb2d597912f2a2364c2f7916718a6b10823 reproduction/nunchaku_backend/diagnose_kernel.py
49
+ 7fcb10f22c5191eabb82222e160ba7434f02099d644c057d74aa25f75cd227fa reproduction/nunchaku_backend/export_v3.py
50
+ 0d43c9c3435dcfd3f821c7e4ce8a7a2cfdc5fc9a329dfca0340756e0550d1471 reproduction/nunchaku_backend/export_v3_test.py
51
+ 7aea9e2f87be048d43a43f9c83a42e14cd313d2d53588612533d44fbd840cf9b reproduction/nunchaku_backend/gptq_v3.py
52
+ a4c48c1b1f6d4abd4d4a8236719049e53827d6da1613c32e3eb2820bf317949b reproduction/nunchaku_backend/gptq_v3_test.py
53
+ 26965ca14b8e4a8f9332bf93a3859f0f22e386313de6fab61995cef95e0995b3 reproduction/nunchaku_backend/hybrid_v3.py
54
+ 73141bec2b023b8342459ef3a07f95f7bb1e74dc03f29cacb956c8dfd791b3bf reproduction/nunchaku_backend/iterations100_probe.md
55
+ e70362903fe14b3a71bafadcb89d11d5b0e80866f296f19f6eca0c529957d28b reproduction/nunchaku_backend/kernel_probe.py
56
+ ac4a8ecc41312dbd675ea309e1532c53fd2d267ac6e60f5c4f9f4fbc4d2db073 reproduction/nunchaku_backend/layout.py
57
+ cc0bfa981c0b615f6e7548f7c583d72cea834f7677cc5167225c655b20c5b4b1 reproduction/nunchaku_backend/mlp_parent_probe_v3.py
58
+ 097acd9b5e792155af27d33930fb3663ec4337d6cd4dba192b28eb09082125a8 reproduction/nunchaku_backend/mlp_parent_probe_v3_test.py
59
+ 770de03be86c5a83b1376f2e612a7459dac2e2e7a99aba43528a986d069a9d5b reproduction/nunchaku_backend/optimize_v3.py
60
+ 6b21f991b863ed8500b2808a709a79fa09ea1c99f19a072dbfcd4b0e2f48b9d0 reproduction/nunchaku_backend/packing.py
61
+ 8e9c2ce022a6569800ea575a8968c7bea413271e1bd83a3183c71574998b55a1 reproduction/nunchaku_backend/probe_iterations_v3.sh
62
+ f7dc00d8f6d643b9a2983acd164f2b0c03abd9379f84421356a582dd46b85dd7 reproduction/nunchaku_backend/rank_probe_v3.md
63
+ d7a04a4d5a573fe7c74d6e0e409cfaabdf414ccfc22c305e8c7fc2dfc2d4d737 reproduction/nunchaku_backend/rank_probe_v3.py
64
+ d42a9be112a8d47e91cd6ef2f940137bb5f670240e54bf793ef4f2291a280746 reproduction/nunchaku_backend/rank_probe_v3_test.py
65
+ 1cee4d9da46125e7f362ad8253ac22921e142209464e41276fcaec84c3f066e7 reproduction/nunchaku_backend/rank_upgrade_helpers.py
66
+ aef04a267f40dbf75003689fda7d274c25081aee27da7eb3e896e0592e256139 reproduction/nunchaku_backend/rank_upgrade_helpers_test.py
67
+ c35a12c28cb982e81c1c5600782db379d4541fedcc54c34ba960472c32b46a94 reproduction/nunchaku_backend/refine_checkpoint_v3.py
68
+ dac9996d5df928a22c719e572256233f1d6331a0d1f93faa994c3930d29852c3 reproduction/nunchaku_backend/refine_v3_test.py
69
+ 67d239d98e03e0fcb8b6d2b6031978b35f0ab39676633898d58e881d2c418ae6 reproduction/nunchaku_backend/refinement100_ready.md
70
+ 8fec8d2e829561c9372f92105fed9c384d1e25fa969f741172f492d74e4a20a8 reproduction/nunchaku_backend/role_sweep_v3.py
71
+ 19f4697e8f8bf20c0c528a82d6ff8159ae2527527c768b84753705593f44eb10 reproduction/nunchaku_backend/runner_hybrid_test.py
72
+ bb0cad2bfec1ea2b7ea5962db4831c10bbed9f86aa6ef0b04bf4cd30bf03168e reproduction/nunchaku_backend/runtime.py
73
+ d7c5c968984e10a0c3ba8e861e8d6a4dc1e84d23a8898c799374c923cbc4579d reproduction/nunchaku_backend/summarize_v3.py
74
+ e1d8c87c9bc867b4dfb0ac581d995abe1824202040e94c471f7077590bd11527 reproduction/nunchaku_backend/teacher_v3.py
75
+ 91562daf3bdb60217ff6fa234969cd6e2b2652ddcd29e0039ca6e5e52e047dcb reproduction/nunchaku_backend/upgrade_mlp_rank_v3.md
76
+ db5b42d8949857e2ae77f8cd3bf741a5c51e7d26d0236658ac6fdc93fb392c0d reproduction/nunchaku_backend/upgrade_mlp_rank_v3.py
77
+ ae7b0cd2d1511aaf699192e5760313409866f0b0e48ca8ac211863c1819af458 reproduction/nunchaku_backend/upgrade_mlp_rank_v3_test.py
78
+ 671985da2939b5565ad0b60891ebca8ad8e778421a711ae021e35153710df8c3 reproduction/nunchaku_backend/upstream/deepcompressor-tree.json
79
+ 032e43f8335089bd1ffe112797478c6b7a46e1b204c048d75776289301b0553f reproduction/nunchaku_backend/upstream/nunchaku-tree.json
80
+ 245cc971a635403851fe77986213a59190e244383c2c75741dc0789facf6ed4d reproduction/nunchaku_backend/upstream/nunchaku__models__linear.py
81
+ a3c209cc785afc3d77f6d78b81b24975671344ff324284c899771cc10be6c833 reproduction/nunchaku_backend/upstream/nunchaku__models__transformers__transformer_qwenimage.py
82
+ 3d865dcb286e8296d1a6e3876f2576675136fdba860406139acdd6e086d1cdcc reproduction/nunchaku_backend/upstream/nunchaku__ops__quantize.py
83
+ a369e7db138a950ae9ae4f80dd1d20e37029603d2c5025c371e0a77f2b29e73c reproduction/nunchaku_backend/v3_notes.md
84
+ c939a0dc977f0bcc21a202888996b2d33cf6b8e351848869ca8416450a93873b reproduction/nunchaku_backend/v3_test.py
85
+ 04fdc67a215261ad700c1356d937087ce8bd6d2b1debd1eb7e3f783f84a55d15 reproduction/nunchaku_backend/validate_rank_upgrade_cpu.py
86
+ 27996383f5ae4568d5bec0512e128ff800edeb6abaa4e5ee2c64c0b48b1ebf26 reproduction/nunchaku_backend/validate_v3_cpu.py
87
+ 6f7e1926881a78b0e342667447d9110642715153e00bb3f677a776004c497ef0 reproduction/repeatability.md
88
+ c62923ee98c8a7876e99b0dfeaa7e6553b09b93a8d6004ff1b8b96fe859f23d3 reproduction/runner.py
89
+ a191a4896478811560e3cf679c23a33b9530252c6d6a72e6136e10418d51c803 reproduction/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png
90
+ 4e231258338bcd8fc2a4fdcd0b5c9b6092f069170fde3137b1c862c1cd95872e reproduction/server.py
91
+ 5ece0c1ef602659ad3a71eb01a05bc951848430cd688141685673fc793f7be6b runtime/API.md
92
+ 8dc973f024ff95966bea25866efa443fd16776dcb1001e681e3d467ea572b28d runtime/LICENSE
93
+ 0bed3a6a716e52829e0d78110339d6c028bb4d3893c7221cf9efe2ba485f8bed runtime/NOTICE
94
+ affa01b1499e11cc0fd209fcc6775e659bcf773959e708e268e27c69be808114 runtime/THIRD_PARTY_NOTICES.md
95
+ d26107b7df25fea9792afdce862b1f5ac7a5d3512fdbfe3575c9876a17666ab1 runtime/lean_encoder.py
96
+ a48779c4bf66738834d01d31da0fbaf817be9a614a5e79d7b1b4ca2341fa2481 runtime/nunchaku_backend/__init__.py
97
+ f0242706d5b5439aee5641b6e16cb576825992d12f3c85037928e783aa7f4b82 runtime/nunchaku_backend/layout.py
98
+ 5c15b3b3459c26e4c614fe3efdda5e1a1220cbb4f63e340ff726b8f28881dc6d runtime/nunchaku_backend/runtime.py
99
+ 1f63fa88b385ca4747043ea6d71645aff5d03a3c943700b4163f79b4dba77498 runtime/runner.py
100
+ 8e0c34d3bafc0ffaef52095d640da340ad90534fb6ab121a03ee22fc6538510f runtime/runpod/Dockerfile
101
+ c799f3ee5bb8d176018dbe96d8f07db0c6e1bad6a79b5f72fe08a20154734268 runtime/runpod/download_models.py
102
+ ba3be5f3ae7945c15aba7acfbc7d41065d6d88993ccf89ae4b41f6abacd25d90 runtime/runpod/handler.py
103
+ d6f05ef501106f7a082f8585e028c8eb9955ae02c4cbcccf6824ac62f1b51204 runtime/runpod/requirements.txt
104
+ c21172fa44d6865969fe4de8f24e709c954a73a5b8714db64024cdb578ee18a9 runtime/server.py
105
+ b73113f657b5ea5aa7be0b6c7bffa671e3e91fb565991ce875cd9ebb7a9737d7 scheduler/scheduler_config.json
106
+ c0bc48f2d1a50def0a306a328236b7e1a260b13da8282ec1673405c12dbd1fca text_encoder/config.json
107
+ 81207a236dbd29cd0df5208c43225ff6d0155e3fe63a40ab2f7331fcf04f4834 text_encoder/generation_config.json
108
+ 61c75987e552c49b9bb804b6665b70e6df10d808397d7dcfe2e6ce94e7292210 text_encoder/model.safetensors
109
+ 3e64be674ff9cf15412ebaa996e7bcbdb1132f0bb3e6e0b6bacbe1dfd78397d3 transformer/config.json
110
+ cf69e83a646981d22ee6df8d5239b46a50df25d8eb73c9f0478feae87323e6cb transformer/manifest.json
111
+ 91cedb20bfd9f986b3dfd666508f7ef59bf2f4eba98784e15071715b9082aba6 transformer/model-boundary.safetensors
112
+ be954118c35254d2512ab5a469765dcc834b9cd9ce23c7f0206de1a63f0d04e5 transformer/model-layer-000.safetensors
113
+ 805910c5b758ae10aa3bd41bd35a0226a0a7ffda8dd5249ef5dc8642b17533f3 transformer/model-layer-001.safetensors
114
+ dee0003b353df79eb67c4fe949e484999b194158c4cbbd144e8d4f7898e57569 transformer/model-layer-002.safetensors
115
+ 3b38108461d9afd2248223c28f208b175822cfdc03d6cac27bd9be8b58ce6b4a transformer/model-layer-003.safetensors
116
+ bf2dedbba5139a3ef770ed583bc6b5391598448fdf81e977f62210830e470493 transformer/model-layer-004.safetensors
117
+ 0d8b6b13096092574df713cfa59333c80d275a8eb983d041cb45b7ef510999c0 transformer/model-layer-005.safetensors
118
+ 43e266de81565062f7e07e8b283ae1678deb0bf3124643e8a1746428a3816098 transformer/model-layer-006.safetensors
119
+ a385b1e5c2bf279906832f96fd028b35939ed4d5e8468ef8a7403695eafb010b transformer/model-layer-007.safetensors
120
+ 438e4c30185fc9f196777196eeafeeca6fbcd0f738b9bbec1d7fc37dd1c51e21 transformer/model-layer-008.safetensors
121
+ f83cbe6924cec25cccd785e226af722ecc1e68a800e57bdfcedbac3d730cb651 transformer/model-layer-009.safetensors
122
+ db6a92ce00cda731450c7b79d5d39fd7df6ce3ffdda6bf58297fd47431fc3bf7 transformer/model-layer-010.safetensors
123
+ 4dc2ec89217357598e6829b46fc4f60c85c46567efbd0f0696cbef036ad11091 transformer/model-layer-011.safetensors
124
+ 54d032380b5660ec2c2b9b1f73ef125fdd05f234164597eb2a6fcc13923945db transformer/model-layer-012.safetensors
125
+ 124ae09e4b97a1667808651d8f445a84b7d8da42bbd84f4d7379b41c6c9d86d2 transformer/model-layer-013.safetensors
126
+ a166d614f6a6941fb7a7508d2a611610abc7bf7f6d1527f52f9c7d7db435f688 transformer/model-layer-014.safetensors
127
+ 4cf90f6c54659fbb2f259b1454da2b9d457f4b12bba9834568fd43bdf8e6acdb transformer/model-layer-015.safetensors
128
+ 4fe7a02c345da061343ac51feea997dfbbee408588d1894ce52ca45804ee7554 transformer/model-layer-016.safetensors
129
+ 403f07d5311ae7da8ed97ace3ece9d3d56284d0a764cee35d9419df1ce79a515 transformer/model-layer-017.safetensors
130
+ fa11beb0cfdb4eaa571a6968872be1f75e25d06e00efef6e44179fc72e04a021 transformer/model-layer-018.safetensors
131
+ 4adcdb2c60bda8d4ec3c96c715cad036dd1c43da11d0ab225dfe72d09ca0dacb transformer/model-layer-019.safetensors
132
+ 95593972d08bd06bba59cc6a7ec9ee2a61a78d6b53ef5e0d01de34719b18cb9e transformer/model-layer-020.safetensors
133
+ e68b95ecfc1df57b3b91dc813d57740acbf40302fa5e5f6989634ea44bd80d26 transformer/model-layer-021.safetensors
134
+ 6c13a089c4784a26862cfaa59312f09fc71f95b72c4d26b8065ee54b6bc7c421 transformer/model-layer-022.safetensors
135
+ 121526e8517c376058d032eecbc2445df4e26faa26fa04361473d83fb94ec960 transformer/model-layer-023.safetensors
136
+ 238667fe2e13404a8adaf981a3283f2593ae859096d26bce454bf574d0a61434 transformer/model-layer-024.safetensors
137
+ e3f3f259fb31dc98060e1de725cad827ca1b0eae0bf5bfa3ab4bc50ff0d11f55 transformer/model-layer-025.safetensors
138
+ 896efca07b1bcd0272af0cd3d580c64762591e738c9b9bff2eec81feda9b6cf2 transformer/model-layer-026.safetensors
139
+ 1feff8f4d8d07777e3c914c0152443d643257e9155a364829fe14fa52eaca563 transformer/model-layer-027.safetensors
140
+ 562ed669c40ca3bd447109beb66df73ad21c00aa82b7b91d50c5d34475fc88c8 transformer/model-layer-028.safetensors
141
+ eb27895adb5ff1402342f530c9244b02344edcaf5cdb7571b55170dc87aa8b66 transformer/model-layer-029.safetensors
142
+ 7ead55cbc062e1a32f48ca2e5bc17f311f0a80e6125e6bad1d4bd30c762434f3 transformer/model-layer-030.safetensors
143
+ b01bbd72216709fb7888e201f1b25a70505740e34ab23d399c17465e324c75e9 transformer/model-layer-031.safetensors
144
+ 49db7476493465a03fce9fc9085e27e596440246b7a0d182074e569114d52f06 transformer/model-layer-032.safetensors
145
+ c67faa03d75949ce4a209210de02cdf58192877d445a0264b0030c55f7c374b9 transformer/model-layer-033.safetensors
146
+ 38b273218108066350b01c2006a6ab667092868dbbc144b390dddf8550be2988 transformer/model-layer-034.safetensors
147
+ 046ac87ebda321a370208bbc879d62f62d439463b297e05a11bbec12990edb7a transformer/model-layer-035.safetensors
148
+ e1e8afb5ccbb88266124e88a233e7545e27591e9e0ac07c8ca9fb45783569eaa transformer/model-layer-036.safetensors
149
+ d3628d58f6e3394ebf0e790c8dd44685b76ee027154c6d5fff863308b5758ab6 transformer/model-layer-037.safetensors
150
+ e73d297573fb3fb87d909441a81d82ed56eed0facd8a38d42836ed8cc61fbb84 transformer/model-layer-038.safetensors
151
+ 6604928bf9b57f4ec5189fd534249e064758d2c3c319db2231b608cabd968a4e transformer/model-layer-039.safetensors
152
+ 2c07c7123bd68cb52e5f69abdee4e2d4cc945dc8c8436662b6ff004422f821dd transformer/model-layer-040.safetensors
153
+ 4177a378ff702ee8e079a980bd47cc759a8f8e25a57b63d6746110ee9ed5e257 transformer/model-layer-041.safetensors
154
+ a200b9c9383eb3c969ea124941df9fb157821c327e4a2991a1073ca0b274943b transformer/model-layer-042.safetensors
155
+ 681f1584a3b9200fdea7ca0f55fc263ca01f1d8f5d80326fc8a00b09f73bbd9d transformer/model-layer-043.safetensors
156
+ 241e6e0ba34c77390c9173c5b512d4544b74c27c3f71e1834d14d900af1f4cd2 transformer/model-layer-044.safetensors
157
+ 5276d95df07f1e59f18388a1f2d210e1c5cfe95f4136df5ea6aa27d2ce112a3f transformer/model-layer-045.safetensors
158
+ 18b52c52508cfa67e73a8c74db48dee8d9b2453a3234557d97e1f3b2ca1dfb9b transformer/model-layer-046.safetensors
159
+ 543bc08d4fc332aa6a290dc026b62ebfc22d05e181f16346ad36003ae1e8566e transformer/model-layer-047.safetensors
160
+ 8717d187ac5dc06762b8c578b791d932221b2adffc85d499d4b8e3a9103af03c transformer/model-layer-048.safetensors
161
+ 56962839346c87c9dc42328b8af42f56a4d678c7bfbea57d36bc7b3dfc7ec9b1 transformer/model-layer-049.safetensors
162
+ ed43ccd63f133a9a191ac6e2037b43f4effeee1e44887f377ddc6459c9d1ef1f transformer/model-layer-050.safetensors
163
+ e3597b0ff8f3a2b798d0873873b9627f0c7e0ef5ceeac560251fdc66fe8f6195 transformer/model-layer-051.safetensors
164
+ fce9a7a76138781ec49ba8054f7d2f8b770aa7b9bcf78787d3db7d8cde159ff4 transformer/model-layer-052.safetensors
165
+ 22b532bffe7dd1ded582cb37b970fa483fe6eb197acdaa8778d6b700a740393a transformer/model-layer-053.safetensors
166
+ ae24df341390707a611a12cd08763386397a6c271e7b1e0dbf062f5116e56af5 transformer/model-layer-054.safetensors
167
+ 089382d22679fe93818b79bd53a52b413c3984d366cdd1331bde0d4f959d4782 transformer/model-layer-055.safetensors
168
+ 12a9d61e10e9623df25d571e42a1b1006aaf0e3eec573e89f53cf6c21a4bf120 transformer/model-layer-056.safetensors
169
+ 6ca13e70f2ed1616e2c9c5604aff6daeebb226d21b04effeeb9b1de1de1ea99f transformer/model-layer-057.safetensors
170
+ d01cfd242427e7bbe69d73245ec95e1d22e43f62bf45f53595019a703135968d transformer/model-layer-058.safetensors
171
+ 378f3c3fcdb57101f5760e988aff466923f5bf0332584860aedf24c3352b8617 transformer/model-layer-059.safetensors
172
+ 662886be033ee5f2ee3541ed7187d3b752dd8176c4688dbb72890d5aaa891532 transformer/model-layer-060.safetensors
173
+ 64375e1fe523f359aca412649e1c30ab1496269f0737e230fade32adcfbf6347 transformer/model-layer-061.safetensors
174
+ 37658199e6de2fb0e049ae0fe30cb16407db44922274cbc911245cec8fa3332b transformer/model-layer-062.safetensors
175
+ 702e1a78ebfd91fe2b6efbef78fefd1257d0509320636e21e557bd86f5e2153c transformer/model-layer-063.safetensors
176
+ 1c3f65152c3afba62069cda1c22df5798184a29c869de87d835e311798906530 transformer/model-layer-064.safetensors
177
+ 610ba7bd5f30ad75be63f2368b5e76320007ded4d3551873c0366d9602d681ca transformer/model-layer-065.safetensors
178
+ f910131d9dbdc89949e03f087ac979b01fbfd1f43eed7ae1db583107d89de328 transformer/model-layer-066.safetensors
179
+ bdeee3b356428a7e75a19982522273328c20708cc572f12fe442f1dde57a3d15 transformer/model-layer-067.safetensors
180
+ 3fc30d0a256df124ca9d254f2c677e01c114b729239a589848b3039289de947c transformer/model-layer-068.safetensors
181
+ 9b6ca9886ac5cf675000ad407d008857f46c0b50617f926d53af1a028a00e278 transformer/model-layer-069.safetensors
182
+ 40ad6cdbd4a624bba006875b65055bfcb78dbf8a3dd22e719f2a5afafdde36d3 transformer/model-layer-070.safetensors
183
+ 333c2026b5f2c6f2746241824a1ef7560fac6a2bb9930c160cca9218b5c0e49d transformer/model-layer-071.safetensors
184
+ 852177773dcff10bae0917265572f7658cd6cb06512a197b6cbba770f269c703 transformer/model-layer-072.safetensors
185
+ 7ec3d0b4a46c5683f3dedd572298dab524cf3f6c22d2db4483786f2f4f1b7f4c transformer/model-layer-073.safetensors
186
+ e042abf43b93871fe6b0a0b3a6374d1d2435308643df8fe720d7e9899f25b8f5 transformer/model-layer-074.safetensors
187
+ ad29ca50a411ef8070919e4217520d1de0f41ca4666c6a2a3b475595a9e0e9ae transformer/model-layer-075.safetensors
188
+ 04471891428d103172959be5afc598594537ce52929fb8ef4e007cd869726a81 transformer/model-layer-076.safetensors
189
+ 2b94f5d716a82f220d3f846b70402346610671e48617084843ec161e558155ea transformer/model-layer-077.safetensors
190
+ 8f686e842a4abea031616d9be089dc679f2b953b2b3bf575a1e109df7eeedbaf transformer/model-layer-078.safetensors
191
+ f080bed05b77c27fa76a2e4f5b7394f59a3a7732c853e358788f95d564ab62f3 transformer/model-layer-079.safetensors
192
+ 60cdb41cdfecd42bcc2313e0c30c03109deaf88349472234a1f065ebe6165c96 transformer/model-layer-080.safetensors
193
+ c2fff5071e36336544a28ea5bab59f34a739b1887b201b1c0c77e4af4112d9ab transformer/model-layer-081.safetensors
194
+ 622a1e8c1e6ebb3acff58b676d822cd3ab7cf849da5d692090f6dbf18e123fbd transformer/model-layer-082.safetensors
195
+ b213327da54a43dd807d91785c035be6b5f1ff651b0e523e8a7d4215231517c8 transformer/model-layer-083.safetensors
196
+ ca594ed744f93b6c247582ce8b238f3f7741f92b11b8de175cd2f898c3fd1201 transformer/model-layer-084.safetensors
197
+ 6178d8df840e6cd804a85a93400ab524a723aa1779109278ed40c1915bf58176 transformer/model-layer-085.safetensors
198
+ cc0fbfa403743d2b45586649e5842c25993c46e69e41842a4c0e95494791b4d5 transformer/model-layer-086.safetensors
199
+ 5aaea51447d8f892bed089a2ed4bd30cffe8a017430423493a2e9fdeaa806fd8 transformer/model-layer-087.safetensors
200
+ 1fb2952585109eb0e33e7fb280f950850b685207c694603f0e241ba19a0efc2c transformer/model-layer-088.safetensors
201
+ 3595b8921f6308f9df9a8bc73c68d28b0628ee3346c2b755a9b0e152ad50dbff transformer/model-layer-089.safetensors
202
+ 779cba9de87994d34841be62fb9fb3098edfea5f8a32719a192820a95b04b570 transformer/model-layer-090.safetensors
203
+ 1d8d72051cf8d046e9d3804ad00c0feb887e8fa89adcd3c74190ebdb510b6241 transformer/model-layer-091.safetensors
204
+ c83ff768a390c18f08ff89120787f05661ba2646a88eb128a80cdefad310aa13 transformer/model-layer-092.safetensors
205
+ b39f0766395452bcd08186a57659bdf8fde0e14302ed6d486913794281ff9cc5 transformer/model-layer-093.safetensors
206
+ 7ca843ab6f7c7c503f31370980a6509e092081518df1c8a5a190bafe761b46d9 transformer/model-layer-094.safetensors
207
+ 4780e7617421b97b54d3fbd997a15410009298cd8ddf901a32dc3662fb6c0c45 transformer/model-layer-095.safetensors
208
+ ef31e5488181a2e5a8efffaa15ab993ea1fbd698aa282a0aff12e787d702aa58 transformer/model-layer-096.safetensors
209
+ fe65aa51d7ae66bd32c7647a32a9e6a6f8e205247da5ecedee6b820b42b457af transformer/model-layer-097.safetensors
210
+ d9530418ef182c32baf1ebbf6597d4f395795e9597cc0ae3480f5a7bc3ae589f transformer/model-layer-098.safetensors
211
+ f061a9bd99376abf7896896f5e3fa39cc75b5dfee8dd3112dc30fd76cd6c99df transformer/model-layer-099.safetensors
212
+ ef0d94bf7d96ddcf977127d461593fd1b1fe92287ebec8b1c039b72337ffbc49 transformer/model-layer-100.safetensors
213
+ eb9e524affd0d6fff922656c540c695f09d5d0ef0dc3bda142c8155b2e03a33c transformer/model-layer-101.safetensors
214
+ 17d356995ea24b6b3f78e421ef5dae1b3e4614f5688c761e0ac44782f97b3587 transformer/model-layer-102.safetensors
215
+ e943711c2e0f0c182336c85bbf3cea24a5c0e736498b54dd87e8f3b56fa7850f transformer/model-layer-103.safetensors
216
+ 7b89be9f61d558b436aa50cb76791fa27acee5ec2b4d5b39b9f81b52e40d2ba9 transformer/model-layer-104.safetensors
217
+ 997415569ddeae044e68348e149237ce72997afe5807eeed9f153b8cb90ea9f6 transformer/model-layer-105.safetensors
218
+ cb839042bf02639fb0f8d7f30245a644ab718865f8219b069eb11ef52a4b3700 transformer/model-layer-106.safetensors
219
+ 37a4279e3632da1979e7e228bb56fa251f2af33b32423685139680ae37e3ed79 transformer/model-layer-107.safetensors
220
+ a72a4ca1452fde96faeb4913cabeb3990afd41fc7a59881fadad9b449e707869 transformer/model-layer-108.safetensors
221
+ cdb26f30293cf565b39dcc48432cb34a1127680ddcd98d5a7843d4529026e583 transformer/model-layer-109.safetensors
222
+ 1640282696d83ecd960d9e53e494a2a0c258c572c0baf40a5d7293e171bf9d70 transformer/model-layer-110.safetensors
223
+ c9fdb01cbb69e977e15d19fd5c676a978bd8de8e77e4d87b128fe4e112f1443b transformer/model-layer-111.safetensors
224
+ bc8406984a60b35a95cc3d3df9fa4043598fd4c4a7369ba46701c1c82e295329 transformer/model-layer-112.safetensors
225
+ 153830f28286160f9c24d532bc060ff5a88f66377ee5599a88fa0b75d18206b8 transformer/model-layer-113.safetensors
226
+ 7c716c2b9efde0599f0d6d5a9c8b3fa372a3ccd35caf1c2f7bae9d59895ce810 transformer/model-layer-114.safetensors
227
+ 1dc07c10fe6e9ae35c6fb80e6a1db2074a170a149129924908845e610c949414 transformer/model-layer-115.safetensors
228
+ a8a6a19494147a2b0b54d1b0c1a478ccb7124da5c85589bfa432f2dd2a79d705 transformer/model-layer-116.safetensors
229
+ 47073f8778699492cd7c3efd5f518cf43d594f42a5e2bb73163eb5562a0df56d transformer/model-layer-117.safetensors
230
+ 1e8a2eeee518d9b399cc6d462d86950fd82e2dcec2a1782e49b4f800365b10a5 transformer/model-layer-118.safetensors
231
+ 9ab2de6d358176842de9cb233f23eb72c979e3d4b930d6d1d1c810ed163b6b47 transformer/model-layer-119.safetensors
232
+ 1b70e6dcd33f38c84f407d528e1244e392af653b9b20320b7e1b08426d7126d0 transformer/model-layer-120.safetensors
233
+ 7c25bcdec9d6343adccc80d1d8736e41513043592df3c7d51b843ab9f4f42a56 transformer/model-layer-121.safetensors
234
+ 09e0aede5e349c62df8cb67a4fe4f384394cc632dabe683c6d60bd011b691b51 transformer/model-layer-122.safetensors
235
+ 7406a84e2cb4384b5f91f7d096da029c90f05f61c27c433444cd6ab7eb29f0f5 transformer/model-layer-123.safetensors
236
+ 7f554df5ac153938fe5f19c8da9b386e4e7d0c302cc41cb8c00c9bed86356809 transformer/model-layer-124.safetensors
237
+ 0d379ea34ac4cb43b0395ccadc262c8d4f3c2467d8705358fbc2ee6ef80abc7e transformer/model-layer-125.safetensors
238
+ f6e6bdfd817ab781c7a897d7a1278baf418990d129c5e2aa9fa53da7c6b54445 transformer/model-layer-126.safetensors
239
+ 6a053d3ce96af8761049b45ba3e9973260c589241075fb6c7aad97af85261d19 transformer/model-layer-127.safetensors
240
+ b3ee8d8f7c634f20269b03f2ed4f9b07aa66322b3a1778551ad4cc895eac2f64 transformer/model-layer-128.safetensors
241
+ fc5d83671933ce3687969b89fa4d3cbe417d87791f6a8eff115df5b5d5d69507 transformer/model-layer-129.safetensors
242
+ bbd42cd7cf806884036cc47f3eb8ce63a39225074069dd937f358df36375f6f7 transformer/model-layer-130.safetensors
243
+ 35399d61dbfef25d553d2ad4a9ab9aaddaf09d34c7a8c1fb26dd81d0359a3b32 transformer/model-layer-131.safetensors
244
+ 76891b27ccbe864d5dc81287f1efcc62f24cf9bf2cad8a95873d836ab3df1362 transformer/model-layer-132.safetensors
245
+ f652df6c51d36d2f84ac3fb2c64f31c90854b61fa8bf9f1316f541533e52c4b9 transformer/model-layer-133.safetensors
246
+ 8092cc3d159db04987ec29ea2dc66f311e1dab723891a98be65c28564dd119b1 transformer/model-layer-134.safetensors
247
+ 86a53245dd56b67d3c6d0f82ead2b958499224c8cc4d4413856380ef5e7c59aa transformer/model-layer-135.safetensors
248
+ d7bd44c6263219fd40557e7ee70e0b6f32798933adb6c2a9f6a610905fb228c8 transformer/model-layer-136.safetensors
249
+ 8d1fa8656256b8c3de2e20012d8c417922d23c2dc2364d52c8e799bc61ef23c9 transformer/model-layer-137.safetensors
250
+ 1dabfe4025ba54ab98fbadc3186358c3e8a2d8f3f35e88f16fbc8a740345e284 transformer/model-layer-138.safetensors
251
+ b5b778bef0930e115b05f6fdafa5f15af628dc0893b9e123d0057ad0c7354dcb transformer/model-layer-139.safetensors
252
+ 9a2caab7233b4b3e6cba26bdd5cad20e481068d7711edd80ebda3a1818a02bbd transformer/model-layer-140.safetensors
253
+ 6f185e3ca0aecd04909b919c4b2f27828ca38142707a2db68d37b874a5ae3325 transformer/model-layer-141.safetensors
254
+ b8216d42ab6e5a2a8c9acc2ef8e5792c3033aff240b7c7f1c5852a2657f9328c transformer/model-layer-142.safetensors
255
+ 83a99ec434c8d942ffcdb01095c57eea23c69d936e505c758987a2cb8f83802e transformer/model-layer-143.safetensors
256
+ 1f2f38a7b782158b024e780af5061c0a925f85217655457302507392d4d7fa1c transformer/model-layer-144.safetensors
257
+ 55deabfd19590e3c31d09c0289341c28e10a3022f7664cb5b3027beb5b173dc9 transformer/model-layer-145.safetensors
258
+ fbdec697030590d2918d16a57cb4df0bd0ed985f86b2cc6e9fc8ecf3d6ec9b74 transformer/model-layer-146.safetensors
259
+ 79d19289ab817171111ad3ebf21f0d899828d33d7d8294d18acde940b457278f transformer/model-layer-147.safetensors
260
+ 5bff9dcdbc570a1a708d1e885916384c5250567a6ff9485cbfc5a1acb6ba4e92 transformer/model-layer-148.safetensors
261
+ a07bb1a5c12fb027651d81e97be8bb5d1d5f9c0941a293e11aaf352416465feb transformer/model-layer-149.safetensors
262
+ 595b454b5adf39d47d578425d0565ee58b99b7c15e0ee8460a930718857b2adc transformer/model-layer-150.safetensors
263
+ ebb02d3899c2c549bf745ee9ce4bd5a0399c7dcfcece4d5ec095bc97476d0611 transformer/model-layer-151.safetensors
264
+ 3dde89f1f23a5f29ff717be2a2a466a1fba52770eac4708e4d5c1fc60b0ea6bd transformer/model-layer-152.safetensors
265
+ f4c08a3a7f9c4db79f222e1a0bd4292570b1639a4c92077cc3a67731b35fabb0 transformer/model-layer-153.safetensors
266
+ 7aecccd706d0a48e73ed91c47bf48241a1dab1dcdc35c1211bf7662745e3ab77 transformer/model-layer-154.safetensors
267
+ a6d713ad190ddf60d32ccbcce641fa2a4624a34fb7721e4ff7f06c5a591edce7 transformer/model-layer-155.safetensors
268
+ aca79eaf47f044ece19a2700c253ed460d14af7f010ecffda533f0f39e21a1af transformer/model-layer-156.safetensors
269
+ 1a50e43104053ceb8044f7eb23429eb65617aecc4dab6630493256dbedf9f90a transformer/model-layer-157.safetensors
270
+ 0d6e0b4a265ff0e5b74efc3d2c1531033fdd88215d7f98e12353ed6dcf53bf65 transformer/model-layer-158.safetensors
271
+ 4ceae84b1d40e869902c0389cdb4cdda6c0e7f25c655df54639a903b5a1dcb23 transformer/model-layer-159.safetensors
272
+ f5480155d2336e66accfdc3ff68bdd7add76e2ee69a52d4b79f84008122f4775 transformer/model-layer-160.safetensors
273
+ d90c15eda925b9e4078346c3d6f8ab244acc22f7564284ceecaa8da534c792d6 transformer/model-layer-161.safetensors
274
+ 937689f2e2b07dd3eb5c676de842613df063c09d919512a5a0dd4527595d1a73 transformer/model-layer-162.safetensors
275
+ 0c0510d9bb9133bbc158ef839b613c405d2c48527efdc4233d40d72bf12729e7 transformer/model-layer-163.safetensors
276
+ 2304466eb2d328c41b90ea9ccf36a0ce9f8c6c546c61b16b927d2d31b0fa6f3f transformer/model-layer-164.safetensors
277
+ 2d4c60764ea4134625ba6a53bf680916401dd2d542726773eb23ae98d675d939 transformer/model-layer-165.safetensors
278
+ 2af0f011d764cb17330079edc8bcd81579008b435ccfa09535d2d617f2540e40 transformer/model-layer-166.safetensors
279
+ bfb70cc86e8346d1e8ec99b0f1169c7f09045c6dcafc9f4c83e37e792d8ccc1a transformer/model-layer-167.safetensors
280
+ 6b62b906fd2e95b94cb49ff99a30ccab6de09f43ab90d822073f06065ef34848 transformer/model-layer-168.safetensors
281
+ bd2c6280cb61f2df44303881c2f03cca3b8144085f3103abef7cde079b6ba61c transformer/model-layer-169.safetensors
282
+ f6e436975ca7f716bb2cef12cb6f580c04ace3407c9ac93fdcca38dec253fad2 transformer/model-layer-170.safetensors
283
+ 63d8cdeac345b59db7c7d13ef181db297f722f482ba403221087d88586ca51ae transformer/model-layer-171.safetensors
284
+ 076ffdc562304bb46f13a63153c276fed0c5e7a016869bb8abff4aa96bc8d73a transformer/model-layer-172.safetensors
285
+ 92560f36cb2caf425f0f582a59d9cb881dbed4cb9b24b7e269295756e14857b7 transformer/model-layer-173.safetensors
286
+ 96f258672dea3940c766f99102d2e7875eefdb122bb470c9917b03544a71a7f8 transformer/model-layer-174.safetensors
287
+ 07a4006bfdf83fa3edf1d76a91d32569d7d01ab6d6df8094f7610695fb12eb5c transformer/model-layer-175.safetensors
288
+ a7a469067f641186b3ee98d3599018146123476ea8cf17f34f564fd2b1be7331 transformer/model-layer-176.safetensors
289
+ 1068a02cb088b6ff81828bfe41ddc29e5ed39d5110fcd0765329eb9eceef13de transformer/model-layer-177.safetensors
290
+ 232d48d785782768233dabd00869e5fc830e1ed3afdbdfc1584c4e4998e269e6 transformer/model-layer-178.safetensors
291
+ 65dfba28d0eb981c4511519315124786d17b7dda06a862b950c42e64e257bef3 transformer/model-layer-179.safetensors
292
+ 154299e21c12fc6f6ffe0c925b680b037e9770ba1f4c290b92ae7e9b6d82ad07 transformer/model-layer-180.safetensors
293
+ f8c8f0218cddb0766c81c8e53ed75a381c2dbe6caa51c37a648ecd72566ec9c5 transformer/model-layer-181.safetensors
294
+ 569cfa0dc65e1f0962ecadd00014ffeafdca746a962b6125fe0a3b695ceecb7b transformer/model-layer-182.safetensors
295
+ 9a1e032e18799849da0cdcfd12212dd8c64f0d30cfd845c01210b58ca65eda4b transformer/model-layer-183.safetensors
296
+ a6eadd8851c292ed4720e86fb1843b4a6aac42c6eb28de8d1655a6c3986337db transformer/model-layer-184.safetensors
297
+ 538fe7cd6635b778c45ca08f7325345f5ca8011394d22ecd14552fbcdae1b156 transformer/model-layer-185.safetensors
298
+ aaea8e82f725e47bfaae1c203e43bb72726925a0f6c7723c71fe9324d96e49dc transformer/model-layer-186.safetensors
299
+ 65491dfc924613b72091a7023eda6a1bfcfddc935f55fdbb8e9167ae9e926c01 transformer/model-layer-187.safetensors
300
+ ee2a07daae8304502a47b24814e4acf020c1db8447d11d9d3b915f1df736f071 transformer/model-layer-188.safetensors
301
+ 097fd99046f597a1c420a5631c329c570d28d43d9e39246022a04391606aacee transformer/model-layer-189.safetensors
302
+ e455979028643775c982bcca099c238373eb398890470844bd752a92935570d1 transformer/model-layer-190.safetensors
303
+ f80d4b0a029bf21df4fcb2d1f5dbe64bdca08b944f4f855330b6bed858f62b45 transformer/model-layer-191.safetensors
304
+ fc8620ef733372e04139aa28cb66e79521d822f62d0d5a9fae0b8331cc6058f7 transformer/model-layer-192.safetensors
305
+ 040ceabbe19d1bb7ba6d3ffa003911c04fb9ae058d4904a94eef63214c972ed3 transformer/model-layer-193.safetensors
306
+ 67d90b1f8937a77cef64eb98f7d4524b7bfb63ca8091e37b5e3d5b9cb7de983f transformer/model-layer-194.safetensors
307
+ aa44539d208f33cec16ff224233ed529e099dc3c3ce55171ffe0471df564ab74 transformer/model-layer-195.safetensors
308
+ 530a027205a2ececbbb5fc4d1d7b583203c02d52cc1510be952d97b688d4a588 transformer/model-layer-196.safetensors
309
+ b22addafc782d0026fe5c8a38fe86cde27142db717f5d3814bc21eb5e7c96f72 transformer/model-layer-197.safetensors
310
+ d35be6bf6784806af8c1fec6f42fb0c334d26dd7f3f6b0d5eda436ba6eee2f12 transformer/model-layer-198.safetensors
311
+ 02e7643b75d7b32f660fe066cc27fb25a15a4bbd2de7f7b34893cea28d39c83c transformer/model-layer-199.safetensors
312
+ 5e130350f6946bfde73715a0772685f89172fc06ffc7887d4e913edd3ca31a64 transformer/model-layer-200.safetensors
313
+ 4e530d8d190724ff383d8c64672bf4c6c641f3e5690d6078e231d882285c6401 transformer/model-layer-201.safetensors
314
+ cd314577248b0ad6f33f2b73e783fd7190298d9d130a9963c91559264f5135ac transformer/model-layer-202.safetensors
315
+ 7ebaae021b537a48f8d5a819920a6c3ecc5b2313c9f172d53d8db89c90693178 transformer/model-layer-203.safetensors
316
+ c3c9287c6a83dcde1b0e60e08276bbd6ea7524c5a663b1d495044a2a6f80354b transformer/model-layer-204.safetensors
317
+ 1270ede15fe5c4591d2fd57abce23b2a7fa7541bb772e344a1c49ec93d0472c3 transformer/model-layer-205.safetensors
318
+ e8cf69b2c8ddf028a77f16d64063431ec46a208f6245e0a769412757e4724d65 transformer/model-layer-206.safetensors
319
+ 72b08fa0e31cf3bd99e6a6236df6ad282093046df19a01498c78d551ac86251a transformer/model-layer-207.safetensors
320
+ e7576cdccbe9919168cd79afc6b123669b103db4a22f85aee43669f0208b6add transformer/model-layer-208.safetensors
321
+ 406764ed05b591c51c6de957ba08c14b70c3005d7ed7f51b43d0669558c46d32 transformer/model-layer-209.safetensors
322
+ cc0fdc115f94600e85fb67716296156b8e961749d5611ea4d20c08f2c53c3e9f transformer/model-layer-210.safetensors
323
+ 261f7f9c98592065b59de6ab65436755f6e51cb89a1225a9d2617f89dbca90c2 transformer/model-layer-211.safetensors
324
+ b0e4e8d429ed025f7861f0d7e24778db961e3e457afdbed0b3b1dd9359a00606 transformer/model-layer-212.safetensors
325
+ 958b6bc3ed202a331ac2c7356b36a44d15133dabb6dfd9df9a00bedd973e4b37 transformer/model-layer-213.safetensors
326
+ 4b11f13bd803c408fccf70880038ed91b2bdfdb7dbb9af55e44ffd86d948665b transformer/model-layer-214.safetensors
327
+ 85d5055784d4dd7b742d080672e1fd48d7de3ad2281c85f7b59136aaaa9ec886 transformer/model-layer-215.safetensors
328
+ 9d3f778c492229cfffbe3f89fbfb81d70c482289081f102b981f22415eba35ed transformer/model-layer-216.safetensors
329
+ b67a198870946e16bf78c3437ef8944224a300da262efacf565742f3b95bd8fa transformer/model-layer-217.safetensors
330
+ 68c03ece759f1c35bafbadf5aae72a12d747825cef792b335d58f1fa8a029cac transformer/model-layer-218.safetensors
331
+ 203bdb319411779aebb2bf6d2a897d2b8bc25d5139dc854c7d3e269e9ed61d17 transformer/model-layer-219.safetensors
332
+ 145609d723d1d92fa9caaec598bb1a4dda4da2fe597d28b01a1e715dee387619 transformer/model-layer-220.safetensors
333
+ 197f0cac2cc9128fe788c1d97293c1f9bd01166cda4503dc976ba2f322f8a7af transformer/model-layer-221.safetensors
334
+ 27fe4be4ba4b522fc21c32cb74c93b39217fcb7af95eabb3d9788c54dcd456ce transformer/model-layer-222.safetensors
335
+ ba5619d3690bcb7b364efe3868a9e96125a2ca283e3a50ef370b524973fb19dd transformer/model-layer-223.safetensors
336
+ 9a00de7da981c4e7b52dad8af812688bcf02ad5d21533d5a412b95e9a0815a4b transformer/model.safetensors.index.json
337
+ 15c66897aa5b7d34158680d5cb8e0710ae8a40c7e06fdec6fc0174bfafbe81b0 vae/config.json
338
+ 71879ffd5321e6d10c3c87513e2b474b1252efa7f3dec2969214a9bf06a6dd5c vae/diffusion_pytorch_model.safetensors
THIRD_PARTY_NOTICES.md ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party components
2
+
3
+ - Qwen Image 2.1, revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`: [Qwen Research License](https://huggingface.co/Qwen/Qwen-Image-2.1/blob/b3179ad355be050328e483a9dfdd9e60cd62adfa/LICENSE). Full local copy: LICENSE.
4
+ - Nunchaku 1.2.1: [upstream source and license](https://github.com/nunchaku-tech/nunchaku/tree/v1.2.1). Generic SVDQW4A4Linear runtime; no claim of official Qwen Image 2.1 support.
5
+ - DeepCompressor `69f3473f5e1c1504bae35cc50c7858ef900a9b17`: [source and license](https://github.com/mit-han-lab/deepcompressor/tree/69f3473f5e1c1504bae35cc50c7858ef900a9b17). Conversion only.
6
+ - Diffusers `80c7ed262aeffbeb43ef13ae04baeb9b84515a69`: [Apache-2.0 source](https://github.com/huggingface/diffusers/tree/80c7ed262aeffbeb43ef13ae04baeb9b84515a69).
7
+ - Transformers 5.17.0: [Apache-2.0 source](https://github.com/huggingface/transformers).
8
+ - bitsandbytes 0.50.2: [MIT source](https://github.com/bitsandbytes-foundation/bitsandbytes).
9
+
10
+ Dependency packages in the container include their own license metadata. These software licenses do not replace the model's research-only terms.
model_index.json ADDED
@@ -0,0 +1,25 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_class_name": "QwenImage21Pipeline",
3
+ "_diffusers_version": "0.41.0.dev0",
4
+ "_name_or_path": "Qwen/Qwen-Image-2.1",
5
+ "processor": [
6
+ "transformers",
7
+ "Qwen3VLProcessor"
8
+ ],
9
+ "scheduler": [
10
+ "diffusers",
11
+ "FlowMatchEulerDiscreteScheduler"
12
+ ],
13
+ "text_encoder": [
14
+ "transformers",
15
+ "Qwen3VLForConditionalGeneration"
16
+ ],
17
+ "transformer": [
18
+ "diffusers",
19
+ "QwenImage21Transformer2DModel"
20
+ ],
21
+ "vae": [
22
+ "diffusers",
23
+ "AutoencoderKLQwenImage21"
24
+ ]
25
+ }
processor/chat_template.jinja ADDED
@@ -0,0 +1,120 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- if tools %}
2
+ {{- '<|im_start|>system\n' }}
3
+ {%- if messages[0].role == 'system' %}
4
+ {%- if messages[0].content is string %}
5
+ {{- messages[0].content }}
6
+ {%- else %}
7
+ {%- for content in messages[0].content %}
8
+ {%- if 'text' in content %}
9
+ {{- content.text }}
10
+ {%- endif %}
11
+ {%- endfor %}
12
+ {%- endif %}
13
+ {{- '\n\n' }}
14
+ {%- endif %}
15
+ {{- "# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within <tools></tools> XML tags:\n<tools>" }}
16
+ {%- for tool in tools %}
17
+ {{- "\n" }}
18
+ {{- tool | tojson }}
19
+ {%- endfor %}
20
+ {{- "\n</tools>\n\nFor each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
21
+ {%- else %}
22
+ {%- if messages[0].role == 'system' %}
23
+ {{- '<|im_start|>system\n' }}
24
+ {%- if messages[0].content is string %}
25
+ {{- messages[0].content }}
26
+ {%- else %}
27
+ {%- for content in messages[0].content %}
28
+ {%- if 'text' in content %}
29
+ {{- content.text }}
30
+ {%- endif %}
31
+ {%- endfor %}
32
+ {%- endif %}
33
+ {{- '<|im_end|>\n' }}
34
+ {%- endif %}
35
+ {%- endif %}
36
+ {%- set image_count = namespace(value=0) %}
37
+ {%- set video_count = namespace(value=0) %}
38
+ {%- for message in messages %}
39
+ {%- if message.role == "user" %}
40
+ {{- '<|im_start|>' + message.role + '\n' }}
41
+ {%- if message.content is string %}
42
+ {{- message.content }}
43
+ {%- else %}
44
+ {%- for content in message.content %}
45
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
46
+ {%- set image_count.value = image_count.value + 1 %}
47
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
48
+ <|vision_start|><|image_pad|><|vision_end|>
49
+ {%- elif content.type == 'video' or 'video' in content %}
50
+ {%- set video_count.value = video_count.value + 1 %}
51
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
52
+ <|vision_start|><|video_pad|><|vision_end|>
53
+ {%- elif 'text' in content %}
54
+ {{- content.text }}
55
+ {%- endif %}
56
+ {%- endfor %}
57
+ {%- endif %}
58
+ {{- '<|im_end|>\n' }}
59
+ {%- elif message.role == "assistant" %}
60
+ {{- '<|im_start|>' + message.role + '\n' }}
61
+ {%- if message.content is string %}
62
+ {{- message.content }}
63
+ {%- else %}
64
+ {%- for content_item in message.content %}
65
+ {%- if 'text' in content_item %}
66
+ {{- content_item.text }}
67
+ {%- endif %}
68
+ {%- endfor %}
69
+ {%- endif %}
70
+ {%- if message.tool_calls %}
71
+ {%- for tool_call in message.tool_calls %}
72
+ {%- if (loop.first and message.content) or (not loop.first) %}
73
+ {{- '\n' }}
74
+ {%- endif %}
75
+ {%- if tool_call.function %}
76
+ {%- set tool_call = tool_call.function %}
77
+ {%- endif %}
78
+ {{- '<tool_call>\n{"name": "' }}
79
+ {{- tool_call.name }}
80
+ {{- '", "arguments": ' }}
81
+ {%- if tool_call.arguments is string %}
82
+ {{- tool_call.arguments }}
83
+ {%- else %}
84
+ {{- tool_call.arguments | tojson }}
85
+ {%- endif %}
86
+ {{- '}\n</tool_call>' }}
87
+ {%- endfor %}
88
+ {%- endif %}
89
+ {{- '<|im_end|>\n' }}
90
+ {%- elif message.role == "tool" %}
91
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
92
+ {{- '<|im_start|>user' }}
93
+ {%- endif %}
94
+ {{- '\n<tool_response>\n' }}
95
+ {%- if message.content is string %}
96
+ {{- message.content }}
97
+ {%- else %}
98
+ {%- for content in message.content %}
99
+ {%- if content.type == 'image' or 'image' in content or 'image_url' in content %}
100
+ {%- set image_count.value = image_count.value + 1 %}
101
+ {%- if add_vision_id %}Picture {{ image_count.value }}: {% endif -%}
102
+ <|vision_start|><|image_pad|><|vision_end|>
103
+ {%- elif content.type == 'video' or 'video' in content %}
104
+ {%- set video_count.value = video_count.value + 1 %}
105
+ {%- if add_vision_id %}Video {{ video_count.value }}: {% endif -%}
106
+ <|vision_start|><|video_pad|><|vision_end|>
107
+ {%- elif 'text' in content %}
108
+ {{- content.text }}
109
+ {%- endif %}
110
+ {%- endfor %}
111
+ {%- endif %}
112
+ {{- '\n</tool_response>' }}
113
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
114
+ {{- '<|im_end|>\n' }}
115
+ {%- endif %}
116
+ {%- endif %}
117
+ {%- endfor %}
118
+ {%- if add_generation_prompt %}
119
+ {{- '<|im_start|>assistant\n' }}
120
+ {%- endif %}
processor/processor_config.json ADDED
@@ -0,0 +1,65 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "image_processor": {
3
+ "data_format": "channels_first",
4
+ "default_to_square": true,
5
+ "do_convert_rgb": true,
6
+ "do_normalize": true,
7
+ "do_rescale": true,
8
+ "do_resize": true,
9
+ "image_mean": [
10
+ 0.5,
11
+ 0.5,
12
+ 0.5
13
+ ],
14
+ "image_processor_type": "Qwen2VLImageProcessor",
15
+ "image_std": [
16
+ 0.5,
17
+ 0.5,
18
+ 0.5
19
+ ],
20
+ "merge_size": 2,
21
+ "patch_size": 16,
22
+ "resample": 3,
23
+ "rescale_factor": 0.00392156862745098,
24
+ "size": {
25
+ "longest_edge": 16777216,
26
+ "shortest_edge": 65536
27
+ },
28
+ "temporal_patch_size": 2
29
+ },
30
+ "processor_class": "Qwen3VLProcessor",
31
+ "video_processor": {
32
+ "data_format": "channels_first",
33
+ "default_to_square": true,
34
+ "do_convert_rgb": true,
35
+ "do_normalize": true,
36
+ "do_rescale": true,
37
+ "do_resize": true,
38
+ "do_sample_frames": true,
39
+ "fps": 2,
40
+ "image_mean": [
41
+ 0.5,
42
+ 0.5,
43
+ 0.5
44
+ ],
45
+ "image_std": [
46
+ 0.5,
47
+ 0.5,
48
+ 0.5
49
+ ],
50
+ "max_frames": 768,
51
+ "max_video_tokens": 768,
52
+ "merge_size": 2,
53
+ "min_frames": 4,
54
+ "patch_size": 16,
55
+ "resample": 3,
56
+ "rescale_factor": 0.00392156862745098,
57
+ "return_metadata": false,
58
+ "size": {
59
+ "longest_edge": 25165824,
60
+ "shortest_edge": 4096
61
+ },
62
+ "temporal_patch_size": 2,
63
+ "video_processor_type": "Qwen3VLVideoProcessor"
64
+ }
65
+ }
processor/tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:be75606093db2094d7cd20f3c2f385c212750648bd6ea4fb2bf507a6a4c55506
3
+ size 11422650
processor/tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": null,
5
+ "clean_up_tokenization_spaces": false,
6
+ "eos_token": "<|im_end|>",
7
+ "errors": "replace",
8
+ "is_local": true,
9
+ "local_files_only": false,
10
+ "model_max_length": 262144,
11
+ "pad_token": "<|endoftext|>",
12
+ "processor_class": "Qwen3VLProcessor",
13
+ "split_special_tokens": false,
14
+ "tokenizer_class": "Qwen2Tokenizer",
15
+ "unk_token": null
16
+ }
reproduction/Dockerfile ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Cached home-PC base; immutable digest preserves torch 2.8.0 / CUDA 12.8.
2
+ FROM mesmerlord/flux2-klein-runpod@sha256:c6622a7307520c3001afe7969bdb44858f3b27d100a4066285a5f54e14a4af0e
3
+ USER root
4
+ WORKDIR /poc
5
+ RUN python -m pip install --no-cache-dir transformers==5.17.0 bitsandbytes==0.50.2 accelerate==1.12.0 huggingface-hub==1.32.0 tokenizers==0.23.2 safetensors==0.8.0 pydantic==2.13.4 httpx==0.28.1 pillow==12.2.0 starlette==0.47.3 fastapi==0.116.1 uvicorn==0.35.0 \
6
+ 'diffusers @ https://github.com/huggingface/diffusers/archive/80c7ed262aeffbeb43ef13ae04baeb9b84515a69.zip'
7
+ RUN python -m pip install --no-cache-dir --no-deps torchvision==0.23.0 --index-url https://download.pytorch.org/whl/cu128
8
+ ADD https://github.com/mit-han-lab/deepcompressor/archive/69f3473f5e1c1504bae35cc50c7858ef900a9b17.tar.gz /tmp/deepcompressor.tar.gz
9
+ RUN mkdir -p /opt/deepcompressor && tar -xzf /tmp/deepcompressor.tar.gz --strip-components=1 -C /opt/deepcompressor && rm /tmp/deepcompressor.tar.gz
10
+ COPY runner.py lean_encoder.py server.py /poc/
11
+ COPY nunchaku_backend/ /poc/nunchaku_backend/
12
+ COPY scripts/download.py /poc/scripts/
13
+ ENV KLEIN_LIGHT_IMPORTS=0 HF_HUB_OFFLINE=0 TRANSFORMERS_OFFLINE=0 HF_HOME=/cache/huggingface TORCHINDUCTOR_CACHE_DIR=/cache/torchinductor TRITON_CACHE_DIR=/cache/triton PYTHONUNBUFFERED=1
14
+ ENV PYTHONPATH=/opt/deepcompressor:/poc
15
+ ENTRYPOINT []
16
+ HEALTHCHECK --interval=30s --timeout=5s --start-period=120s --retries=3 CMD python -c "import urllib.request; urllib.request.urlopen('http://127.0.0.1:8091/readyz',timeout=3)"
17
+ CMD ["uvicorn", "server:app", "--host", "0.0.0.0", "--port", "8091"]
reproduction/LICENSE ADDED
@@ -0,0 +1,55 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen RESEARCH LICENSE AGREEMENT
2
+
3
+ Qwen RESEARCH LICENSE AGREEMENT Release Date: September 20, 2026
4
+
5
+ By clicking to agree or by using or distributing any portion or element of the Qwen Materials, you will be deemed to have recognized and accepted the content of this Agreement, which is effective immediately.
6
+
7
+ 1. Definitions
8
+ a. This Qwen RESEARCH LICENSE AGREEMENT (this "Agreement") shall mean the terms and conditions for use, reproduction, distribution and modification of the Materials as defined by this Agreement.
9
+ b. "We" (or "Us") shall mean Hangzhou Tongyi Laboratory Technology Co., Ltd.
10
+ c. "You" (or "Your") shall mean a natural person or legal entity exercising the rights granted by this Agreement and/or using the Materials for any purpose and in any field of use.
11
+ d. "Third Parties" shall mean individuals or legal entities that are not under common control with us or you.
12
+ e. "Qwen" shall mean the large language models, diffusion models, and software and algorithms, consisting of trained model weights, parameters (including optimizer states), machine-learning model code, inference-enabling code, training-enabling code, fine-tuning enabling code and other elements of the foregoing distributed by us.
13
+ f. "Materials" shall mean, collectively, our proprietary Qwen and Documentation (and any portion thereof) made available under this Agreement.
14
+ g. "Source" form shall mean the preferred form for making modifications, including but not limited to model source code, documentation source, and configuration files.
15
+ h. "Object" form shall mean any form resulting from mechanical transformation or translation of a Source form, including but not limited to compiled object code, generated documentation, and conversions to other media types.
16
+ i. "Non-Commercial" shall mean for research or evaluation purposes only.
17
+
18
+ 2. Grant of Rights
19
+ a. You are granted a non-exclusive, worldwide, non-transferable and royalty-free limited license under our intellectual property or other rights owned by us embodied in the Materials to use, reproduce, distribute, copy, create derivative works of, and make modifications to the Materials FOR NON-COMMERCIAL PURPOSES ONLY.
20
+ b. You shall not use the Materials for any commercial purpose without obtaining a separate commercial license from us. If you wish to use the Materials commercially, you shall request a license from us at model-business@notice.qwencloud.com.
21
+
22
+ 3. Redistribution
23
+ Subject to Section 2 (Grant of Rights), you may distribute copies or make the Materials, or derivative works thereof, available as part of a product or service that contains any of them, with or without modifications, and in Source or Object form, provided that you meet the following conditions:
24
+ a. You shall give any other recipients of the Materials or derivative works a copy of this Agreement;
25
+ b. You shall cause any modified files to carry prominent notices stating that you changed the files;
26
+ c. You shall retain in all copies of the Materials that you distribute the following attribution notices within a "Notice" text file distributed as a part of such copies: "Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved."; and
27
+ d. You may add your own copyright statement to your modifications and may provide additional or different license terms and conditions for use, reproduction, or distribution of your modifications, or for any such derivative works as a whole, provided your use, reproduction, and distribution of the work otherwise complies with the terms and conditions of this Agreement.
28
+
29
+ 4. Rules of use
30
+ a. The Materials may be subject to export controls or restrictions in China, the United States or other countries or regions. You shall comply with applicable laws and regulations in your use of the Materials.
31
+ b. If you use the Materials or any outputs or results therefrom to create, train, fine-tune, or improve an AI model that is distributed or made available, you shall prominently display “Built with Qwen” or “Improved using Qwen” in the related product documentation.
32
+ c. You shall not use "Qwen" as the primary name or identifier of any derivative works or products; reasonable descriptive use (e.g., "fine-tuned from Qwen Image") is permitted.
33
+
34
+ 5. Intellectual Property
35
+ a. We retain ownership of all intellectual property rights in and to the Materials and derivatives made by or for us. Conditioned upon compliance with the terms and conditions of this Agreement, with respect to any derivative works and modifications of the Materials that are made by you, you are and will be the owner of such derivative works and modifications.
36
+ b. No trademark license is granted to use the trade names, trademarks, service marks, or product names of us, except as required to fulfill notice requirements under this Agreement or as required for reasonable and customary use in describing and redistributing the Materials.
37
+ c. If you commence a lawsuit or other proceedings (including a cross-claim or counterclaim in a lawsuit) against us or any entity alleging that the Materials or any output therefrom, or any part of the foregoing, infringe any intellectual property or other right owned or licensable by you, then all licenses granted to you under this Agreement shall terminate as of the date such lawsuit or other proceeding is commenced or brought.
38
+
39
+ 6. Disclaimer of Warranty and Limitation of Liability
40
+ a. We are not obligated to support, update, provide training for, or develop any further version of the Qwen Materials or to grant any license thereto.
41
+ b. THE MATERIALS ARE PROVIDED "AS IS" WITHOUT ANY EXPRESS OR IMPLIED WARRANTY OF ANY KIND INCLUDING WARRANTIES OF MERCHANTABILITY, NONINFRINGEMENT, OR FITNESS FOR A PARTICULAR PURPOSE. WE MAKE NO WARRANTY AND ASSUME NO RESPONSIBILITY FOR THE SAFETY OR STABILITY OF THE MATERIALS AND ANY OUTPUT THEREFROM.
42
+ c. IN NO EVENT SHALL WE BE LIABLE TO YOU FOR ANY DAMAGES, INCLUDING, BUT NOT LIMITED TO ANY DIRECT, OR INDIRECT, SPECIAL OR CONSEQUENTIAL DAMAGES ARISING FROM YOUR USE OR INABILITY TO USE THE MATERIALS OR ANY OUTPUT OF IT, NO MATTER HOW IT’S CAUSED.
43
+ d. You will defend, indemnify and hold harmless us from and against any claim by any third party arising out of or related to your use or distribution of the Materials.
44
+
45
+ 7. Survival and Termination.
46
+ a. The term of this Agreement shall commence upon your acceptance of this Agreement or access to the Materials and will continue in full force and effect until terminated in accordance with the terms and conditions herein.
47
+ b. We may terminate this Agreement if you breach any of the terms or conditions of this Agreement. Upon termination of this Agreement, you must delete and cease use of the Materials. Sections 6 and 8 shall survive the termination of this Agreement.
48
+
49
+ 8. Governing Law and Jurisdiction.
50
+ a. This Agreement and any dispute arising out of or relating to it will be governed by the laws of China, without regard to conflict of law principles, and the UN Convention on Contracts for the International Sale of Goods does not apply to this Agreement.
51
+ b. The People's Courts in Hangzhou City shall have exclusive jurisdiction over any dispute arising out of this Agreement.
52
+
53
+ 9. Other Terms and Conditions.
54
+ a. Any arrangements, understandings, or agreements regarding the Material not stated herein are separate from and independent of the terms and conditions of this Agreement. You shall request a separate license from us, if you use the Materials in ways not expressly agreed to in this Agreement.
55
+ b. We shall not be bound by any additional or different terms or conditions communicated by you unless expressly agreed.
reproduction/NOTICE ADDED
@@ -0,0 +1,9 @@
 
 
 
 
 
 
 
 
 
 
1
+ Built with Qwen
2
+
3
+ Qwen is licensed under the Qwen RESEARCH LICENSE AGREEMENT, Copyright (c) 2026 Hangzhou Tongyi Laboratory Technology Co., Ltd. All Rights Reserved.
4
+
5
+ Mesmer Image 21 Nunchaku is a modified, independently calibrated quantization of Qwen Image 2.1. The transformer uses Nunchaku signed INT4 weights and activations with BF16 rank128 residual branches. The text encoder is serialized in bitsandbytes NF4; processor, scheduler and VAE originate from the pinned upstream release. These modifications are by MesmerTech, September 2026, and are not an official Qwen or Nunchaku release.
6
+
7
+ The model and derivatives are for non-commercial research and evaluation under the accompanying LICENSE. Commercial use requires a separate license from Qwen.
8
+
9
+ Runtime dependencies retain their respective licenses. The custom linear runtime uses MIT HAN Lab Nunchaku; the conversion packing adapter uses DeepCompressor. See THIRD_PARTY_NOTICES.md for pinned source and license links.
reproduction/REPORT.md ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Qwen Image 2.1 fidelity revision — September 21, 2026
2
+
3
+ The new evaluation contains 21 scenarios at both 25 and 40 steps. It includes interacting groups with specific roles, a wheelchair award ceremony, a family reunion, a radio interview, dense bilingual text, a three-panel comic, museum geometry, product layouts, local preservation edits and two-reference composition. Initial samples remain archived; all new comparisons are in `samples/fidelity-v3/`.
4
+
5
+ The serving choice is calibrated rank 128. The larger selective rank 512 candidate is retained as an experiment: despite better numerical error, direct image inspection found extra objects and geometry regressions. All 126 comparison images (42 each for the BF16-transformer teacher, calibrated rank 128 and selective rank 512) are complete, locally verified and covered by direct visual reviews. These are 42 matched jobs, not 126 independent prompts. The selected API is running and its functional generation/edit/cache checks are complete; strict same-seed repeatability failed, as documented below.
6
+
7
+ ## What changed in the converter
8
+
9
+ The previous one-pass rank 32 approximation was replaced by calibration against actual Nunchaku CUDA output. The new search selects smoothing and low-rank residual fits using validation output error, checks held-out rows after selection, and accepts optional GPTQ correction only when validation improves. Rank 128 is used across 224 quantized linear layers. Each linear remains W4A4 plus a BF16 low-rank branch; original normalization, positional encoding, attention and KV-cache behavior remain in the Diffusers implementation.
10
+
11
+ The SVDQuant paper and official Nunchaku/DeepCompressor implementation motivated the calibration and decomposition work. This is a custom Qwen Image 2.1 adapter and bounded calibration implementation, not a reproduction of every official calibration setting. The exact shipped Klein calibration recipe is not public. See the [SVDQuant source review](../../research/svdquant-paper-v3.md) and [converter audit](../../research/nunchaku-v3-audit.md).
12
+
13
+ A follow-up sensitivity sweep found that the MLP projection and output layers contribute substantially to denoiser error. Temporarily restoring both roles to BF16 reduced the three diagnostic relative-L2 errors to 5.61%, 2.74% and 9.28%, but adds roughly 4.15 GiB of model state. This is a diagnostic, not a measured image-quality or safe two-reference serving configuration.
14
+
15
+ The practical candidate instead raises only the 32 MLP projection low-rank branches from 128 to 512. All 32 selected 512 by validation. Other 192 quantized linears and 193 untouched shards remain byte-identical. It retains all 224 INT4 kernels and adds 384 MiB of model tensors. The complete checkpoint has 5,057,830,912 bytes (4.710 GiB) of model tensors. Independent CPU loading verified 1,417 finite state tensors, every output/source shard hash and all 193 untouched shard identities; load time was 3.48 seconds. Across the 32 upgraded layers, held-out MSE ratios to actual rank 128 range from 0.531 to 0.656, with median 0.642. Model state is not peak VRAM.
16
+
17
+ ## Numerical evidence and rejected changes
18
+
19
+ The first rank 128 converter improved held-out linear MSE against a regenerated one-pass rank 32 recipe in all 224 linears; median ratio was 0.6064. That numerical baseline is not the exact archived v2 checkpoint. Actual v2 images and whole-denoiser checks use its real saved checkpoint.
20
+
21
+ Three fixed-recipe runs allowing 100 fitting iterations did not justify a full conversion: one control stayed unchanged, one slightly better validation result worsened held-out error, and another slightly regressed. More iterations were rejected instead of presumed beneficial.
22
+
23
+ The rank 512 probe improved held-out projection MSE by 35.5–38.6% in three representative layers. Through the complete MLP, improvement was smaller: 7.9–12.2%. Block 11 rank 256 improved raw projection error while slightly worsening the parent MLP's validation error. These checks demonstrate why linear error alone cannot establish image fidelity.
24
+
25
+ | Actual checkpoint | First-step denoiser relative L2 | Middle step | Final step |
26
+ |---|---:|---:|---:|
27
+ | Archived v2 |12.48%|6.54%|21.87%|
28
+ | Calibrated rank 128 |8.85%|5.08%|19.43%|
29
+ | Selective MLP rank 512 |8.46%|4.72%|17.93%|
30
+
31
+ This is one disjoint development editing prompt at steps 0/20/39 of 40. Every checkpoint builds its own prefix KV cache; no teacher cache is reused. These are conditional predictions on the teacher trajectory, not free-running image scores.
32
+
33
+ ## Image review and step count
34
+
35
+ Forty steps is the [official starting recommendation](https://huggingface.co/docs/diffusers/main/api/pipelines/qwenimage21). Both 25 and 40 are evaluated here at 1024×1024, with fixed seeds, CFG 1 and prefix KV caching. More steps do not reliably fix incorrect counts or relationships.
36
+
37
+ The rank 128 review covers all 42 teacher/candidate pairs. It restores the fantasy telescope and improves the old pancake-count error, retains all 14 bilingual poster strings and six comic dialogue lines, and performs the local coat/poster edits with strong preservation. It still changes some faces, hand positions, poses, object scale and design details. Its 25-step chess banner loses a digit in 2026; 40 steps restores it. Its museum drawings add tables and confuse routes at both step counts. The product campaign improves requested cup placement relative to the teacher, demonstrating that visual similarity and prompt compliance are different measurements.
38
+
39
+ The BF16 teacher also has real limitations: its 25-step chess title reads CIT CHES FINAL, the comic does not stage keys under the chair, the product relocation edit leaves its cup on the pedestal, and the two-reference portrait looks off-camera. These are shared model/instruction failures, not evidence that every candidate difference is caused by quantization. The teacher uses the same NF4 encoder, so it isolates transformer approximation rather than representing a fully BF16 pipeline.
40
+
41
+ Direct observations and image hashes are recorded in the [rank 128 scorecard](../../research/quality-v3-scorecard.json), with root versus independent reviewer attribution, and the [complete selective rank 512 review index](../../research/quality-v3-rank512-review-index.md). Similarity metrics compare the same labels/seeds and unaligned 1024×1024 images. SSIM and PSNR measure resemblance, not a percentage of semantic quality.
42
+
43
+ The **matched original 18-job cohort** is available for all four candidates. Each is compared with the corresponding BF16-transformer image; these medians pool nine scenarios at both step counts.
44
+
45
+ | Candidate | Images | Median SSIM | Median PSNR (dB) |
46
+ |---|---:|---:|---:|
47
+ | Archived Nunchaku v2 | 18 | 0.7608 | 18.52 |
48
+ | Calibrated rank 128 | 18 | 0.8508 | 20.57 |
49
+ | Selective rank 512 | 18 | 0.8287 | 19.96 |
50
+ | NF4 transformer | 18 | 0.8237 | 20.84 |
51
+
52
+ The **full matched 42-job cohort** is available for the two new Nunchaku candidates. It includes the original 18 jobs plus 24 expanded jobs.
53
+
54
+ | Candidate | Images | Median SSIM | Median PSNR (dB) |
55
+ |---|---:|---:|---:|
56
+ | Calibrated rank 128 | 42 | 0.8269 | 19.99 |
57
+ | Selective rank 512 | 42 | 0.8207 | 19.62 |
58
+
59
+ Rank 128 has higher median SSIM than NF4 on the original 18 jobs, while NF4 has higher median PSNR. Selective rank 512 trails rank 128 on both medians in both matched cohorts despite its better denoiser probe. NF4 was not measured on the full 42-job cohort. Comparisons across the two tables would mix different prompt sets. Exact labels and per-image metrics are in the [original 18-job results](../../results/compare-fidelity-v3-original18.json) and [full 42-job results](../../results/compare-fidelity-v3-full42.json).
60
+
61
+ ## Selective rank 512 decision
62
+
63
+ The larger branch is not promoted. All 42 selective outputs now have direct review coverage, including fixed-input checks for all 14 edits. In the 12 original generation jobs it improves the thin perfume peel and restores a more teacher-like towel grip at 40 steps, but produces seven blueberries and two knives in the 40-step breakfast, two cats in the 25-step fantasy scene, and a distorted horn-like telescope at 40. Botanical wording and lemon counts remain correct.
64
+
65
+ The 16 expanded generation jobs show further mixed effects. Selective rank 512 restores the chess year and camera-at-eye action, but adds six kitchen rolls at 25 steps and six/five cafe tables at 25/40. The 40-step museum also gains at least three shelf groups instead of two. Bilingual text and comic dialogue remain readable, while the comic still fails to stage the keys under the chair. Product cups sit correctly on the tabletop at 25 but return to the pedestal at 40, matching a teacher error and losing rank 128's adherence improvement. Forty steps trades failures rather than consistently repairing them.
66
+
67
+ Localized towel, coat and poster edits retain strong visual preservation, with small texture and typography changes and no decisive broad improvement over rank 128. The 40-step two-reference portrait restores the unobscured bottle emblem and supporting grip more closely to the teacher; off-camera gaze and rendering drift remain. Product-relocation edits still leave the cup on the pedestal in all compared backends. These observations support retaining the smaller checkpoint for this POC, not a claim that rank 128 is universally better or near-lossless.
68
+
69
+ ## Timing, memory and reproducibility
70
+
71
+ For the selected rank 128 checkpoint, median generation takes **15.50 s at 25 steps / 22.82 s at 40**, versus **27.51 s / 42.76 s** for the streamed BF16 transformer. These medians use the same 14 generation scenarios per step count. Median one-reference edit inference takes **19.48 s / 28.43 s** across six matched scenarios per step count. The single two-reference portrait takes **23.76 s / 34.35 s**; these are individual runs, not repeated-run medians. Sampled board occupancy reaches **11,994 MiB (11.71 GiB)** in that two-reference case. Selective rank 512 is slightly slower (**16.06 s / 23.56 s** median generation) and reaches **12,374 MiB** with two references.
72
+
73
+ See [performance.md](performance.md) and the [per-job measurement summary](../../results/fidelity-v3-summary.json) for verified per-job measurements. CLI comparisons disable prompt/reference LRU reuse, retain per-request prefix KV, use an untiled BF16 VAE and release dead KV before decoding. Timings cover engine inference, excluding process startup and image file loading/saving. Board occupancy is sampled separately from allocator peaks; initial teacher jobs without board telemetry are explicitly missing those values.
74
+
75
+ GPU 0 is an RTX 4070 Ti SUPER with 16,376 MiB. Klein remains on its existing GPU 1. GPU jobs run serially. The implementation stages the large NF4 encoder on CPU between requests; it retains all 36 decoder layers and removes only the unused LM-head/final-normalization path after exact feature-parity verification. Persistent compilation caches and optional compilation remain available; this Nunchaku quality revision uses eager inference.
76
+
77
+ Reproduction commands are in the experiment README, including calibration, checkpoint conversion, selective rank upgrade and the frozen 42-job image suite. Docker/model/library revisions are pinned. Saved source state, service rollback information and historical failures remain preserved.
78
+
79
+ ## Service verification
80
+
81
+ The calibrated rank128 API is healthy on PC 2 loopback port 8091. Four real 25-step requests completed: bilingual generation at 18.22 s, cached repeat 14.81 s, two-reference portrait 26.42 s, and cached repeat 22.93 s. These are client wall times from one run each. The prompt cache hit on both repeats and the reference-latent cache hit twice on the two-reference repeat. All four responses were COMPLETED with valid 1024×1024 PNGs. Engine startup measured 8.78 s from the saved checkpoint, including ML imports; this is a cached-disk process start, not machine boot. Maximum sampled API board occupancy was 11,986 MiB. See [final API measurements](../../results/api-fidelity-v3-final.json).
82
+
83
+ **Strict byte-identical repeatability failed.** The failure is preserved, not converted into a tolerance pass. A controlled four-run diagnostic compared two uncached runs and a cache miss/hit: all prompt tensors and first-transformer inputs were exactly equal in values, shapes, strides and dtypes, and stored cache tensors remained unchanged. Nevertheless the first transformer outputs differed even without caching. Uncached repeats had RGB MAE 5.06/255; cache miss/hit had 6.48/255. This isolates the observed divergence to the transformer forward, but does not identify the exact kernel or operation. Official low-rank atomic reductions are a hypothesis, not a confirmed sole cause. See [diagnosis](../../research/api-cache-v3-diagnosis.md) and [input fingerprints](../../results/repeatability-diagnostic/diagnostic.json).
84
+
85
+ The final API verification used an explicit record-differences mode so it could preserve and inspect every returned image while retaining `strict_repeatability_passed=false`. Both posters retained the requested wording; the two-reference repeats changed hand placement, bench and suitcase position while retaining the person/product attributes. Same-seed variation can affect composition, not merely file bytes. The image-suite comparisons therefore describe individual recorded realizations; they do not establish repeatable equivalence or isolate every visual difference solely to quantization.
86
+
87
+ Klein remains healthy with the same container and original start time on GPU1; Qwen alone occupies GPU0 for this POC. Higgs and GPU0 Comfy remain stopped under the user's authorization, GPU1 Comfy and connectivity remain intact, and experiment telemetry is stopped. [Final service state](../../results/final-state-v3.json) and `scripts/rollback.sh` preserve the recovery path.
88
+
89
+ ## Limits
90
+
91
+ This is a 1024×1024 POC below the release's recommended native 2K resolution. It does not establish near-lossless quality, superiority to Klein, full-BF16 encoder fidelity, ten-reference operation, native 2K speed or a production-router rollout. Calibration is small and edit splits share one source photo; prefix/reference-token coverage is limited. The visual suite was reused after earlier candidate results informed further work, so these are diagnostic comparisons rather than a fresh unseen acceptance set. Joint attention and full-block optimization are not implemented by this converter. The model uses the Qwen Research License; the downloaded license is retained with the release research.
reproduction/THIRD_PARTY_NOTICES.md ADDED
@@ -0,0 +1,10 @@
 
 
 
 
 
 
 
 
 
 
 
1
+ # Third-party components
2
+
3
+ - Qwen Image 2.1, revision `b3179ad355be050328e483a9dfdd9e60cd62adfa`: [Qwen Research License](https://huggingface.co/Qwen/Qwen-Image-2.1/blob/b3179ad355be050328e483a9dfdd9e60cd62adfa/LICENSE). Full local copy: LICENSE.
4
+ - Nunchaku 1.2.1: [upstream source and license](https://github.com/nunchaku-tech/nunchaku/tree/v1.2.1). Generic SVDQW4A4Linear runtime; no claim of official Qwen Image 2.1 support.
5
+ - DeepCompressor `69f3473f5e1c1504bae35cc50c7858ef900a9b17`: [source and license](https://github.com/mit-han-lab/deepcompressor/tree/69f3473f5e1c1504bae35cc50c7858ef900a9b17). Conversion only.
6
+ - Diffusers `80c7ed262aeffbeb43ef13ae04baeb9b84515a69`: [Apache-2.0 source](https://github.com/huggingface/diffusers/tree/80c7ed262aeffbeb43ef13ae04baeb9b84515a69).
7
+ - Transformers 5.17.0: [Apache-2.0 source](https://github.com/huggingface/transformers).
8
+ - bitsandbytes 0.50.2: [MIT source](https://github.com/bitsandbytes-foundation/bitsandbytes).
9
+
10
+ Dependency packages in the container include their own license metadata. These software licenses do not replace the model's research-only terms.
reproduction/artifacts/qwen21-calibration/activation_stats.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7f032d0930881f8352309e10084b3e5e4418ac81fb1c69a9bc7457f5fa731018
3
+ size 4743896
reproduction/artifacts/qwen21-calibration/calibration.json ADDED
@@ -0,0 +1,293 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "format": "qwen21-input-absmax-v1",
3
+ "source": "nf4-trajectories-approximate",
4
+ "observations": {
5
+ "transformer_blocks.0.attn.to_q": 82130,
6
+ "transformer_blocks.0.attn.to_k": 82130,
7
+ "transformer_blocks.0.attn.to_v": 82130,
8
+ "transformer_blocks.0.attn.to_out.0": 82130,
9
+ "transformer_blocks.0.img_mlp.gate_layer": 82130,
10
+ "transformer_blocks.0.img_mlp.proj": 82130,
11
+ "transformer_blocks.0.img_mlp.out": 82130,
12
+ "transformer_blocks.1.attn.to_q": 82130,
13
+ "transformer_blocks.1.attn.to_k": 82130,
14
+ "transformer_blocks.1.attn.to_v": 82130,
15
+ "transformer_blocks.1.attn.to_out.0": 82130,
16
+ "transformer_blocks.1.img_mlp.gate_layer": 82130,
17
+ "transformer_blocks.1.img_mlp.proj": 82130,
18
+ "transformer_blocks.1.img_mlp.out": 82130,
19
+ "transformer_blocks.2.attn.to_q": 82130,
20
+ "transformer_blocks.2.attn.to_k": 82130,
21
+ "transformer_blocks.2.attn.to_v": 82130,
22
+ "transformer_blocks.2.attn.to_out.0": 82130,
23
+ "transformer_blocks.2.img_mlp.gate_layer": 82130,
24
+ "transformer_blocks.2.img_mlp.proj": 82130,
25
+ "transformer_blocks.2.img_mlp.out": 82130,
26
+ "transformer_blocks.3.attn.to_q": 82130,
27
+ "transformer_blocks.3.attn.to_k": 82130,
28
+ "transformer_blocks.3.attn.to_v": 82130,
29
+ "transformer_blocks.3.attn.to_out.0": 82130,
30
+ "transformer_blocks.3.img_mlp.gate_layer": 82130,
31
+ "transformer_blocks.3.img_mlp.proj": 82130,
32
+ "transformer_blocks.3.img_mlp.out": 82130,
33
+ "transformer_blocks.4.attn.to_q": 82130,
34
+ "transformer_blocks.4.attn.to_k": 82130,
35
+ "transformer_blocks.4.attn.to_v": 82130,
36
+ "transformer_blocks.4.attn.to_out.0": 82130,
37
+ "transformer_blocks.4.img_mlp.gate_layer": 82130,
38
+ "transformer_blocks.4.img_mlp.proj": 82130,
39
+ "transformer_blocks.4.img_mlp.out": 82130,
40
+ "transformer_blocks.5.attn.to_q": 82130,
41
+ "transformer_blocks.5.attn.to_k": 82130,
42
+ "transformer_blocks.5.attn.to_v": 82130,
43
+ "transformer_blocks.5.attn.to_out.0": 82130,
44
+ "transformer_blocks.5.img_mlp.gate_layer": 82130,
45
+ "transformer_blocks.5.img_mlp.proj": 82130,
46
+ "transformer_blocks.5.img_mlp.out": 82130,
47
+ "transformer_blocks.6.attn.to_q": 82130,
48
+ "transformer_blocks.6.attn.to_k": 82130,
49
+ "transformer_blocks.6.attn.to_v": 82130,
50
+ "transformer_blocks.6.attn.to_out.0": 82130,
51
+ "transformer_blocks.6.img_mlp.gate_layer": 82130,
52
+ "transformer_blocks.6.img_mlp.proj": 82130,
53
+ "transformer_blocks.6.img_mlp.out": 82130,
54
+ "transformer_blocks.7.attn.to_q": 82130,
55
+ "transformer_blocks.7.attn.to_k": 82130,
56
+ "transformer_blocks.7.attn.to_v": 82130,
57
+ "transformer_blocks.7.attn.to_out.0": 82130,
58
+ "transformer_blocks.7.img_mlp.gate_layer": 82130,
59
+ "transformer_blocks.7.img_mlp.proj": 82130,
60
+ "transformer_blocks.7.img_mlp.out": 82130,
61
+ "transformer_blocks.8.attn.to_q": 82130,
62
+ "transformer_blocks.8.attn.to_k": 82130,
63
+ "transformer_blocks.8.attn.to_v": 82130,
64
+ "transformer_blocks.8.attn.to_out.0": 82130,
65
+ "transformer_blocks.8.img_mlp.gate_layer": 82130,
66
+ "transformer_blocks.8.img_mlp.proj": 82130,
67
+ "transformer_blocks.8.img_mlp.out": 82130,
68
+ "transformer_blocks.9.attn.to_q": 82130,
69
+ "transformer_blocks.9.attn.to_k": 82130,
70
+ "transformer_blocks.9.attn.to_v": 82130,
71
+ "transformer_blocks.9.attn.to_out.0": 82130,
72
+ "transformer_blocks.9.img_mlp.gate_layer": 82130,
73
+ "transformer_blocks.9.img_mlp.proj": 82130,
74
+ "transformer_blocks.9.img_mlp.out": 82130,
75
+ "transformer_blocks.10.attn.to_q": 82130,
76
+ "transformer_blocks.10.attn.to_k": 82130,
77
+ "transformer_blocks.10.attn.to_v": 82130,
78
+ "transformer_blocks.10.attn.to_out.0": 82130,
79
+ "transformer_blocks.10.img_mlp.gate_layer": 82130,
80
+ "transformer_blocks.10.img_mlp.proj": 82130,
81
+ "transformer_blocks.10.img_mlp.out": 82130,
82
+ "transformer_blocks.11.attn.to_q": 82130,
83
+ "transformer_blocks.11.attn.to_k": 82130,
84
+ "transformer_blocks.11.attn.to_v": 82130,
85
+ "transformer_blocks.11.attn.to_out.0": 82130,
86
+ "transformer_blocks.11.img_mlp.gate_layer": 82130,
87
+ "transformer_blocks.11.img_mlp.proj": 82130,
88
+ "transformer_blocks.11.img_mlp.out": 82130,
89
+ "transformer_blocks.12.attn.to_q": 82130,
90
+ "transformer_blocks.12.attn.to_k": 82130,
91
+ "transformer_blocks.12.attn.to_v": 82130,
92
+ "transformer_blocks.12.attn.to_out.0": 82130,
93
+ "transformer_blocks.12.img_mlp.gate_layer": 82130,
94
+ "transformer_blocks.12.img_mlp.proj": 82130,
95
+ "transformer_blocks.12.img_mlp.out": 82130,
96
+ "transformer_blocks.13.attn.to_q": 82130,
97
+ "transformer_blocks.13.attn.to_k": 82130,
98
+ "transformer_blocks.13.attn.to_v": 82130,
99
+ "transformer_blocks.13.attn.to_out.0": 82130,
100
+ "transformer_blocks.13.img_mlp.gate_layer": 82130,
101
+ "transformer_blocks.13.img_mlp.proj": 82130,
102
+ "transformer_blocks.13.img_mlp.out": 82130,
103
+ "transformer_blocks.14.attn.to_q": 82130,
104
+ "transformer_blocks.14.attn.to_k": 82130,
105
+ "transformer_blocks.14.attn.to_v": 82130,
106
+ "transformer_blocks.14.attn.to_out.0": 82130,
107
+ "transformer_blocks.14.img_mlp.gate_layer": 82130,
108
+ "transformer_blocks.14.img_mlp.proj": 82130,
109
+ "transformer_blocks.14.img_mlp.out": 82130,
110
+ "transformer_blocks.15.attn.to_q": 82130,
111
+ "transformer_blocks.15.attn.to_k": 82130,
112
+ "transformer_blocks.15.attn.to_v": 82130,
113
+ "transformer_blocks.15.attn.to_out.0": 82130,
114
+ "transformer_blocks.15.img_mlp.gate_layer": 82130,
115
+ "transformer_blocks.15.img_mlp.proj": 82130,
116
+ "transformer_blocks.15.img_mlp.out": 82130,
117
+ "transformer_blocks.16.attn.to_q": 82130,
118
+ "transformer_blocks.16.attn.to_k": 82130,
119
+ "transformer_blocks.16.attn.to_v": 82130,
120
+ "transformer_blocks.16.attn.to_out.0": 82130,
121
+ "transformer_blocks.16.img_mlp.gate_layer": 82130,
122
+ "transformer_blocks.16.img_mlp.proj": 82130,
123
+ "transformer_blocks.16.img_mlp.out": 82130,
124
+ "transformer_blocks.17.attn.to_q": 82130,
125
+ "transformer_blocks.17.attn.to_k": 82130,
126
+ "transformer_blocks.17.attn.to_v": 82130,
127
+ "transformer_blocks.17.attn.to_out.0": 82130,
128
+ "transformer_blocks.17.img_mlp.gate_layer": 82130,
129
+ "transformer_blocks.17.img_mlp.proj": 82130,
130
+ "transformer_blocks.17.img_mlp.out": 82130,
131
+ "transformer_blocks.18.attn.to_q": 82130,
132
+ "transformer_blocks.18.attn.to_k": 82130,
133
+ "transformer_blocks.18.attn.to_v": 82130,
134
+ "transformer_blocks.18.attn.to_out.0": 82130,
135
+ "transformer_blocks.18.img_mlp.gate_layer": 82130,
136
+ "transformer_blocks.18.img_mlp.proj": 82130,
137
+ "transformer_blocks.18.img_mlp.out": 82130,
138
+ "transformer_blocks.19.attn.to_q": 82130,
139
+ "transformer_blocks.19.attn.to_k": 82130,
140
+ "transformer_blocks.19.attn.to_v": 82130,
141
+ "transformer_blocks.19.attn.to_out.0": 82130,
142
+ "transformer_blocks.19.img_mlp.gate_layer": 82130,
143
+ "transformer_blocks.19.img_mlp.proj": 82130,
144
+ "transformer_blocks.19.img_mlp.out": 82130,
145
+ "transformer_blocks.20.attn.to_q": 82130,
146
+ "transformer_blocks.20.attn.to_k": 82130,
147
+ "transformer_blocks.20.attn.to_v": 82130,
148
+ "transformer_blocks.20.attn.to_out.0": 82130,
149
+ "transformer_blocks.20.img_mlp.gate_layer": 82130,
150
+ "transformer_blocks.20.img_mlp.proj": 82130,
151
+ "transformer_blocks.20.img_mlp.out": 82130,
152
+ "transformer_blocks.21.attn.to_q": 82130,
153
+ "transformer_blocks.21.attn.to_k": 82130,
154
+ "transformer_blocks.21.attn.to_v": 82130,
155
+ "transformer_blocks.21.attn.to_out.0": 82130,
156
+ "transformer_blocks.21.img_mlp.gate_layer": 82130,
157
+ "transformer_blocks.21.img_mlp.proj": 82130,
158
+ "transformer_blocks.21.img_mlp.out": 82130,
159
+ "transformer_blocks.22.attn.to_q": 82130,
160
+ "transformer_blocks.22.attn.to_k": 82130,
161
+ "transformer_blocks.22.attn.to_v": 82130,
162
+ "transformer_blocks.22.attn.to_out.0": 82130,
163
+ "transformer_blocks.22.img_mlp.gate_layer": 82130,
164
+ "transformer_blocks.22.img_mlp.proj": 82130,
165
+ "transformer_blocks.22.img_mlp.out": 82130,
166
+ "transformer_blocks.23.attn.to_q": 82130,
167
+ "transformer_blocks.23.attn.to_k": 82130,
168
+ "transformer_blocks.23.attn.to_v": 82130,
169
+ "transformer_blocks.23.attn.to_out.0": 82130,
170
+ "transformer_blocks.23.img_mlp.gate_layer": 82130,
171
+ "transformer_blocks.23.img_mlp.proj": 82130,
172
+ "transformer_blocks.23.img_mlp.out": 82130,
173
+ "transformer_blocks.24.attn.to_q": 82130,
174
+ "transformer_blocks.24.attn.to_k": 82130,
175
+ "transformer_blocks.24.attn.to_v": 82130,
176
+ "transformer_blocks.24.attn.to_out.0": 82130,
177
+ "transformer_blocks.24.img_mlp.gate_layer": 82130,
178
+ "transformer_blocks.24.img_mlp.proj": 82130,
179
+ "transformer_blocks.24.img_mlp.out": 82130,
180
+ "transformer_blocks.25.attn.to_q": 82130,
181
+ "transformer_blocks.25.attn.to_k": 82130,
182
+ "transformer_blocks.25.attn.to_v": 82130,
183
+ "transformer_blocks.25.attn.to_out.0": 82130,
184
+ "transformer_blocks.25.img_mlp.gate_layer": 82130,
185
+ "transformer_blocks.25.img_mlp.proj": 82130,
186
+ "transformer_blocks.25.img_mlp.out": 82130,
187
+ "transformer_blocks.26.attn.to_q": 82130,
188
+ "transformer_blocks.26.attn.to_k": 82130,
189
+ "transformer_blocks.26.attn.to_v": 82130,
190
+ "transformer_blocks.26.attn.to_out.0": 82130,
191
+ "transformer_blocks.26.img_mlp.gate_layer": 82130,
192
+ "transformer_blocks.26.img_mlp.proj": 82130,
193
+ "transformer_blocks.26.img_mlp.out": 82130,
194
+ "transformer_blocks.27.attn.to_q": 82130,
195
+ "transformer_blocks.27.attn.to_k": 82130,
196
+ "transformer_blocks.27.attn.to_v": 82130,
197
+ "transformer_blocks.27.attn.to_out.0": 82130,
198
+ "transformer_blocks.27.img_mlp.gate_layer": 82130,
199
+ "transformer_blocks.27.img_mlp.proj": 82130,
200
+ "transformer_blocks.27.img_mlp.out": 82130,
201
+ "transformer_blocks.28.attn.to_q": 82130,
202
+ "transformer_blocks.28.attn.to_k": 82130,
203
+ "transformer_blocks.28.attn.to_v": 82130,
204
+ "transformer_blocks.28.attn.to_out.0": 82130,
205
+ "transformer_blocks.28.img_mlp.gate_layer": 82130,
206
+ "transformer_blocks.28.img_mlp.proj": 82130,
207
+ "transformer_blocks.28.img_mlp.out": 82130,
208
+ "transformer_blocks.29.attn.to_q": 82130,
209
+ "transformer_blocks.29.attn.to_k": 82130,
210
+ "transformer_blocks.29.attn.to_v": 82130,
211
+ "transformer_blocks.29.attn.to_out.0": 82130,
212
+ "transformer_blocks.29.img_mlp.gate_layer": 82130,
213
+ "transformer_blocks.29.img_mlp.proj": 82130,
214
+ "transformer_blocks.29.img_mlp.out": 82130,
215
+ "transformer_blocks.30.attn.to_q": 82130,
216
+ "transformer_blocks.30.attn.to_k": 82130,
217
+ "transformer_blocks.30.attn.to_v": 82130,
218
+ "transformer_blocks.30.attn.to_out.0": 82130,
219
+ "transformer_blocks.30.img_mlp.gate_layer": 82130,
220
+ "transformer_blocks.30.img_mlp.proj": 82130,
221
+ "transformer_blocks.30.img_mlp.out": 82130,
222
+ "transformer_blocks.31.attn.to_q": 82130,
223
+ "transformer_blocks.31.attn.to_k": 82130,
224
+ "transformer_blocks.31.attn.to_v": 82130,
225
+ "transformer_blocks.31.attn.to_out.0": 82130,
226
+ "transformer_blocks.31.img_mlp.gate_layer": 82130,
227
+ "transformer_blocks.31.img_mlp.proj": 82130,
228
+ "transformer_blocks.31.img_mlp.out": 82130
229
+ },
230
+ "metadata": {
231
+ "jobs_file": "/poc/nunchaku_backend/calibration_jobs.json",
232
+ "jobs": [
233
+ {
234
+ "label": "calibration-landscape",
235
+ "prompt": "Photorealistic mountain valley at sunrise, a winding river and pine forest, clouds behind snowy peaks, crisp natural colors.",
236
+ "width": 1024,
237
+ "height": 1024,
238
+ "seed": 731
239
+ },
240
+ {
241
+ "label": "calibration-chess-portrait",
242
+ "prompt": "Editorial photograph of an elderly chess player concentrating over a wooden chessboard in a quiet cafe, soft side lighting, natural facial texture.",
243
+ "width": 1024,
244
+ "height": 1024,
245
+ "seed": 732
246
+ },
247
+ {
248
+ "label": "calibration-subway",
249
+ "prompt": "A candid documentary photograph of a modern subway platform with commuters, silver train arriving, overhead fluorescent lighting and deep perspective.",
250
+ "width": 1024,
251
+ "height": 1024,
252
+ "seed": 733
253
+ },
254
+ {
255
+ "label": "calibration-fruit",
256
+ "prompt": "A still life of oranges, green pears and a cut pomegranate in a blue ceramic bowl on a linen cloth, realistic daylight and detailed fruit texture.",
257
+ "width": 1024,
258
+ "height": 1024,
259
+ "seed": 734
260
+ },
261
+ {
262
+ "label": "calibration-mug-white",
263
+ "prompt": "Change the mug to matte white ceramic. Preserve the composition, spoon, table and lighting.",
264
+ "images": [
265
+ "/poc/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png"
266
+ ],
267
+ "width": 1024,
268
+ "height": 1024,
269
+ "seed": 735
270
+ },
271
+ {
272
+ "label": "calibration-mug-background",
273
+ "prompt": "Add a small green houseplant in a simple pot behind the mug near the wall. Preserve the mug, spoon and table.",
274
+ "images": [
275
+ "/poc/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png"
276
+ ],
277
+ "width": 1024,
278
+ "height": 1024,
279
+ "seed": 736
280
+ }
281
+ ],
282
+ "steps": 8,
283
+ "recorded_steps": [
284
+ 0,
285
+ 4,
286
+ 7
287
+ ],
288
+ "teacher": "NF4 approximate trajectories",
289
+ "evaluation_prompts_and_references_held_out": true
290
+ },
291
+ "instrumented_latency": true,
292
+ "layers": 224
293
+ }
reproduction/environment.freeze.txt ADDED
@@ -0,0 +1,193 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ accelerate==1.12.0
2
+ aiodns==4.0.4
3
+ aiohappyeyeballs==2.6.2
4
+ aiohttp==3.14.1
5
+ aiohttp-retry==2.9.1
6
+ aiosignal==1.4.0
7
+ annotated-doc==0.0.4
8
+ annotated-types==0.7.0
9
+ anyio==4.14.0
10
+ archspec @ file:///home/conda/feedstock_root/build_artifacts/archspec_1737352602016/work
11
+ asttokens @ file:///home/conda/feedstock_root/build_artifacts/asttokens_1733250440834/work
12
+ astunparse==1.6.3
13
+ attrs @ file:///home/conda/feedstock_root/build_artifacts/attrs_1741918516150/work
14
+ backoff==2.2.1
15
+ backports.zstd==1.6.0
16
+ bcrypt==5.0.0
17
+ beautifulsoup4 @ file:///home/conda/feedstock_root/build_artifacts/beautifulsoup4_1744783198182/work
18
+ bitsandbytes==0.50.2
19
+ boltons @ file:///home/conda/feedstock_root/build_artifacts/boltons_1749686179973/work
20
+ boto3==1.43.34
21
+ botocore==1.43.34
22
+ brotli==1.2.0
23
+ certifi @ file:///home/conda/feedstock_root/build_artifacts/certifi_1754231422783/work/certifi
24
+ cffi==2.0.0
25
+ chardet @ file:///home/conda/feedstock_root/build_artifacts/chardet_1741797914774/work
26
+ charset-normalizer @ file:///home/conda/feedstock_root/build_artifacts/charset-normalizer_1746214863626/work
27
+ click==8.5.0
28
+ cmake==4.0.3
29
+ colorama @ file:///home/conda/feedstock_root/build_artifacts/colorama_1733218098505/work
30
+ conda @ file:///home/conda/feedstock_root/build_artifacts/conda_1754405241914/work/conda-src
31
+ conda-build @ file:///home/conda/feedstock_root/build_artifacts/conda-build_1754316273870/work
32
+ conda-libmamba-solver @ file:///home/conda/feedstock_root/build_artifacts/conda-libmamba-solver_1742219570693/work/src
33
+ conda-package-handling @ file:///home/conda/feedstock_root/build_artifacts/conda-package-handling_1736345463896/work
34
+ conda_index @ file:///home/conda/feedstock_root/build_artifacts/conda-index_1748375757308/work
35
+ conda_package_streaming @ file:///home/conda/feedstock_root/build_artifacts/conda-package-streaming_1751548120229/work
36
+ cryptography==46.0.7
37
+ decorator @ file:///home/conda/feedstock_root/build_artifacts/decorator_1740384970518/work
38
+ detect-installer==0.1.0
39
+ diffusers @ https://github.com/huggingface/diffusers/archive/80c7ed262aeffbeb43ef13ae04baeb9b84515a69.zip#sha256=7b0d00b5b44f5ca1745d477ae4a0c2da125d58b89cff8470c422d3181a797a09
40
+ distro @ file:///home/conda/feedstock_root/build_artifacts/distro_1734729835256/work
41
+ dnspython==2.7.0
42
+ einops==0.8.2
43
+ email-validator==2.3.0
44
+ evalidate @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_evalidate_1746793833/work
45
+ exceptiongroup @ file:///home/conda/feedstock_root/build_artifacts/exceptiongroup_1746947292760/work
46
+ executing @ file:///home/conda/feedstock_root/build_artifacts/executing_1745502089858/work
47
+ expecttest==0.3.0
48
+ fastapi==0.116.1
49
+ fastapi-cli==0.0.27
50
+ fastapi-cloud-cli==0.20.0
51
+ fastar==0.11.0
52
+ filelock @ file:///home/conda/feedstock_root/build_artifacts/filelock_1741969488311/work
53
+ frozendict @ file:///home/conda/feedstock_root/build_artifacts/frozendict_1728841334936/work
54
+ frozenlist==1.8.0
55
+ fsspec==2025.7.0
56
+ h11==0.16.0
57
+ h2 @ file:///home/conda/feedstock_root/build_artifacts/h2_1738578511449/work
58
+ hf-xet==1.6.0
59
+ hpack @ file:///home/conda/feedstock_root/build_artifacts/hpack_1737618293087/work
60
+ httpcore==1.0.9
61
+ httptools==0.8.0
62
+ httpx==0.28.1
63
+ huggingface_hub==1.32.0
64
+ hyperframe @ file:///home/conda/feedstock_root/build_artifacts/hyperframe_1737618333194/work
65
+ hypothesis==6.137.1
66
+ idna @ file:///home/conda/feedstock_root/build_artifacts/idna_1733211830134/work
67
+ importlib_metadata==9.0.0
68
+ inquirerpy==0.3.4
69
+ invoke==3.0.3
70
+ ipython @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_ipython_1751465044/work
71
+ ipython_pygments_lexers @ file:///home/conda/feedstock_root/build_artifacts/ipython_pygments_lexers_1737123620466/work
72
+ itsdangerous==2.2.0
73
+ jedi @ file:///home/conda/feedstock_root/build_artifacts/jedi_1733300866624/work
74
+ Jinja2 @ file:///home/conda/feedstock_root/build_artifacts/jinja2_1741263328855/work
75
+ jmespath==1.1.0
76
+ jsonpatch @ file:///home/conda/feedstock_root/build_artifacts/jsonpatch_1733814567314/work
77
+ jsonpointer @ file:///home/conda/feedstock_root/build_artifacts/jsonpointer_1725302941992/work
78
+ jsonschema @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_jsonschema_1752925388/work
79
+ jsonschema-specifications @ file:///tmp/tmpuvkyqc9y/src
80
+ libarchive-c @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_python-libarchive-c_1747927321/work
81
+ libmambapy @ file:///home/conda/feedstock_root/build_artifacts/mamba-split_1746515836725/work/libmambapy
82
+ lief @ file:///home/conda/feedstock_root/build_artifacts/lief_1750151383011/work/api/python
83
+ lintrunner==0.12.7
84
+ markdown-it-py==4.2.0
85
+ MarkupSafe @ file:///home/conda/feedstock_root/build_artifacts/markupsafe_1733219680183/work
86
+ matplotlib-inline @ file:///home/conda/feedstock_root/build_artifacts/matplotlib-inline_1733416936468/work
87
+ mdurl==0.1.2
88
+ menuinst @ file:///home/conda/feedstock_root/build_artifacts/menuinst_1753546279984/work
89
+ mpmath==1.3.0
90
+ msgpack @ file:///home/conda/feedstock_root/build_artifacts/msgpack-python_1749813202382/work
91
+ multidict==6.7.1
92
+ networkx==3.5
93
+ ninja==1.11.1.4
94
+ numpy @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_numpy_1753401560/work/dist/numpy-2.3.2-cp311-cp311-linux_x86_64.whl#sha256=469440884ff65cdd5601d2ff9e1c5b9bcc2d8176c37e37ec78a0666703a54157
95
+ nunchaku @ https://github.com/nunchaku-tech/nunchaku/releases/download/v1.2.1/nunchaku-1.2.1+cu12.8torch2.8-cp311-cp311-linux_x86_64.whl#sha256=77dab1a3abdff16d5cbff70e26a5700e6fccb77b4207c3d53127d5f0224bd82d
96
+ nvidia-cublas-cu12==12.8.4.1
97
+ nvidia-cuda-cupti-cu12==12.8.90
98
+ nvidia-cuda-nvrtc-cu12==12.8.93
99
+ nvidia-cuda-runtime-cu12==12.8.90
100
+ nvidia-cudnn-cu12==9.10.2.21
101
+ nvidia-cufft-cu12==11.3.3.83
102
+ nvidia-cufile-cu12==1.13.1.3
103
+ nvidia-curand-cu12==10.3.9.90
104
+ nvidia-cusolver-cu12==11.7.3.90
105
+ nvidia-cusparse-cu12==12.5.8.93
106
+ nvidia-cusparselt-cu12==0.7.1
107
+ nvidia-nccl-cu12==2.27.3
108
+ nvidia-nvjitlink-cu12==12.8.93
109
+ nvidia-nvtx-cu12==12.8.90
110
+ optree==0.17.0
111
+ packaging @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_packaging_1745345660/work
112
+ paramiko==5.0.0
113
+ parso @ file:///home/conda/feedstock_root/build_artifacts/parso_1733271261340/work
114
+ peft==0.19.1
115
+ pexpect @ file:///home/conda/feedstock_root/build_artifacts/pexpect_1733301927746/work
116
+ pfzy==0.3.4
117
+ pickleshare @ file:///home/conda/feedstock_root/build_artifacts/pickleshare_1733327343728/work
118
+ pillow==12.2.0
119
+ pkginfo @ file:///home/conda/feedstock_root/build_artifacts/pkginfo_1739984581450/work
120
+ platformdirs @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_platformdirs_1746710438/work
121
+ pluggy @ file:///home/conda/feedstock_root/build_artifacts/pluggy_1747339660894/work
122
+ prettytable==3.17.0
123
+ prompt_toolkit @ file:///home/conda/feedstock_root/build_artifacts/prompt-toolkit_1744724089886/work
124
+ propcache==0.5.2
125
+ protobuf==7.35.1
126
+ psutil @ file:///home/conda/feedstock_root/build_artifacts/psutil_1740663149797/work
127
+ ptyprocess @ file:///home/conda/feedstock_root/build_artifacts/ptyprocess_1733302279685/work/dist/ptyprocess-0.7.0-py2.py3-none-any.whl#sha256=92c32ff62b5fd8cf325bec5ab90d7be3d2a8ca8c8a3813ff487a8d2002630d1f
128
+ pure_eval @ file:///home/conda/feedstock_root/build_artifacts/pure_eval_1733569405015/work
129
+ py-cpuinfo==9.0.0
130
+ pycares==5.0.1
131
+ pycosat @ file:///home/conda/feedstock_root/build_artifacts/pycosat_1732588400443/work
132
+ pycparser @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_pycparser_1733195786/work
133
+ pydantic==2.13.4
134
+ pydantic-extra-types==2.11.1
135
+ pydantic-settings==2.14.2
136
+ pydantic_core==2.46.4
137
+ Pygments @ file:///home/conda/feedstock_root/build_artifacts/pygments_1750615794071/work
138
+ PyNaCl==1.6.2
139
+ PySocks @ file:///home/conda/feedstock_root/build_artifacts/pysocks_1733217236728/work
140
+ python-dateutil==2.9.0.post0
141
+ python-dotenv==1.2.2
142
+ python-etcd==0.4.5
143
+ python-multipart==0.0.32
144
+ pytz @ file:///home/conda/feedstock_root/build_artifacts/pytz_1742920838005/work
145
+ PyYAML @ file:///home/conda/feedstock_root/build_artifacts/pyyaml_1737454647378/work
146
+ referencing @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_referencing_1737836872/work
147
+ regex==2026.5.9
148
+ requests==2.34.2
149
+ rich==15.0.0
150
+ rich-toolkit==0.20.1
151
+ rignore==0.7.6
152
+ rpds-py @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_rpds-py_1751468291/work
153
+ ruamel.yaml @ file:///home/conda/feedstock_root/build_artifacts/ruamel.yaml_1749479918291/work
154
+ ruamel.yaml.clib @ file:///home/conda/feedstock_root/build_artifacts/ruamel.yaml.clib_1728724459810/work
155
+ runpod==1.9.1
156
+ s3transfer==0.19.0
157
+ safetensors==0.8.0
158
+ sentencepiece==0.2.1
159
+ sentry-sdk==2.63.0
160
+ shellingham==1.5.4
161
+ six==1.17.0
162
+ sortedcontainers==2.4.0
163
+ soupsieve @ file:///home/conda/feedstock_root/build_artifacts/soupsieve_1746563585861/work
164
+ stack_data @ file:///home/conda/feedstock_root/build_artifacts/stack_data_1733569443808/work
165
+ starlette==0.47.3
166
+ sympy==1.14.0
167
+ tokenizers==0.23.2
168
+ tomli==2.4.1
169
+ tomlkit==0.15.0
170
+ torch==2.8.0+cu128
171
+ torchaudio==2.8.0+cu128
172
+ torchelastic==0.2.2
173
+ torchvision==0.23.0+cu128
174
+ tqdm @ file:///home/conda/feedstock_root/build_artifacts/tqdm_1735661334605/work
175
+ tqdm-loggable==0.4.1
176
+ traitlets @ file:///home/conda/feedstock_root/build_artifacts/traitlets_1733367359838/work
177
+ transformers==5.17.0
178
+ triton==3.4.0
179
+ truststore @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_truststore_1739009763/work
180
+ typer==0.25.1
181
+ types-dataclasses==0.6.6
182
+ typing-inspection==0.4.2
183
+ typing_extensions @ file:///home/conda/feedstock_root/build_artifacts/bld/rattler-build_typing_extensions_1751643513/work
184
+ urllib3 @ file:///home/conda/feedstock_root/build_artifacts/urllib3_1750271362675/work
185
+ uvicorn==0.35.0
186
+ uvloop==0.22.1
187
+ watchdog==6.0.0
188
+ watchfiles==1.2.0
189
+ wcwidth @ file:///home/conda/feedstock_root/build_artifacts/wcwidth_1733231326287/work
190
+ websockets==16.0
191
+ yarl==1.24.2
192
+ zipp==4.1.0
193
+ zstandard==0.23.0
reproduction/experiments/fidelity-v3/README.md ADDED
@@ -0,0 +1,82 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Qwen Image 2.1 fidelity revision
2
+
3
+ This experiment improves the custom Nunchaku converter against the original BF16 transformer. The encoder stays NF4 and the VAE stays BF16 in every comparison, so the reference isolates transformer quantization; it is not a fully BF16 pipeline.
4
+
5
+ The previous results remain under `samples/varied-v2/`. Initial POC samples remain archived under `samples/archive/2026-09-20-initial-poc/`. New outputs go only under `samples/fidelity-v3/`.
6
+
7
+ ## Calibration and evaluation separation
8
+
9
+ `calibration-jobs.json` contains four training, two validation and two held-out calibration jobs. Each split contains generation and editing. Editing instructions differ, but these calibration splits share one source photograph; this is a limitation. None of the final evaluation prompts or reference photographs is used for calibration.
10
+
11
+ The initial three-layer probe used 16-step BF16 trajectories. The final collection uses 40 steps and samples invocations 0, 10, 20, 30 and 39. First-step sampling explicitly covers text, reference-image and target-image tokens. Later sampling covers target tokens. Per-layer files retain 512 training, 256 validation and 256 held-out rows, plus full-token training channel maxima. The manifest records actual jobs, sampling, model revision and timestamps.
12
+
13
+ `quality-jobs-all.json` freezes 21 scenarios at both 25 and 40 steps: the original six generation prompts and three edits, plus eight new generation and four new editing scenarios in `expanded-jobs.json`. There are 42 jobs per evaluated backend. The new cases cover role-specific groups, coordinated hand interactions, a wheelchair award presentation, a family reunion, dense English/Chinese type, dialogue continuity, diagram geometry, a product campaign, a newsroom, local preservation and two-reference identity/product composition. `expanded-rubric.md` defines their checks. `quality-jobs.json` remains the original nine-job 40-step subset.
14
+
15
+ The expanded edits use fixed BF16 reference images generated by earlier jobs. Their file contents, including any source-model imperfections, must be held identical across candidates. Inspect counting, anatomy, geometry, lettering, materials and edit preservation separately from image-similarity scores.
16
+
17
+ ## Reference integrity
18
+
19
+ `teacher_v3.py` uses streamed block offloading to fit the original BF16 transformer on GPU 0. A resident-versus-offloaded 512-pixel, four-step test produced byte-identical PNGs; the measured result is `results/teacher-v3-parity.json`. That is an implementation parity check, not a broad image-quality result.
20
+
21
+ ## Reproduction
22
+
23
+ Use the pinned Dockerfile and the same GPU-isolated Docker invocation as the main POC. Mount the project at `/poc` and its cache at `/cache`. Only physical GPU UUID `GPU-65bcac75-a854-7804-3e12-aca221530033` is exposed. Klein remains on physical GPU 1. Run GPU experiments serially.
24
+
25
+ Inside the container:
26
+
27
+ ```bash
28
+ python -m nunchaku_backend.collect_v3 \
29
+ --jobs /poc/experiments/fidelity-v3/calibration-jobs.json \
30
+ --out /cache/qwen21-activation-v3-40 --steps 40
31
+
32
+ python -m nunchaku_backend.teacher_v3 \
33
+ --jobs /poc/experiments/fidelity-v3/quality-jobs-all.json \
34
+ --sample-dir /poc/samples/fidelity-v3/bf16-dit
35
+
36
+ python -m nunchaku_backend.export_v3 \
37
+ --model-path /cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/b3179ad355be050328e483a9dfdd9e60cd62adfa \
38
+ --activations /cache/qwen21-activation-v3-40 \
39
+ --baseline-calibration /cache/qwen21-calibration \
40
+ --out /cache/qwen-nunchaku-v3-r128 --device cuda:0 \
41
+ --ranks 128 --alphas .25 .5 .75 \
42
+ --families activation_only smoothquant --weighting none \
43
+ --iterations 16 --final-gptq --factorization balanced
44
+
45
+ python /poc/runner.py --backend nunchaku \
46
+ --nunchaku-checkpoint /cache/qwen-nunchaku-v3-r128 \
47
+ --prequant /cache/qwen-nf4 --no-cache \
48
+ --jobs /poc/experiments/fidelity-v3/quality-jobs-all.json \
49
+ --sample-dir /poc/samples/fidelity-v3/nunchaku-v3-r128
50
+ ```
51
+
52
+ The subsequent selective upgrade keeps the other 192 quantized projections and BF16 boundary tensors byte-identical, while choosing rank128, 256 or 512 for each of the 32 MLP projection layers using validation kernel output error. All 32 selected rank512 in this run. This adds 384 MiB of model tensors; it is not a peak-VRAM estimate.
53
+
54
+ ```bash
55
+ python -m nunchaku_backend.upgrade_mlp_rank_v3 \
56
+ --source-checkpoint /cache/qwen-nunchaku-v3-r128 \
57
+ --model-path /cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/b3179ad355be050328e483a9dfdd9e60cd62adfa \
58
+ --activations /cache/qwen21-activation-v3-40 \
59
+ --baseline-calibration /cache/qwen21-calibration \
60
+ --out /cache/qwen-nunchaku-v3-mlpproj-upgrade \
61
+ --device cuda:0 --ranks 256 512 --iterations 16
62
+
63
+ python /poc/runner.py --backend nunchaku \
64
+ --nunchaku-checkpoint /cache/qwen-nunchaku-v3-mlpproj-upgrade \
65
+ --prequant /cache/qwen-nf4 --no-cache \
66
+ --jobs /poc/experiments/fidelity-v3/quality-jobs-all.json \
67
+ --sample-dir /poc/samples/fidelity-v3/nunchaku-v3-mlpproj512
68
+ ```
69
+
70
+ The checkpoint exporter records its exact search settings, activation-file hashes, per-layer candidate errors and selection decisions in the checkpoint manifest. Selection uses validation outputs from the actual Nunchaku kernel; held-out rows are measured only after selection. GPTQ is retained only when validation improves. The original one-pass recipe remains a candidate, although rerunning randomized SVD is not necessarily a byte-for-byte reproduction of the archived v2 checkpoint.
71
+
72
+ After generation, run `scripts/build_fidelity_gallery.py` locally to rebuild the offline comparison. Results must distinguish measured numerical approximation, visible image quality, latency and VRAM. No near-lossless claim follows from per-layer error alone.
73
+
74
+ ## Sources
75
+
76
+ See `research/svdquant-paper-v3.md` for the paper and official implementation audit, and `research/nunchaku-v3-audit.md` for the shipped Klein checkpoint inspection. The new converter uses a bounded search and randomized SVD; RMS weighting and ridge correction are explicitly experimental extensions, and its GPTQ implementation retains original channel order. It does not reproduce every official calibration detail, especially joint attention/block-level selection.
77
+
78
+ ## Selected serving configuration
79
+
80
+ Use `docker compose -f compose.yaml -f compose.nunchaku-v3.yaml up -d` for calibrated rank128. The selective512 override remains experimental and is not selected. Both candidates complete42/42 jobs; the comparison gallery contains126 new quality images plus the archived comparison controls.
81
+
82
+ FinalAPI functional checks are complete, while strict same-seed repeatability failed. Reproduce the full diagnostic recording with `python /poc/scripts/api_fidelity_smoke.py --record-repeat-differences --sample-dir samples/fidelity-v3/api-new-run --report results/api-new-run.json`; use a fresh output directory. Omitting the flag preserves the strict assertion after all images/differences are saved. This is not a tolerance-based repeatability pass. See REPORT.md for measured limitations.
reproduction/experiments/fidelity-v3/REPORT.md ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Qwen Image 2.1 fidelity revision — September 21, 2026
2
+
3
+ The new evaluation contains 21 scenarios at both 25 and 40 steps. It includes interacting groups with specific roles, a wheelchair award ceremony, a family reunion, a radio interview, dense bilingual text, a three-panel comic, museum geometry, product layouts, local preservation edits and two-reference composition. Initial samples remain archived; all new comparisons are in `samples/fidelity-v3/`.
4
+
5
+ The serving choice is calibrated rank 128. The larger selective rank 512 candidate is retained as an experiment: despite better numerical error, direct image inspection found extra objects and geometry regressions. All 126 comparison images (42 each for the BF16-transformer teacher, calibrated rank 128 and selective rank 512) are complete, locally verified and covered by direct visual reviews. These are 42 matched jobs, not 126 independent prompts. The selected API is running and its functional generation/edit/cache checks are complete; strict same-seed repeatability failed, as documented below.
6
+
7
+ ## What changed in the converter
8
+
9
+ The previous one-pass rank 32 approximation was replaced by calibration against actual Nunchaku CUDA output. The new search selects smoothing and low-rank residual fits using validation output error, checks held-out rows after selection, and accepts optional GPTQ correction only when validation improves. Rank 128 is used across 224 quantized linear layers. Each linear remains W4A4 plus a BF16 low-rank branch; original normalization, positional encoding, attention and KV-cache behavior remain in the Diffusers implementation.
10
+
11
+ The SVDQuant paper and official Nunchaku/DeepCompressor implementation motivated the calibration and decomposition work. This is a custom Qwen Image 2.1 adapter and bounded calibration implementation, not a reproduction of every official calibration setting. The exact shipped Klein calibration recipe is not public. See the [SVDQuant source review](../../research/svdquant-paper-v3.md) and [converter audit](../../research/nunchaku-v3-audit.md).
12
+
13
+ A follow-up sensitivity sweep found that the MLP projection and output layers contribute substantially to denoiser error. Temporarily restoring both roles to BF16 reduced the three diagnostic relative-L2 errors to 5.61%, 2.74% and 9.28%, but adds roughly 4.15 GiB of model state. This is a diagnostic, not a measured image-quality or safe two-reference serving configuration.
14
+
15
+ The practical candidate instead raises only the 32 MLP projection low-rank branches from 128 to 512. All 32 selected 512 by validation. Other 192 quantized linears and 193 untouched shards remain byte-identical. It retains all 224 INT4 kernels and adds 384 MiB of model tensors. The complete checkpoint has 5,057,830,912 bytes (4.710 GiB) of model tensors. Independent CPU loading verified 1,417 finite state tensors, every output/source shard hash and all 193 untouched shard identities; load time was 3.48 seconds. Across the 32 upgraded layers, held-out MSE ratios to actual rank 128 range from 0.531 to 0.656, with median 0.642. Model state is not peak VRAM.
16
+
17
+ ## Numerical evidence and rejected changes
18
+
19
+ The first rank 128 converter improved held-out linear MSE against a regenerated one-pass rank 32 recipe in all 224 linears; median ratio was 0.6064. That numerical baseline is not the exact archived v2 checkpoint. Actual v2 images and whole-denoiser checks use its real saved checkpoint.
20
+
21
+ Three fixed-recipe runs allowing 100 fitting iterations did not justify a full conversion: one control stayed unchanged, one slightly better validation result worsened held-out error, and another slightly regressed. More iterations were rejected instead of presumed beneficial.
22
+
23
+ The rank 512 probe improved held-out projection MSE by 35.5–38.6% in three representative layers. Through the complete MLP, improvement was smaller: 7.9–12.2%. Block 11 rank 256 improved raw projection error while slightly worsening the parent MLP's validation error. These checks demonstrate why linear error alone cannot establish image fidelity.
24
+
25
+ | Actual checkpoint | First-step denoiser relative L2 | Middle step | Final step |
26
+ |---|---:|---:|---:|
27
+ | Archived v2 |12.48%|6.54%|21.87%|
28
+ | Calibrated rank 128 |8.85%|5.08%|19.43%|
29
+ | Selective MLP rank 512 |8.46%|4.72%|17.93%|
30
+
31
+ This is one disjoint development editing prompt at steps 0/20/39 of 40. Every checkpoint builds its own prefix KV cache; no teacher cache is reused. These are conditional predictions on the teacher trajectory, not free-running image scores.
32
+
33
+ ## Image review and step count
34
+
35
+ Forty steps is the [official starting recommendation](https://huggingface.co/docs/diffusers/main/api/pipelines/qwenimage21). Both 25 and 40 are evaluated here at 1024×1024, with fixed seeds, CFG 1 and prefix KV caching. More steps do not reliably fix incorrect counts or relationships.
36
+
37
+ The rank 128 review covers all 42 teacher/candidate pairs. It restores the fantasy telescope and improves the old pancake-count error, retains all 14 bilingual poster strings and six comic dialogue lines, and performs the local coat/poster edits with strong preservation. It still changes some faces, hand positions, poses, object scale and design details. Its 25-step chess banner loses a digit in 2026; 40 steps restores it. Its museum drawings add tables and confuse routes at both step counts. The product campaign improves requested cup placement relative to the teacher, demonstrating that visual similarity and prompt compliance are different measurements.
38
+
39
+ The BF16 teacher also has real limitations: its 25-step chess title reads CIT CHES FINAL, the comic does not stage keys under the chair, the product relocation edit leaves its cup on the pedestal, and the two-reference portrait looks off-camera. These are shared model/instruction failures, not evidence that every candidate difference is caused by quantization. The teacher uses the same NF4 encoder, so it isolates transformer approximation rather than representing a fully BF16 pipeline.
40
+
41
+ Direct observations and image hashes are recorded in the [rank 128 scorecard](../../research/quality-v3-scorecard.json), with root versus independent reviewer attribution, and the [complete selective rank 512 review index](../../research/quality-v3-rank512-review-index.md). Similarity metrics compare the same labels/seeds and unaligned 1024×1024 images. SSIM and PSNR measure resemblance, not a percentage of semantic quality.
42
+
43
+ The **matched original 18-job cohort** is available for all four candidates. Each is compared with the corresponding BF16-transformer image; these medians pool nine scenarios at both step counts.
44
+
45
+ | Candidate | Images | Median SSIM | Median PSNR (dB) |
46
+ |---|---:|---:|---:|
47
+ | Archived Nunchaku v2 | 18 | 0.7608 | 18.52 |
48
+ | Calibrated rank 128 | 18 | 0.8508 | 20.57 |
49
+ | Selective rank 512 | 18 | 0.8287 | 19.96 |
50
+ | NF4 transformer | 18 | 0.8237 | 20.84 |
51
+
52
+ The **full matched 42-job cohort** is available for the two new Nunchaku candidates. It includes the original 18 jobs plus 24 expanded jobs.
53
+
54
+ | Candidate | Images | Median SSIM | Median PSNR (dB) |
55
+ |---|---:|---:|---:|
56
+ | Calibrated rank 128 | 42 | 0.8269 | 19.99 |
57
+ | Selective rank 512 | 42 | 0.8207 | 19.62 |
58
+
59
+ Rank 128 has higher median SSIM than NF4 on the original 18 jobs, while NF4 has higher median PSNR. Selective rank 512 trails rank 128 on both medians in both matched cohorts despite its better denoiser probe. NF4 was not measured on the full 42-job cohort. Comparisons across the two tables would mix different prompt sets. Exact labels and per-image metrics are in the [original 18-job results](../../results/compare-fidelity-v3-original18.json) and [full 42-job results](../../results/compare-fidelity-v3-full42.json).
60
+
61
+ ## Selective rank 512 decision
62
+
63
+ The larger branch is not promoted. All 42 selective outputs now have direct review coverage, including fixed-input checks for all 14 edits. In the 12 original generation jobs it improves the thin perfume peel and restores a more teacher-like towel grip at 40 steps, but produces seven blueberries and two knives in the 40-step breakfast, two cats in the 25-step fantasy scene, and a distorted horn-like telescope at 40. Botanical wording and lemon counts remain correct.
64
+
65
+ The 16 expanded generation jobs show further mixed effects. Selective rank 512 restores the chess year and camera-at-eye action, but adds six kitchen rolls at 25 steps and six/five cafe tables at 25/40. The 40-step museum also gains at least three shelf groups instead of two. Bilingual text and comic dialogue remain readable, while the comic still fails to stage the keys under the chair. Product cups sit correctly on the tabletop at 25 but return to the pedestal at 40, matching a teacher error and losing rank 128's adherence improvement. Forty steps trades failures rather than consistently repairing them.
66
+
67
+ Localized towel, coat and poster edits retain strong visual preservation, with small texture and typography changes and no decisive broad improvement over rank 128. The 40-step two-reference portrait restores the unobscured bottle emblem and supporting grip more closely to the teacher; off-camera gaze and rendering drift remain. Product-relocation edits still leave the cup on the pedestal in all compared backends. These observations support retaining the smaller checkpoint for this POC, not a claim that rank 128 is universally better or near-lossless.
68
+
69
+ ## Timing, memory and reproducibility
70
+
71
+ For the selected rank 128 checkpoint, median generation takes **15.50 s at 25 steps / 22.82 s at 40**, versus **27.51 s / 42.76 s** for the streamed BF16 transformer. These medians use the same 14 generation scenarios per step count. Median one-reference edit inference takes **19.48 s / 28.43 s** across six matched scenarios per step count. The single two-reference portrait takes **23.76 s / 34.35 s**; these are individual runs, not repeated-run medians. Sampled board occupancy reaches **11,994 MiB (11.71 GiB)** in that two-reference case. Selective rank 512 is slightly slower (**16.06 s / 23.56 s** median generation) and reaches **12,374 MiB** with two references.
72
+
73
+ See [performance.md](performance.md) and the [per-job measurement summary](../../results/fidelity-v3-summary.json) for verified per-job measurements. CLI comparisons disable prompt/reference LRU reuse, retain per-request prefix KV, use an untiled BF16 VAE and release dead KV before decoding. Timings cover engine inference, excluding process startup and image file loading/saving. Board occupancy is sampled separately from allocator peaks; initial teacher jobs without board telemetry are explicitly missing those values.
74
+
75
+ GPU 0 is an RTX 4070 Ti SUPER with 16,376 MiB. Klein remains on its existing GPU 1. GPU jobs run serially. The implementation stages the large NF4 encoder on CPU between requests; it retains all 36 decoder layers and removes only the unused LM-head/final-normalization path after exact feature-parity verification. Persistent compilation caches and optional compilation remain available; this Nunchaku quality revision uses eager inference.
76
+
77
+ Reproduction commands are in the experiment README, including calibration, checkpoint conversion, selective rank upgrade and the frozen 42-job image suite. Docker/model/library revisions are pinned. Saved source state, service rollback information and historical failures remain preserved.
78
+
79
+ ## Service verification
80
+
81
+ The calibrated rank128 API is healthy on PC 2 loopback port 8091. Four real 25-step requests completed: bilingual generation at 18.22 s, cached repeat 14.81 s, two-reference portrait 26.42 s, and cached repeat 22.93 s. These are client wall times from one run each. The prompt cache hit on both repeats and the reference-latent cache hit twice on the two-reference repeat. All four responses were COMPLETED with valid 1024×1024 PNGs. Engine startup measured 8.78 s from the saved checkpoint, including ML imports; this is a cached-disk process start, not machine boot. Maximum sampled API board occupancy was 11,986 MiB. See [final API measurements](../../results/api-fidelity-v3-final.json).
82
+
83
+ **Strict byte-identical repeatability failed.** The failure is preserved, not converted into a tolerance pass. A controlled four-run diagnostic compared two uncached runs and a cache miss/hit: all prompt tensors and first-transformer inputs were exactly equal in values, shapes, strides and dtypes, and stored cache tensors remained unchanged. Nevertheless the first transformer outputs differed even without caching. Uncached repeats had RGB MAE 5.06/255; cache miss/hit had 6.48/255. This isolates the observed divergence to the transformer forward, but does not identify the exact kernel or operation. Official low-rank atomic reductions are a hypothesis, not a confirmed sole cause. See [diagnosis](../../research/api-cache-v3-diagnosis.md) and [input fingerprints](../../results/repeatability-diagnostic/diagnostic.json).
84
+
85
+ The final API verification used an explicit record-differences mode so it could preserve and inspect every returned image while retaining `strict_repeatability_passed=false`. Both posters retained the requested wording; the two-reference repeats changed hand placement, bench and suitcase position while retaining the person/product attributes. Same-seed variation can affect composition, not merely file bytes. The image-suite comparisons therefore describe individual recorded realizations; they do not establish repeatable equivalence or isolate every visual difference solely to quantization.
86
+
87
+ Klein remains healthy with the same container and original start time on GPU1; Qwen alone occupies GPU0 for this POC. Higgs and GPU0 Comfy remain stopped under the user's authorization, GPU1 Comfy and connectivity remain intact, and experiment telemetry is stopped. [Final service state](../../results/final-state-v3.json) and `scripts/rollback.sh` preserve the recovery path.
88
+
89
+ ## Limits
90
+
91
+ This is a 1024×1024 POC below the release's recommended native 2K resolution. It does not establish near-lossless quality, superiority to Klein, full-BF16 encoder fidelity, ten-reference operation, native 2K speed or a production-router rollout. Calibration is small and edit splits share one source photo; prefix/reference-token coverage is limited. The visual suite was reused after earlier candidate results informed further work, so these are diagnostic comparisons rather than a fresh unseen acceptance set. Joint attention and full-block optimization are not implemented by this converter. The model uses the Qwen Research License; the downloaded license is retained with the release research.
reproduction/experiments/fidelity-v3/calibration-jobs.json ADDED
@@ -0,0 +1,75 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "calibration-landscape",
4
+ "prompt": "Photorealistic mountain valley at sunrise, a winding river and pine forest, clouds behind snowy peaks, crisp natural colors.",
5
+ "width": 1024,
6
+ "height": 1024,
7
+ "seed": 731,
8
+ "split": "train"
9
+ },
10
+ {
11
+ "label": "calibration-chess-portrait",
12
+ "prompt": "Editorial photograph of an elderly chess player concentrating over a wooden chessboard in a quiet cafe, soft side lighting, natural facial texture.",
13
+ "width": 1024,
14
+ "height": 1024,
15
+ "seed": 732,
16
+ "split": "train"
17
+ },
18
+ {
19
+ "label": "calibration-subway",
20
+ "prompt": "A candid documentary photograph of a modern subway platform with commuters, silver train arriving, overhead fluorescent lighting and deep perspective.",
21
+ "width": 1024,
22
+ "height": 1024,
23
+ "seed": 733,
24
+ "split": "train"
25
+ },
26
+ {
27
+ "label": "calibration-mug-white",
28
+ "prompt": "Change the mug to matte white ceramic. Preserve the composition, spoon, table and lighting.",
29
+ "images": [
30
+ "/poc/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png"
31
+ ],
32
+ "width": 1024,
33
+ "height": 1024,
34
+ "seed": 735,
35
+ "split": "train"
36
+ },
37
+ {
38
+ "label": "validation-market",
39
+ "prompt": "Documentary photograph of an outdoor farmers market with vegetables, baskets and shoppers beneath striped awnings in soft morning light.",
40
+ "seed": 2401,
41
+ "split": "validation",
42
+ "width": 1024,
43
+ "height": 1024
44
+ },
45
+ {
46
+ "label": "validation-mug-edit",
47
+ "prompt": "Place a folded linen cloth under the mug. Preserve the mug, table, spoon and background.",
48
+ "seed": 2402,
49
+ "images": [
50
+ "/poc/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png"
51
+ ],
52
+ "split": "validation",
53
+ "width": 1024,
54
+ "height": 1024
55
+ },
56
+ {
57
+ "label": "heldout-sailboat",
58
+ "prompt": "A watercolor illustration of two small sailboats moored beside a wooden jetty in a quiet lake, distant mountains and pine trees, overcast sky.",
59
+ "seed": 3401,
60
+ "split": "heldout",
61
+ "width": 1024,
62
+ "height": 1024
63
+ },
64
+ {
65
+ "label": "heldout-mug-edit",
66
+ "prompt": "Change the tabletop to polished dark walnut wood while retaining the mug and spoon and wall with the same framing and lighting.",
67
+ "seed": 3402,
68
+ "images": [
69
+ "/poc/samples/archive/2026-09-20-initial-poc/reference-clean-rgb.png"
70
+ ],
71
+ "split": "heldout",
72
+ "width": 1024,
73
+ "height": 1024
74
+ }
75
+ ]
reproduction/experiments/fidelity-v3/candidate-remaining.json ADDED
@@ -0,0 +1,326 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "editorial-25",
4
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
5
+ "steps": 25,
6
+ "seed": 118,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "rainy-city-40",
12
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
13
+ "steps": 40,
14
+ "seed": 227,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "rainy-city-25",
20
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
21
+ "steps": 25,
22
+ "seed": 227,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "botanical-poster-25",
28
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
29
+ "steps": 25,
30
+ "seed": 336,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "botanical-poster-40",
36
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
37
+ "steps": 40,
38
+ "seed": 336,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "food-spatial-25",
44
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
45
+ "steps": 25,
46
+ "seed": 445,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "fantasy-cutaway-25",
52
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
53
+ "steps": 25,
54
+ "seed": 554,
55
+ "width": 1024,
56
+ "height": 1024
57
+ },
58
+ {
59
+ "label": "glass-product-40",
60
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
61
+ "steps": 40,
62
+ "seed": 663,
63
+ "width": 1024,
64
+ "height": 1024
65
+ },
66
+ {
67
+ "label": "glass-product-25",
68
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
69
+ "steps": 25,
70
+ "seed": 663,
71
+ "width": 1024,
72
+ "height": 1024
73
+ },
74
+ {
75
+ "label": "editorial-edit-25",
76
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
77
+ "steps": 25,
78
+ "seed": 774,
79
+ "width": 1024,
80
+ "height": 1024,
81
+ "images": [
82
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
83
+ ]
84
+ },
85
+ {
86
+ "label": "editorial-edit-40",
87
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
88
+ "steps": 40,
89
+ "seed": 774,
90
+ "width": 1024,
91
+ "height": 1024,
92
+ "images": [
93
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
94
+ ]
95
+ },
96
+ {
97
+ "label": "poster-edit-25",
98
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
99
+ "steps": 25,
100
+ "seed": 885,
101
+ "width": 1024,
102
+ "height": 1024,
103
+ "images": [
104
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
105
+ ]
106
+ },
107
+ {
108
+ "label": "poster-edit-40",
109
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
110
+ "steps": 40,
111
+ "seed": 885,
112
+ "width": 1024,
113
+ "height": 1024,
114
+ "images": [
115
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
116
+ ]
117
+ },
118
+ {
119
+ "label": "city-edit-25",
120
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
121
+ "steps": 25,
122
+ "seed": 996,
123
+ "width": 1024,
124
+ "height": 1024,
125
+ "images": [
126
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
127
+ ]
128
+ },
129
+ {
130
+ "label": "city-edit-40",
131
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
132
+ "steps": 40,
133
+ "seed": 996,
134
+ "width": 1024,
135
+ "height": 1024,
136
+ "images": [
137
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
138
+ ]
139
+ },
140
+ {
141
+ "label": "expanded-community-kitchen-25",
142
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
143
+ "steps": 25,
144
+ "seed": 21011,
145
+ "width": 1024,
146
+ "height": 1024
147
+ },
148
+ {
149
+ "label": "expanded-chess-awards-25",
150
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
151
+ "steps": 25,
152
+ "seed": 21022,
153
+ "width": 1024,
154
+ "height": 1024
155
+ },
156
+ {
157
+ "label": "expanded-station-reunion-25",
158
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
159
+ "steps": 25,
160
+ "seed": 21033,
161
+ "width": 1024,
162
+ "height": 1024
163
+ },
164
+ {
165
+ "label": "expanded-station-reunion-40",
166
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
167
+ "steps": 40,
168
+ "seed": 21033,
169
+ "width": 1024,
170
+ "height": 1024
171
+ },
172
+ {
173
+ "label": "expanded-bilingual-festival-25",
174
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
175
+ "steps": 25,
176
+ "seed": 21044,
177
+ "width": 1024,
178
+ "height": 1024
179
+ },
180
+ {
181
+ "label": "expanded-library-comic-25",
182
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
183
+ "steps": 25,
184
+ "seed": 21055,
185
+ "width": 1024,
186
+ "height": 1024
187
+ },
188
+ {
189
+ "label": "expanded-museum-plan-25",
190
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
191
+ "steps": 25,
192
+ "seed": 21066,
193
+ "width": 1024,
194
+ "height": 1024
195
+ },
196
+ {
197
+ "label": "expanded-museum-plan-40",
198
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
199
+ "steps": 40,
200
+ "seed": 21066,
201
+ "width": 1024,
202
+ "height": 1024
203
+ },
204
+ {
205
+ "label": "expanded-ridgeline-campaign-25",
206
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
207
+ "steps": 25,
208
+ "seed": 21077,
209
+ "width": 1024,
210
+ "height": 1024
211
+ },
212
+ {
213
+ "label": "expanded-ridgeline-campaign-40",
214
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
215
+ "steps": 40,
216
+ "seed": 21077,
217
+ "width": 1024,
218
+ "height": 1024
219
+ },
220
+ {
221
+ "label": "expanded-newsroom-interview-25",
222
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
223
+ "steps": 25,
224
+ "seed": 21088,
225
+ "width": 1024,
226
+ "height": 1024
227
+ },
228
+ {
229
+ "label": "expanded-newsroom-interview-40",
230
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
231
+ "steps": 40,
232
+ "seed": 21088,
233
+ "width": 1024,
234
+ "height": 1024
235
+ },
236
+ {
237
+ "label": "expanded-station-coat-edit-25",
238
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
239
+ "steps": 25,
240
+ "seed": 21101,
241
+ "width": 1024,
242
+ "height": 1024,
243
+ "images": [
244
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
245
+ ]
246
+ },
247
+ {
248
+ "label": "expanded-station-coat-edit-40",
249
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
250
+ "steps": 40,
251
+ "seed": 21101,
252
+ "width": 1024,
253
+ "height": 1024,
254
+ "images": [
255
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
256
+ ]
257
+ },
258
+ {
259
+ "label": "expanded-festival-type-edit-25",
260
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
261
+ "steps": 25,
262
+ "seed": 21112,
263
+ "width": 1024,
264
+ "height": 1024,
265
+ "images": [
266
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
267
+ ]
268
+ },
269
+ {
270
+ "label": "expanded-festival-type-edit-40",
271
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
272
+ "steps": 40,
273
+ "seed": 21112,
274
+ "width": 1024,
275
+ "height": 1024,
276
+ "images": [
277
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
278
+ ]
279
+ },
280
+ {
281
+ "label": "expanded-multiref-portrait-25",
282
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
283
+ "steps": 25,
284
+ "seed": 21123,
285
+ "width": 1024,
286
+ "height": 1024,
287
+ "images": [
288
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
289
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
290
+ ]
291
+ },
292
+ {
293
+ "label": "expanded-multiref-portrait-40",
294
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
295
+ "steps": 40,
296
+ "seed": 21123,
297
+ "width": 1024,
298
+ "height": 1024,
299
+ "images": [
300
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
301
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
302
+ ]
303
+ },
304
+ {
305
+ "label": "expanded-product-relocation-edit-25",
306
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
307
+ "steps": 25,
308
+ "seed": 21134,
309
+ "width": 1024,
310
+ "height": 1024,
311
+ "images": [
312
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
313
+ ]
314
+ },
315
+ {
316
+ "label": "expanded-product-relocation-edit-40",
317
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
318
+ "steps": 40,
319
+ "seed": 21134,
320
+ "width": 1024,
321
+ "height": 1024,
322
+ "images": [
323
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
324
+ ]
325
+ }
326
+ ]
reproduction/experiments/fidelity-v3/expanded-jobs.json ADDED
@@ -0,0 +1,220 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "expanded-community-kitchen-25",
4
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
5
+ "steps": 25,
6
+ "seed": 21011,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "expanded-community-kitchen-40",
12
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
13
+ "steps": 40,
14
+ "seed": 21011,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "expanded-chess-awards-25",
20
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
21
+ "steps": 25,
22
+ "seed": 21022,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "expanded-chess-awards-40",
28
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
29
+ "steps": 40,
30
+ "seed": 21022,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "expanded-station-reunion-25",
36
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
37
+ "steps": 25,
38
+ "seed": 21033,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "expanded-station-reunion-40",
44
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
45
+ "steps": 40,
46
+ "seed": 21033,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "expanded-bilingual-festival-25",
52
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
53
+ "steps": 25,
54
+ "seed": 21044,
55
+ "width": 1024,
56
+ "height": 1024
57
+ },
58
+ {
59
+ "label": "expanded-bilingual-festival-40",
60
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
61
+ "steps": 40,
62
+ "seed": 21044,
63
+ "width": 1024,
64
+ "height": 1024
65
+ },
66
+ {
67
+ "label": "expanded-library-comic-25",
68
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
69
+ "steps": 25,
70
+ "seed": 21055,
71
+ "width": 1024,
72
+ "height": 1024
73
+ },
74
+ {
75
+ "label": "expanded-library-comic-40",
76
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
77
+ "steps": 40,
78
+ "seed": 21055,
79
+ "width": 1024,
80
+ "height": 1024
81
+ },
82
+ {
83
+ "label": "expanded-museum-plan-25",
84
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
85
+ "steps": 25,
86
+ "seed": 21066,
87
+ "width": 1024,
88
+ "height": 1024
89
+ },
90
+ {
91
+ "label": "expanded-museum-plan-40",
92
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
93
+ "steps": 40,
94
+ "seed": 21066,
95
+ "width": 1024,
96
+ "height": 1024
97
+ },
98
+ {
99
+ "label": "expanded-ridgeline-campaign-25",
100
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
101
+ "steps": 25,
102
+ "seed": 21077,
103
+ "width": 1024,
104
+ "height": 1024
105
+ },
106
+ {
107
+ "label": "expanded-ridgeline-campaign-40",
108
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
109
+ "steps": 40,
110
+ "seed": 21077,
111
+ "width": 1024,
112
+ "height": 1024
113
+ },
114
+ {
115
+ "label": "expanded-newsroom-interview-25",
116
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
117
+ "steps": 25,
118
+ "seed": 21088,
119
+ "width": 1024,
120
+ "height": 1024
121
+ },
122
+ {
123
+ "label": "expanded-newsroom-interview-40",
124
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
125
+ "steps": 40,
126
+ "seed": 21088,
127
+ "width": 1024,
128
+ "height": 1024
129
+ },
130
+ {
131
+ "label": "expanded-station-coat-edit-25",
132
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
133
+ "steps": 25,
134
+ "seed": 21101,
135
+ "width": 1024,
136
+ "height": 1024,
137
+ "images": [
138
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
139
+ ]
140
+ },
141
+ {
142
+ "label": "expanded-station-coat-edit-40",
143
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
144
+ "steps": 40,
145
+ "seed": 21101,
146
+ "width": 1024,
147
+ "height": 1024,
148
+ "images": [
149
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
150
+ ]
151
+ },
152
+ {
153
+ "label": "expanded-festival-type-edit-25",
154
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
155
+ "steps": 25,
156
+ "seed": 21112,
157
+ "width": 1024,
158
+ "height": 1024,
159
+ "images": [
160
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
161
+ ]
162
+ },
163
+ {
164
+ "label": "expanded-festival-type-edit-40",
165
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
166
+ "steps": 40,
167
+ "seed": 21112,
168
+ "width": 1024,
169
+ "height": 1024,
170
+ "images": [
171
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
172
+ ]
173
+ },
174
+ {
175
+ "label": "expanded-multiref-portrait-25",
176
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
177
+ "steps": 25,
178
+ "seed": 21123,
179
+ "width": 1024,
180
+ "height": 1024,
181
+ "images": [
182
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
183
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
184
+ ]
185
+ },
186
+ {
187
+ "label": "expanded-multiref-portrait-40",
188
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
189
+ "steps": 40,
190
+ "seed": 21123,
191
+ "width": 1024,
192
+ "height": 1024,
193
+ "images": [
194
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
195
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
196
+ ]
197
+ },
198
+ {
199
+ "label": "expanded-product-relocation-edit-25",
200
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
201
+ "steps": 25,
202
+ "seed": 21134,
203
+ "width": 1024,
204
+ "height": 1024,
205
+ "images": [
206
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
207
+ ]
208
+ },
209
+ {
210
+ "label": "expanded-product-relocation-edit-40",
211
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
212
+ "steps": 40,
213
+ "seed": 21134,
214
+ "width": 1024,
215
+ "height": 1024,
216
+ "images": [
217
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
218
+ ]
219
+ }
220
+ ]
reproduction/experiments/fidelity-v3/expanded-rubric.md ADDED
@@ -0,0 +1,92 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Expanded Qwen fidelity evaluation
2
+
3
+ This is an **evaluation-only** set. Do not use its prompts, seeds, teacher images, editing references or captured activations to fit, calibrate or select a quantization configuration. Freeze candidate settings before running it. If these results motivate another change, report that reuse and obtain a fresh independent evaluation set for the next acceptance claim.
4
+
5
+ `expanded-jobs.json` contains 12 scenarios, each at 25 and 40 steps, for 24 jobs at 1024×1024: eight text-to-image scenarios followed by four editing scenarios. Each pair has identical prompt, seed, size and references; only the step count and label differ. Seeds are new and fixed. The manifest uses the existing runner schema.
6
+
7
+ ## Order, controls and evidence
8
+
9
+ 1. Generate the BF16-transformer teacher for the eight generation scenarios first. The three required 40-step reference images are `expanded-station-reunion-40.png`, `expanded-bilingual-festival-40.png`, and `expanded-ridgeline-campaign-40.png`, all under `/poc/samples/fidelity-v3/bf16-dit/`.
10
+ 2. Use these exact reference bytes for every teacher and candidate editing job, including both step counts. Never substitute a candidate's own generated reference. Record reference SHA256 values with the run. The multi-reference job receives the station image first and the product image second.
11
+ 3. Keep seed, scheduler, KV mode, text encoder, VAE, resolution and preprocessing fixed within teacher/candidate comparisons. Record any exception. This teacher isolates transformer precision; it is not an entirely BF16 pipeline if its encoder is quantized.
12
+ 4. Inspect the actual teacher references before scoring edits. If a required source feature is absent or unreadable, mark that edit criterion **source-limited** and describe what is visible. Do not claim preservation of an identity, text or product feature the reference never established.
13
+ 5. Review at full resolution. Record adherence, visual defects and drift from the teacher separately. A teacher can fail the prompt; a candidate can fix a prompt error while becoming less pixel-similar. SSIM or PSNR cannot adjudicate that distinction.
14
+
15
+ For each criterion use **pass / partial / fail / source-limited**, with one sentence of visible evidence. Partial means the intended result is recognizable but incomplete or ambiguous; identify the exact ambiguity. Record failures independently rather than averaging unrelated properties into an unexplained score. Review 25 and 40 steps separately, then state whether the additional steps visibly helped, hurt or made no material difference for that case.
16
+
17
+ ## Generation scenarios
18
+
19
+ | Scenario and seed | Critical requested content | Composition and physical plausibility | Secondary review |
20
+ |---|---|---|---|
21
+ | `expanded-community-kitchen`, 21011 | Exactly six adults with six distinct roles: green-apron woman chopping carrots at left; silver-haired man ladling soup at center; mustard-cardigan woman passing a tray to navy-shirt man at center-right; red-apron man rinsing greens at right; glasses/purple-sweater woman reading a recipe behind center. The one tray contains four bread rolls. | Passing and receiving hands touch opposite edges of the same tray; arms attach to the right people. Knife/holding hand and ladle/pot contact are plausible. All six are visible at least waist-up and engaged rather than posed. | Face separation, anatomy, kitchen tools, reflections, food texture, coordinated documentary appearance. Note any background figure/reflection counted as a spurious person. |
22
+ | `expanded-chess-awards`, 21022 | Exactly five adults: seated woman winner at center, older male presenter immediately left, cream-sweater male runner-up immediately right, kneeling female photographer far left, burgundy-suited host far right. Winner and presenter jointly hold one gold trophy; runner-up wears one silver medal and applauds. Banner is `CITY CHESS FINAL` with `2026` below. | Winner sits in a recognizable manual wheelchair with coherent wheels/footrests; all five fit fully in frame. Trophy hand contact, applauding hands, camera held to eye and microphone/cue-card grips are plausible. | Small rear-left chess table/board; actual stage presentation; no additional audience. Do not penalize assistive-device use or natural bodily variation; score incoherent rendering and missed requested geometry. |
23
+ | `expanded-station-reunion`, 21033 | Exactly four people: mother left with dark low bun, oval face, gold earrings, navy coat/yellow scarf; red-coated girl at center holding mother's hand and carrying a stuffed rabbit; gray-jacket/red-backpack father at right embracing teal-coated gray-haired grandmother. One green suitcase by mother's left leg. | Clear handholding and embrace, correct arm ownership, believable eye contact. Full bodies/feet in frame. Mother and girl watch the reunion; father and grandmother face each other. | Preserve a usable mother identity for later edits. Train/platform realism, coat/scarf details, coherent suitcase, no duplicate people or cropped feet. |
24
+ | `expanded-bilingual-festival`, 21044 | All 14 specified text strings below are readable and correct. Three program rows have aligned time/title columns, each English title directly above its corresponding Chinese title. | Header/date, thin rule, program, bridge illustration and two footer lines form a clear hierarchy without collisions. Cream/navy/vermilion flat square poster, generous margins. | Chinese glyph integrity, Latin spelling, punctuation and typography. Check letters at native resolution; decorative pseudo-text is a failure even when the poster looks polished. |
25
+ | `expanded-library-comic`, 21055 | Exactly three equal vertical panels left-to-right; the same Maya (black bob/glasses/red sweater) and Leo (curly brown hair/blue shirt) in each. Exactly two speech bubbles per panel with the six exact lines below and correct speakers. Sequence: searching bag; empty pocket plus keys under chair; Maya holding recovered keys. | Bubble tails identify their speakers; reading order is clear. Same reading-room/chair continuity. Hands, pointing, bag, pocket and key ring form a coherent causal sequence. | Expressions change appropriately. No extra panels, characters, captions or missing key ring. A plausible illustration with scrambled dialogue or event order fails those criteria. |
26
+ | `expanded-museum-plan`, 21066 | Exactly five labeled spaces: Gallery A upper left, Gallery B upper right, Main Hall central full width, Cafe lower left, Shop lower right. Two display plinths per gallery, three round cafe tables, two shop shelves. All eight label strings below correct. | Bottom-center entrance aisle separates Cafe/Shop and joins the hall; each corner room has one doorway into hall. One continuous blue route goes entrance → aisle → hall → Gallery A through doorways, crossing no walls. | Orthographic readable geometry, distinct requested room colors, wall/door consistency, arrow direction, no extra rooms or perspective. Do not count entrance aisle as a sixth room. |
27
+ | `expanded-ridgeline-campaign`, 21077 | Exactly three products: central cobalt bottle on pedestal, yellow cup left, turquoise cup right. Bottle has black cap, one narrow orange stripe and a small white mountain-triangle emblem near its base. All three exact text lines below correct. | Separate fully visible objects, coherent cap/bottle form, plausible contact shadows, stable front three-quarter view, clear type hierarchy and negative space. | Material rendering, stripe/emblem clarity, usable reference product identity, consistent studio lighting. Additional props/products or illegible bottom text are explicit failures. |
28
+ | `expanded-newsroom-interview`, 21088 | Exactly four adults: ponytailed woman journalist left with notebook and gesture; bearded male guest right; blue-sweater engineer at console behind glass; standing green-shirt producer beside engineer holding `2 MIN`. Red sign reads `ON AIR`. Exactly two desk microphones, one aimed at each seated speaker. | Journalist/guest look at each other; engineer's hand reaches a fader; producer faces studio. Coherent microphone booms/cables, notebook grip, cue-card grip, glass and reflections without duplicate people. | Studio realism, faces/hands, desk objects, seated/staff depth separation and readable signs. |
29
+
30
+ ### Exact text checks
31
+
32
+ For each string, record correct / incorrect / absent. Record wrong characters or punctuation explicitly. Wrapping and harmless whitespace changes may be noted separately from actual spelling errors; do not silently normalize wording, numeral changes or Chinese glyph substitutions.
33
+
34
+ **Bilingual festival: 14 strings**
35
+
36
+ | Position | Exact text |
37
+ |---|---|
38
+ | English headline | `RIVERLIGHT FILM WEEK` |
39
+ | Chinese headline | `河畔电影周` |
40
+ | Date | `18–20 SEPTEMBER 2026` |
41
+ | Row 1 time | `18:00` |
42
+ | Row 1 English | `A Quiet River` |
43
+ | Row 1 Chinese | `静静的河` |
44
+ | Row 2 time | `19:30` |
45
+ | Row 2 English | `Night Market` |
46
+ | Row 2 Chinese | `夜市` |
47
+ | Row 3 time | `21:00` |
48
+ | Row 3 English | `Home Again` |
49
+ | Row 3 Chinese | `再次回家` |
50
+ | Venue footer | `RIVERSIDE CINEMA / 河畔影院` |
51
+ | Admission footer | `FREE ENTRY / 免费入场` |
52
+
53
+ **Comic: six bubbles, attribution matters**
54
+
55
+ | Panel | Maya | Leo |
56
+ |---|---|---|
57
+ | 1 | `Where are my keys?` | `Check your pocket.` |
58
+ | 2 | `Not here!` | `Under the chair.` |
59
+ | 3 | `Found them. Thanks!` | `You're welcome.` |
60
+
61
+ **Museum: eight strings:** `MUSEUM VISITOR MAP`, `GALLERY A`, `GALLERY B`, `MAIN HALL`, `CAFE`, `SHOP`, `ENTRANCE`, `SUGGESTED ROUTE`.
62
+
63
+ **Product campaign: three strings:** `TAKE THE LONG WAY`, `RIDGELINE / EVERYDAY ADVENTURE`, `750 mL · BUILT TO REUSE`.
64
+
65
+ ## Editing scenarios
66
+
67
+ | Scenario and seed | Fixed teacher reference(s) | Required change | Protected evidence / failure modes |
68
+ |---|---|---|---|
69
+ | `expanded-station-coat-edit`, 21101 | `expanded-station-reunion-40.png` | Only mother's navy coat becomes burgundy, preserving cut, length, folds and fabric. | Mother identity/face/gaze/pose; yellow scarf; exact handholding; all four people and their hands/clothing; father/grandmother embrace; rabbit; suitcase; train/platform/crop/light. Report recoloring spill, reconstructed fingers/faces, changed pose, unintended wardrobe changes and background redraw. Use original reference as the preservation baseline. |
70
+ | `expanded-festival-type-edit`, 21112 | `expanded-bilingual-festival-40.png` | Headline becomes `RIVERLIGHT FILM NIGHTS`; date becomes `25–27 SEPTEMBER 2026`. Headline may resize sensibly within its original area. | Chinese headline, all nine program strings, both footer lines and their alignment; illustration, rules, margins, background/colors. Check every original string that was actually legible. Fail undesired translation, missing rows, font/style redesign or collateral Chinese corruption separately from the two requested replacements. |
71
+ | `expanded-multiref-portrait`, 21123 | First `expanded-station-reunion-40.png`; second `expanded-ridgeline-campaign-40.png` | One seated mother from reference 1 holds one bottle from reference 2 in both hands in a waist-up railway-platform portrait. She smiles toward the camera; one green suitcase is partly visible by the bench. | Recognizable mother face, low bun, gold earrings, navy coat/yellow scarf; recognizable bottle shape, black cap, orange stripe/emblem. New pose/background are intentional. Hands must grip plausibly without hiding cap/stripe/emblem. Check coherent scale/light, exactly one person/bottle, no imported other people/cups/pedestal/poster text. Judge identity and product fidelity separately; whole-frame pixel similarity is not the target here. |
72
+ | `expanded-product-relocation-edit`, 21134 | `expanded-ridgeline-campaign-40.png` | Remove yellow cup; relocate turquoise cup from right to far left of pedestal with a visible gap. Final products: unchanged bottle plus one turquoise cup, exactly two. Former cup locations should be empty/naturally restored. | Bottle location/scale/shape/cap/stripe/emblem, pedestal, all three text strings, crop/light/background. Check no ghost cup, extra handle, duplicate cup, right-side residual object or typography redraw. Moved cup must contact tabletop with consistent shadow. |
73
+
74
+ ## Review record
75
+
76
+ Use one row per backend/scenario/step count. Add per-string or per-role details when the summary would hide a failure.
77
+
78
+ | Backend/checkpoint | Scenario / steps | Adherence findings | Anatomical / physical defects | Drift from teacher | Edit protected-region findings | Source-limited criteria | Overall evidence, no unsupported ranking |
79
+ |---|---|---|---|---|---|---|---|
80
+ | Pending | — | — | — | — | — | — | — |
81
+
82
+ Record visually meaningful candidate improvements as well as regressions. Keep text accuracy, people interactions, diagram logic, editing preservation and image fidelity distinct in the final report. A count of passed criteria may summarize this finite set, but cannot establish general model quality or statistical significance.
83
+
84
+ To measure image fidelity after matching teacher/candidate images exist, pass this manifest explicitly to the CPU comparison tool:
85
+
86
+ ```sh
87
+ /tmp/qwen-api-test-py311/bin/python scripts/compare_fidelity_v3.py \
88
+ --jobs experiments/fidelity-v3/expanded-jobs.json \
89
+ --output results/compare-fidelity-v3-expanded.json
90
+ ```
91
+
92
+ The prior v2 folders have no outputs for these new labels and should remain explicitly missing. For a direct new-candidate comparison, specify only actual candidate folders with repeated `--candidate NAME=FOLDER`; do not mix different available-case cohorts into a ranking. No measurements or image-quality claims have been made in this rubric.
reproduction/experiments/fidelity-v3/original-18-jobs.json ADDED
@@ -0,0 +1,164 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "editorial-25",
4
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
5
+ "steps": 25,
6
+ "seed": 118,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "editorial-40",
12
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
13
+ "steps": 40,
14
+ "seed": 118,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "rainy-city-40",
20
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
21
+ "steps": 40,
22
+ "seed": 227,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "rainy-city-25",
28
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
29
+ "steps": 25,
30
+ "seed": 227,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "botanical-poster-25",
36
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
37
+ "steps": 25,
38
+ "seed": 336,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "botanical-poster-40",
44
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
45
+ "steps": 40,
46
+ "seed": 336,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "food-spatial-40",
52
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
53
+ "steps": 40,
54
+ "seed": 445,
55
+ "width": 1024,
56
+ "height": 1024
57
+ },
58
+ {
59
+ "label": "food-spatial-25",
60
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
61
+ "steps": 25,
62
+ "seed": 445,
63
+ "width": 1024,
64
+ "height": 1024
65
+ },
66
+ {
67
+ "label": "fantasy-cutaway-25",
68
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
69
+ "steps": 25,
70
+ "seed": 554,
71
+ "width": 1024,
72
+ "height": 1024
73
+ },
74
+ {
75
+ "label": "fantasy-cutaway-40",
76
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
77
+ "steps": 40,
78
+ "seed": 554,
79
+ "width": 1024,
80
+ "height": 1024
81
+ },
82
+ {
83
+ "label": "glass-product-40",
84
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
85
+ "steps": 40,
86
+ "seed": 663,
87
+ "width": 1024,
88
+ "height": 1024
89
+ },
90
+ {
91
+ "label": "glass-product-25",
92
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
93
+ "steps": 25,
94
+ "seed": 663,
95
+ "width": 1024,
96
+ "height": 1024
97
+ },
98
+ {
99
+ "label": "editorial-edit-25",
100
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
101
+ "steps": 25,
102
+ "seed": 774,
103
+ "width": 1024,
104
+ "height": 1024,
105
+ "images": [
106
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
107
+ ]
108
+ },
109
+ {
110
+ "label": "editorial-edit-40",
111
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
112
+ "steps": 40,
113
+ "seed": 774,
114
+ "width": 1024,
115
+ "height": 1024,
116
+ "images": [
117
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
118
+ ]
119
+ },
120
+ {
121
+ "label": "poster-edit-25",
122
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
123
+ "steps": 25,
124
+ "seed": 885,
125
+ "width": 1024,
126
+ "height": 1024,
127
+ "images": [
128
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
129
+ ]
130
+ },
131
+ {
132
+ "label": "poster-edit-40",
133
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
134
+ "steps": 40,
135
+ "seed": 885,
136
+ "width": 1024,
137
+ "height": 1024,
138
+ "images": [
139
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
140
+ ]
141
+ },
142
+ {
143
+ "label": "city-edit-25",
144
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
145
+ "steps": 25,
146
+ "seed": 996,
147
+ "width": 1024,
148
+ "height": 1024,
149
+ "images": [
150
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
151
+ ]
152
+ },
153
+ {
154
+ "label": "city-edit-40",
155
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
156
+ "steps": 40,
157
+ "seed": 996,
158
+ "width": 1024,
159
+ "height": 1024,
160
+ "images": [
161
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
162
+ ]
163
+ }
164
+ ]
reproduction/experiments/fidelity-v3/performance.md ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Fidelity experiment performance
2
+
3
+ Serial engine inference, excludes image-file loading/saving and process startup; no prompt/reference LRU reuse. Per-request prefix KV remains enabled.
4
+
5
+ Original BF16 DiT; NF4 text/vision encoder and BF16 untiled VAE fixed across candidates.
6
+
7
+ | Backend | Reference images | Steps | Images | Median inference (s) | Median transformer (s) | Max allocated (MiB) | Sampled board max (MiB) |
8
+ |---|---:|---:|---:|---:|---:|---:|---:|
9
+ | bf16-dit | 0 | 25 | 14 | 27.51 | 25.31 | 7387 | 9346 |
10
+ | bf16-dit | 0 | 40 | 14 | 42.76 | 40.54 | 7387 | 9348 |
11
+ | bf16-dit | 1 | 25 | 6 | 31.98 | 29.01 | 7391 | 12030 |
12
+ | bf16-dit | 1 | 40 | 6 | 48.70 | 45.74 | 7391 | 12030 |
13
+ | bf16-dit | 2 | 25 | 1 | 36.79 | 33.14 | 7398 | 15120 |
14
+ | bf16-dit | 2 | 40 | 1 | 55.28 | 51.62 | 7400 | 14826 |
15
+ | nunchaku-v3-mlpproj512 | 0 | 25 | 14 | 16.06 | 13.21 | 7388 | 9344 |
16
+ | nunchaku-v3-mlpproj512 | 0 | 40 | 14 | 23.56 | 20.68 | 7388 | 9346 |
17
+ | nunchaku-v3-mlpproj512 | 1 | 25 | 6 | 20.02 | 16.49 | 8242 | 9488 |
18
+ | nunchaku-v3-mlpproj512 | 1 | 40 | 6 | 29.23 | 25.72 | 8242 | 9488 |
19
+ | nunchaku-v3-mlpproj512 | 2 | 25 | 1 | 24.41 | 20.08 | 10911 | 12374 |
20
+ | nunchaku-v3-mlpproj512 | 2 | 40 | 1 | 35.35 | 31.04 | 10911 | 12374 |
21
+ | nunchaku-v3-r128 | 0 | 25 | 14 | 15.50 | 12.71 | 7388 | 9344 |
22
+ | nunchaku-v3-r128 | 0 | 40 | 14 | 22.82 | 19.93 | 7388 | 9629 |
23
+ | nunchaku-v3-r128 | 1 | 25 | 6 | 19.48 | 16.04 | 7862 | 9346 |
24
+ | nunchaku-v3-r128 | 1 | 40 | 6 | 28.43 | 24.92 | 7862 | 9344 |
25
+ | nunchaku-v3-r128 | 2 | 25 | 1 | 23.76 | 19.54 | 10531 | 11994 |
26
+ | nunchaku-v3-r128 | 2 | 40 | 1 | 34.35 | 30.19 | 10531 | 11994 |
27
+
28
+ Complete paired run: True. See the JSON report for missing jobs, validation errors, exact settings, timestamps and image/reference hashes.
29
+
30
+ These timings describe this fixed 1024-pixel suite on physical GPU 0. The original v2 timings used different LRU-cache states and must not be compared as controlled end-to-end speed measurements. Model-state byte costs do not substitute for measured peak VRAM.
reproduction/experiments/fidelity-v3/quality-jobs-all.json ADDED
@@ -0,0 +1,382 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "editorial-25",
4
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
5
+ "steps": 25,
6
+ "seed": 118,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "editorial-40",
12
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
13
+ "steps": 40,
14
+ "seed": 118,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "rainy-city-40",
20
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
21
+ "steps": 40,
22
+ "seed": 227,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "rainy-city-25",
28
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
29
+ "steps": 25,
30
+ "seed": 227,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "botanical-poster-25",
36
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
37
+ "steps": 25,
38
+ "seed": 336,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "botanical-poster-40",
44
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
45
+ "steps": 40,
46
+ "seed": 336,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "food-spatial-40",
52
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
53
+ "steps": 40,
54
+ "seed": 445,
55
+ "width": 1024,
56
+ "height": 1024
57
+ },
58
+ {
59
+ "label": "food-spatial-25",
60
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
61
+ "steps": 25,
62
+ "seed": 445,
63
+ "width": 1024,
64
+ "height": 1024
65
+ },
66
+ {
67
+ "label": "fantasy-cutaway-25",
68
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
69
+ "steps": 25,
70
+ "seed": 554,
71
+ "width": 1024,
72
+ "height": 1024
73
+ },
74
+ {
75
+ "label": "fantasy-cutaway-40",
76
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
77
+ "steps": 40,
78
+ "seed": 554,
79
+ "width": 1024,
80
+ "height": 1024
81
+ },
82
+ {
83
+ "label": "glass-product-40",
84
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
85
+ "steps": 40,
86
+ "seed": 663,
87
+ "width": 1024,
88
+ "height": 1024
89
+ },
90
+ {
91
+ "label": "glass-product-25",
92
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
93
+ "steps": 25,
94
+ "seed": 663,
95
+ "width": 1024,
96
+ "height": 1024
97
+ },
98
+ {
99
+ "label": "editorial-edit-25",
100
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
101
+ "steps": 25,
102
+ "seed": 774,
103
+ "width": 1024,
104
+ "height": 1024,
105
+ "images": [
106
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
107
+ ]
108
+ },
109
+ {
110
+ "label": "editorial-edit-40",
111
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
112
+ "steps": 40,
113
+ "seed": 774,
114
+ "width": 1024,
115
+ "height": 1024,
116
+ "images": [
117
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
118
+ ]
119
+ },
120
+ {
121
+ "label": "poster-edit-25",
122
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
123
+ "steps": 25,
124
+ "seed": 885,
125
+ "width": 1024,
126
+ "height": 1024,
127
+ "images": [
128
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
129
+ ]
130
+ },
131
+ {
132
+ "label": "poster-edit-40",
133
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
134
+ "steps": 40,
135
+ "seed": 885,
136
+ "width": 1024,
137
+ "height": 1024,
138
+ "images": [
139
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
140
+ ]
141
+ },
142
+ {
143
+ "label": "city-edit-25",
144
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
145
+ "steps": 25,
146
+ "seed": 996,
147
+ "width": 1024,
148
+ "height": 1024,
149
+ "images": [
150
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
151
+ ]
152
+ },
153
+ {
154
+ "label": "city-edit-40",
155
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
156
+ "steps": 40,
157
+ "seed": 996,
158
+ "width": 1024,
159
+ "height": 1024,
160
+ "images": [
161
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
162
+ ]
163
+ },
164
+ {
165
+ "label": "expanded-community-kitchen-25",
166
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
167
+ "steps": 25,
168
+ "seed": 21011,
169
+ "width": 1024,
170
+ "height": 1024
171
+ },
172
+ {
173
+ "label": "expanded-community-kitchen-40",
174
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
175
+ "steps": 40,
176
+ "seed": 21011,
177
+ "width": 1024,
178
+ "height": 1024
179
+ },
180
+ {
181
+ "label": "expanded-chess-awards-25",
182
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
183
+ "steps": 25,
184
+ "seed": 21022,
185
+ "width": 1024,
186
+ "height": 1024
187
+ },
188
+ {
189
+ "label": "expanded-chess-awards-40",
190
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
191
+ "steps": 40,
192
+ "seed": 21022,
193
+ "width": 1024,
194
+ "height": 1024
195
+ },
196
+ {
197
+ "label": "expanded-station-reunion-25",
198
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
199
+ "steps": 25,
200
+ "seed": 21033,
201
+ "width": 1024,
202
+ "height": 1024
203
+ },
204
+ {
205
+ "label": "expanded-station-reunion-40",
206
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
207
+ "steps": 40,
208
+ "seed": 21033,
209
+ "width": 1024,
210
+ "height": 1024
211
+ },
212
+ {
213
+ "label": "expanded-bilingual-festival-25",
214
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
215
+ "steps": 25,
216
+ "seed": 21044,
217
+ "width": 1024,
218
+ "height": 1024
219
+ },
220
+ {
221
+ "label": "expanded-bilingual-festival-40",
222
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
223
+ "steps": 40,
224
+ "seed": 21044,
225
+ "width": 1024,
226
+ "height": 1024
227
+ },
228
+ {
229
+ "label": "expanded-library-comic-25",
230
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
231
+ "steps": 25,
232
+ "seed": 21055,
233
+ "width": 1024,
234
+ "height": 1024
235
+ },
236
+ {
237
+ "label": "expanded-library-comic-40",
238
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
239
+ "steps": 40,
240
+ "seed": 21055,
241
+ "width": 1024,
242
+ "height": 1024
243
+ },
244
+ {
245
+ "label": "expanded-museum-plan-25",
246
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
247
+ "steps": 25,
248
+ "seed": 21066,
249
+ "width": 1024,
250
+ "height": 1024
251
+ },
252
+ {
253
+ "label": "expanded-museum-plan-40",
254
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
255
+ "steps": 40,
256
+ "seed": 21066,
257
+ "width": 1024,
258
+ "height": 1024
259
+ },
260
+ {
261
+ "label": "expanded-ridgeline-campaign-25",
262
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
263
+ "steps": 25,
264
+ "seed": 21077,
265
+ "width": 1024,
266
+ "height": 1024
267
+ },
268
+ {
269
+ "label": "expanded-ridgeline-campaign-40",
270
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
271
+ "steps": 40,
272
+ "seed": 21077,
273
+ "width": 1024,
274
+ "height": 1024
275
+ },
276
+ {
277
+ "label": "expanded-newsroom-interview-25",
278
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
279
+ "steps": 25,
280
+ "seed": 21088,
281
+ "width": 1024,
282
+ "height": 1024
283
+ },
284
+ {
285
+ "label": "expanded-newsroom-interview-40",
286
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
287
+ "steps": 40,
288
+ "seed": 21088,
289
+ "width": 1024,
290
+ "height": 1024
291
+ },
292
+ {
293
+ "label": "expanded-station-coat-edit-25",
294
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
295
+ "steps": 25,
296
+ "seed": 21101,
297
+ "width": 1024,
298
+ "height": 1024,
299
+ "images": [
300
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
301
+ ]
302
+ },
303
+ {
304
+ "label": "expanded-station-coat-edit-40",
305
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
306
+ "steps": 40,
307
+ "seed": 21101,
308
+ "width": 1024,
309
+ "height": 1024,
310
+ "images": [
311
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
312
+ ]
313
+ },
314
+ {
315
+ "label": "expanded-festival-type-edit-25",
316
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
317
+ "steps": 25,
318
+ "seed": 21112,
319
+ "width": 1024,
320
+ "height": 1024,
321
+ "images": [
322
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
323
+ ]
324
+ },
325
+ {
326
+ "label": "expanded-festival-type-edit-40",
327
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
328
+ "steps": 40,
329
+ "seed": 21112,
330
+ "width": 1024,
331
+ "height": 1024,
332
+ "images": [
333
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
334
+ ]
335
+ },
336
+ {
337
+ "label": "expanded-multiref-portrait-25",
338
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
339
+ "steps": 25,
340
+ "seed": 21123,
341
+ "width": 1024,
342
+ "height": 1024,
343
+ "images": [
344
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
345
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
346
+ ]
347
+ },
348
+ {
349
+ "label": "expanded-multiref-portrait-40",
350
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
351
+ "steps": 40,
352
+ "seed": 21123,
353
+ "width": 1024,
354
+ "height": 1024,
355
+ "images": [
356
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
357
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
358
+ ]
359
+ },
360
+ {
361
+ "label": "expanded-product-relocation-edit-25",
362
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
363
+ "steps": 25,
364
+ "seed": 21134,
365
+ "width": 1024,
366
+ "height": 1024,
367
+ "images": [
368
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
369
+ ]
370
+ },
371
+ {
372
+ "label": "expanded-product-relocation-edit-40",
373
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
374
+ "steps": 40,
375
+ "seed": 21134,
376
+ "width": 1024,
377
+ "height": 1024,
378
+ "images": [
379
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
380
+ ]
381
+ }
382
+ ]
reproduction/experiments/fidelity-v3/quality-jobs.json ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "editorial-40",
4
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
5
+ "steps": 40,
6
+ "seed": 118,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "rainy-city-40",
12
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
13
+ "steps": 40,
14
+ "seed": 227,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "botanical-poster-40",
20
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
21
+ "steps": 40,
22
+ "seed": 336,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "food-spatial-40",
28
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
29
+ "steps": 40,
30
+ "seed": 445,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "fantasy-cutaway-40",
36
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
37
+ "steps": 40,
38
+ "seed": 554,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "glass-product-40",
44
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
45
+ "steps": 40,
46
+ "seed": 663,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "editorial-edit-40",
52
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
53
+ "steps": 40,
54
+ "seed": 774,
55
+ "width": 1024,
56
+ "height": 1024,
57
+ "images": [
58
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
59
+ ]
60
+ },
61
+ {
62
+ "label": "poster-edit-40",
63
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
64
+ "steps": 40,
65
+ "seed": 885,
66
+ "width": 1024,
67
+ "height": 1024,
68
+ "images": [
69
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
70
+ ]
71
+ },
72
+ {
73
+ "label": "city-edit-40",
74
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
75
+ "steps": 40,
76
+ "seed": 996,
77
+ "width": 1024,
78
+ "height": 1024,
79
+ "images": [
80
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
81
+ ]
82
+ }
83
+ ]
reproduction/experiments/fidelity-v3/selection.json ADDED
@@ -0,0 +1,15 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "selected": "nunchaku-v3-r128",
3
+ "checkpoint": "/cache/qwen-nunchaku-v3-r128",
4
+ "manifest_sha256": "cf69e83a646981d22ee6df8d5239b46a50df25d8eb73c9f0478feae87323e6cb",
5
+ "compose_override": "compose.nunchaku-v3.yaml",
6
+ "decision": "Retain calibrated rank128; selective MLP rank512 is experimental, not promoted.",
7
+ "reason": "Selective512 improves numerical prediction error but has new free-running image count and geometry regressions. Rank128 is faster and smaller and already has a complete42-image visual review.",
8
+ "counterexamples": [
9
+ "food-spatial-40: seven blueberries and two knives",
10
+ "fantasy-cutaway-25: two cats",
11
+ "fantasy-cutaway-40: distorted telescope",
12
+ "expanded-community-kitchen-25: six bread rolls"
13
+ ],
14
+ "limitations": "Neither checkpoint is established near-lossless or uniformly superior; serving verification recorded separately."
15
+ }
reproduction/experiments/fidelity-v3/teacher-priority-bf16.json ADDED
@@ -0,0 +1,26 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "food-spatial-40",
4
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
5
+ "steps": 40,
6
+ "seed": 445,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "fantasy-cutaway-40",
12
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
13
+ "steps": 40,
14
+ "seed": 554,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "editorial-40",
20
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
21
+ "steps": 40,
22
+ "seed": 118,
23
+ "width": 1024,
24
+ "height": 1024
25
+ }
26
+ ]
reproduction/experiments/fidelity-v3/teacher-priority.json ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "food-spatial-40",
4
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
5
+ "steps": 40,
6
+ "seed": 445,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "fantasy-cutaway-40",
12
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
13
+ "steps": 40,
14
+ "seed": 554,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "editorial-40",
20
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
21
+ "steps": 40,
22
+ "seed": 118,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "expanded-community-kitchen-40",
28
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
29
+ "steps": 40,
30
+ "seed": 21011,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "expanded-chess-awards-40",
36
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
37
+ "steps": 40,
38
+ "seed": 21022,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "expanded-bilingual-festival-40",
44
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
45
+ "steps": 40,
46
+ "seed": 21044,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "expanded-library-comic-40",
52
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
53
+ "steps": 40,
54
+ "seed": 21055,
55
+ "width": 1024,
56
+ "height": 1024
57
+ }
58
+ ]
reproduction/experiments/fidelity-v3/teacher-remaining.json ADDED
@@ -0,0 +1,358 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ [
2
+ {
3
+ "label": "expanded-community-kitchen-40",
4
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
5
+ "steps": 40,
6
+ "seed": 21011,
7
+ "width": 1024,
8
+ "height": 1024
9
+ },
10
+ {
11
+ "label": "expanded-chess-awards-40",
12
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
13
+ "steps": 40,
14
+ "seed": 21022,
15
+ "width": 1024,
16
+ "height": 1024
17
+ },
18
+ {
19
+ "label": "expanded-station-reunion-40",
20
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
21
+ "steps": 40,
22
+ "seed": 21033,
23
+ "width": 1024,
24
+ "height": 1024
25
+ },
26
+ {
27
+ "label": "expanded-bilingual-festival-40",
28
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
29
+ "steps": 40,
30
+ "seed": 21044,
31
+ "width": 1024,
32
+ "height": 1024
33
+ },
34
+ {
35
+ "label": "expanded-library-comic-40",
36
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
37
+ "steps": 40,
38
+ "seed": 21055,
39
+ "width": 1024,
40
+ "height": 1024
41
+ },
42
+ {
43
+ "label": "expanded-museum-plan-40",
44
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
45
+ "steps": 40,
46
+ "seed": 21066,
47
+ "width": 1024,
48
+ "height": 1024
49
+ },
50
+ {
51
+ "label": "expanded-ridgeline-campaign-40",
52
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
53
+ "steps": 40,
54
+ "seed": 21077,
55
+ "width": 1024,
56
+ "height": 1024
57
+ },
58
+ {
59
+ "label": "expanded-newsroom-interview-40",
60
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
61
+ "steps": 40,
62
+ "seed": 21088,
63
+ "width": 1024,
64
+ "height": 1024
65
+ },
66
+ {
67
+ "label": "expanded-community-kitchen-25",
68
+ "prompt": "Documentary photograph for a community cooking program, wide eye-level composition in a bright, practical teaching kitchen. Exactly six adults, all visible from at least the waist up. At the left counter, a woman with short black hair in a green apron chops carrots on a wooden board, with one hand holding the knife and the other safely curled on the carrot. At center, a silver-haired man in a white apron holds a ladle over a large soup pot. Immediately to his right, a woman in a mustard cardigan passes a rectangular tray of four bread rolls to a younger man in a navy shirt, who receives it with both hands; their hands meet opposite edges of the same tray. At the far right, a man in a red apron rinses a colander of leafy greens at the sink. Behind the central counter, a woman wearing glasses and a purple sweater reads a paper recipe card. Natural varied expressions, clear individual faces, plausible arms and hands attached to their owners, no extra people, realistic morning window light and stainless-steel reflections. No readable signage. The image should feel like a real coordinated cooking session, not a posed group portrait.",
69
+ "steps": 25,
70
+ "seed": 21011,
71
+ "width": 1024,
72
+ "height": 1024
73
+ },
74
+ {
75
+ "label": "expanded-chess-awards-25",
76
+ "prompt": "Editorial event photograph of a small chess championship award presentation on a low indoor stage. Exactly five adults. At center, a smiling woman with a short dark bob sits in a clearly recognizable manual wheelchair; she and the presenter standing immediately to her left jointly hold one gold trophy between them. The presenter is an older man wearing a charcoal suit. Immediately to the winner's right, the runner-up, a young man in a cream sweater, wears one silver medal and applauds. At the far left, a female photographer kneels with a camera held to her eye and aims toward the winner. At the far right, a host in a burgundy suit holds a microphone and a small cue card. All five people fit fully in frame, with believable hands, wheelchair wheels and footrests. Behind them, a clean white stage banner reads exactly \"CITY CHESS FINAL\" with \"2026\" centered on the line below. A small chess table with a board stands at the rear left. Warm indoor event lighting, realistic fabric and skin, no audience or additional figures.",
77
+ "steps": 25,
78
+ "seed": 21022,
79
+ "width": 1024,
80
+ "height": 1024
81
+ },
82
+ {
83
+ "label": "expanded-station-reunion-25",
84
+ "prompt": "Natural editorial photograph of a family reunion beside a stationary passenger train on a quiet outdoor railway platform. Exactly four people, full bodies in frame, no other passengers. On the left is a woman in her thirties with an oval face, dark curly hair tied in a low bun, small round gold earrings, a navy knee-length coat and a mustard yellow scarf. She holds the left hand of a young girl standing at center; the girl wears a red raincoat and carries a small stuffed rabbit in her free hand. On the right, a man with close-cropped brown hair, a light gray jacket and a red backpack bends slightly to embrace an older gray-haired woman wearing a teal coat; the older woman faces him with one arm around his shoulder. The mother and girl watch them with happy, understated expressions. One upright green suitcase stands beside the mother's left leg. Anatomically plausible hands and embraces with clear ownership of each arm, natural eye contact, soft overcast daylight, authentic train windows and platform paving. Do not crop feet or add duplicate people. No readable logos or signs.",
85
+ "steps": 25,
86
+ "seed": 21033,
87
+ "width": 1024,
88
+ "height": 1024
89
+ },
90
+ {
91
+ "label": "expanded-bilingual-festival-25",
92
+ "prompt": "A polished bilingual neighborhood film festival poster, square format, flat front-on print artwork with generous margins, warm cream paper and deep navy typography accented by vermilion. The large centered English title at the top reads exactly \"RIVERLIGHT FILM WEEK\". Directly beneath it, an equally clear Chinese title reads exactly \"河畔电影周\". The next centered line reads exactly \"18–20 SEPTEMBER 2026\". Below a thin vermilion rule, a tidy two-column program occupies the middle: the left column contains times, and the right column contains one English film title with its Chinese title directly underneath. Row one: \"18:00\", \"A Quiet River\", \"静静的河\". Row two: \"19:30\", \"Night Market\", \"夜市\". Row three: \"21:00\", \"Home Again\", \"再次回家\". Maintain aligned columns and clearly separated rows. At the bottom, centered on separate lines, print exactly \"RIVERSIDE CINEMA / 河畔影院\" and \"FREE ENTRY / 免费入场\". A small restrained illustration of a red bridge over two blue river lines sits between the program and footer without touching any letters. Sharp readable Latin and Chinese lettering, consistent type hierarchy, no additional text, no mockup perspective.",
93
+ "steps": 25,
94
+ "seed": 21044,
95
+ "width": 1024,
96
+ "height": 1024
97
+ },
98
+ {
99
+ "label": "expanded-library-comic-25",
100
+ "prompt": "A polished, friendly three-panel comic strip in a square canvas, with exactly three equal vertical panels arranged left to right, clear white gutters and a thin dark border around each panel. Keep the same two adult characters and library reading-room setting in all panels: Maya has a chin-length black bob, round glasses and a red sweater; Leo has short curly brown hair and a blue button-up shirt. Each panel contains exactly two white speech bubbles, with clear tails pointing to the correct speaker. Panel 1: Maya stands beside a wooden reading chair, looks worried and searches her bag. Maya says exactly \"Where are my keys?\" Leo, standing to her right, says exactly \"Check your pocket.\" Panel 2: Maya turns out an empty coat pocket and says exactly \"Not here!\" Leo points down at a small key ring visible under that same chair and says exactly \"Under the chair.\" Panel 3: Maya stands upright holding that key ring up, smiles and says exactly \"Found them. Thanks!\" Leo smiles and says exactly \"You're welcome.\" Crisp ink lines, restrained warm colors, expressive natural poses, consistent faces and clothing, believable hands, a coherent sequence with the chair in the same room. Readable dialogue with no narration captions and no additional panels or characters.",
101
+ "steps": 25,
102
+ "seed": 21055,
103
+ "width": 1024,
104
+ "height": 1024
105
+ },
106
+ {
107
+ "label": "expanded-museum-plan-25",
108
+ "prompt": "A clear visitor floor-plan diagram for a small museum, square format, flat orthographic top-down vector design on white, with thick charcoal exterior walls, thinner interior walls and clearly visible doorway gaps. Title centered above the plan: \"MUSEUM VISITOR MAP\". Inside one rectangular building, arrange five labeled spaces: \"GALLERY A\" in a large upper-left room, \"GALLERY B\" in a large upper-right room, \"MAIN HALL\" in a wide horizontal central space spanning the building, \"CAFE\" in the lower-left room, and \"SHOP\" in the lower-right room. A narrow unlabeled central entrance aisle runs from a doorway at the bottom edge into the main hall, separating the cafe and shop. Label that bottom doorway \"ENTRANCE\" just outside the building. Each of the four corner rooms has one doorway directly into the main hall. Give Gallery A a pale blue fill, Gallery B pale green, the cafe pale orange and the shop pale violet; keep the hall and entrance aisle white. Put exactly two simple rectangular display plinth symbols in each gallery, exactly three round table symbols in the cafe, and exactly two parallel shelf symbols in the shop. A single continuous blue visitor-route line with arrowheads begins outside the entrance, goes through the entrance aisle and main hall, and ends inside Gallery A, passing through actual doorway openings without crossing any walls. Small bottom legend: a blue arrow icon followed by exactly \"SUGGESTED ROUTE\". Spacious legible labeling, precise geometry, no perspective, no people and no extra rooms.",
109
+ "steps": 25,
110
+ "seed": 21066,
111
+ "width": 1024,
112
+ "height": 1024
113
+ },
114
+ {
115
+ "label": "expanded-ridgeline-campaign-25",
116
+ "prompt": "Premium studio product campaign for an insulated water bottle, square print advertisement. Center one tall matte cobalt-blue metal bottle on a low pale stone rectangular pedestal. The bottle has a black screw cap, one narrow orange vertical stripe on its front, and a small white mountain-triangle emblem near its base. On the tabletop to the left of the pedestal is one short yellow enamel cup; to the right is one short turquoise enamel cup. There are exactly three products in total: the bottle and the two cups, all fully visible and separate. Clean warm gray seamless backdrop, soft directional daylight from the upper left, realistic brushed metal and enamel, restrained shadows. At the top, large elegant dark navy sans-serif lettering reads exactly \"TAKE THE LONG WAY\". Directly underneath it, smaller lettering reads exactly \"RIDGELINE / EVERYDAY ADVENTURE\". At the bottom center, small text reads exactly \"750 mL · BUILT TO REUSE\". Plenty of negative space around the type, crisp product edges, no additional props, hands or extra words. Front three-quarter product view with the orange stripe and mountain emblem clearly visible.",
117
+ "steps": 25,
118
+ "seed": 21077,
119
+ "width": 1024,
120
+ "height": 1024
121
+ },
122
+ {
123
+ "label": "expanded-newsroom-interview-25",
124
+ "prompt": "Behind-the-scenes editorial photograph of a radio newsroom recording an interview, wide eye-level square composition. Exactly four adults with clearly separated faces and bodies. On the left, a woman journalist with a dark ponytail, white shirt and black headphones sits at a small table and holds an open notebook in her left hand while gesturing toward the guest with her right. On the right, a middle-aged man with a trimmed beard, brown jacket and silver headphones sits opposite her, speaking toward one black microphone on an articulated desk boom; a second matching microphone points toward the journalist. Behind a glass partition at center, a sound engineer in a blue sweater sits at a mixing console with one hand on a fader. Next to the engineer, a producer wearing a green shirt stands holding up a white card that reads exactly \"2 MIN\". Above the studio window, a red illuminated sign reads exactly \"ON AIR\". The two seated people look at each other, the engineer watches the controls, and the producer faces the studio. Plausible microphone arms and cables, convincing glass with subtle reflections that do not duplicate faces, believable fingers, soft warm studio lighting, documentary realism and no other people.",
125
+ "steps": 25,
126
+ "seed": 21088,
127
+ "width": 1024,
128
+ "height": 1024
129
+ },
130
+ {
131
+ "label": "expanded-station-coat-edit-25",
132
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
133
+ "steps": 25,
134
+ "seed": 21101,
135
+ "width": 1024,
136
+ "height": 1024,
137
+ "images": [
138
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
139
+ ]
140
+ },
141
+ {
142
+ "label": "expanded-station-coat-edit-40",
143
+ "prompt": "Change only the mother's navy knee-length coat to a rich burgundy red fabric of the same cut, length, folds and texture. She is the woman on the left with dark curly hair in a low bun, gold earrings and a mustard yellow scarf. Preserve her exact face and identity, expression, gaze, body pose and handholding with the girl. Preserve the yellow scarf, all other clothing, the girl and stuffed rabbit, the father and grandmother's embrace, every person's hands and face, the green suitcase, train, platform, lighting, crop and photographic style. This is a localized garment recoloring, with the rest of the photograph retained.",
144
+ "steps": 40,
145
+ "seed": 21101,
146
+ "width": 1024,
147
+ "height": 1024,
148
+ "images": [
149
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png"
150
+ ]
151
+ },
152
+ {
153
+ "label": "expanded-festival-type-edit-25",
154
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
155
+ "steps": 25,
156
+ "seed": 21112,
157
+ "width": 1024,
158
+ "height": 1024,
159
+ "images": [
160
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
161
+ ]
162
+ },
163
+ {
164
+ "label": "expanded-festival-type-edit-40",
165
+ "prompt": "Update only two text lines in this festival poster. Replace the top English headline \"RIVERLIGHT FILM WEEK\" with exactly \"RIVERLIGHT FILM NIGHTS\", keeping it centered in the same headline area with the same navy font style and a sensible size adjustment if needed. Replace the date line \"18–20 SEPTEMBER 2026\" with exactly \"25–27 SEPTEMBER 2026\". Preserve the Chinese headline \"河畔电影周\" exactly, every program time and English/Chinese film title, both footer lines, the two-column alignment, row spacing, bridge illustration, rules, colors, margins and paper background. Do not translate or reword any other text, and do not redesign the poster.",
166
+ "steps": 40,
167
+ "seed": 21112,
168
+ "width": 1024,
169
+ "height": 1024,
170
+ "images": [
171
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-bilingual-festival-40.png"
172
+ ]
173
+ },
174
+ {
175
+ "label": "expanded-multiref-portrait-25",
176
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
177
+ "steps": 25,
178
+ "seed": 21123,
179
+ "width": 1024,
180
+ "height": 1024,
181
+ "images": [
182
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
183
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
184
+ ]
185
+ },
186
+ {
187
+ "label": "expanded-multiref-portrait-40",
188
+ "prompt": "Create a natural lifestyle campaign photograph using both references. From the first image, use only the mother on the left: preserve her recognizable oval face, dark curly hair tied in a low bun, small round gold earrings, navy coat and mustard yellow scarf. From the second image, use the central cobalt-blue insulated bottle: preserve its tall shape, black screw cap, single narrow orange front stripe and small white mountain-triangle emblem near the base. Show this same woman seated alone on a wooden railway-platform bench, waist-up, smiling gently toward the camera while holding that bottle upright with both hands around its lower half; the cap, orange stripe and emblem should remain visible. Put one green suitcase on the ground beside the bench, partly visible in frame. Use soft overcast daylight and a stationary passenger train softly out of focus behind her. Realistic hands and fingers, recognizable identity, coherent scale and contact shadows. Do not include the other people from the first reference, the cups or pedestal from the second reference, or any poster lettering. The intended result is one coherent photograph with exactly one woman and one bottle.",
189
+ "steps": 40,
190
+ "seed": 21123,
191
+ "width": 1024,
192
+ "height": 1024,
193
+ "images": [
194
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-station-reunion-40.png",
195
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
196
+ ]
197
+ },
198
+ {
199
+ "label": "expanded-product-relocation-edit-25",
200
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
201
+ "steps": 25,
202
+ "seed": 21134,
203
+ "width": 1024,
204
+ "height": 1024,
205
+ "images": [
206
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
207
+ ]
208
+ },
209
+ {
210
+ "label": "expanded-product-relocation-edit-40",
211
+ "prompt": "Edit the arrangement of the small cups in this product campaign while preserving the central bottle and all typography. Remove the yellow cup that is currently on the left, restoring the tabletop and its lighting naturally. Move the turquoise cup from the right side to the far left of the pedestal, fully visible with a small clear gap from the pedestal; leave the former right-hand position empty. The final image must contain exactly two products: the unchanged cobalt-blue bottle on its original pedestal and the one turquoise cup on the left. Preserve the bottle's shape, cap, orange stripe, white mountain emblem, location and scale; preserve the pedestal, backdrop, framing, light direction, and the exact text \"TAKE THE LONG WAY\", \"RIDGELINE / EVERYDAY ADVENTURE\", and \"750 mL · BUILT TO REUSE\". Give the moved cup a physically consistent contact shadow. Do not introduce any extra objects or redesign the campaign.",
212
+ "steps": 40,
213
+ "seed": 21134,
214
+ "width": 1024,
215
+ "height": 1024,
216
+ "images": [
217
+ "/poc/samples/fidelity-v3/bf16-dit/expanded-ridgeline-campaign-40.png"
218
+ ]
219
+ },
220
+ {
221
+ "label": "editorial-25",
222
+ "prompt": "Candid editorial photograph in a sunlit pottery studio. Exactly two adults: an older woman with short silver hair on the left demonstrates shaping a clay bowl on a pottery wheel; a younger man with curly dark hair on the right watches and holds a folded blue towel with both hands. Their hands are anatomically natural and clearly visible, with clay on the woman's fingertips. Wooden shelves hold irregular ceramic vases, one trailing green plant hangs high in the back right, and soft morning light enters from a large window on the left. Waist-up environmental composition, documentary photography, natural skin texture, restrained warm colors, realistic depth of field. No lettering.",
223
+ "steps": 25,
224
+ "seed": 118,
225
+ "width": 1024,
226
+ "height": 1024
227
+ },
228
+ {
229
+ "label": "rainy-city-40",
230
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
231
+ "steps": 40,
232
+ "seed": 227,
233
+ "width": 1024,
234
+ "height": 1024
235
+ },
236
+ {
237
+ "label": "rainy-city-25",
238
+ "prompt": "Cinematic street-level architectural photograph of a narrow Tokyo side street at blue hour after rain. A tiny warmly lit ramen restaurant occupies the left foreground, with a striped navy awning and a vertical sign that reads \"RAMEN\". Exactly three bicycles are parked along the right wall. A person holding a transparent umbrella walks away at center, wearing a mustard yellow raincoat. Overhead wires cross between weathered three-story buildings, red lanterns glow farther down the street, and puddles reflect the signs. Strong one-point perspective, realistic glass and wet asphalt, fine distant details, balanced blue and amber lighting.",
239
+ "steps": 25,
240
+ "seed": 227,
241
+ "width": 1024,
242
+ "height": 1024
243
+ },
244
+ {
245
+ "label": "botanical-poster-25",
246
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
247
+ "steps": 25,
248
+ "seed": 336,
249
+ "width": 1024,
250
+ "height": 1024
251
+ },
252
+ {
253
+ "label": "botanical-poster-40",
254
+ "prompt": "Sophisticated botanical exhibition poster on warm ivory paper, flat front-on view, crisp print design. At the top the large dark green serif headline reads exactly \"THE SECRET GARDEN\" on two centered lines. Under it a smaller line reads exactly \"BOTANICAL STUDIES / 2026\". The center contains a delicate detailed watercolor illustration of a lemon branch with exactly two yellow lemons, white blossoms and dark green leaves, surrounded by generous negative space. Along the bottom three evenly spaced text blocks read \"12 OCTOBER\", \"10 AM - 6 PM\", and \"GLASSHOUSE No. 4\". A thin rectangular green border surrounds the entire design. Elegant typography, no other text, no mockup, no shadows outside the paper.",
255
+ "steps": 40,
256
+ "seed": 336,
257
+ "width": 1024,
258
+ "height": 1024
259
+ },
260
+ {
261
+ "label": "food-spatial-25",
262
+ "prompt": "Overhead high-end food photograph of a neatly arranged breakfast on a pale stone table. A large round white plate is centered. On the plate, exactly three pancakes form a vertical stack, topped by exactly five blueberries and a small square of butter. To the left of the plate is a silver fork, to the right is a silver knife with its blade facing the plate. A clear glass of orange juice sits at the upper right, a small white bowl containing sliced strawberries sits at the upper left, and a folded rust-colored linen napkin lies beneath the fork. A little maple syrup runs down only the right edge of the pancake stack. Soft natural light from the upper left, realistic food texture, all objects fully inside frame.",
263
+ "steps": 25,
264
+ "seed": 445,
265
+ "width": 1024,
266
+ "height": 1024
267
+ },
268
+ {
269
+ "label": "fantasy-cutaway-25",
270
+ "prompt": "Intricate isometric cutaway illustration of a cozy three-level library built inside an enormous ancient tree, in a hand-painted storybook style. Ground floor: a round green entrance door, a circular rug, and a sleeping orange cat beside a small wood stove. Middle floor: curved bookshelves filled with colorful books, a writing desk by an oval window, and a brass telescope pointing outside. Top floor: a glass-domed reading nook with two red armchairs and a hanging lantern. A single wooden spiral staircase visibly connects all three floors. Exposed roots wrap around mossy rocks, tiny mushrooms cluster near the door, and the leafy crown surrounds the dome without hiding it. Consistent isometric perspective, warm interior lighting, cool twilight forest background, detailed but readable, no text.",
271
+ "steps": 25,
272
+ "seed": 554,
273
+ "width": 1024,
274
+ "height": 1024
275
+ },
276
+ {
277
+ "label": "glass-product-40",
278
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
279
+ "steps": 40,
280
+ "seed": 663,
281
+ "width": 1024,
282
+ "height": 1024
283
+ },
284
+ {
285
+ "label": "glass-product-25",
286
+ "prompt": "Luxury studio still life photograph on a polished dark green marble plinth. A rectangular clear glass perfume bottle half filled with pale amber liquid stands in the center, with a brushed gold cylindrical cap and a small cream label reading exactly \"LUMEN\". A thin curved strip of orange peel lies in front of the bottle. Behind and to the left is a translucent ribbed glass sphere; behind and to the right are two glossy dark green leaves. Hard afternoon sunlight from the upper left creates realistic caustics, transparent overlapping shadows, precise refraction through the bottle and a soft reflection on the marble. Deep emerald background, controlled commercial composition, realistic materials, no additional text.",
287
+ "steps": 25,
288
+ "seed": 663,
289
+ "width": 1024,
290
+ "height": 1024
291
+ },
292
+ {
293
+ "label": "editorial-edit-25",
294
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
295
+ "steps": 25,
296
+ "seed": 774,
297
+ "width": 1024,
298
+ "height": 1024,
299
+ "images": [
300
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
301
+ ]
302
+ },
303
+ {
304
+ "label": "editorial-edit-40",
305
+ "prompt": "Replace only the blue towel held by the younger man with a bright red towel of the same size and folded shape. Preserve both people's identities, facial expressions, hands, clothing, clay bowl, studio shelves, window light, framing and photographic style.",
306
+ "steps": 40,
307
+ "seed": 774,
308
+ "width": 1024,
309
+ "height": 1024,
310
+ "images": [
311
+ "/poc/samples/varied-v2/nf4/editorial-40.png"
312
+ ]
313
+ },
314
+ {
315
+ "label": "poster-edit-25",
316
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
317
+ "steps": 25,
318
+ "seed": 885,
319
+ "width": 1024,
320
+ "height": 1024,
321
+ "images": [
322
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
323
+ ]
324
+ },
325
+ {
326
+ "label": "poster-edit-40",
327
+ "prompt": "Edit only the text at the top: replace \"THE SECRET GARDEN\" with \"THE LEMON HOUSE\". Keep the same dark green serif type style and centered placement. Preserve all other text exactly, the lemon branch illustration, paper background, border, spacing and overall design.",
328
+ "steps": 40,
329
+ "seed": 885,
330
+ "width": 1024,
331
+ "height": 1024,
332
+ "images": [
333
+ "/poc/samples/varied-v2/nf4/botanical-poster-40.png"
334
+ ]
335
+ },
336
+ {
337
+ "label": "city-edit-25",
338
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
339
+ "steps": 25,
340
+ "seed": 996,
341
+ "width": 1024,
342
+ "height": 1024,
343
+ "images": [
344
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
345
+ ]
346
+ },
347
+ {
348
+ "label": "city-edit-40",
349
+ "prompt": "Transform this rainy Tokyo street photograph into a richly textured hand-painted watercolor illustration. Preserve the layout of the buildings, the ramen restaurant on the left, all three bicycles on the right, the central person with the transparent umbrella and yellow raincoat, the overhead wires, blue-hour lighting, and puddle reflections. Keep the RAMEN sign legible.",
350
+ "steps": 40,
351
+ "seed": 996,
352
+ "width": 1024,
353
+ "height": 1024,
354
+ "images": [
355
+ "/poc/samples/varied-v2/nf4/rainy-city-40.png"
356
+ ]
357
+ }
358
+ ]
reproduction/lean_encoder.py ADDED
@@ -0,0 +1,163 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Optional Qwen3-VL encoder-only adapter for the pinned Qwen Image 2.1 POC.
2
+
3
+ Save a normal (including quantized) checkpoint BEFORE calling enable_lean_encoder.
4
+ The adapted instance no longer has the architecture's lm_head; do not pass it to
5
+ save_pretrained or pipeline.save_pretrained. Reload the normal checkpoint and
6
+ reapply this adapter at runtime. No decoder layers or normalization are changed.
7
+
8
+ Typical use, after loading and before serving concurrent requests::
9
+
10
+ report = verify_encoder_parity(te, processor, prompt="A red teapot")
11
+ assert report["passed"], report
12
+ enable_lean_encoder(te)
13
+
14
+ The parity helper is optional, runs two encoder forwards, and temporarily changes
15
+ forward dispatch. Run it only while the engine is idle. The application must
16
+ serialize enablement and parity checks with generation requests.
17
+ """
18
+ from types import MethodType
19
+
20
+
21
+ def _base_model_forward(self, *args, **kwargs):
22
+ # Image generation reads features, never autoregressive vocabulary logits.
23
+ kwargs["use_cache"] = False
24
+ return self.model(*args, **kwargs)
25
+
26
+
27
+ def _dispatch_attribute(te):
28
+ # Accelerate wraps forward and calls _old_forward. Replace its inner function
29
+ # to preserve onload/offload hooks; future remove/reinstall cycles retain it.
30
+ return "_old_forward" if hasattr(te, "_hf_hook") and hasattr(te, "_old_forward") else "forward"
31
+
32
+
33
+ def enable_lean_encoder(te):
34
+ """Remove only lm_head and forward through the existing multimodal base model.
35
+
36
+ Returns small metadata; mutates te in place and is idempotent. This preserves
37
+ te.model.language_model so Diffusers' final-RMSNorm hook keeps working.
38
+ Existing Accelerate CPU-offload hooks are retained. Save checkpoints first.
39
+ """
40
+ if getattr(te, "_qwen21_lean_encoder", False):
41
+ return {"enabled": True, "already_enabled": True}
42
+ if not hasattr(te, "model") or not hasattr(te.model, "language_model"):
43
+ raise TypeError("Expected Qwen3VLForConditionalGeneration with model.language_model")
44
+ if not hasattr(te, "lm_head"):
45
+ raise ValueError("lm_head is missing; load an unmodified checkpoint before enabling this adapter")
46
+ removed_parameters = sum(p.numel() for p in te.lm_head.parameters())
47
+ dispatch = _dispatch_attribute(te)
48
+ setattr(te, dispatch, MethodType(_base_model_forward, te))
49
+ te.config.use_cache = False
50
+ te.config.text_config.use_cache = False
51
+ del te.lm_head
52
+ te._qwen21_lean_encoder = True
53
+ return {
54
+ "enabled": True,
55
+ "already_enabled": False,
56
+ "removed_parameters": removed_parameters,
57
+ "dispatch_attribute": dispatch,
58
+ "checkpoint_save_supported": False,
59
+ }
60
+
61
+
62
+ def _prepare_inputs(processor, prompt, images, device):
63
+ from PIL import Image
64
+ system = "Comprehend and analyze the provided prompt."
65
+ vision_images = []
66
+ for img in images or []:
67
+ if not isinstance(img, Image.Image):
68
+ raise TypeError("Parity references must be PIL images, already resized as for inference")
69
+ if img.mode == "RGBA":
70
+ white = Image.new("RGB", img.size, (255, 255, 255))
71
+ white.paste(img, mask=img.getchannel("A"))
72
+ img = white
73
+ vision_images.append(img)
74
+ prefix = " ".join(
75
+ f"<image{i}><|vision_start|><|image_pad|><|vision_end|>"
76
+ for i in range(1, len(vision_images) + 1)
77
+ )
78
+ text = (
79
+ f"<|im_start|>system\n{system}<|im_end|>\n"
80
+ f"<|im_start|>user\n{prefix}{prompt or ' '}<|im_end|>\n"
81
+ "<|im_start|>assistant\n"
82
+ )
83
+ kwargs = {"text": [text], "padding": True, "padding_side": "left", "return_tensors": "pt"}
84
+ if vision_images:
85
+ kwargs["images"] = vision_images
86
+ encoded = processor(**kwargs).to(device)
87
+ forward = {"input_ids": encoded.input_ids, "attention_mask": encoded.attention_mask}
88
+ for key in ("pixel_values", "image_grid_thw", "mm_token_type_ids"):
89
+ if key in encoded:
90
+ forward[key] = encoded[key]
91
+ return forward
92
+
93
+
94
+ def verify_encoder_parity(te, processor=None, prompt="A red teapot", images=None,
95
+ model_inputs=None, device=None, atol=0.0, rtol=0.0):
96
+ """Compare full final pre-norm features with/without vocabulary projection.
97
+
98
+ Does not remove lm_head or permanently change dispatch/configuration. Supply
99
+ either model_inputs (prepared forward kwargs, including reference pixels) or
100
+ processor/prompt/optional PIL images. Images should already have the same
101
+ resize used by the pipeline. CPU-offloaded encoders use their hook's execution
102
+ device by default. Returns a parity report; enable only when passed is true.
103
+
104
+ Exact equality is the default because both paths execute the same base model.
105
+ Tolerances can be supplied explicitly if the runtime is nondeterministic.
106
+ The comparison mimics the pinned pipeline's norm hook and output_hidden_states
107
+ request. It compares full features rather than only the prompt-trimmed suffix.
108
+ """
109
+ import torch
110
+ if getattr(te, "_qwen21_lean_encoder", False) or not hasattr(te, "lm_head"):
111
+ raise ValueError("Verify parity before enabling lean encoding")
112
+ if not hasattr(te, "model") or not hasattr(te.model, "language_model"):
113
+ raise TypeError("Expected Qwen3VLForConditionalGeneration with model.language_model")
114
+ if device is None:
115
+ device = getattr(getattr(te, "_hf_hook", None), "execution_device", None)
116
+ if device is None:
117
+ device = next(te.parameters()).device
118
+ if model_inputs is None:
119
+ if processor is None:
120
+ raise ValueError("Provide processor or prepared model_inputs")
121
+ model_inputs = _prepare_inputs(processor, prompt, images, device)
122
+ else:
123
+ model_inputs = {
124
+ k: v.to(device) if isinstance(v, torch.Tensor) else v
125
+ for k, v in dict(model_inputs).items()
126
+ }
127
+ model_inputs = {**model_inputs, "use_cache": False, "output_hidden_states": True, "return_dict": True}
128
+ # These conditional-generation-only options do not belong to base-model input.
129
+ for key in ("labels", "logits_to_keep"):
130
+ model_inputs.pop(key, None)
131
+ dispatch = _dispatch_attribute(te)
132
+ original_forward = getattr(te, dispatch)
133
+ was_training = te.training
134
+ norm = te.model.language_model.norm
135
+ handle = norm.register_forward_hook(lambda module, args, output: args[0])
136
+ try:
137
+ te.eval()
138
+ with torch.inference_mode():
139
+ baseline = te(**model_inputs).hidden_states[-1].detach().float().cpu().clone()
140
+ setattr(te, dispatch, MethodType(_base_model_forward, te))
141
+ candidate = te(**model_inputs).hidden_states[-1].detach().float().cpu()
142
+ finally:
143
+ setattr(te, dispatch, original_forward)
144
+ handle.remove()
145
+ if was_training:
146
+ te.train()
147
+ if baseline.shape != candidate.shape:
148
+ return {"passed": False, "reason": "shape_mismatch", "baseline_shape": list(baseline.shape),
149
+ "candidate_shape": list(candidate.shape)}
150
+ delta = candidate - baseline
151
+ finite = bool(torch.isfinite(baseline).all() and torch.isfinite(candidate).all())
152
+ return {
153
+ "passed": finite and bool(torch.allclose(baseline, candidate, atol=atol, rtol=rtol)),
154
+ "finite": finite,
155
+ "shape": list(baseline.shape),
156
+ "max_abs_difference": float(delta.abs().max()),
157
+ "rms_difference": float(delta.square().mean().sqrt()),
158
+ "baseline_rms": float(baseline.square().mean().sqrt()),
159
+ "atol": atol,
160
+ "rtol": rtol,
161
+ "reference_count": len(images or []) if images is not None else None,
162
+ "dispatch_attribute": dispatch,
163
+ }
reproduction/nunchaku_backend/README.md ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Calibrated Qwen Image 2.1 Nunchaku backend
2
+
3
+ The selected backend is **calibrated v3 rank 128**, using Nunchaku's actual signed INT4 residual weights/activations and BF16 low-rank correction. The pinned Diffusers Qwen2.1 model keeps attention, prefix KV caching, positional embeddings and boundary projections. Only its 224 block linears are replaced. This is a custom backend, not an official Nunchaku Qwen2.1 release or a near-lossless quality claim.
4
+
5
+ The superseded NF4-trajectory absmax collector and one-pass whole-model exporter have been removed from this active package. Their unchanged source and the complete pre-cleanup backend snapshot are in [the dated archive](../archive/2026-09-21-superseded-implementations/README.md), with file hashes. Existing samples, reports and checkpoint manifests are preserved.
6
+
7
+ ## Runtime dependency boundary
8
+
9
+ Loading a completed calibrated checkpoint requires only these project modules:
10
+
11
+ - `__init__.py`: backend format and public loader.
12
+ - `runtime.py`: strict packed-checkpoint validation and Nunchaku layer replacement.
13
+ - `layout.py`: block-linear name pattern; no Torch import.
14
+
15
+ Runtime also requires the pinned Torch, Diffusers, Transformers, Nunchaku and safetensors dependencies. It does **not** need DeepCompressor, the archived collector/exporter, or optimization modules. Optional Engine BF16 overrides continue to use `hybrid_v3.py` and its shared checkpoint helpers; that optional path was retained for compatibility and is not the selected rank-128 configuration.
16
+
17
+ ## Calibration and selected conversion recipe
18
+
19
+ The final calibration collected original BF16 DiT activations on 40-step trajectories at steps **0, 10, 20, 30, 39**, with the NF4 text encoder held fixed. Disjoint train/validation/held-out jobs are in `experiments/fidelity-v3/calibration-jobs.json`. Collection saves per-layer activation files and provenance. GPU commands below must run in the isolated POC environment on the reserved non-Klein GPU.
20
+
21
+ ```bash
22
+ python -m nunchaku_backend.collect_v3 \
23
+ --jobs /poc/experiments/fidelity-v3/calibration-jobs.json \
24
+ --out /cache/qwen21-activation-v3-40 --steps 40
25
+
26
+ python -m nunchaku_backend.export_v3 \
27
+ --model-path /cache/huggingface/hub/models--Qwen--Qwen-Image-2.1/snapshots/b3179ad355be050328e483a9dfdd9e60cd62adfa \
28
+ --activations /cache/qwen21-activation-v3-40 \
29
+ --baseline-calibration /cache/qwen21-calibration \
30
+ --out /cache/qwen-nunchaku-v3-r128-reproduction \
31
+ --ranks 128 --alphas 0.25 0.5 0.75 \
32
+ --families activation_only smoothquant --weighting none \
33
+ --iterations 16 --final-gptq --gptq-damp 0.01 \
34
+ --factorization balanced --ridge 0.01 --baseline-rank 32 \
35
+ --seed 1947 --niter 4 --oversample 16 --threads 8 \
36
+ --device cuda:0 --objective-backend auto
37
+ ```
38
+
39
+ Use the preserved baseline calibration artifact for exact historical reproduction: its SHA-256 is `7f032d0930881f8352309e10084b3e5e4418ac81fb1c69a9bc7457f5fa731018`. It influences the baseline candidate, even though new search candidates use BF16 activation rows. Without that artifact, omitting the flag is a new conversion recipe and must not be described as reproducing the saved checkpoint. The old collector can be run from a separate restored archive workspace when historical re-collection is needed; do not restore it into a deployment package.
40
+
41
+ The final activation fingerprint is `be755525bef86b1be059e0b5667fb8f38305472a6775506b3363d9477f630b25`. It includes paths as well as archive contents, so preserve paths when comparing identity. The saved manifest in `results/nunchaku-v3-manifest.json.gz` is authoritative for settings. Collection/re-export is costly and was not rerun during cleanup; mathematical extraction equivalence and CPU tests verify this refactor, not bitwise GPU reproduction.
42
+
43
+ ## Active implementation
44
+
45
+ - `collect_v3.py` / `teacher_v3.py`: bounded teacher activation capture.
46
+ - `export_v3.py` / `optimize_v3.py` / `gptq_v3.py`: activation-output scoring, smoothing search, alternating low-rank/residual refinement, output correction and optional final GPTQ.
47
+ - `packing.py`: pinned DeepCompressor packing adapter and group size.
48
+ - `baseline_candidate.py`: unchanged one-pass **internal fallback**. The final optimizer evaluates this candidate; deleting it would change the recipe. There is no active one-pass whole-checkpoint CLI.
49
+ - `checkpoint_io.py`: lazy safetensors reads and atomic JSON writes.
50
+ - `convert_activation_reference.py` / `convert_reference.py`: numerical kernel reference, retained for validation.
51
+
52
+ DeepCompressor packing commit: `69f3473f5e1c1504bae35cc50c7858ef900a9b17`; Diffusers commit: `80c7ed262aeffbeb43ef13ae04baeb9b84515a69`. Conversion expects the pinned DeepCompressor checkout on `PYTHONPATH` (`/opt/deepcompressor` in the POC image). Loading does not.
53
+
54
+ Rank-512 upgrade, mixed-BF16 and denoiser/parent-MLP probes remain optional research tools with their existing module paths so historical diagnostics and the Engine's optional overrides remain usable. They are not selected release defaults. See [the fidelity report](../experiments/fidelity-v3/REPORT.md) and [rank-512 visual review](../research/quality-v3-rank512-original-generation.md) for measured limitations. Cleanup does not change old source-hash manifests; new refinement manifests name the extracted modules.
reproduction/nunchaku_backend/__init__.py ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Experimental Qwen Image2.1 SVDQuant/Nunchaku integration.
2
+
3
+ This package does not alias NF4: its runtime uses packed signed INT4 weights,
4
+ INT4 activations, and a BF16 low-rank branch through Nunchaku CUDA kernels.
5
+ """
6
+ FORMAT = "qwen21-nunchaku-svdq-int4-v1"
7
+
8
+
9
+ def load_transformer(*args, **kwargs):
10
+ from .runtime import load_transformer as implementation
11
+ return implementation(*args, **kwargs)
reproduction/nunchaku_backend/baseline_candidate.py ADDED
@@ -0,0 +1,265 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """One-pass baseline candidate retained for faithful calibrated-v3 reproduction.
2
+
3
+ This is an internal fallback evaluated by optimize_v3, not a supported standalone
4
+ whole-model exporter. The superseded converter and CLI are in the dated archive.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import time
9
+ from typing import Any
10
+ import torch
11
+ from .packing import DEEPCOMPRESSOR_COMMIT, GROUP_SIZE, _upstream_converter
12
+
13
+ def _working_float(tensor: torch.Tensor, name: str, device: torch.device) -> torch.Tensor:
14
+ if not tensor.is_floating_point():
15
+ raise ValueError(f"{name} must be floating point")
16
+ result = tensor.detach().to(device=device, dtype=torch.float32)
17
+ if not torch.isfinite(result).all():
18
+ raise ValueError(f"{name} contains nonfinite values")
19
+ return result
20
+
21
+
22
+ def _smoothing(
23
+ weight: torch.Tensor,
24
+ smooth: torch.Tensor | None,
25
+ input_absmax: torch.Tensor | None,
26
+ alpha: float,
27
+ dtype: torch.dtype,
28
+ ) -> tuple[torch.Tensor, str]:
29
+ if smooth is not None and input_absmax is not None:
30
+ raise ValueError("Pass smooth or input_absmax, not both")
31
+ if not 0 <= alpha <= 1:
32
+ raise ValueError("smooth_alpha must be in [0, 1]")
33
+ ic = weight.shape[1]
34
+ if smooth is not None:
35
+ factor = _working_float(smooth, "smooth", weight.device)
36
+ if factor.shape != (ic,) or not (factor > 0).all():
37
+ raise ValueError("smooth must be a positive vector of length in_features")
38
+ source = "caller_provided"
39
+ elif input_absmax is not None:
40
+ amax = _working_float(input_absmax, "input_absmax", weight.device)
41
+ if amax.shape != (ic,) or not (amax >= 0).all():
42
+ raise ValueError("input_absmax must be a nonnegative vector of length in_features")
43
+ # SmoothQuant-style starting point, without per-layer alpha search.
44
+ wmax = weight.abs().amax(dim=0).clamp_min(1e-5)
45
+ factor = (amax.clamp_min(1e-5).pow(alpha) / wmax.pow(1 - alpha)).clamp(1e-4, 1e4)
46
+ source = "activation_absmax"
47
+ else:
48
+ factor = torch.ones(ic, dtype=torch.float32, device=weight.device)
49
+ source = "identity_uncalibrated"
50
+ # Decompose using the exact factors that the kernel will load.
51
+ factor = factor.to(dtype).float()
52
+ if not torch.isfinite(factor).all() or not (factor > 0).all():
53
+ raise ValueError("smooth cannot be represented as positive finite values in parameter dtype")
54
+ return factor, source
55
+
56
+
57
+ @torch.no_grad()
58
+ def convert_linear_weight(
59
+ weight: torch.Tensor,
60
+ bias: torch.Tensor | None = None,
61
+ rank: int = 32,
62
+ *,
63
+ smooth: torch.Tensor | None = None,
64
+ input_absmax: torch.Tensor | None = None,
65
+ smooth_alpha: float = 0.5,
66
+ svd_method: str = "randomized",
67
+ seed: int = 0,
68
+ niter: int = 4,
69
+ oversample: int = 16,
70
+ torch_dtype: torch.dtype = torch.bfloat16,
71
+ return_reference: bool = False,
72
+ conversion_device: str | torch.device = "cpu",
73
+ ) -> tuple:
74
+ """Return packed ``SVDQW4A4Linear`` state and JSON-compatible statistics.
75
+
76
+ Input/output widths must be multiples of 128; rank must be a positive
77
+ multiple of 16. Weight shape is [out_features, in_features]. No source
78
+ tensors are mutated. Smoothing has the convention X/s and W*s; the
79
+ low-rank down matrix is divided by s because that kernel branch sees X.
80
+
81
+ ``input_absmax`` is the per-input-channel maximum absolute activation
82
+ collected from representative denoising calls. No activations means an
83
+ explicitly uncalibrated conversion. Provided statistics do not establish
84
+ representative coverage or image quality. ``return_reference=True`` adds
85
+ a third dictionary of unpacked tensors for numerical kernel probing.
86
+ Conversion runs on CPU unless ``conversion_device='cuda:0'`` is explicitly
87
+ selected. All returned tensors remain on the conversion device.
88
+ """
89
+ start = time.perf_counter()
90
+ pack = _upstream_converter()
91
+ if torch_dtype not in (torch.bfloat16, torch.float16):
92
+ raise ValueError("torch_dtype must be bfloat16 or float16")
93
+ device = torch.device(conversion_device)
94
+ if device.type not in ("cpu", "cuda"):
95
+ raise ValueError("conversion_device must be cpu or an explicit CUDA device")
96
+ if device.type == "cuda" and device.index is None:
97
+ raise ValueError("Specify a CUDA device index, e.g. conversion_device='cuda:0'")
98
+ original = _working_float(weight, "weight", device)
99
+ if original.ndim != 2:
100
+ raise ValueError("weight must be two-dimensional")
101
+ oc, ic = original.shape
102
+ if not oc or not ic or oc % 128 or ic % 128:
103
+ raise ValueError("in_features and out_features must be positive multiples of 128")
104
+ if rank <= 0 or rank % 16 or rank > min(oc, ic):
105
+ raise ValueError("rank must be a positive multiple of 16, at most min(in_features, out_features)")
106
+ if niter < 0 or oversample < 0:
107
+ raise ValueError("niter and oversample must be nonnegative")
108
+ if bias is not None:
109
+ bias = _working_float(bias, "bias", device)
110
+ if bias.shape != (oc,):
111
+ raise ValueError("bias must have shape [out_features]")
112
+ bias = bias.to(torch_dtype)
113
+ if not torch.isfinite(bias).all():
114
+ raise ValueError("bias overflows parameter dtype")
115
+
116
+ factor, source = _smoothing(original, smooth, input_absmax, smooth_alpha, torch_dtype)
117
+ smoothed = original * factor.unsqueeze(0)
118
+ if not torch.isfinite(smoothed).all():
119
+ raise ValueError("smoothed weight overflowed")
120
+ if svd_method == "full":
121
+ u, singular, vh = torch.linalg.svd(smoothed, full_matrices=False)
122
+ u, singular, v = u[:, :rank], singular[:rank], vh[:rank, :].T
123
+ elif svd_method == "randomized":
124
+ # CPU mode never touches CUDA generators. Explicit CUDA conversion
125
+ # saves/restores only that device's RNG, plus the CPU stream.
126
+ with torch.random.fork_rng(devices=[device.index] if device.type == "cuda" else []):
127
+ torch.random.default_generator.manual_seed(seed)
128
+ if device.type == "cuda":
129
+ torch.cuda.default_generators[device.index].manual_seed(seed)
130
+ u, singular, v = torch.svd_lowrank(smoothed, q=min(rank + oversample, oc, ic), niter=niter)
131
+ u, singular, v = u[:, :rank], singular[:rank], v[:, :rank]
132
+ else:
133
+ raise ValueError("svd_method must be 'randomized' or 'full'")
134
+
135
+ root = singular.clamp_min(0).sqrt()
136
+ up = (u * root.unsqueeze(0)).to(torch_dtype)
137
+ down_smoothed = (root.unsqueeze(1) * v.T).to(torch_dtype)
138
+ down = (down_smoothed.float() / factor.unsqueeze(0)).to(torch_dtype)
139
+ if not torch.isfinite(up).all() or not torch.isfinite(down).all():
140
+ raise ValueError("low-rank factors overflow parameter dtype")
141
+ # Include the final inverse-smoothing BF16 rounding in the residual too.
142
+ # The exported low-rank branch represents up@down in original coordinates.
143
+ residual = smoothed - up.float() @ (down.float() * factor.unsqueeze(0))
144
+
145
+ grouped = residual.reshape(oc, ic // GROUP_SIZE, GROUP_SIZE)
146
+ scales = (grouped.abs().amax(dim=-1) / 7).clamp_min(torch.finfo(torch_dtype).tiny).to(torch_dtype)
147
+ if not torch.isfinite(scales).all():
148
+ raise ValueError("residual scales overflow parameter dtype")
149
+ quantized = (grouped / scales.float().unsqueeze(-1)).round().clamp(-7, 7)
150
+ dequantized = (quantized * scales.float().unsqueeze(-1)).reshape(oc, ic).to(torch_dtype)
151
+ # Upstream handles all tensor-core tile permutations, nibble packing,
152
+ # packed scales/bias/smoothing, and separate low-rank matrix layouts.
153
+ qweight, wscales, packed_bias, packed_smooth, lora, subscale = pack(
154
+ dequantized,
155
+ scales.reshape(oc, 1, ic // GROUP_SIZE, 1),
156
+ bias=bias,
157
+ smooth=factor.to(torch_dtype),
158
+ lora=(down, up),
159
+ float_point=False,
160
+ )
161
+ assert lora is not None and subscale is None
162
+ state = {
163
+ "qweight": qweight.contiguous(),
164
+ "wscales": wscales.contiguous(),
165
+ "smooth_factor": packed_smooth.contiguous(),
166
+ "smooth_factor_orig": packed_smooth.clone().contiguous(),
167
+ "proj_down": lora[0].contiguous(),
168
+ "proj_up": lora[1].contiguous(),
169
+ }
170
+ if bias is not None:
171
+ state["bias"] = packed_bias.contiguous()
172
+
173
+ # Representation error uses integer*stored_scale, matching the kernel's
174
+ # weight values (the dequantized BF16 intermediary is only for export).
175
+ error_sq, weight_sq = 0.0, 0.0
176
+ for offset in range(0, oc, 256):
177
+ sl = slice(offset, offset + 256)
178
+ recovered = (quantized[sl] * scales[sl].float().unsqueeze(-1)).reshape(-1, ic)
179
+ recovered = recovered / factor + up[sl].float() @ down.float()
180
+ error_sq += (recovered - original[sl]).double().square().sum().item()
181
+ weight_sq += original[sl].double().square().sum().item()
182
+ stats = {
183
+ "algorithm": "one_pass_smoothed_svd_signed_int4",
184
+ "deepcompressor_commit": DEEPCOMPRESSOR_COMMIT,
185
+ "in_features": ic,
186
+ "out_features": oc,
187
+ "rank": rank,
188
+ "group_size": GROUP_SIZE,
189
+ "precision": "int4",
190
+ "torch_dtype": str(torch_dtype),
191
+ "conversion_device": str(device),
192
+ "smoothing_source": source,
193
+ "calibrated": input_absmax is not None,
194
+ "caller_smoothing_provided": smooth is not None,
195
+ "smooth_alpha": smooth_alpha if input_absmax is not None else None,
196
+ "smooth_min": factor.min().item(),
197
+ "smooth_max": factor.max().item(),
198
+ "svd_method": svd_method,
199
+ "seed": seed,
200
+ "niter": niter,
201
+ "oversample": oversample,
202
+ "weight_relative_l2_error": (error_sq / weight_sq) ** 0.5 if weight_sq else 0.0,
203
+ "packed_bytes": sum(t.numel() * t.element_size() for t in state.values()),
204
+ "conversion_seconds": time.perf_counter() - start,
205
+ "limitations": "No activation rounding simulation, alpha search, GPTQ, iterative residual fit, or image quality guarantee.",
206
+ }
207
+ if return_reference:
208
+ reference = {
209
+ "residual_dequant": (quantized * scales.float().unsqueeze(-1)).reshape(oc, ic).contiguous(),
210
+ "down_unpacked": down.contiguous(),
211
+ "up_unpacked": up.contiguous(),
212
+ "smooth": factor.to(torch_dtype).contiguous(),
213
+ }
214
+ if bias is not None:
215
+ reference["bias"] = bias.contiguous()
216
+ return state, stats, reference
217
+ return state, stats
218
+
219
+
220
+ def convert_linear(linear: torch.nn.Linear, **kwargs):
221
+ """Convenience wrapper; the original module is left unchanged."""
222
+ return convert_linear_weight(linear.weight, linear.bias, **kwargs)
223
+
224
+
225
+ @torch.no_grad()
226
+ def reference_forward(
227
+ x: torch.Tensor,
228
+ reference: dict[str, torch.Tensor],
229
+ *,
230
+ quantize_activations: bool = True,
231
+ ) -> torch.Tensor:
232
+ """FP32 numerical reference using exported weights and kernel A4 rules.
233
+
234
+ Smoothing rounds to input BF16/FP16; per-token group64 activation scales
235
+ are calculated in FP32. Rounding to integer uses the *unrounded* FP32
236
+ scale, while dequantization uses the BF16/FP16 stored scale. The low-rank
237
+ branch consumes original input. Residual, bias and low-rank activation
238
+ rounding follow kernel epilogue boundaries. Returns compute-dtype-rounded
239
+ values in FP32. CUDA approximate reciprocals and accumulation order are
240
+ not bit-exact here. All-zero activation groups dequantize to zero.
241
+ """
242
+ if x.dtype not in (torch.bfloat16, torch.float16):
243
+ raise ValueError("reference input must match the kernel BF16/FP16 input dtype")
244
+ dtype = x.dtype
245
+ smooth = reference["smooth"].to(device=x.device, dtype=torch.float32)
246
+ original = x.float()
247
+ smoothed = (original / smooth).to(dtype).float()
248
+ if smoothed.shape[-1] % GROUP_SIZE:
249
+ raise ValueError("input width must be divisible by 64")
250
+ if quantize_activations:
251
+ grouped = smoothed.reshape(*smoothed.shape[:-1], smoothed.shape[-1] // GROUP_SIZE, GROUP_SIZE)
252
+ # Mirrors gemm_w4a4.cuh: max*(1/7), FP32 reciprocal for rounding,
253
+ # cvt.rni (ties-to-even), signed saturation; store scales in BF16.
254
+ scale = grouped.abs().amax(-1, keepdim=True) * (1.0 / 7.0)
255
+ reciprocal = torch.where(scale > 0, scale.reciprocal(), torch.zeros_like(scale))
256
+ q = (grouped * reciprocal).round().clamp(-8, 7)
257
+ smoothed = (q * scale.to(dtype).float()).reshape_as(smoothed)
258
+ residual = reference["residual_dequant"].to(device=x.device, dtype=torch.float32)
259
+ down = reference["down_unpacked"].to(device=x.device, dtype=torch.float32)
260
+ up = reference["up_unpacked"].to(device=x.device, dtype=torch.float32)
261
+ output = (smoothed @ residual.T).to(dtype).float()
262
+ if "bias" in reference:
263
+ output = (output + reference["bias"].to(device=x.device, dtype=torch.float32)).to(dtype).float()
264
+ hidden = (original @ down.T).to(dtype).float()
265
+ return (output + hidden @ up.T).to(dtype).float()
reproduction/nunchaku_backend/baseline_candidate_test.py ADDED
@@ -0,0 +1,141 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CPU layout, numerical, smoothing and validation tests (no Nunchaku import)."""
2
+
3
+ import json
4
+ import unittest
5
+
6
+ import torch
7
+ from deepcompressor.backend.nunchaku.utils import NunchakuWeightPacker
8
+
9
+ if __package__:
10
+ from .baseline_candidate import convert_linear_weight, reference_forward
11
+ else:
12
+ from baseline_candidate import convert_linear_weight, reference_forward
13
+
14
+
15
+ def unpack_qweight(packed):
16
+ """Invert upstream's documented INT4 warp128 memory order independently."""
17
+ oc, half_ic = packed.shape
18
+ ic = half_ic * 2
19
+ words = packed.contiguous().view(torch.int32)
20
+ q = ((words.unsqueeze(-1) >> torch.arange(0, 32, 4, dtype=torch.int32)) & 15)
21
+ q = q.reshape(oc // 128, ic // 64, 1, 8, 8, 4, 2, 2, 1, 8)
22
+ q = q.permute(0, 3, 6, 4, 8, 1, 2, 7, 5, 9).contiguous().reshape(oc, ic)
23
+ return torch.where(q >= 8, q - 16, q)
24
+
25
+
26
+ def unpack_scale(packed, oc):
27
+ groups = packed.numel() // oc
28
+ scale = packed.reshape(oc // 128, groups, 1, 8, 4, 2, 2)
29
+ return scale.permute(0, 2, 3, 5, 4, 6, 1).contiguous().reshape(oc, groups)
30
+
31
+
32
+ def unpack_state(state):
33
+ oc, half_ic = state["qweight"].shape
34
+ ic = half_ic * 2
35
+ q = unpack_qweight(state["qweight"])
36
+ scales = unpack_scale(state["wscales"], oc).float()
37
+ residual = (q.reshape(oc, ic // 64, 64).float() * scales.unsqueeze(-1)).reshape(oc, ic)
38
+ smooth = unpack_scale(state["smooth_factor"], ic).reshape(ic).float()
39
+ packer = NunchakuWeightPacker(4)
40
+ down = packer.unpack_lowrank_weight(state["proj_down"], down=True).float()
41
+ up = packer.unpack_lowrank_weight(state["proj_up"], down=False).float()
42
+ return residual, smooth, down, up
43
+
44
+
45
+ class ConversionTests(unittest.TestCase):
46
+ def setUp(self):
47
+ torch.manual_seed(21)
48
+ torch.set_num_threads(2)
49
+
50
+ def test_layout_roundtrip_signed_nibbles_and_scales(self):
51
+ packer = NunchakuWeightPacker(4)
52
+ original = torch.arange(256 * 384, dtype=torch.int32).reshape(256, 384) % 16 - 8
53
+ packed = packer.pack_weight(original.clone())
54
+ self.assertTrue(torch.equal(unpack_qweight(packed), original))
55
+ scales = torch.arange(256 * 6).reshape(256, 1, 6, 1).to(torch.bfloat16)
56
+ self.assertTrue(torch.equal(unpack_scale(packer.pack_scale(scales, 64), 256), scales.reshape(256, 6)))
57
+
58
+ def test_rank32_reconstructs_dominant_lowrank_better_than_rtn(self):
59
+ weight = (torch.randn(256, 16) @ torch.randn(16, 256) + torch.randn(256, 256) * 0.02).bfloat16()
60
+ original = weight.clone()
61
+ bias = torch.randn(256).bfloat16()
62
+ state, stats = convert_linear_weight(weight, bias, rank=32)
63
+ residual, smooth, down, up = unpack_state(state)
64
+ recovered = residual / smooth + up @ down
65
+ error = (recovered - weight.float()).norm() / weight.float().norm()
66
+ grouped = weight.float().reshape(256, 4, 64)
67
+ scale = grouped.abs().amax(-1, keepdim=True) / 7
68
+ rtn = ((grouped / scale).round() * scale).reshape_as(weight)
69
+ rtn_error = (rtn - weight.float()).norm() / weight.float().norm()
70
+ self.assertLess(error, rtn_error * 0.15)
71
+ self.assertAlmostEqual(error.item(), stats["weight_relative_l2_error"], places=6)
72
+ self.assertTrue(torch.equal(weight, original))
73
+ self.assertTrue(torch.equal(unpack_scale(state["bias"], 256).flatten(), bias))
74
+ self.assertFalse(stats["calibrated"])
75
+ self.assertEqual(state["wscales"].shape, (4, 256))
76
+ self.assertEqual(state["proj_down"].shape, (256, 32))
77
+ json.dumps(stats, allow_nan=False)
78
+
79
+ def test_nonuniform_smoothing_uses_original_input_for_lowrank(self):
80
+ weight = (torch.randn(128, 8) @ torch.randn(8, 256)).bfloat16()
81
+ smooth_input = torch.logspace(-1, 1, 256)
82
+ state, stats = convert_linear_weight(weight, smooth=smooth_input, rank=32, svd_method="full")
83
+ residual, smooth, down, up = unpack_state(state)
84
+ x = torch.randn(17, 256)
85
+ expected = x @ weight.float().T
86
+ actual = (x / smooth) @ residual.T + (x @ down.T) @ up.T
87
+ error = (actual - expected).norm() / expected.norm()
88
+ self.assertLess(error, 0.015)
89
+ # A regression to smoothing the lowrank branch must be detectable.
90
+ wrong = (x / smooth) @ residual.T + ((x / smooth) @ down.T) @ up.T
91
+ self.assertGreater(((wrong - expected).norm() / expected.norm()).item(), 0.5)
92
+ self.assertFalse(stats["calibrated"])
93
+ self.assertTrue(stats["caller_smoothing_provided"])
94
+
95
+ def test_activation_statistics_and_rng_are_preserved(self):
96
+ weight = torch.randn(128, 256).bfloat16()
97
+ absmax = torch.logspace(-2, 2, 256)
98
+ rng = torch.random.get_rng_state().clone()
99
+ state, stats = convert_linear_weight(weight, input_absmax=absmax, seed=187)
100
+ self.assertTrue(torch.equal(torch.random.get_rng_state(), rng))
101
+ self.assertEqual(stats["smoothing_source"], "activation_absmax")
102
+ expected = (absmax.sqrt() / weight.float().abs().amax(0).sqrt()).clamp(1e-4, 1e4).bfloat16().float()
103
+ self.assertTrue(torch.equal(unpack_scale(state["smooth_factor"], 256).flatten().float(), expected))
104
+ self.assertNotIn("bias", state)
105
+
106
+ def test_zero_weight_and_fail_closed_shapes(self):
107
+ weight = torch.zeros(128, 128, dtype=torch.bfloat16)
108
+ state, stats = convert_linear_weight(weight)
109
+ self.assertEqual(stats["weight_relative_l2_error"], 0)
110
+ self.assertTrue((state["qweight"] == 0).all())
111
+ for kwargs in ({"rank": 17}, {"smooth": torch.zeros(128)}, {"input_absmax": -torch.ones(128)}, {"smooth_alpha": 1.1}):
112
+ with self.assertRaises(ValueError):
113
+ convert_linear_weight(weight, **kwargs)
114
+ with self.assertRaises(ValueError):
115
+ convert_linear_weight(torch.zeros(127, 128))
116
+ with self.assertRaises(ValueError):
117
+ convert_linear_weight(torch.full((128, 128), float("nan")))
118
+
119
+ def test_reference_tensors_match_packed_export(self):
120
+ weight = torch.randn(128, 256).bfloat16()
121
+ bias = torch.randn(128).bfloat16()
122
+ state, stats, reference = convert_linear_weight(weight, bias, smooth=torch.logspace(-1, 1, 256), return_reference=True)
123
+ residual, smooth, down, up = unpack_state(state)
124
+ self.assertTrue(torch.equal(residual, reference["residual_dequant"]))
125
+ self.assertTrue(torch.equal(down, reference["down_unpacked"].float()))
126
+ self.assertTrue(torch.equal(up, reference["up_unpacked"].float()))
127
+ x = torch.randn(1, 17, 256).bfloat16()
128
+ base = ((x.float() / smooth).bfloat16().float() @ residual.T).bfloat16().float()
129
+ base = (base + bias.float()).bfloat16().float()
130
+ hidden = (x.float() @ down.T).bfloat16().float()
131
+ expected = (base + hidden @ up.T).bfloat16().float()
132
+ self.assertTrue(torch.equal(reference_forward(x, reference, quantize_activations=False), expected))
133
+ actual = reference_forward(x, reference)
134
+ self.assertTrue(torch.isfinite(actual).all())
135
+ self.assertEqual(actual.shape, (1, 17, 128))
136
+ zeros = reference_forward(torch.zeros_like(x), reference)
137
+ self.assertTrue(torch.equal(zeros, bias.float().expand_as(zeros)))
138
+
139
+
140
+ if __name__ == "__main__":
141
+ unittest.main(verbosity=2)
reproduction/nunchaku_backend/cache_repeat_diagnostic.py ADDED
@@ -0,0 +1,146 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Four controlled generation runs; instrumentation only, no runtime edits."""
2
+ import argparse
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+ import time
7
+
8
+
9
+ def tensor_fingerprint(tensor):
10
+ if tensor is None:
11
+ return None
12
+ import torch
13
+ return {"shape": list(tensor.shape), "stride": list(tensor.stride()),
14
+ "dtype": str(tensor.dtype), "device": str(tensor.device),
15
+ "sha256": hashlib.sha256(tensor.detach().contiguous().cpu().view(torch.uint8).numpy().tobytes()).hexdigest()}
16
+
17
+
18
+ def prompt_cache_fingerprint(engine):
19
+ return [{"key_sha256": hashlib.sha256(repr(key).encode()).hexdigest(),
20
+ "tensors": [tensor_fingerprint(tensor) for tensor in value]}
21
+ for key, value in engine.prompt_cache.items()]
22
+
23
+
24
+ def pixel_comparison(first, second):
25
+ import numpy as np
26
+ from PIL import Image
27
+ with Image.open(first) as image:
28
+ a = np.array(image.convert("RGBA"))
29
+ with Image.open(second) as image:
30
+ b = np.array(image.convert("RGBA"))
31
+ if a.shape != b.shape:
32
+ raise ValueError("Repeated image shape changed")
33
+ error = a.astype(np.float64) - b.astype(np.float64)
34
+ rgb = error[:, :, :3]
35
+ return {"first": str(first), "second": str(second),
36
+ "encoded_bytes_equal": Path(first).read_bytes() == Path(second).read_bytes(),
37
+ "pixels_equal": bool(np.array_equal(a, b)),
38
+ "rgb_mae_255": float(np.abs(rgb).mean()), "rgb_rmse_255": float(np.mean(rgb ** 2) ** .5),
39
+ "max_rgb_difference_255": int(np.abs(rgb).max()),
40
+ "changed_rgb_pixels_fraction": float(np.any(rgb != 0, axis=-1).mean()),
41
+ "changed_alpha_pixels": int(np.count_nonzero(error[:, :, 3]))}
42
+
43
+
44
+ def compare_fingerprints(a, b):
45
+ if a is None or b is None:
46
+ return a is None and b is None
47
+ return a["shape"] == b["shape"] and a["dtype"] == b["dtype"] and a["sha256"] == b["sha256"]
48
+
49
+
50
+ def main():
51
+ parser = argparse.ArgumentParser(description=__doc__)
52
+ parser.add_argument("--checkpoint", default="/cache/qwen-nunchaku-v3-r128")
53
+ parser.add_argument("--prequant", default="/cache/qwen-nf4")
54
+ parser.add_argument("--jobs", type=Path, default=Path("/poc/experiments/fidelity-v3/quality-jobs-all.json"))
55
+ parser.add_argument("--job", default="expanded-bilingual-festival-25")
56
+ parser.add_argument("--out", type=Path, default=Path("/poc/results/repeatability-diagnostic"))
57
+ parser.add_argument("--dry-run", action="store_true")
58
+ args = parser.parse_args()
59
+ from runner import Engine, _parse_cli_args
60
+ import torch
61
+ jobs = {job["label"]: job for job in json.loads(args.jobs.read_text())}
62
+ job = {**jobs[args.job], "cfg": 1.0, "kv_cache": True}
63
+ if job.get("images"):
64
+ raise ValueError("This controlled diagnosis intentionally isolates text-to-image prompt caching")
65
+ if args.out.exists():
66
+ raise ValueError("Use a new diagnostic output directory; prior evidence must be preserved")
67
+ engine_args = _parse_cli_args(["--backend", "nunchaku", "--nunchaku-checkpoint", args.checkpoint,
68
+ "--prequant", args.prequant, "--sample-dir", str(args.out / "samples"),
69
+ "--cache", "--no-tiling", "--lean-encoder", "--stage-offload", "--release-kv"])
70
+ if engine_args.restore_roles or engine_args.bf16_source:
71
+ raise ValueError("Remove inherited BF16 restoration environment for this exact rank128 diagnosis")
72
+ phases = [("no-cache-1", False), ("no-cache-2", False), ("cache-miss", True), ("cache-hit", True)]
73
+ if args.dry_run:
74
+ print(json.dumps({"dry_run": True, "job": job, "engine_args": vars(engine_args), "phases": phases,
75
+ "cuda_initialized": torch.cuda.is_initialized(), "gpu_work_executed": False}))
76
+ return
77
+ args.out.mkdir(parents=True)
78
+ report_path = args.out / "diagnostic.json"
79
+ report = {"complete": False, "job": job, "engine_args": vars(engine_args), "runs": [],
80
+ "timing_limitation": "Tensor hashing adds CPU/GPU transfers and synchronization; diagnostic times are not benchmark latencies",
81
+ "selection": "two no-cache runs then cache-miss and cache-hit; one Engine, identical explicit generation seed"}
82
+ def save():
83
+ temporary = report_path.with_suffix(".tmp")
84
+ temporary.write_text(json.dumps(report, indent=2) + "\n")
85
+ temporary.replace(report_path)
86
+ save()
87
+ engine = Engine(engine_args)
88
+ current = {}
89
+ original_prompt = engine.pipe._get_qwen_prompt_embeds
90
+ def prompt_wrapper(*positional, **kwargs):
91
+ outputs = original_prompt(*positional, **kwargs)
92
+ current.setdefault("prompt_calls", []).append([tensor_fingerprint(tensor) for tensor in outputs])
93
+ return outputs
94
+ engine.pipe._get_qwen_prompt_embeds = prompt_wrapper
95
+ def before_transformer(module, positional, kwargs):
96
+ current["transformer_calls"] = current.get("transformer_calls", 0) + 1
97
+ if current["transformer_calls"] == 1:
98
+ current["first_transformer_inputs"] = {name: tensor_fingerprint(kwargs.get(name)) for name in
99
+ ("hidden_states", "encoder_hidden_states", "encoder_hidden_states_mask", "img_mask", "timestep")}
100
+ current["first_transformer_metadata"] = {"img_shapes": kwargs.get("img_shapes"), "kv_cache_mode": kwargs.get("kv_cache_mode")}
101
+ def after_transformer(module, positional, output):
102
+ if current["transformer_calls"] == 1:
103
+ value = output[0] if isinstance(output, (tuple, list)) else output.sample
104
+ current["first_transformer_output"] = tensor_fingerprint(value)
105
+ pre = engine.pipe.transformer.register_forward_pre_hook(before_transformer, with_kwargs=True)
106
+ post = engine.pipe.transformer.register_forward_hook(after_transformer)
107
+ try:
108
+ for phase, enabled in phases:
109
+ engine.args.cache = enabled
110
+ if phase != "cache-hit":
111
+ engine.prompt_cache.clear()
112
+ engine.vae_cache.clear()
113
+ current = {"phase": phase, "cache_enabled": enabled, "cache_before": prompt_cache_fingerprint(engine)}
114
+ started = time.perf_counter()
115
+ metrics = engine.generate({**job, "label": "repeatability-" + phase})
116
+ current["cache_after"] = prompt_cache_fingerprint(engine)
117
+ path = Path(metrics["path"])
118
+ current.update(metrics=json.loads(json.dumps(metrics)), path=str(path), sha256=hashlib.sha256(path.read_bytes()).hexdigest(),
119
+ wall_seconds=time.perf_counter() - started)
120
+ report["runs"].append(current)
121
+ save()
122
+ print(json.dumps({"phase": phase, "sha256": current["sha256"], "prompt_cache_hits": metrics.get("prompt_cache_hits", 0)}), flush=True)
123
+ comparisons = []
124
+ for first, second in ((0, 1), (1, 2), (2, 3), (0, 3)):
125
+ a, b = report["runs"][first], report["runs"][second]
126
+ row = pixel_comparison(a["path"], b["path"])
127
+ row.update(phases=[a["phase"], b["phase"]],
128
+ prompt_tensor_equal=[compare_fingerprints(x, y) for x, y in zip(a["prompt_calls"][0], b["prompt_calls"][0])],
129
+ first_transformer_input_equal={name: compare_fingerprints(a["first_transformer_inputs"][name], b["first_transformer_inputs"][name]) for name in a["first_transformer_inputs"]},
130
+ first_transformer_output_equal=compare_fingerprints(a["first_transformer_output"], b["first_transformer_output"]))
131
+ comparisons.append(row)
132
+ report["comparisons"] = comparisons
133
+ report["cache_hit_stored_tensors_unchanged"] = report["runs"][3]["cache_before"] == report["runs"][3]["cache_after"]
134
+ report["cache_miss_after_matches_hit_before"] = report["runs"][2]["cache_after"] == report["runs"][3]["cache_before"]
135
+ report["complete"] = True
136
+ save()
137
+ print(json.dumps({"complete": True, "comparisons": comparisons,
138
+ "cache_hit_stored_tensors_unchanged": report["cache_hit_stored_tensors_unchanged"]}), flush=True)
139
+ finally:
140
+ pre.remove()
141
+ post.remove()
142
+ engine.pipe._get_qwen_prompt_embeds = original_prompt
143
+
144
+
145
+ if __name__ == "__main__":
146
+ main()
reproduction/nunchaku_backend/checkpoint_io.py ADDED
@@ -0,0 +1,28 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Shared streaming checkpoint I/O; no conversion CLI or eager Torch import."""
2
+ import json
3
+
4
+ def _json_write(path, content):
5
+ temporary = path.with_name(path.name + ".tmp")
6
+ temporary.write_text(json.dumps(content, indent=2, sort_keys=True) + "\n")
7
+ temporary.replace(path)
8
+
9
+
10
+ def _source_index(directory):
11
+ from safetensors import safe_open
12
+ result = {}
13
+ for path in sorted(directory.glob("*.safetensors")):
14
+ with safe_open(str(path), framework="pt", device="cpu") as handle:
15
+ for key in handle.keys():
16
+ if key in result:
17
+ raise ValueError(f"Duplicate source tensor {key}; provide one model variant only")
18
+ result[key] = path
19
+ if not result:
20
+ raise ValueError(f"No source safetensors in {directory}")
21
+ return result
22
+
23
+
24
+ def _read_tensor(index, key):
25
+ from safetensors import safe_open
26
+ with safe_open(str(index[key]), framework="pt", device="cpu") as handle:
27
+ return handle.get_tensor(key)
28
+
reproduction/nunchaku_backend/cleanup_test.py ADDED
@@ -0,0 +1,74 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CPU-only safety checks for archived CLI removal and runtime packaging."""
2
+ import ast
3
+ import hashlib
4
+ import json
5
+ from pathlib import Path
6
+ import shutil
7
+ import subprocess
8
+ import sys
9
+ import tempfile
10
+ import unittest
11
+
12
+
13
+ PACKAGE = Path(__file__).resolve().parent
14
+ ARCHIVE = PACKAGE.parent / "archive/2026-09-21-superseded-implementations"
15
+
16
+
17
+ def functions(path):
18
+ return {n.name: ast.dump(n, include_attributes=False) for n in ast.parse(path.read_text()).body
19
+ if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef))}
20
+
21
+
22
+ class CleanupTests(unittest.TestCase):
23
+ def test_historical_sources_are_byte_for_byte_preserved(self):
24
+ hashes = json.loads((ARCHIVE / "SHA256SUMS.json").read_text())
25
+ self.assertGreaterEqual(len(hashes), 50)
26
+ for name, expected in hashes.items():
27
+ with self.subTest(name=name):
28
+ self.assertEqual(hashlib.sha256((ARCHIVE / name).read_bytes()).hexdigest(), expected)
29
+
30
+ def test_candidate_packing_and_io_math_unchanged(self):
31
+ old = ARCHIVE / "nunchaku_backend"
32
+ original = functions(old / "convert.py")
33
+ extracted = functions(PACKAGE / "baseline_candidate.py") | functions(PACKAGE / "packing.py")
34
+ self.assertEqual(original, extracted)
35
+ old_io = functions(old / "export_checkpoint.py")
36
+ new_io = functions(PACKAGE / "checkpoint_io.py")
37
+ self.assertEqual(set(new_io), {"_json_write", "_source_index", "_read_tensor"})
38
+ for name in new_io:
39
+ self.assertEqual(old_io[name], new_io[name])
40
+
41
+ def test_active_relative_imports_resolve_without_archived_modules(self):
42
+ forbidden = {"convert", "export_checkpoint", "calibration", "calibrate"}
43
+ for path in PACKAGE.glob("*.py"):
44
+ for node in ast.walk(ast.parse(path.read_text())):
45
+ if isinstance(node, ast.ImportFrom) and node.level and node.module:
46
+ first = node.module.split(".")[0]
47
+ with self.subTest(file=path.name, dependency=first):
48
+ self.assertNotIn(first, forbidden)
49
+ self.assertTrue((PACKAGE / (first + ".py")).is_file() or (PACKAGE / first).is_dir())
50
+ for name in forbidden:
51
+ self.assertFalse((PACKAGE / (name + ".py")).exists())
52
+
53
+ def test_runtime_imports_with_only_three_release_files(self):
54
+ with tempfile.TemporaryDirectory() as directory:
55
+ target = Path(directory) / "nunchaku_backend"
56
+ target.mkdir()
57
+ for name in ("__init__.py", "runtime.py", "layout.py"):
58
+ shutil.copyfile(PACKAGE / name, target / name)
59
+ code = ("import sys; sys.path.insert(0, " + repr(directory) + "); "
60
+ "import nunchaku_backend.runtime; from nunchaku_backend.layout import BLOCK_LINEAR_PATTERN; "
61
+ "assert BLOCK_LINEAR_PATTERN.fullmatch('transformer_blocks.31.img_mlp.proj'); "
62
+ "assert not BLOCK_LINEAR_PATTERN.fullmatch('proj_out'); "
63
+ "assert 'torch' not in sys.modules; assert 'deepcompressor' not in sys.modules")
64
+ subprocess.run([sys.executable, "-I", "-c", code], check=True, capture_output=True, text=True)
65
+
66
+ def test_final_collection_and_export_cli_entry_points_still_parse(self):
67
+ for module in ("collect_v3", "export_v3"):
68
+ result = subprocess.run([sys.executable, "-m", "nunchaku_backend." + module, "--help"],
69
+ cwd=PACKAGE.parent, check=True, capture_output=True, text=True)
70
+ self.assertIn("--out", result.stdout)
71
+
72
+
73
+ if __name__ == "__main__":
74
+ unittest.main()
reproduction/nunchaku_backend/collect_v3.py ADDED
@@ -0,0 +1,81 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Collect deterministic BF16 teacher activations with disjoint prompt splits."""
2
+ import argparse,hashlib,json,math,time
3
+ from pathlib import Path
4
+ from .layout import BLOCK_LINEAR_PATTERN
5
+
6
+
7
+ def main():
8
+ p=argparse.ArgumentParser();p.add_argument('--jobs',required=True);p.add_argument('--out',required=True);p.add_argument('--steps',type=int,default=16)
9
+ args=p.parse_args()
10
+ import torch
11
+ from PIL import Image
12
+ from safetensors.torch import save_file
13
+ from .teacher_v3 import make_teacher
14
+ engine=make_teacher('/poc/samples/fidelity-v3/calibration-unused')
15
+ jobs=json.loads(Path(args.jobs).read_text());splits={'train':512,'validation':256,'heldout':256}
16
+ selected_steps=sorted({0,args.steps//4,args.steps//2,3*args.steps//4,args.steps-1})
17
+ split_counts={s:sum(j['split']==s for j in jobs) for s in splits}
18
+ samples={};amax={};provenance=[];handles=[]
19
+ state={'job':None,'step':-1,'job_index':0,'token_groups':None}
20
+ def transformer_pre(module,inputs,kwargs):
21
+ state['step']+=1;state['token_groups']=None
22
+ if state['step']==0:
23
+ mask=kwargs['img_mask'][0].detach().cpu().bool()
24
+ expanded=mask.repeat_interleave(torch.where(mask,4,1))
25
+ target=math.prod(kwargs['img_shapes'][0][-1]);prefix=len(expanded)-target
26
+ positions=torch.arange(len(expanded))
27
+ state['token_groups']=[positions[(~expanded)&(positions<prefix)],
28
+ positions[expanded&(positions<prefix)],positions[positions>=prefix]]
29
+ handles.append(engine.pipe.transformer.register_forward_pre_hook(transformer_pre,with_kwargs=True))
30
+ def hook(name):
31
+ def capture(module,inputs):
32
+ if state['step'] not in selected_steps:return
33
+ job=state['job'];split=job['split'];x=inputs[0].detach().reshape(-1,inputs[0].shape[-1])
34
+ if split=='train':
35
+ current=x.abs().amax(0).float().cpu()
36
+ amax[name]=torch.maximum(amax[name],current) if name in amax else current
37
+ quota=math.ceil(splits[split]/(split_counts[split]*len(selected_steps)))
38
+ # Same row selection for projections sharing a token sequence.
39
+ seed=123456+state['job_index']*1000+state['step']
40
+ g=torch.Generator(device='cpu').manual_seed(seed)
41
+ groups=state['token_groups']
42
+ if groups is not None:
43
+ groups=[group for group in groups if len(group)]
44
+ selected=[];remaining=quota
45
+ for i,group in enumerate(groups):
46
+ take=min(len(group),math.ceil(remaining/(len(groups)-i)))
47
+ selected.append(group[torch.randperm(len(group),generator=g)[:take]])
48
+ remaining-=take
49
+ idx=torch.cat(selected).to(x.device)
50
+ else:idx=torch.randperm(x.shape[0],generator=g)[:quota].to(x.device)
51
+ samples.setdefault(name,{}).setdefault(split,[]).append(x.index_select(0,idx).cpu().contiguous())
52
+ return capture
53
+ for name,module in engine.pipe.transformer.named_modules():
54
+ if BLOCK_LINEAR_PATTERN.fullmatch(name):handles.append(module.register_forward_pre_hook(hook(name)))
55
+ try:
56
+ for i,job in enumerate(jobs):
57
+ state.update(job=job,step=-1,job_index=i)
58
+ engine.metrics={};engine._kv_caches={}
59
+ refs=[Image.open(path).copy() for path in job.get('images',[])]
60
+ t=time.perf_counter()
61
+ output=engine.pipe(prompt=job['prompt'],image=refs or None,width=job.get('width',1024),height=job.get('height',1024),output_resolution=1024,
62
+ num_inference_steps=args.steps,true_cfg_scale=1.,use_kv_cache=True,output_type='latent',generator=torch.Generator(device='cuda').manual_seed(job['seed']))
63
+ torch.cuda.synchronize();del output;engine._kv_caches.clear();torch.cuda.empty_cache()
64
+ record={**job,'seconds':time.perf_counter()-t,'captured_steps':selected_steps}
65
+ provenance.append(record);print(json.dumps({'event':'teacher_calibration_job',**record}),flush=True)
66
+ out=Path(args.out);out.mkdir(parents=True,exist_ok=True);files={}
67
+ for name,ss in samples.items():
68
+ values={split:torch.cat(ss[split])[:count].contiguous() for split,count in splits.items()}
69
+ values['input_absmax']=amax[name]
70
+ for split,count in splits.items():
71
+ if values[split].shape[0]!=count:raise RuntimeError(f'Insufficient rows {name} {split}')
72
+ if not torch.isfinite(values[split]).all():raise RuntimeError('Nonfinite teacher activation')
73
+ filename=name+'.safetensors';save_file(values,str(out/filename))
74
+ files[name]={'file':filename,'rows':{s:len(values[s]) for s in splits},'in_features':values['train'].shape[-1]}
75
+ if len(files)!=224:raise RuntimeError('Incomplete teacher calibration')
76
+ report={'format':'qwen21-activation-v3','teacher':'BF16 DiT, same NF4 encoder; streamed-group1 offload','model_revision':'b3179ad355be050328e483a9dfdd9e60cd62adfa','steps':args.steps,'sampling':'equal job/timestep quotas; first step stratified across text, reference-image and target-image tokens; later steps uniform target tokens','split_limitation':'Calibration editing prompts differ across splits but share the same source photograph. Final image-evaluation references differ from calibration.','jobs':provenance,'layers':files,'created_unix':time.time(),'evaluation_prompts_and_references_held_out':True}
77
+ (out/'manifest.json').write_text(json.dumps(report,indent=2)+'\n');print(json.dumps({'event':'teacher_calibration_complete','layers':len(files),'out':str(out)}),flush=True)
78
+ finally:
79
+ for h in handles:h.remove()
80
+
81
+ if __name__=='__main__':main()
reproduction/nunchaku_backend/compare_iterations_v3.py ADDED
@@ -0,0 +1,79 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Compare isolated longer-fit probes against the ACTUAL saved V3 metrics."""
2
+ import argparse
3
+ import gzip
4
+ import json
5
+ from pathlib import Path
6
+
7
+
8
+ def read_manifest(path):
9
+ opener = gzip.open if str(path).endswith(".gz") else open
10
+ with opener(path, "rt") as handle:
11
+ return json.load(handle)
12
+
13
+
14
+ def layer_stats(manifest):
15
+ return {name: row for shard in manifest["files"].values() for name, row in shard.get("layer_stats", {}).items()}
16
+
17
+
18
+ def compare(v3, probes):
19
+ original = layer_stats(v3)
20
+ results = []
21
+ for probe in probes:
22
+ identity = probe["conversion_identity"]
23
+ for key in ("activation_fingerprint", "baseline_calibration_sha256", "source_config_sha256"):
24
+ if identity[key] != v3["conversion_identity"][key]:
25
+ raise ValueError(f"Probe does not match V3 {key}")
26
+ for name, current in layer_stats(probe).items():
27
+ old = original[name]
28
+ settings = current["search"]
29
+ if settings["seed"] != old["search"]["seed"]:
30
+ raise ValueError(f"Probe seed changed: {name}")
31
+ selected = old["selected"]
32
+ if settings.get("fixed_smoothing") != selected["family"] or (selected["family"] != "identity" and settings["alphas"] != [selected["alpha"]]):
33
+ raise ValueError(f"Probe smoothing differs from V3 selection: {name}")
34
+ if settings["ranks"] != [selected["rank"]] or settings["weighting"] != [selected["weighting"]]:
35
+ raise ValueError(f"Probe rank/weighting changed: {name}")
36
+ for key_name in ("niter", "oversample", "ridge", "factorization", "final_gptq", "gptq_damp"):
37
+ if settings[key_name] != old["search"][key_name]:
38
+ raise ValueError(f"Probe {key_name} changed: {name}")
39
+ def key(row):
40
+ return tuple(row.get(k) for k in ("family", "alpha", "rank", "weighting", "iteration", "output_correction", "factorization"))
41
+ old_history = {key(r): r for r in old["history"] if not r.get("gptq") and r["family"] != "one_pass_baseline"}
42
+ current_history = [r for r in current["history"] if not r.get("gptq") and r["family"] != "one_pass_baseline"]
43
+ prefix_differences = [abs(r["validation"]["mse"] - old_history[key(r)]["validation"]["mse"]) /
44
+ max(old_history[key(r)]["validation"]["mse"], 1e-30)
45
+ for r in current_history if key(r) in old_history]
46
+ if not prefix_differences:
47
+ raise ValueError(f"No matching deterministic recurrence prefix: {name}")
48
+ validation_ratio = current["validation"]["mse"] / max(old["validation"]["mse"], 1e-30)
49
+ heldout_ratio = current["heldout"]["mse"] / max(old["heldout"]["mse"], 1e-30)
50
+ results.append({"layer": name, "seed": settings["seed"], "v3_selected": old["selected"],
51
+ "probe_selected": current["selected"], "v3_hit_cap": selected["iteration"] == 15,
52
+ "max_iteration_evaluated": max(r["iteration"] for r in current_history),
53
+ "prefix_compared_candidates": len(prefix_differences),
54
+ "prefix_max_relative_validation_mse_difference": max(prefix_differences),
55
+ "v3_validation": old["validation"], "probe_validation": current["validation"],
56
+ "validation_mse_ratio_to_actual_v3": validation_ratio,
57
+ "v3_heldout": old["heldout"], "probe_heldout": current["heldout"],
58
+ "heldout_mse_ratio_to_actual_v3": heldout_ratio,
59
+ "validation_prefers_probe": validation_ratio < 1,
60
+ "heldout_improvement_exceeds_5pct": heldout_ratio < 0.95,
61
+ "probe_seconds": current["seconds"]})
62
+ return {"comparison": "100-iteration-limit fixed-recipe replay versus actual saved V3 metrics", "layers": results,
63
+ "limitations": "Validation chooses candidates; heldout is reporting only. Early stopping is unchanged and can end well before100. These are deterministic replays fromiteration0, not restored optimizer states. Final GPTQ is tried only on the selected raw candidate, so old V3 may still beat the new probe; retain V3 in that case. No full export or deployment is authorized by this report."}
64
+
65
+
66
+ def main():
67
+ parser = argparse.ArgumentParser()
68
+ parser.add_argument("--v3", type=Path, required=True)
69
+ parser.add_argument("--probes", nargs="+", type=Path, required=True)
70
+ parser.add_argument("--out", type=Path, required=True)
71
+ args = parser.parse_args()
72
+ result = compare(read_manifest(args.v3), [read_manifest(path) for path in args.probes])
73
+ args.out.parent.mkdir(parents=True, exist_ok=True)
74
+ args.out.write_text(json.dumps(result, indent=2) + "\n")
75
+ print(json.dumps(result, indent=2))
76
+
77
+
78
+ if __name__ == "__main__":
79
+ main()
reproduction/nunchaku_backend/convert_activation_reference.py ADDED
@@ -0,0 +1,102 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Independent activation-quantization reference using Nunchaku's PTX math.
2
+
3
+ No Nunchaku kernel or packed layout is reused. This launches only when called
4
+ explicitly on CUDA tensors. Source: nunchaku-ai/nunchaku v1.2.1,
5
+ src/kernels/zgemm/gemm_utils.cuh h2div/cuda_frcp/quantize_float2 and
6
+ src/kernels/zgemm/gemm_w4a4.cuh quantize_w4a4_from_fpsum_warp.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import torch
12
+
13
+ try:
14
+ import triton
15
+ import triton.language as tl
16
+ except ImportError:
17
+ triton = None
18
+ tl = None
19
+
20
+
21
+ if triton is not None:
22
+
23
+ @triton.jit
24
+ def _quantize_activation_ptx(X, Smooth, QA, SA, K: tl.constexpr, GROUPS: tl.constexpr):
25
+ block = tl.program_id(0)
26
+ row = block // GROUPS
27
+ group = block % GROUPS
28
+ channel = group * 64 + tl.arange(0, 64)
29
+ values = tl.load(X + row * K + channel).to(tl.float32)
30
+ smooth = tl.load(Smooth + channel).to(tl.float32)
31
+
32
+ # CUDA h2div uses __fdividef, then float22half2. Explicit PTX avoids
33
+ # Triton changing division to a differently-rounded reciprocal/mul.
34
+ divided = tl.inline_asm_elementwise(
35
+ "div.approx.ftz.f32 $0, $1, $2;",
36
+ constraints="=f,f,f", args=[values, smooth], dtype=tl.float32,
37
+ is_pure=True, pack=1,
38
+ )
39
+ divided = divided.to(X.dtype.element_ty).to(tl.float32)
40
+ amax = tl.max(tl.abs(divided), axis=0)
41
+
42
+ # Match C++ float RECPI_QVALUE_MAX_SIGNED = 1 / 7.0f. Keeping mul
43
+ # and reciprocal as distinct PTX operations prevents reassociation.
44
+ scale = tl.inline_asm_elementwise(
45
+ "mul.rn.f32 $0, $1, $2;",
46
+ constraints="=f,f,f",
47
+ args=[amax, tl.full((), 0.14285714285714285, tl.float32)],
48
+ dtype=tl.float32, is_pure=True, pack=1,
49
+ )
50
+ reciprocal = tl.inline_asm_elementwise(
51
+ "rcp.approx.ftz.f32 $0, $1;",
52
+ constraints="=f,f", args=[scale], dtype=tl.float32,
53
+ is_pure=True, pack=1,
54
+ )
55
+ normalized = tl.inline_asm_elementwise(
56
+ "mul.rn.f32 $0, $1, $2;",
57
+ constraints="=f,f,f", args=[divided, reciprocal], dtype=tl.float32,
58
+ is_pure=True, pack=1,
59
+ )
60
+ integer = tl.inline_asm_elementwise(
61
+ "cvt.rni.s32.f32 $0, $1;",
62
+ constraints="=r,f", args=[normalized], dtype=tl.int32,
63
+ is_pure=True, pack=1,
64
+ )
65
+ integer = tl.maximum(tl.minimum(integer, 7), -8)
66
+ # An all-zero group intentionally follows PTX's NaN->INT_MIN->-8
67
+ # conversion; its stored scale is zero, so it dequantizes to zero.
68
+ tl.store(QA + block * 64 + tl.arange(0, 64), integer.to(tl.float32))
69
+ tl.store(SA + block, scale)
70
+
71
+
72
+ @torch.no_grad()
73
+ def quantize_activations_ptx(x: torch.Tensor, smooth: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]:
74
+ """Return unpacked QA FP32[N,K/64,64], scales BF16/FP16[N,K/64].
75
+
76
+ Inputs must already be on CUDA. The function independently reproduces
77
+ Nunchaku's signed A4 mathematics; it does not call its quantizer or read
78
+ its packed tensors. This source was syntax-checked on CPU; actual GPU
79
+ validation belongs to the caller's numerical probe.
80
+ """
81
+ if triton is None:
82
+ raise ImportError("Triton is required for the explicit PTX activation reference")
83
+ if x.device.type != "cuda" or smooth.device.type != "cuda":
84
+ raise ValueError("Activation reference requires explicit CUDA input and smoothing tensors")
85
+ if x.dtype not in (torch.bfloat16, torch.float16):
86
+ raise ValueError("Input must be BF16 or FP16")
87
+ if x.ndim < 2 or x.shape[-1] % 64:
88
+ raise ValueError("Input must have at least two dimensions and width divisible by 64")
89
+ if smooth.shape != (x.shape[-1],) or smooth.device != x.device or smooth.dtype != x.dtype:
90
+ raise ValueError("Smoothing must have matching device/dtype and shape [input_width]")
91
+ k = x.shape[-1]
92
+ flattened = x.reshape(-1, k).contiguous()
93
+ groups = k // 64
94
+ qa = torch.empty((flattened.shape[0], groups, 64), device=x.device, dtype=torch.float32)
95
+ sa = torch.empty((flattened.shape[0], groups), device=x.device, dtype=x.dtype)
96
+ if flattened.shape[0] == 0:
97
+ return qa, sa
98
+ _quantize_activation_ptx[(flattened.shape[0] * groups,)](
99
+ flattened, smooth.contiguous(), qa, sa, K=k, GROUPS=groups,
100
+ num_warps=4, num_stages=1,
101
+ )
102
+ return qa, sa
reproduction/nunchaku_backend/convert_recipe_v3.md ADDED
@@ -0,0 +1,160 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Pinned DeepCompressor SVDQuant recipe and Qwen v3 implications
2
+
3
+ Source review of DeepCompressor commit
4
+ `69f3473f5e1c1504bae35cc50c7858ef900a9b17`. This is a description of source
5
+ behavior, not an image-quality or latency measurement. No stable converter,
6
+ checkpoint, or GPU workload was modified for this review.
7
+
8
+ ## What the official example configuration actually requests
9
+
10
+ The base SVDQuant example selects rank 32, output-error calibration, up to
11
+ 100 low-rank iterations, and early stopping. Smoothing uses absolute channel
12
+ maxima for activations and weights, a grid search, low-rank-aware evaluation,
13
+ and `fuse_when_possible=false`.
14
+
15
+ Its `alpha: 0.5`, `beta: -2`, `num_grids: 20` do **not** specify one alpha=0.5
16
+ conversion. Those flags generate 39 smoothing candidates:
17
+
18
+ - identity `(alpha,beta)=(0,0)`;
19
+ - 19 activation-only candidates `(alpha,0)` with alpha=0.05 through 0.95;
20
+ - 19 balanced candidates `(alpha,1-alpha)` on the same grid.
21
+
22
+ The scale is `amax(X)^alpha / amax(W)^beta`. Invalid/zero scale handling in
23
+ the source differs from our bounded positive-clamp implementation. The fast
24
+ configuration uses 10 grids and 64 calibration samples; the general default
25
+ uses 128 calibration samples.
26
+
27
+ Sources: [base SVDQuant YAML](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/examples/diffusion/configs/svdquant/__default__.yaml),
28
+ [alpha/beta candidate generation](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/calib/config/smooth.py#L134),
29
+ [fast override](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/examples/diffusion/configs/svdquant/fast.yaml).
30
+
31
+ ## Exact iterative low-rank algorithm
32
+
33
+ In the following, W already includes any preceding smoothing. Q denotes a
34
+ dequantized quantized residual; L denotes the effective low-rank weight.
35
+
36
+ ```text
37
+ Q = 0 # default compensate=false
38
+ # alternate optional initialization: Q = Quantize(W) when compensate=true
39
+ best = none
40
+ repeat up to num_iters:
41
+ L = rank_r_SVD(W - Q)
42
+ Q = Quantize(W - L, kernel=None)
43
+ evaluate candidate module with weight Q, low-rank branch L,
44
+ and configured activation quantization
45
+ retain L if output-error sum <= best_error
46
+ stop on the first worse candidate when early_stop=true
47
+ return best L
48
+ ```
49
+
50
+ This is alternating quantization and SVD, not gradient training or an
51
+ activation-weighted SVD. The next SVD acts on `W - previous_Q`, so rerunning
52
+ SVD on W is not equivalent. Candidate generation continues from the current Q,
53
+ not from the best Q, unless early stopping ends the loop.
54
+
55
+ The actual `LowRankBranch` implementation computes full FP64 SVD and stores
56
+ `down=Vh[:r]`, `up=U[:,:r]*singular_values[:r]` in the original parameter
57
+ dtype. It does not split sqrt(singular_values) between the factors. Both
58
+ factorizations have the same ideal matrix product, but their BF16 hidden
59
+ activation rounding differs. The current custom converter's randomized SVD
60
+ and balanced factors are implementation choices, not exact copies of this
61
+ reference recipe.
62
+
63
+ For the output-error objective, the candidate module's residual weight is Q.
64
+ A branch hook captures input before activation-quantizer hooks and adds the
65
+ low-rank output afterward. The branch therefore sees unquantized input.
66
+ For weight/product objectives the candidate representation can instead be
67
+ assembled as Q+L; the default SVDQuant recipe uses output error.
68
+
69
+ The winning branch is later subtracted from the real module weight, and the
70
+ remaining weight is quantized during the final quantization phase. There is
71
+ no guarantee that selecting by weight reconstruction error picks the same
72
+ branch as selecting by output error.
73
+
74
+ Sources: [low-rank calibrator](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/calib/lowrank.py#L80),
75
+ [SVD branch factorization](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/nn/patch/lowrank.py#L31),
76
+ [input-capturing branch hook](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/utils/hooks/branch.py#L28),
77
+ [applying the selected branch](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/app/diffusion/quant/weight.py#L105).
78
+
79
+ ## What smoothing and output-error evaluation include
80
+
81
+ For each smoothing candidate, weights are scaled by s and inputs are divided
82
+ by s. With `allow_low_rank=true`, a low-rank branch and quantized residual are
83
+ constructed for the candidate. This inner smoothing evaluation uses
84
+ `quantize_with_low_rank`, which performs one SVD/residual quantization, not the
85
+ whole up-to-100-iteration low-rank search. The activation quantizer is installed
86
+ after the low-rank input capture.
87
+
88
+ The generic output-error search evaluates the selected module on cached
89
+ inputs and sums squared differences from its original outputs. For Q/K
90
+ projections, the diffusion wrapper can evaluate the enclosing attention or
91
+ parallel transformer module instead of only the linear projection. Shared-input
92
+ Q/K/V weights can be concatenated for a shared down projection when
93
+ `exclusive=false`. Other linears are usually evaluated individually.
94
+
95
+ The pinned recipe's PyTorch quantization simulation does not reproduce our
96
+ observed Nunchaku groupwise BF16 accumulation and PTX reciprocal bit for bit.
97
+ Using the verified custom numerical reference to select Qwen candidates is a
98
+ justified adaptation; it must be described as such.
99
+
100
+ Sources: [low-rank-aware smoothing module evaluation](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/calib/smooth.py#L578),
101
+ [one-pass quantize_with_low_rank](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/quantizer/processor.py#L215),
102
+ [output-error sum](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/calib/search.py#L817),
103
+ [attention evaluation scope](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/app/diffusion/quant/weight.py#L50).
104
+
105
+ ## GPTQ is optional and comes later
106
+
107
+ The optional GPTQ configuration uses damping 0.01, block size 128, up to
108
+ 250 inversion attempts, and a Hessian **sample accumulation chunk** of 512.
109
+ `hessian_block_size=512` does not make the Hessian block diagonal: the source
110
+ still allocates the full K×K Hessian.
111
+
112
+ The implementation builds `H = sum(2/N * X.T @ X)` from cached input samples,
113
+ handles zero diagonal channels, sorts channels by descending Hessian diagonal,
114
+ damps the diagonal, and takes an upper Cholesky factor of the inverse.
115
+ It quantizes columns sequentially, propagates each column's quantization error
116
+ through the inverse-Hessian factor, propagates completed block error to the
117
+ remaining columns, then restores the original channel order. Quantization
118
+ scales are addressed using original column-group indices even while columns
119
+ are permuted.
120
+
121
+ Both smoothing and iterative low-rank candidate quantization explicitly pass
122
+ `kernel=None`; GPTQ therefore is not active inside those searches. The final
123
+ weight quantization call can use the configured GPTQ kernel after the selected
124
+ branch has been subtracted. It minimizes weight-product error using input
125
+ covariance; it does not directly optimize the error from rounding inputs to A4.
126
+
127
+ For Qwen's K=12288 MLP output, one full FP32 Hessian is 576 MiB, before copies,
128
+ Cholesky workspace or weights. SVD, covariance storage and inversion deserve
129
+ their own timing and peak-memory measurements.
130
+
131
+ Sources: [optional GPTQ YAML](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/examples/diffusion/configs/svdquant/gptq.yaml),
132
+ [GPTQ kernel](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/quantizer/kernel/gptq.py#L129),
133
+ [final quantization call](https://github.com/nunchaku-ai/deepcompressor/blob/69f3473f5e1c1504bae35cc50c7858ef900a9b17/deepcompressor/app/diffusion/quant/weight.py#L239).
134
+
135
+ ## Practical Qwen v3 priorities
136
+
137
+ 1. Search identity, activation-only smoothing and balanced smoothing with
138
+ representative generation/editing activations. Keep separate held-out
139
+ prompts/timesteps for selection verification; a large token count from a
140
+ single request is not equivalent to diverse requests.
141
+ 2. Score the complete A4 residual plus BF16 low-rank output, including actual
142
+ rounding boundaries. Preserve the best candidate and iteration, not the
143
+ last. Add alternating `L=SVD(W-Q)` iterations after one-pass smoothing
144
+ selection; start with a bounded iteration budget and measured early stopping.
145
+ 3. Compare upstream-style `(U*S,Vh)` and balanced factorization under that
146
+ output metric. Retain the final BF16 inverse-smoothing correction. Consider
147
+ higher ranks only with measured held-out improvement and VRAM/latency cost.
148
+ 4. Diagnose the worst linears separately. If activation error dominates, GPTQ
149
+ alone will not solve it. A held-out-scored fit of the low-rank output to
150
+ residual output error can be tested, but that is an additional method, not
151
+ the pinned DeepCompressor algorithm.
152
+ 5. Add optional final residual GPTQ only after the simpler calibrated baseline
153
+ passes. Preserve original channel order and group64 scales for Nunchaku;
154
+ build the Hessian in the correct smoothed coordinate system.
155
+
156
+ The official INT4 example also enables activation shifting and unsigned
157
+ quantization. Do not copy that by toggling generic
158
+ `SVDQW4A4Linear.act_unsigned`: its standalone quantizer instantiates signed
159
+ activation quantization and does not accept that flag. Unsigned inference needs
160
+ a compatible quantizer and correct shift/bias handling, not just a GEMM flag.
reproduction/nunchaku_backend/convert_reference.py ADDED
@@ -0,0 +1,72 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Groupwise BF16 CUDA arithmetic reference; does not alter exported weights.
2
+
3
+ Nunchaku gemm_w4a4_kernel selects USE_FP32_ACCUM=false. The kernel rounds the
4
+ INT32 dot and the product of scales to BF16/FP16, then accumulates with a
5
+ BF16/FP16 FMA after *each* group64. This differs from one FP32 GEMM rounded
6
+ only at the end. See gemm_base.cuh apply_scales(fpsum_warp&) and
7
+ gemm_w4a4.cuh gemm_w4a4_kernel at Nunchaku commit
8
+ 302e0e97024ebd68688fe890e5df83731edf7b54.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import torch
14
+
15
+
16
+ @torch.no_grad()
17
+ def reference_forward_groupwise(x: torch.Tensor, reference: dict[str, torch.Tensor]) -> torch.Tensor:
18
+ """Emulate signed INT4 group64 arithmetic and LoRA epilogue boundaries.
19
+
20
+ Returns compute-dtype-rounded FP32 output. CUDA reciprocal approximation
21
+ and low-rank atomic reduction order still need not be bit-identical.
22
+ This accepts the existing converter's unpacked reference without changing
23
+ the conversion/checkpoint. Weight scales recover exactly from nonzero
24
+ groups because our symmetric converter always includes a +/-7 extremum.
25
+ A guard rejects reference tensors where that assumption does not hold.
26
+ """
27
+ if x.dtype not in (torch.bfloat16, torch.float16):
28
+ raise ValueError("BF16/FP16 input required")
29
+ dtype, device = x.dtype, x.device
30
+ ic = x.shape[-1]
31
+ if ic % 64:
32
+ raise ValueError("Input width must be a multiple of 64")
33
+ smooth = reference["smooth"].to(device=device, dtype=torch.float32)
34
+ original = x.float()
35
+ smoothed = (original / smooth).to(dtype).float()
36
+ grouped = smoothed.reshape(-1, ic // 64, 64)
37
+ scales_fp32 = grouped.abs().amax(-1) * (1.0 / 7.0)
38
+ reciprocal = torch.where(scales_fp32 > 0, scales_fp32.reciprocal(), torch.zeros_like(scales_fp32))
39
+ qa = (grouped * reciprocal.unsqueeze(-1)).round().clamp(-8, 7)
40
+ sa = scales_fp32.to(dtype).float()
41
+ if device.type == "cuda":
42
+ from .convert_activation_reference import quantize_activations_ptx
43
+ qa, sa = quantize_activations_ptx(x, smooth.to(dtype))
44
+ sa = sa.float()
45
+
46
+ residual = reference["residual_dequant"].to(device=device, dtype=torch.float32)
47
+ oc = residual.shape[0]
48
+ grouped_weight = residual.reshape(oc, ic // 64, 64)
49
+ sw = (grouped_weight.abs().amax(-1) / 7.0).to(dtype).float()
50
+ sw = torch.where(sw > 0, sw, torch.ones_like(sw))
51
+ qw = (grouped_weight / sw.unsqueeze(-1)).round()
52
+ if (qw.abs() > 7).any() or not torch.equal(qw * sw.unsqueeze(-1), grouped_weight):
53
+ raise ValueError("Reference residual does not match symmetric group64 +/-7 quantization")
54
+
55
+ output = torch.zeros((grouped.shape[0], oc), dtype=torch.float32, device=device)
56
+ for group in range(ic // 64):
57
+ # Small integer values are exact in FP32 dot accumulation here.
58
+ int_dot = qa[:, group] @ qw[:, group].T
59
+ rounded_dot = int_dot.to(dtype).float()
60
+ scale_product = (sa[:, group, None] * sw[None, :, group]).to(dtype).float()
61
+ # CUDA __hfma2 has a single output rounding, not BF16 mul then add.
62
+ # FP64 represents these BF16/FP16 operands closely enough to avoid
63
+ # an extra FP32 rounding before the final compute-dtype conversion.
64
+ output = (rounded_dot.double() * scale_product.double() + output.double()).to(dtype).float()
65
+
66
+ output = output.reshape(*x.shape[:-1], oc)
67
+ if "bias" in reference:
68
+ output = (output + reference["bias"].to(device=device, dtype=torch.float32)).to(dtype).float()
69
+ down = reference["down_unpacked"].to(device=device, dtype=torch.float32)
70
+ up = reference["up_unpacked"].to(device=device, dtype=torch.float32)
71
+ hidden = (original @ down.T).to(dtype).float()
72
+ return (output + hidden @ up.T).to(dtype).float()
reproduction/nunchaku_backend/convert_reference_notes.md ADDED
@@ -0,0 +1,54 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Groupwise Nunchaku arithmetic reference
2
+
3
+ The first GPU probe reported finite output but actual/reference relative L2
4
+ differences of roughly 0.009–0.016, increasing with input width. The original
5
+ FP32 reference omitted a concrete kernel behavior: the INT4 residual branch
6
+ uses BF16 accumulation after every 64-channel group.
7
+
8
+ `convert_reference.py::reference_forward_groupwise` is a standalone replacement
9
+ reference. It does not modify conversion or existing checkpoints. The GPU
10
+ probe still needs rerunning against this reference before attributing the
11
+ observed discrepancy to accumulation alone.
12
+
13
+ For each group, the source computes:
14
+
15
+ ```text
16
+ integer_dot = INT32 dot(qA_group, qW_group)
17
+ dot = BF16(integer_dot)
18
+ scale = BF16(BF16_activation_scale * BF16_weight_scale)
19
+ running = BF16_FMA(dot, scale, running)
20
+ ```
21
+
22
+ Groups accumulate sequentially from 0 through K/64 - 1. The two-stage loop
23
+ prefetches into alternating slots but does not reorder the arithmetic.
24
+ After that, the existing BF16 bias and low-rank epilogue boundaries apply.
25
+
26
+ The emulator uses FP64 multiplication plus addition followed by BF16 rounding
27
+ to represent a single BF16 fused multiply-add without introducing a separate
28
+ rounded multiply. It recovers INT4 weight groups and scales from the existing
29
+ unpacked residual, using the converter's nonzero-group +/-7 extremum. An exact
30
+ reconstruction guard rejects tensors for which that assumption fails.
31
+
32
+ CPU tests passed with Torch 2.14.0. One regression constructs sequential group
33
+ contributions near 256, 1 and 1: BF16 running accumulation gives 256, while
34
+ the old full FP32 sum rounded at the end gives 258. CUDA reciprocal
35
+ approximation and low-rank atomic reduction order can still create residual
36
+ differences; this reference is not claimed bit-exact before a GPU comparison.
37
+
38
+ Source anchors at Nunchaku commit 302e0e97024ebd68688fe890e5df83731edf7b54:
39
+
40
+ - [Kernel explicitly selects USE_FP32_ACCUM=false, near line 1080](https://github.com/nunchaku-ai/nunchaku/blob/302e0e97024ebd68688fe890e5df83731edf7b54/src/kernels/zgemm/gemm_w4a4.cuh#L1080)
41
+ - [Per-group BF16 dot conversion, scale multiplication and fused addition](https://github.com/nunchaku-ai/nunchaku/blob/302e0e97024ebd68688fe890e5df83731edf7b54/src/kernels/zgemm/gemm_base.cuh#L369)
42
+ - [Sequential two-stage group accumulation](https://github.com/nunchaku-ai/nunchaku/blob/302e0e97024ebd68688fe890e5df83731edf7b54/src/kernels/zgemm/gemm_w4a4.cuh#L884)
43
+
44
+ ## Subsequent PTX validation
45
+
46
+ The groupwise correction reduced the mismatch but retained activation-boundary
47
+ rounding differences. The CUDA reference now delegates activation math to
48
+ `convert_activation_reference.py`, an independent Triton implementation using
49
+ PTX `div.approx.ftz.f32`, `rcp.approx.ftz.f32`, and integer ties-to-even conversion.
50
+ This resolved the discrepancy without checkpoint changes or tolerance increases:
51
+ all nine GPU cases passed, relative L2 from 1.19e-6 to 5.57e-5. The worst absolute
52
+ error point in every case differed by one BF16 ULP. The full result is preserved
53
+ in `../results/nunchaku-kernel-probe-ptx.json`; earlier failed references remain
54
+ alongside it for comparison.
reproduction/nunchaku_backend/convert_reference_test.py ADDED
@@ -0,0 +1,59 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """CPU sanity checks for the groupwise CUDA arithmetic emulator."""
2
+
3
+ import unittest
4
+
5
+ import torch
6
+
7
+ if __package__:
8
+ from .baseline_candidate import convert_linear_weight, reference_forward
9
+ from .convert_reference import reference_forward_groupwise
10
+ else:
11
+ from baseline_candidate import convert_linear_weight, reference_forward
12
+ from convert_reference import reference_forward_groupwise
13
+
14
+
15
+ class GroupwiseReferenceTests(unittest.TestCase):
16
+ def test_emulator_handles_export_zero_groups_and_nonuniform_smoothing(self):
17
+ torch.set_num_threads(2)
18
+ torch.manual_seed(88)
19
+ weight = torch.randn(128, 256).bfloat16()
20
+ bias = torch.randn(128).bfloat16()
21
+ _, _, ref = convert_linear_weight(weight, bias, smooth=torch.logspace(-1, 1, 256), return_reference=True)
22
+ x = torch.randn(1, 17, 256).bfloat16()
23
+ out = reference_forward_groupwise(x, ref)
24
+ self.assertEqual(out.shape, (1, 17, 128))
25
+ self.assertTrue(torch.isfinite(out).all())
26
+ self.assertTrue(torch.equal(reference_forward_groupwise(torch.zeros_like(x), ref), bias.float().expand_as(out)))
27
+ # Group rounding should produce a small, nonzero change from the
28
+ # previous approximation, not a different scale or packing order.
29
+ old = reference_forward(x, ref)
30
+ difference = (out - old).norm() / old.norm()
31
+ self.assertGreater(difference.item(), 0)
32
+ self.assertLess(difference.item(), 0.02)
33
+
34
+ def test_per_group_accumulator_rounding_is_observable(self):
35
+ # Group0 contributes 256; groups1/2 each add1. BF16 running FMA
36
+ # ties-to-even rounds 257 back to256 twice. Full FP32 sum gives258.
37
+ q = torch.zeros(128, 192)
38
+ q[:, 0] = 7
39
+ q[:, 64] = 7
40
+ q[:, 128] = 7
41
+ sw = torch.tensor([256.0 / 49, 1.0 / 49, 1.0 / 49]).bfloat16().float()
42
+ ref = {
43
+ "residual_dequant": (q.reshape(128, 3, 64) * sw[None, :, None]).reshape(128, 192),
44
+ "smooth": torch.ones(192, dtype=torch.bfloat16),
45
+ "down_unpacked": torch.zeros(32, 192, dtype=torch.bfloat16),
46
+ "up_unpacked": torch.zeros(128, 32, dtype=torch.bfloat16),
47
+ }
48
+ x = torch.zeros(1, 1, 192, dtype=torch.bfloat16)
49
+ x[..., 0] = 7
50
+ x[..., 64] = 7
51
+ x[..., 128] = 7
52
+ out = reference_forward_groupwise(x, ref)
53
+ old = reference_forward(x, ref)
54
+ self.assertTrue((out == 256).all())
55
+ self.assertTrue((old == 258).all())
56
+
57
+
58
+ if __name__ == "__main__":
59
+ unittest.main(verbosity=2)
reproduction/nunchaku_backend/denoiser_probe_v3.py ADDED
@@ -0,0 +1,391 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Bounded BF16-versus-Nunchaku denoiser prediction diagnostic.
2
+
3
+ Run only in the isolated POC container after its GPU has been reserved::
4
+
5
+ python -m nunchaku_backend.denoiser_probe_v3 capture \
6
+ --jobs experiments/fidelity-v3/jobs.json --label heldout-example \
7
+ --out /cache/denoiser-probe-v3
8
+ python -m nunchaku_backend.denoiser_probe_v3 compare \
9
+ --capture /cache/denoiser-probe-v3 \
10
+ --checkpoint v2=/cache/qwen21-nunchaku-r32 \
11
+ --checkpoint v3=/cache/qwen21-nunchaku-v3 \
12
+ --out results/denoiser-probe-v3.json
13
+
14
+ Capture retains exactly first/middle/last transformer inputs and target-token
15
+ predictions. It NEVER stores teacher KV. Each cached-step comparison rebuilds
16
+ a fresh prefix using the selected quantized checkpoint's first-step forward.
17
+ Later inputs remain teacher-trajectory inputs: this measures conditional
18
+ prediction error, not accumulated free-running trajectory/image error.
19
+
20
+ Importing this file does not import Torch or start GPU work. The `capture` and
21
+ `compare` subcommands explicitly execute on the container's cuda:0 only.
22
+ """
23
+ from __future__ import annotations
24
+
25
+ import argparse
26
+ import gc
27
+ import hashlib
28
+ import json
29
+ import math
30
+ from pathlib import Path
31
+ import time
32
+
33
+
34
+ FORMAT = "qwen21-denoiser-teacher-capture-v3"
35
+
36
+
37
+ def _write_json(path, value):
38
+ path = Path(path)
39
+ path.parent.mkdir(parents=True, exist_ok=True)
40
+ temporary = path.with_suffix(path.suffix + ".tmp")
41
+ temporary.write_text(json.dumps(value, indent=2, sort_keys=True) + "\n")
42
+ temporary.replace(path)
43
+
44
+
45
+ def _hash_file(path):
46
+ digest = hashlib.sha256()
47
+ with Path(path).open("rb") as handle:
48
+ for block in iter(lambda: handle.read(1024 * 1024), b""):
49
+ digest.update(block)
50
+ return digest.hexdigest()
51
+
52
+
53
+ def _tree(value, torch, device):
54
+ """Copy supported forward data; reject accidentally captured cache objects."""
55
+ if isinstance(value, torch.Tensor):
56
+ return value.detach().to(device=device, copy=True).contiguous()
57
+ if isinstance(value, dict):
58
+ if "kv_cache" in value:
59
+ raise ValueError("A teacher KV cache must never enter a saved input tree")
60
+ return {key: _tree(item, torch, device) for key, item in value.items()}
61
+ if isinstance(value, tuple):
62
+ return tuple(_tree(item, torch, device) for item in value)
63
+ if isinstance(value, list):
64
+ return [_tree(item, torch, device) for item in value]
65
+ if value is None or isinstance(value, (str, int, float, bool)):
66
+ return value
67
+ raise TypeError(f"Unsupported forward data: {type(value).__name__}")
68
+
69
+
70
+ def _tensor_bytes(value, torch):
71
+ if isinstance(value, torch.Tensor):
72
+ return value.numel() * value.element_size()
73
+ if isinstance(value, dict):
74
+ return sum(_tensor_bytes(item, torch) for item in value.values())
75
+ if isinstance(value, (list, tuple)):
76
+ return sum(_tensor_bytes(item, torch) for item in value)
77
+ return 0
78
+
79
+
80
+ def _target_prediction(output, kwargs):
81
+ """Match pipeline noise_pred[:, -latents.size(1):], including edit prefill."""
82
+ prediction = output[0] if isinstance(output, tuple) else output.sample
83
+ target_tokens = math.prod(kwargs["img_shapes"][0][-1])
84
+ if prediction.ndim != 3 or prediction.shape[1] < target_tokens:
85
+ raise ValueError(f"Invalid denoiser prediction shape {prediction.shape}; target {target_tokens}")
86
+ return prediction[:, -target_tokens:]
87
+
88
+
89
+ def _clear_cache(cache):
90
+ if cache is not None:
91
+ for layer in cache.layer_caches:
92
+ layer.k = None
93
+ layer.v = None
94
+
95
+
96
+ def _cache_summary(cache):
97
+ layers = cache.layer_caches
98
+ if not layers or any(layer.k is None or layer.v is None for layer in layers):
99
+ raise RuntimeError("Quantized prefill did not populate every prefix KV layer")
100
+ return {
101
+ "layers": len(layers),
102
+ "tensor_bytes": sum(t.numel() * t.element_size() for layer in layers for t in (layer.k, layer.v)),
103
+ "first_key_shape": list(layers[0].k.shape),
104
+ "dtype": str(layers[0].k.dtype),
105
+ }
106
+
107
+
108
+ def _metrics(actual, teacher, torch):
109
+ if actual.shape != teacher.shape:
110
+ raise ValueError(f"Prediction shapes differ: {actual.shape} versus {teacher.shape}")
111
+ actual = actual.detach().to(device="cpu", dtype=torch.float64).reshape(-1)
112
+ teacher = teacher.detach().to(device="cpu", dtype=torch.float64).reshape(-1)
113
+ if not bool(torch.isfinite(actual).all() and torch.isfinite(teacher).all()):
114
+ raise RuntimeError("Non-finite denoiser prediction")
115
+ delta = actual - teacher
116
+ pred_norm, teacher_norm = actual.norm().item(), teacher.norm().item()
117
+ return {
118
+ "relative_l2": delta.norm().item() / max(teacher_norm, 1e-30),
119
+ "cosine_similarity": float(torch.dot(actual, teacher)) / max(pred_norm * teacher_norm, 1e-30),
120
+ "max_abs_error": delta.abs().max().item(),
121
+ "mse": delta.square().mean().item(),
122
+ "teacher_l2_norm": teacher_norm,
123
+ "prediction_l2_norm": pred_norm,
124
+ "teacher_rms": teacher.square().mean().sqrt().item(),
125
+ "prediction_rms": actual.square().mean().sqrt().item(),
126
+ "teacher_max_abs": teacher.abs().max().item(),
127
+ "prediction_max_abs": actual.abs().max().item(),
128
+ "elements": actual.numel(),
129
+ "finite": True,
130
+ }
131
+
132
+
133
+ def capture(args):
134
+ import torch
135
+ from PIL import Image
136
+ from .teacher_v3 import make_teacher
137
+
138
+ jobs = json.loads(Path(args.jobs).read_text())
139
+ matches = [job for job in jobs if job.get("label") == args.label]
140
+ if len(matches) != 1:
141
+ raise ValueError("Choose exactly one job by its unique --label")
142
+ job = matches[0]
143
+ steps = int(job.get("steps", 40))
144
+ if steps < 3 or float(job.get("cfg", 1.0)) != 1.0 or job.get("kv_cache", True) is not True:
145
+ raise ValueError("This bounded probe requires at least 3 steps, CFG 1, and KV caching")
146
+ if (job.get("width", 1024), job.get("height", 1024)) != (1024, 1024):
147
+ raise ValueError("This held-out diagnostic deliberately requires a 1024×1024 job")
148
+ directory = Path(args.out)
149
+ directory.mkdir(parents=True, exist_ok=True)
150
+ if any(directory.iterdir()):
151
+ raise ValueError("Capture output must be a new empty directory")
152
+ if args.max_capture_mib <= 0:
153
+ raise ValueError("Capture memory bound must be positive")
154
+ selected = sorted({0, steps // 2, steps - 1})
155
+ metadata = {
156
+ "format": FORMAT, "complete": False, "job": job,
157
+ "jobs_file": str(Path(args.jobs).resolve()), "jobs_sha256": _hash_file(args.jobs),
158
+ "selected_steps_zero_based": selected, "teacher": "original BF16 DiT; fixed NF4 encoder",
159
+ "teacher_offload": "streamed group1" if not args.no_stream else "synchronous group4",
160
+ "cache_policy": "Teacher KV never saved; independently rebuilt by every compared backend",
161
+ "maximum_capture_mib": args.max_capture_mib, "captures": [],
162
+ "reference_sha256": {str(path): _hash_file(path) for path in job.get("images", [])},
163
+ "claim": "Conditional prediction diagnostic; not free-running image quality or clean inference timing",
164
+ }
165
+ _write_json(directory / "capture.json", metadata)
166
+ engine = make_teacher(stream=not args.no_stream)
167
+ engine.metrics = {}
168
+ engine._kv_caches = {}
169
+ transformer = engine.pipe.transformer
170
+ if not transformer.config.causal_condition:
171
+ raise ValueError("Prefix replay requires causal_condition=True")
172
+ pending = None
173
+ call_index = -1
174
+ bytes_captured = 0
175
+
176
+ def before(module, positional, kwargs):
177
+ nonlocal call_index, pending
178
+ call_index += 1
179
+ if positional:
180
+ raise ValueError("Pinned pipeline must pass transformer inputs by keyword")
181
+ expected = "extract" if call_index == 0 else "cached"
182
+ if kwargs.get("kv_cache_mode") != expected or kwargs.get("kv_cache") is None:
183
+ raise ValueError("Unexpected teacher call/cache sequence; CFG or pipeline contract changed")
184
+ if call_index not in selected:
185
+ return
186
+ allowed = {key: value for key, value in kwargs.items() if key != "kv_cache"}
187
+ # Check size before copying; three bounded snapshots are the only
188
+ # trajectory data retained, and each is written immediately to disk.
189
+ input_bytes = _tensor_bytes(allowed, torch)
190
+ if bytes_captured + input_bytes > args.max_capture_mib * 2**20:
191
+ raise MemoryError("Teacher capture input exceeds the configured bound")
192
+ pending = _tree(allowed, torch, "cpu")
193
+
194
+ def after(module, positional, kwargs, output):
195
+ nonlocal pending, bytes_captured
196
+ if call_index not in selected:
197
+ return
198
+ if pending is None:
199
+ raise RuntimeError("Missing bounded input snapshot")
200
+ prediction = _target_prediction(output, kwargs)
201
+ new_bytes = _tensor_bytes(pending, torch) + prediction.numel() * prediction.element_size()
202
+ if bytes_captured + new_bytes > args.max_capture_mib * 2**20:
203
+ raise MemoryError("Teacher capture output exceeds the configured bound")
204
+ payload = {"kwargs": pending, "teacher_target_prediction": prediction.detach().cpu().clone(),
205
+ "step_zero_based": call_index}
206
+ filename = f"step-{call_index:03d}.pt"
207
+ torch.save(payload, directory / filename)
208
+ bytes_captured += new_bytes
209
+ metadata["captures"].append({
210
+ "step_zero_based": call_index, "file": filename, "sha256": _hash_file(directory / filename),
211
+ "timestep": pending["timestep"].tolist(), "cache_mode": pending["kv_cache_mode"],
212
+ "target_shape": list(prediction.shape), "tensor_bytes": new_bytes,
213
+ })
214
+ _write_json(directory / "capture.json", metadata)
215
+ print(json.dumps({"event": "denoiser_teacher_capture", **metadata["captures"][-1]}), flush=True)
216
+ pending = None
217
+
218
+ handles = [transformer.register_forward_pre_hook(before, with_kwargs=True),
219
+ transformer.register_forward_hook(after, with_kwargs=True)]
220
+ references = []
221
+ for path in job.get("images", []):
222
+ with Image.open(path) as image:
223
+ references.append(image.copy())
224
+ started = time.perf_counter()
225
+ try:
226
+ with torch.inference_mode():
227
+ result = engine.pipe(
228
+ prompt=job["prompt"], image=references or None, width=1024, height=1024,
229
+ output_resolution=1024, num_inference_steps=steps, true_cfg_scale=1.0,
230
+ generator=torch.Generator(device="cuda:0").manual_seed(job.get("seed", 42)),
231
+ use_kv_cache=True, output_type="latent",
232
+ )
233
+ del result
234
+ torch.cuda.synchronize()
235
+ if call_index + 1 != steps or [row["step_zero_based"] for row in metadata["captures"]] != selected:
236
+ raise RuntimeError("Teacher trajectory/capture count does not match requested steps")
237
+ metadata.update(complete=True, tensor_bytes=bytes_captured,
238
+ instrumented_seconds=time.perf_counter() - started,
239
+ transformer_calls=call_index + 1)
240
+ _write_json(directory / "capture.json", metadata)
241
+ finally:
242
+ for handle in handles:
243
+ handle.remove()
244
+ for cache in engine._kv_caches.values():
245
+ _clear_cache(cache)
246
+ engine._kv_caches.clear()
247
+ engine.pipe.maybe_free_model_hooks()
248
+ pending = None
249
+ gc.collect()
250
+ torch.cuda.empty_cache()
251
+
252
+
253
+ def _load_capture(directory, record, torch):
254
+ path = directory / record["file"]
255
+ if path.parent.resolve() != directory.resolve() or _hash_file(path) != record["sha256"]:
256
+ raise ValueError("Capture path or checksum mismatch")
257
+ payload = torch.load(path, map_location="cpu", weights_only=True)
258
+ if "kv_cache" in payload["kwargs"]:
259
+ raise ValueError("Refuse to reuse captured teacher KV")
260
+ if payload["step_zero_based"] != record["step_zero_based"]:
261
+ raise ValueError("Capture step metadata mismatch")
262
+ return payload
263
+
264
+
265
+ def _checkpoint_spec(spec):
266
+ name, separator, path = spec.partition("=")
267
+ if not separator or not name or not path:
268
+ raise ValueError("Checkpoint must be NAME=/absolute/checkpoint/path")
269
+ return name, Path(path)
270
+
271
+
272
+ def compare(args):
273
+ import torch
274
+ from diffusers.models.transformers.transformer_qwenimage21 import QwenImage21KVCache
275
+ from .runtime import load_transformer
276
+
277
+ directory = Path(args.capture)
278
+ metadata = json.loads((directory / "capture.json").read_text())
279
+ if metadata.get("format") != FORMAT or metadata.get("complete") is not True:
280
+ raise ValueError("Teacher capture is incomplete or incompatible")
281
+ records = sorted(metadata["captures"], key=lambda row: row["step_zero_based"])
282
+ if len(records) != 3 or records[0]["step_zero_based"] != 0:
283
+ raise ValueError("Expected exactly three captures, beginning at step0")
284
+ checkpoints = [_checkpoint_spec(spec) for spec in args.checkpoint]
285
+ if len({name for name, _ in checkpoints}) != len(checkpoints):
286
+ raise ValueError("Checkpoint names must be unique")
287
+ restore_roles = getattr(args, "restore_role", [])
288
+ restore_names = getattr(args, "restore_name", [])
289
+ bf16_source = getattr(args, "bf16_source", None)
290
+ if bool(restore_roles or restore_names) != bool(bf16_source):
291
+ raise ValueError("Hybrid diagnostics require both --bf16-source and at least one restoration selector")
292
+ if bf16_source:
293
+ from .hybrid_v3 import _selected_names
294
+ _selected_names(restore_roles, restore_names)
295
+ report = {
296
+ "format": "qwen21-denoiser-comparison-v3", "complete": False,
297
+ "capture_manifest_sha256": _hash_file(directory / "capture.json"),
298
+ "teacher_capture": metadata, "backends": [],
299
+ "cache_policy": "Fresh backend-owned prefix rebuilt from step0 separately before every cached-step probe",
300
+ "claim": "Conditional denoiser prediction error on teacher trajectory; final image evaluation still required",
301
+ }
302
+ _write_json(args.out, report)
303
+ for name, checkpoint in checkpoints:
304
+ hybrid_report = None
305
+ model = load_transformer(checkpoint, device="cpu" if bf16_source else "cuda:0")
306
+ if bf16_source:
307
+ from .hybrid_v3 import restore_bf16_projections
308
+ hybrid_report = restore_bf16_projections(model, bf16_source, roles=restore_roles, names=restore_names)
309
+ model.to("cuda:0")
310
+ if not model.config.causal_condition:
311
+ raise ValueError("Compared checkpoint is not causal")
312
+ backend = {"name": name, "checkpoint": str(checkpoint),
313
+ "manifest_sha256": _hash_file(checkpoint / "manifest.json"), "probes": []}
314
+ if hybrid_report is not None:
315
+ backend["hybrid_override"] = hybrid_report
316
+ report["backends"].append(backend)
317
+ try:
318
+ for record in records:
319
+ cache = QwenImage21KVCache(len(model.transformer_blocks))
320
+ inputs = None
321
+ prefill = None
322
+ prediction = None
323
+ started = time.perf_counter()
324
+ torch.cuda.reset_peak_memory_stats()
325
+ try:
326
+ with torch.inference_mode():
327
+ if record["step_zero_based"] != 0:
328
+ prefill = _load_capture(directory, records[0], torch)
329
+ prefill_inputs = _tree(prefill["kwargs"], torch, "cuda:0")
330
+ prefill_inputs.update(kv_cache=cache, kv_cache_mode="extract", return_dict=False)
331
+ with model.cache_context("cond"):
332
+ prefill_output = model(**prefill_inputs)
333
+ _cache_summary(cache) # Must be fully backend-populated before decode.
334
+ del prefill_output, prefill_inputs, prefill
335
+ prefill = None
336
+ payload = _load_capture(directory, record, torch)
337
+ inputs = _tree(payload["kwargs"], torch, "cuda:0")
338
+ mode = "extract" if record["step_zero_based"] == 0 else "cached"
339
+ if inputs["kv_cache_mode"] != mode:
340
+ raise ValueError("Captured cache mode does not match selected step")
341
+ inputs.update(kv_cache=cache, kv_cache_mode=mode, return_dict=False)
342
+ with model.cache_context("cond"):
343
+ output = model(**inputs)
344
+ prediction = _target_prediction(output, inputs)
345
+ metrics = _metrics(prediction, payload["teacher_target_prediction"], torch)
346
+ cache_info = _cache_summary(cache)
347
+ torch.cuda.synchronize()
348
+ row = {"step_zero_based": record["step_zero_based"], "timestep": record["timestep"],
349
+ "cache_mode": mode, "fresh_prefill_calls": 1, "cache": cache_info,
350
+ "metrics": metrics, "instrumented_seconds_including_prefill": time.perf_counter() - started,
351
+ "peak_allocated_mib": torch.cuda.max_memory_allocated() / 2**20,
352
+ "peak_reserved_mib": torch.cuda.max_memory_reserved() / 2**20}
353
+ backend["probes"].append(row)
354
+ print(json.dumps({"event": "denoiser_probe", "backend": name, **row}), flush=True)
355
+ _write_json(args.out, report)
356
+ del output, payload
357
+ finally:
358
+ _clear_cache(cache)
359
+ cache = inputs = prefill = prediction = None
360
+ gc.collect()
361
+ torch.cuda.empty_cache()
362
+ finally:
363
+ del model
364
+ gc.collect()
365
+ torch.cuda.empty_cache()
366
+ report["complete"] = True
367
+ _write_json(args.out, report)
368
+
369
+
370
+ def main():
371
+ parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
372
+ commands = parser.add_subparsers(dest="command", required=True)
373
+ cap = commands.add_parser("capture", help="Run one BF16 teacher trajectory and store only three probes")
374
+ cap.add_argument("--jobs", required=True)
375
+ cap.add_argument("--label", required=True)
376
+ cap.add_argument("--out", required=True)
377
+ cap.add_argument("--no-stream", action="store_true")
378
+ cap.add_argument("--max-capture-mib", type=int, default=512)
379
+ comp = commands.add_parser("compare", help="Compare checkpoints sequentially with independent prefix caches")
380
+ comp.add_argument("--capture", required=True)
381
+ comp.add_argument("--checkpoint", action="append", required=True)
382
+ comp.add_argument("--out", required=True)
383
+ comp.add_argument("--bf16-source", help="Optional original BF16 snapshot for hybrid projection diagnostics")
384
+ comp.add_argument("--restore-role", action="append", default=[], help="Restore a role in all 32 blocks, e.g. attn.to_q")
385
+ comp.add_argument("--restore-name", action="append", default=[], help="Restore one exact transformer projection name")
386
+ args = parser.parse_args()
387
+ (capture if args.command == "capture" else compare)(args)
388
+
389
+
390
+ if __name__ == "__main__":
391
+ main()
reproduction/nunchaku_backend/diagnose_kernel.py ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """GPU-only isolated residual/low-rank diagnosis after a kernel probe failure."""
2
+ import argparse
3
+ import json
4
+ from pathlib import Path
5
+
6
+
7
+ def main():
8
+ p=argparse.ArgumentParser()
9
+ p.add_argument('--model-path',type=Path,required=True)
10
+ p.add_argument('--calibration',type=Path,required=True)
11
+ p.add_argument('--out',type=Path,required=True)
12
+ a=p.parse_args()
13
+ import torch
14
+ from safetensors.torch import load_file
15
+ from nunchaku.models.linear import SVDQW4A4Linear
16
+ from .baseline_candidate import convert_linear_weight,reference_forward
17
+ from .checkpoint_io import _source_index,_read_tensor
18
+ from .kernel_probe import _errors
19
+ torch.backends.cuda.matmul.allow_tf32=False
20
+ torch.set_num_threads(4)
21
+ name='transformer_blocks.0.img_mlp.out'
22
+ index=_source_index(a.model_path/'transformer')
23
+ weight=_read_tensor(index,name+'.weight')
24
+ amax=load_file(str(a.calibration/'activation_stats.safetensors'))[name+'.input_absmax']
25
+ packed,stats,reference=convert_linear_weight(weight,rank=32,input_absmax=amax,seed=102,return_reference=True,conversion_device='cuda:0')
26
+ layer=SVDQW4A4Linear(weight.shape[1],weight.shape[0],rank=32,bias=False,torch_dtype=torch.bfloat16,device='cuda:0')
27
+ layer.load_state_dict(packed)
28
+ layer.eval().requires_grad_(False)
29
+ reference={k:v.cuda() for k,v in reference.items()}
30
+ gen=torch.Generator(device='cpu').manual_seed(173)
31
+ x=torch.randn(1,257,weight.shape[1],generator=gen).to(torch.bfloat16).cuda()
32
+ rows={}
33
+ with torch.inference_mode():
34
+ rows['full']=_errors(layer(x),reference_forward(x,reference))
35
+ layer.proj_up.zero_()
36
+ ref0={**reference,'up_unpacked':torch.zeros_like(reference['up_unpacked'])}
37
+ rows['residual_only']=_errors(layer(x),reference_forward(x,ref0))
38
+ layer.load_state_dict(packed)
39
+ layer.qweight.zero_()
40
+ ref1={**reference,'residual_dequant':torch.zeros_like(reference['residual_dequant'])}
41
+ rows['lowrank_only']=_errors(layer(x),reference_forward(x,ref1))
42
+ a.out.write_text(json.dumps(rows,indent=2)+'\n')
43
+ print(json.dumps(rows),flush=True)
44
+
45
+ if __name__=='__main__':main()