fyb1214 commited on
Commit
62e0aea
·
verified ·
1 Parent(s): f71f3d8

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -1,35 +1,36 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
- *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
- *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
- *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
1
+ *.7z filter=lfs diff=lfs merge=lfs -text
2
+ *.arrow filter=lfs diff=lfs merge=lfs -text
3
+ *.bin filter=lfs diff=lfs merge=lfs -text
4
+ *.bz2 filter=lfs diff=lfs merge=lfs -text
5
+ *.ckpt filter=lfs diff=lfs merge=lfs -text
6
+ *.ftz filter=lfs diff=lfs merge=lfs -text
7
+ *.gz filter=lfs diff=lfs merge=lfs -text
8
+ *.h5 filter=lfs diff=lfs merge=lfs -text
9
+ *.joblib filter=lfs diff=lfs merge=lfs -text
10
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
+ *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
+ *.model filter=lfs diff=lfs merge=lfs -text
13
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
14
+ *.npy filter=lfs diff=lfs merge=lfs -text
15
+ *.npz filter=lfs diff=lfs merge=lfs -text
16
+ *.onnx filter=lfs diff=lfs merge=lfs -text
17
+ *.ot filter=lfs diff=lfs merge=lfs -text
18
+ *.parquet filter=lfs diff=lfs merge=lfs -text
19
+ *.pb filter=lfs diff=lfs merge=lfs -text
20
+ *.pickle filter=lfs diff=lfs merge=lfs -text
21
+ *.pkl filter=lfs diff=lfs merge=lfs -text
22
+ *.pt filter=lfs diff=lfs merge=lfs -text
23
+ *.pth filter=lfs diff=lfs merge=lfs -text
24
+ *.rar filter=lfs diff=lfs merge=lfs -text
25
+ *.safetensors filter=lfs diff=lfs merge=lfs -text
26
+ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
+ *.tar.* filter=lfs diff=lfs merge=lfs -text
28
+ *.tar filter=lfs diff=lfs merge=lfs -text
29
+ *.tflite filter=lfs diff=lfs merge=lfs -text
30
+ *.tgz filter=lfs diff=lfs merge=lfs -text
31
+ *.wasm filter=lfs diff=lfs merge=lfs -text
32
+ *.xz filter=lfs diff=lfs merge=lfs -text
33
+ *.zip filter=lfs diff=lfs merge=lfs -text
34
+ *.zst filter=lfs diff=lfs merge=lfs -text
35
+ *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ *.ninfer filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,201 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Apache License
2
+ Version 2.0, January 2004
3
+ http://www.apache.org/licenses/
4
+
5
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
6
+
7
+ 1. Definitions.
8
+
9
+ "License" shall mean the terms and conditions for use, reproduction,
10
+ and distribution as defined by Sections 1 through 9 of this document.
11
+
12
+ "Licensor" shall mean the copyright owner or entity authorized by
13
+ the copyright owner that is granting the License.
14
+
15
+ "Legal Entity" shall mean the union of the acting entity and all
16
+ other entities that control, are controlled by, or are under common
17
+ control with that entity. For the purposes of this definition,
18
+ "control" means (i) the power, direct or indirect, to cause the
19
+ direction or management of such entity, whether by contract or
20
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
21
+ outstanding shares, or (iii) beneficial ownership of such entity.
22
+
23
+ "You" (or "Your") shall mean an individual or Legal Entity
24
+ exercising permissions granted by this License.
25
+
26
+ "Source" form shall mean the preferred form for making modifications,
27
+ including but not limited to software source code, documentation
28
+ source, and configuration files.
29
+
30
+ "Object" form shall mean any form resulting from mechanical
31
+ transformation or translation of a Source form, including but
32
+ not limited to compiled object code, generated documentation,
33
+ and conversions to other media types.
34
+
35
+ "Work" shall mean the work of authorship, whether in Source or
36
+ Object form, made available under the License, as indicated by a
37
+ copyright notice that is included in or attached to the work
38
+ (an example is provided in the Appendix below).
39
+
40
+ "Derivative Works" shall mean any work, whether in Source or Object
41
+ form, that is based on (or derived from) the Work and for which the
42
+ editorial revisions, annotations, elaborations, or other modifications
43
+ represent, as a whole, an original work of authorship. For the purposes
44
+ of this License, Derivative Works shall not include works that remain
45
+ separable from, or merely link (or bind by name) to the interfaces of,
46
+ the Work and Derivative Works thereof.
47
+
48
+ "Contribution" shall mean any work of authorship, including
49
+ the original version of the Work and any modifications or additions
50
+ to that Work or Derivative Works thereof, that is intentionally
51
+ submitted to Licensor for inclusion in the Work by the copyright owner
52
+ or by an individual or Legal Entity authorized to submit on behalf of
53
+ the copyright owner. For the purposes of this definition, "submitted"
54
+ means any form of electronic, verbal, or written communication sent
55
+ to the Licensor or its representatives, including but not limited to
56
+ communication on electronic mailing lists, source code control systems,
57
+ and issue tracking systems that are managed by, or on behalf of, the
58
+ Licensor for the purpose of discussing and improving the Work, but
59
+ excluding communication that is conspicuously marked or otherwise
60
+ designated in writing by the copyright owner as "Not a Contribution."
61
+
62
+ "Contributor" shall mean Licensor and any individual or Legal Entity
63
+ on behalf of whom a Contribution has been received by Licensor and
64
+ subsequently incorporated within the Work.
65
+
66
+ 2. Grant of Copyright License. Subject to the terms and conditions of
67
+ this License, each Contributor hereby grants to You a perpetual,
68
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
69
+ copyright license to reproduce, prepare Derivative Works of,
70
+ publicly display, publicly perform, sublicense, and distribute the
71
+ Work and such Derivative Works in Source or Object form.
72
+
73
+ 3. Grant of Patent License. Subject to the terms and conditions of
74
+ this License, each Contributor hereby grants to You a perpetual,
75
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
76
+ (except as stated in this section) patent license to make, have made,
77
+ use, offer to sell, sell, import, and otherwise transfer the Work,
78
+ where such license applies only to those patent claims licensable
79
+ by such Contributor that are necessarily infringed by their
80
+ Contribution(s) alone or by combination of their Contribution(s)
81
+ with the Work to which such Contribution(s) was submitted. If You
82
+ institute patent litigation against any entity (including a
83
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
84
+ or a Contribution incorporated within the Work constitutes direct
85
+ or contributory patent infringement, then any patent licenses
86
+ granted to You under this License for that Work shall terminate
87
+ as of the date such litigation is filed.
88
+
89
+ 4. Redistribution. You may reproduce and distribute copies of the
90
+ Work or Derivative Works thereof in any medium, with or without
91
+ modifications, and in Source or Object form, provided that You
92
+ meet the following conditions:
93
+
94
+ (a) You must give any other recipients of the Work or
95
+ Derivative Works a copy of this License; and
96
+
97
+ (b) You must cause any modified files to carry prominent notices
98
+ stating that You changed the files; and
99
+
100
+ (c) You must retain, in the Source form of any Derivative Works
101
+ that You distribute, all copyright, patent, trademark, and
102
+ attribution notices from the Source form of the Work,
103
+ excluding those notices that do not pertain to any part of
104
+ the Derivative Works; and
105
+
106
+ (d) If the Work includes a "NOTICE" text file as part of its
107
+ distribution, then any Derivative Works that You distribute must
108
+ include a readable copy of the attribution notices contained
109
+ within such NOTICE file, excluding those notices that do not
110
+ pertain to any part of the Derivative Works, in at least one
111
+ of the following places: within a NOTICE text file distributed
112
+ as part of the Derivative Works; within the Source form or
113
+ documentation, if provided along with the Derivative Works; or,
114
+ within a display generated by the Derivative Works, if and
115
+ wherever such third-party notices normally appear. The contents
116
+ of the NOTICE file are for informational purposes only and
117
+ do not modify the License. You may add Your own attribution
118
+ notices within Derivative Works that You distribute, alongside
119
+ or as an addendum to the NOTICE text from the Work, provided
120
+ that such additional attribution notices cannot be construed
121
+ as modifying the License.
122
+
123
+ You may add Your own copyright statement to Your modifications and
124
+ may provide additional or different license terms and conditions
125
+ for use, reproduction, or distribution of Your modifications, or
126
+ for any such Derivative Works as a whole, provided Your use,
127
+ reproduction, and distribution of the Work otherwise complies with
128
+ the conditions stated in this License.
129
+
130
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
131
+ any Contribution intentionally submitted for inclusion in the Work
132
+ by You to the Licensor shall be under the terms and conditions of
133
+ this License, without any additional terms or conditions.
134
+ Notwithstanding the above, nothing herein shall supersede or modify
135
+ the terms of any separate license agreement you may have executed
136
+ with Licensor regarding such Contributions.
137
+
138
+ 6. Trademarks. This License does not grant permission to use the trade
139
+ names, trademarks, service marks, or product names of the Licensor,
140
+ except as required for reasonable and customary use in describing the
141
+ origin of the Work and reproducing the content of the NOTICE file.
142
+
143
+ 7. Disclaimer of Warranty. Unless required by applicable law or
144
+ agreed to in writing, Licensor provides the Work (and each
145
+ Contributor provides its Contributions) on an "AS IS" BASIS,
146
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
147
+ implied, including, without limitation, any warranties or conditions
148
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
149
+ PARTICULAR PURPOSE. You are solely responsible for determining the
150
+ appropriateness of using or redistributing the Work and assume any
151
+ risks associated with Your exercise of permissions under this License.
152
+
153
+ 8. Limitation of Liability. In no event and under no legal theory,
154
+ whether in tort (including negligence), contract, or otherwise,
155
+ unless required by applicable law (such as deliberate and grossly
156
+ negligent acts) or agreed to in writing, shall any Contributor be
157
+ liable to You for damages, including any direct, indirect, special,
158
+ incidental, or consequential damages of any character arising as a
159
+ result of this License or out of the use or inability to use the
160
+ Work (including but not limited to damages for loss of goodwill,
161
+ work stoppage, computer failure or malfunction, or any and all
162
+ other commercial damages or losses), even if such Contributor
163
+ has been advised of the possibility of such damages.
164
+
165
+ 9. Accepting Warranty or Additional Liability. While redistributing
166
+ the Work or Derivative Works thereof, You may choose to offer,
167
+ and charge a fee for, acceptance of support, warranty, indemnity,
168
+ or other liability obligations and/or rights consistent with this
169
+ License. However, in accepting such obligations, You may act only
170
+ on Your own behalf and on Your sole responsibility, not on behalf
171
+ of any other Contributor, and only if You agree to indemnify,
172
+ defend, and hold each Contributor harmless for any liability
173
+ incurred by, or claims asserted against, such Contributor by reason
174
+ of your accepting any such warranty or additional liability.
175
+
176
+ END OF TERMS AND CONDITIONS
177
+
178
+ APPENDIX: How to apply the Apache License to your work.
179
+
180
+ To apply the Apache License to your work, attach the following
181
+ boilerplate notice, with the fields enclosed by brackets "[]"
182
+ replaced with your own identifying information. (Don't include
183
+ the brackets!) The text should be enclosed in the appropriate
184
+ comment syntax for the file format. We also recommend that a
185
+ file or class name and description of purpose be included on the
186
+ same "printed page" as the copyright notice for easier
187
+ identification within third-party archives.
188
+
189
+ Copyright [yyyy] [name of copyright owner]
190
+
191
+ Licensed under the Apache License, Version 2.0 (the "License");
192
+ you may not use this file except in compliance with the License.
193
+ You may obtain a copy of the License at
194
+
195
+ http://www.apache.org/licenses/LICENSE-2.0
196
+
197
+ Unless required by applicable law or agreed to in writing, software
198
+ distributed under the License is distributed on an "AS IS" BASIS,
199
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
200
+ See the License for the specific language governing permissions and
201
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,129 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Swift-Bonsai-2 27B — NInfer ternary artifacts (PQ2_0_G128 / PTQ1_0_G128)
2
+ =======================================================================
3
+
4
+ This work is licensed under the Apache License, Version 2.0. See the LICENSE file.
5
+
6
+ Every component in the provenance chain is Apache-2.0 licensed **except one link, which is
7
+ disclosed explicitly in section 2 below**. We do not hide it.
8
+
9
+
10
+ 1. PROVENANCE CHAIN
11
+ -------------------
12
+
13
+ Qwen/Qwen3.8-27B Apache-2.0
14
+ Copyright 2026 Alibaba Cloud
15
+ The dense 27B base model. Geometry, tokenizer and the six frontend resources derive here.
16
+
17
+ prism-ml/Ternary-Bonsai-2-27B-gguf Apache-2.0
18
+ Copyright Prism ML, Inc.
19
+ The ternary (1.58-bit) + Hadamard-rotated-basis release of Bonsai 2 27B.
20
+ Defines the `prism.hadamard.*` metadata contract this artifact carries.
21
+
22
+ ukisai/Swift-Bonsai-2-GGUF Apache-2.0
23
+ ukisai — "Swift Bonsai 2", a reasoning-efficient fine-tune.
24
+ >>> THIS IS THE WEIGHT SOURCE OF THE ARTIFACTS IN THIS REPOSITORY. <<<
25
+
26
+ github.com/Neroued/ninfer Apache-2.0
27
+ Upstream engine, converter and the version-2 artifact container format.
28
+
29
+ github.com/Ambolio/ninfer-4090-windows Apache-2.0
30
+ The Ada/Windows source lineage (v1.0.6 / v1.0.8) whose format dialect this artifact uses.
31
+ (The HF copy that existed under the same name is now 404; the lineage persists through forks.)
32
+
33
+ shensanshu/ninfer-ada-ternary Apache-2.0
34
+ The packer `tools/pack.py`, the per-tensor mapping table `tools/MAPPING.json`,
35
+ and the five verification scripts, all redistributed here under Apache-2.0
36
+ with the notice below.
37
+
38
+ Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer Apache-2.0
39
+ Used ONLY as a template: its object list (the skeleton this packer walks) and the
40
+ payloads it donates for vision / mtp / frontend / draft_head.
41
+ **Its text weights are discarded entirely and do not enter this artifact.**
42
+
43
+ Upstream sibling — ukisai/Swift-Qwen3.8-27b swift-open-license-1.0
44
+ This is the BF16 fine-tune checkpoint from which ukisai's Swift derivative family descends.
45
+ It is **not** the direct source of this repository (we pack from ukisai's GGUF release,
46
+ which is itself marked Apache-2.0), but it sits in the same family, and its license is
47
+ **not** Apache. We flag it here rather than omit it so downstream users can make their own
48
+ judgement. If you need a conservative position, treat this chain as carrying that
49
+ upstream license.
50
+
51
+
52
+ 2. SCOPE OF WHAT THIS REPOSITORY DISTRIBUTES
53
+ --------------------------------------------
54
+
55
+ This repository distributes **weight-derived artifacts** (the two `.ninfer` files), not code
56
+ weights per se. Per the packer project's own NOTICE:
57
+
58
+ > 由本项目产出的 `.ninfer` 制品属于权重派生品,其再分发义务以权重原许可为准,与代码许可无关。
59
+
60
+ The direct source (`ukisai/Swift-Bonsai-2-GGUF`) and the base (`prism-ml/Ternary-Bonsai-2-27B-gguf`,
61
+ `Qwen/Qwen3.8-27B`) are all published as **Apache-2.0**, which permits redistribution.
62
+ The un-Apache link noted in section 1 is upstream of the direct source and is disclosed there.
63
+
64
+ **No model weights were retrained, fine-tuned, abliterated or otherwise behaviourally modified
65
+ by us.** Model behaviour — including all refusal characteristics inherited from the upstream
66
+ fine-tune — derives entirely from `ukisai/Swift-Bonsai-2-GGUF`.
67
+
68
+
69
+ STATEMENT OF CHANGES (Apache-2.0 section 4(b))
70
+ ----------------------------------------------
71
+
72
+ Relative to `ukisai/Swift-Bonsai-2-GGUF` (the weight source):
73
+
74
+ * The 402 ternary matrices were **moved byte-for-byte**. The 2-bit codes and their per-128
75
+ FP16 scales were copied verbatim from the GGUF block stream into the artifact's
76
+ `row-split-k128-v1` three-plane layout. **No dequantization and no requantization occurred.**
77
+ * llama.cpp's exporter conventions were undone, as the upstream recipe requires:
78
+ - GDN value heads: tiled order -> grouped order (per-head, `perm48(head)*128 + inner`)
79
+ - zero-centred norms: `1 + w` -> `w` (except `gdn/norm` / `ssm_norm`, whose raw value
80
+ is already ~1)
81
+ - `ssm_a` (`-exp(A_log)`) -> `A_log`
82
+ * `text/hadamard_signs` (28,672 fp32 words) and `text/hadamard_widths` (3 int32) were
83
+ **added**: they restate the GGUF's `prism.hadamard.sign_values` / `sign_widths` in the
84
+ artifact's own object vocabulary. The source GGUF already carries these values; the
85
+ template does not, which is why the object count is 1126 rather than 1124.
86
+ * Non-ternary tensors (norms, GDN A/B projections, convolution, `A_log`, `dt_bias`,
87
+ `token_embedding`'s direct companions) were re-encoded from the same GGUF.
88
+
89
+ Relative to `Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer` (the template):
90
+
91
+ * **Nothing of its text tower is used.** Only its object list and the vision (333), mtp (12),
92
+ frontend (6) and draft_head (2) payloads — 1,116,856,537 bytes — were carried over.
93
+ * Those payloads are the official Qwen3.8-27B components shared across the whole family.
94
+ The packer verified the MTP head against the target at cosine >= 0.99966 with seven norms
95
+ **bit-identical**.
96
+
97
+ Relative to `shensanshu/ninfer-ada-ternary` `tools/pack.py`:
98
+
99
+ * **No functional change.** The only edit was to the default path constants at the top of the
100
+ file — `NINFER_ROOT` (which pointed at the author's own development machine) and `TEMPLATE`
101
+ — both replaced with neutral placeholders (`<NINFER_ROOT>`, `<TEMPLATE>`). The same
102
+ substitution was applied to the five scripts under `tools/verify/` and to `MAPPING.json`'s
103
+ `_verified_by` note. **No logic was altered.**
104
+ * `MAPPING.json` is redistributed verbatim.
105
+ * **No validation, checksum, geometry check or byte-round-trip proof was disabled, relaxed
106
+ or bypassed.** The `check` mode was run to completion against the source GGUF and reported
107
+ `pad = 0` for all eight shape combinations, with `bytes_equal` and `decode_equal` true on
108
+ every sampled tensor.
109
+
110
+ Relative to `github.com/Neroued/ninfer`:
111
+
112
+ * **NO CHANGES.** Upstream was not patched, modified or circumvented by this work.
113
+
114
+
115
+ Trademarks
116
+ ----------
117
+
118
+ "Qwen" is a trademark of Alibaba Cloud. "Bonsai" and "Prism ML" are marks of Prism ML, Inc.
119
+
120
+ This is an unofficial, community-produced derivative. It is **not** endorsed by or affiliated
121
+ with Alibaba Cloud, Prism ML, ukisai, shensanshu, or the NInfer project.
122
+
123
+
124
+ Disclaimer
125
+ ----------
126
+
127
+ Provided **AS IS**, without warranty of any kind. All measurements cited in README.md were taken
128
+ in specific hardware/software environments and will differ across GPU model, driver, CUDA version
129
+ and memory bandwidth.
README.md ADDED
@@ -0,0 +1,293 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ language:
4
+ - en
5
+ - zh
6
+ library_name: ninfer
7
+ pipeline_tag: text-generation
8
+ base_model:
9
+ - prism-ml/Ternary-Bonsai-2-27B-gguf
10
+ - ukisai/Swift-Bonsai-2-GGUF
11
+ - Qwen/Qwen3.8-27B
12
+ base_model_relation: quantized
13
+ tags:
14
+ - ninfer
15
+ - ternary
16
+ - 1.58-bit
17
+ - bonsai
18
+ - qwen3.8
19
+ - hadamard
20
+ - mtp
21
+ - swift
22
+ - text-generation
23
+ - image-text-to-text
24
+ ---
25
+
26
+ # Swift-Bonsai-2 27B — NInfer 三元制品(PQ2 / PTQ1)
27
+
28
+ **English summary at the end.** | [English README](README_en.md)
29
+
30
+ ---
31
+
32
+ ## ⚠️ 先说引擎兼容性(最重要的一节)
33
+
34
+ 本制品是 **`PQ2_0_G128` / `PTQ1_0_G128`** 方言,需要 **`Ambolio/ninfer-4090-windows` 血统**的引擎
35
+ (v1.0.6 / v1.0.8,即"极速档"那条线)。
36
+
37
+ **跟 HF 上其它 Bonsai `.ninfer` 不通用:**
38
+
39
+ | 制品 | 三元格式名 | 需要的引擎 |
40
+ |---|---|---|
41
+ | **本制品** | **`PQ2_0_G128` / `PTQ1_0_G128`** | **Ambolio 血统(v1.0.6 / v1.0.8)** |
42
+ | `WaveCut/Ternary-Bonsai-2-27B-NInfer-v3` | `t2_g128_fp16` | [`iamwavecut/ninfer-all`](https://github.com/iamwavecut/ninfer-all) |
43
+ | `neroued/Qwen3.8-27B-NInfer` | NVFP4 / groupwise-int | [`Neroued/ninfer`](https://github.com/Neroued/ninfer)(上游) |
44
+
45
+ **装错的表现是启动直接被拒(`refuse this file`),不是"慢一点"。**
46
+
47
+ 判方言的方法 —— 读 `.ninfer` 前 1 MB 里的 `"format"` 字段即可,不用解压整个文件。
48
+
49
+ ---
50
+
51
+ ## 这是什么
52
+
53
+ `Swift-Bonsai-2` 的三元 NInfer 制品,两档:
54
+
55
+ | 文件 | 档位 | 大小 | 源 GGUF |
56
+ |---|---|---|---|
57
+ | `bonsai2_27b_swift_pq2.ninfer` | `PQ2_0_G128`(2 bit) | 8,306,927,628 B / 7.736 GiB | `Swift-Bonsai-2-PQ2_0.gguf` |
58
+ | `bonsai2_27b_swift_ptq1.ninfer` | `PTQ1_0_G128`(1.75 bit) | 7,047,407,628 B / 6.563 GiB | `Swift-Bonsai-2-PTQ1_0.gguf` |
59
+
60
+ **组件:text + vision + mtp(不含 dflash2)。**
61
+
62
+ **这是格式转换,不是训练。** 三元码逐字节搬运,没有解量化再重量化。
63
+
64
+ ---
65
+
66
+ ## 产物结构(已核对)
67
+
68
+ ```
69
+ identity {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
70
+ magic NINFER\x00\x02 (version-2 容器)
71
+ objects 1126 = text 775 + vision 333 + mtp 12 + frontend 6
72
+
73
+ PQ2 档 formats BF16 582 | PQ2_0_G128 322 | FP32 97 | Q4G64_F16S 55
74
+ | Q5G64_F16S 54 | W8G32_F16S 7 | I32 2 | Q6G64_F16S 1
75
+ PTQ1 档 formats 同上,仅把 PQ2_0_G128 换成 PTQ1_0_G128
76
+
77
+ layouts row-split-k128-v1 439 | contiguous-le-v1 681
78
+ 新增 text/hadamard_signs、text/hadamard_widths(三元必需)
79
+ 借来的 vision 333 + mtp 12 + frontend 6 + draft_head 2 = 1,116,856,537 B(1.040 GiB)
80
+ ```
81
+
82
+ **借来的那几块是全生态共用的官方件。** 打包器的原话:
83
+
84
+ > vision …**the template's tower is the same official one**
85
+ > mtp …**The template's head is the same official Qwen3.8-27B head**: all twelve tensors
86
+ > matched at cos ≥ 0.99966 with the seven norms **BIT-IDENTICAL**
87
+
88
+ ---
89
+
90
+ ## 怎么用
91
+
92
+ ```powershell
93
+ ninfer-serve.exe bonsai2_27b_swift_pq2.ninfer ^
94
+ --host 127.0.0.1 --port 8087 ^
95
+ --max-context 131072 --kv-capacity 131072 ^
96
+ --kv-dtype int8 ^
97
+ --spec mtp --draft-tokens 3
98
+ ```
99
+
100
+ **KV dtype 按卡的架构选:**
101
+
102
+ | 卡 | 可用 KV |
103
+ |---|---|
104
+ | 30 系(sm_86) | `bf16` / `int8` |
105
+ | 40 系(sm_89) | `bf16` / `int8` / `fp8` / `rk4v4` / `rk4v4-e8` |
106
+ | 50 系(sm_120) | `bf16` / `int8` / `fp8` / `nvfp4` / `k8v4`(`rk4v4` 会被拒) |
107
+
108
+ ### ★ MTP 窗口:先扫,别照抄
109
+
110
+ 上游在 4080S 上测出 **N=2 最优**(N=3/4/5 的接受率落到 40.9 / 33.2 / 22.9%)。
111
+ **但那不是普适值 —— 在 12 GB 的 3060 上,`d3` 才是最好的,且比生产用的 `d4+lm` 平均快 2~5%。**
112
+
113
+ 详见下方「实测数据」的 MTP 表。**建议在你自己卡上扫 1~4 档,两个对照模型交替起服。**
114
+
115
+ ---
116
+
117
+ ## 验收状态
118
+
119
+ | 项 | 状态 |
120
+ |---|---|
121
+ | 源 GGUF 几何 + round-trip | ✅ 8 个形状组合 `pad=0`;5 个真实张量 `bytes_equal` + `decode_equal` 全 True;`zero_share` 0.3277~0.3279(`PQ2_0` 理论指纹 0.3278) |
122
+ | 产出结构对撞 | ✅ 与参考制品 `bonsai2_27b_ternary_v2.ninfer` 的格式分布 **8 项全等**;**总长也逐字节相同**(均 8,306,927,628 B) |
123
+ | 尺寸 | ✅ 两档与上游文档的换算表精确吻合(PQ2 8.31 GB / PTQ1 7.05 GB) |
124
+ | **端到端(RTX 3060 12G / sm_86)** | ✅ **四层全过** —— 启动、输出、前缀复用、视觉、工具调用、采样、思考档位全部正常 |
125
+ | **PPL(同语料 vs base)** | ✅ **略优于 base**,三口径差 −0.07% / −0.04% / −0.01% |
126
+ | **速度 / MTP 接受率** | ✅ **与 base 在 ±2% 内**(测量噪声量级) |
127
+ | 长思考任务上的「思考更短」 | ❓ **未验证** —— 见下 |
128
+
129
+ > **已由发布者在一台 RTX 3060 12G 上完成端到端验收**(2026-09-28)。
130
+ > 原始记录(`l1.txt` / `l23.txt` / `ppl-*.json` / `quiz-*.json`)由测试者保存在本地,**不随本仓分发**。
131
+ >
132
+ > **仍未验证的一项**:Swift 微调宣��的"思考 token 少 ~40%"。本次题集太简单(每题思考仅
133
+ > 100~500 token),**在该量级上得不出显著结论,也无法否证**。要验证需用 AIME / 竞赛级长思考题,
134
+ > 每题 2k~10k 思考 token、≥20 题 × 2 轮。
135
+
136
+ ---
137
+
138
+ ## 实测数据(RTX 3060 12G / sm_86 / 生产参数)
139
+
140
+ ### PPL(同一 `perplexity` 程序、同一份 `pplab-text`、`int8` KV)
141
+
142
+ | 窗口/步长 | Swift | base | 差 |
143
+ |---|---:|---:|---:|
144
+ | 512 / 256 | **8.073669** | 8.079208 | **−0.07%** |
145
+ | 32 / 16 | **26.962015** | 26.972171 | **−0.04%** |
146
+ | 8 / 4 | **148.3656** | 148.3863 | **−0.01%** |
147
+
148
+ **⚠️ 不要拿别处的 PPL 绝对值来比。** PPL 跟语料强相关 —— 上游文档给的参照值(6.448742 /
149
+ 26.049634 / 121.157720)用的是**另一份语料**,与上表**不可直接比较**。上表用的是 `pplab-text`,
150
+ 只有同一语料下的 Swift vs base 差值才有意义。
151
+
152
+ ### 速度与 MTP 接受率
153
+
154
+ 条件:贪心,每组 400 token,1 次热身 + 2 次取中位,**两个模型交替起服**。
155
+ 格式:`t/s(接受率)`。
156
+
157
+ | 配置 | 模型 | 中文 | 英文 | 代码 | 思考 | 平均 |
158
+ |---|---|---:|---:|---:|---:|---:|
159
+ | d1 | Swift | 42.2 (58%) | 44.4 (68%) | 46.0 (82%) | 45.8 (84%) | 44.6 |
160
+ | d1 | base | 43.1 (62%) | 44.9 (71%) | 46.6 (85%) | 45.4 (83%) | 45.0 |
161
+ | d2 | Swift | 46.0 (46%) | 49.5 (56%) | 58.4 (80%) | 55.9 (76%) | 52.5 |
162
+ | d2 | base | 43.7 (42%) | 46.9 (51%) | 59.6 (82%) | 55.7 (76%) | 51.5 |
163
+ | **d3** | Swift | 44.0 (33%) | 49.7 (44%) | 64.2 (70%) | 62.2 (68%) | **55.0** |
164
+ | **d3** | base | 43.4 (33%) | 52.4 (48%) | 64.9 (71%) | 59.9 (65%) | **55.1** |
165
+ | d4+lm(生产) | Swift | 40.4 (28%) | 48.0 (40%) | 65.5 (67%) | 58.6 (59%) | 53.1 |
166
+ | d4+lm(生产) | base | 39.2 (27%) | 48.2 (41%) | 64.6 (66%) | 57.1 (56%) | 52.3 |
167
+
168
+ **三条读法:**
169
+
170
+ 1. **Swift 与 base 在每个配置下都相差 ±2% 以内** —— 这是测量噪声。**借来的 MTP 头与 Swift 主干配合得不比 base 差。**
171
+ 2. **d3 最好,不是上游在 4080S 上说的 N=2。** d2 只在中文上占优。**MTP 窗口的最优值是卡的属性,不是模型的。**
172
+ 3. **d3(不加 `lm-head`)平均比生产用的 `d4+lm` 高 2~5%,两个对照模型都是这个方向。**
173
+ 这是一个**尚未纳入生产配置**的观察 —— 换之前请先用官方 bench 复核。
174
+
175
+ ### 推理题(12 道唯一答案的中英文算术/逻辑题,生产参数,引擎默认采样)
176
+
177
+ | | 第 1 轮 | 第 2 轮 | 合计输出 token |
178
+ |---|---|---|---:|
179
+ | Swift | 12/12 | 12/12 | 4,501 |
180
+ | base | 12/12 | 12/12 | 5,144 |
181
+
182
+ **准确率相同;Swift 输出 token 少 12.5%。但样本太小 —— 逐题分布大量重叠,同一模型两轮之间最多能差 2 倍,得不出显著结论。**
183
+ 所有回答均以 `finish=stop` 正常结束,无死循环或截断。
184
+
185
+ ### 功能面(生产参数:ctx 65536 / `int8` / MTP d4+lm / vision)
186
+
187
+ | 项 | 结果 |
188
+ |---|---|
189
+ | 启动 | ✅ `/v1/models` 返回 `qwen3.8-27b`,`max_model_len` = 65536 |
190
+ | 英文 / 中文 | ✅ 通顺 |
191
+ | 同前缀多轮 ×3 | ✅ `200/200/200/200`,第 2、3 次命中 40 token 缓存,输出一致 |
192
+ | 连续 10 个不同请求 | ✅ 全部 200 |
193
+ | 采样 | ✅ 两次输出不同 |
194
+ | 思考档位 | ✅ `none/low/medium/xhigh` 正常;`high` 按预期返回 400(与 base 一致) |
195
+ | 前缀复用(约 2.4k token,3 轮) | ✅ prompt 耗时 **2848 ms → 258 / 268 ms** |
196
+ | 视觉 | ✅ 正确识别红色方块、`CAT`、蓝色 `42` |
197
+ | 工具调用 | ✅ 返回 `get_weather({"city":"Tokyo"})` |
198
+ | `reasoning_effort=none` 时不调工具 | ⚠️ 与 base 的已知现象相同,非本制品问题 |
199
+
200
+ > **关于前缀复用崩溃**:上游文档记录的"多轮/同前缀第二次请求 500 后全端口 503"在本测试环境**未复现**
201
+ > —— 该引擎已修正此问题。若你的引擎版本较旧且撞上,临时规避是 `--no-prefix-reuse`。
202
+
203
+ ---
204
+
205
+ ## 怎么造出来的(可复现)
206
+
207
+ ```bash
208
+ # 1. 源(ukisai 的 Swift 微调,已三元量化)
209
+ Swift-Bonsai-2-PQ2_0.gguf
210
+ sha256 5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5
211
+
212
+ # 2. 模板 —— 只借它的骨架/对象清单 + vision/MTP/frontend 载荷
213
+ # 它的文本权重被整个丢弃,不进入本制品
214
+ Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
215
+ sha256 8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03
216
+
217
+ # 3. 打包
218
+ python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
219
+ --gguf Swift-Bonsai-2-PQ2_0.gguf \
220
+ --template qwen3_8_27b_huihui_abliterated.ninfer
221
+ # 约 4 分钟(纯 CPU,不需要 GPU)
222
+ ```
223
+
224
+ **完整方法见 `REPRODUCE.md`。** 打包器与逐张量映射表随仓附带(`tools/`)。
225
+
226
+ ---
227
+
228
+ ## 署名(三层,全部必需)
229
+
230
+ ```
231
+ Bonsai 2 27B(原始权重) © Prism ML, Inc. Apache-2.0
232
+ huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
233
+ Qwen3.8-27B(底座) © Alibaba Cloud Apache-2.0
234
+ Swift 微调(���制品权重来源) ukisai Apache-2.0
235
+ huggingface.co/ukisai/Swift-Bonsai-2-GGUF
236
+ 打包工具 pack.py —— shensanshu/ninfer-ada-ternary(Apache-2.0)
237
+ 容器格式 / 引擎 github.com/Neroued/ninfer(Apache-2.0)
238
+ ```
239
+
240
+ 官方请求的原句署名:
241
+
242
+ > **"Created using Bonsai by Prism ML."**
243
+
244
+ **详见 [`NOTICE`](NOTICE)。**
245
+
246
+ ---
247
+
248
+ ## 已知的坑(来自上游文档,非本制品独有)
249
+
250
+ 1. **MTP 窗口 N=2 最优**,N=3/4/5 接受率崩(4080S 实测;不同卡需自测)
251
+ 2. **前缀复用崩溃** —— 多轮/同前缀第二次请求 500 然后全端口 503;临时规避 `--no-prefix-reuse`
252
+ 3. **改 `.h` 后必须 touch 所有 `.cpp`** —— MSVC 依赖扫描坏,改动会静默不生效(`/GS` 崩溃 `0xC0000409`)
253
+ 4. **中文 prompt 走命令行会报 NFC 错** —— 必须用 `--messages <json文件>`(UTF-8 无 BOM)
254
+ 5. **`ninfer-perplexity.exe` 必须跟 `ninfer-serve` 同一批编出来** —— 只重编 serve 会留下旧件,导致误判
255
+
256
+ ---
257
+
258
+ ## 许可
259
+
260
+ **Apache-2.0。** 见 [`LICENSE`](LICENSE) 与 [`NOTICE`](NOTICE)。
261
+
262
+ **商标**:"Qwen" 是 Alibaba Cloud 的商标;"Bonsai" 属 Prism ML。本制品是社区制作的衍生品,
263
+ 与 Alibaba Cloud、Prism ML、ukisai 及 NInfer 项目**无隶属或背书关系**。
264
+
265
+ ---
266
+
267
+ ## English summary
268
+
269
+ **Swift-Bonsai-2 27B, NInfer ternary artifacts** — `PQ2_0_G128` (2-bit) and `PTQ1_0_G128` (1.75-bit).
270
+
271
+ **Engine compatibility:** this artifact uses the **`PQ2_0_G128` / `PTQ1_0_G128`** dialect and requires an
272
+ engine from the **`Ambolio/ninfer-4090-windows` lineage** (v1.0.6 / v1.0.8).
273
+ It is **NOT** interchangeable with `t2_g128_fp16` artifacts (those need `iamwavecut/ninfer-all`).
274
+ Loading the wrong one is refused at startup, not merely slow.
275
+
276
+ **What this is:** a **format conversion**, not a training run. Ternary codes are moved byte-for-byte
277
+ from the source GGUF; nothing is dequantized and requantized. Components: text + vision + mtp
278
+ (**no dflash2**).
279
+
280
+ **Provenance:** Bonsai 2 27B (Prism ML, Apache-2.0) → Swift fine-tune (ukisai, Apache-2.0)
281
+ → packed with `pack.py` from `shensanshu/ninfer-ada-ternary` (Apache-2.0).
282
+
283
+ **Verification status:** structural and format-level checks are green (geometry `pad=0` across all 8
284
+ shape combinations; byte round-trip and decode equality on sampled tensors; `zero_share` 0.3278 —
285
+ the `PQ2_0` fingerprint). **End-to-end validation was completed on an RTX 3060 12G**: the server
286
+ starts, output is coherent, prefix reuse, vision and tool calling all work, and PPL is marginally
287
+ **better** than base on the same corpus (−0.07 % / −0.04 % / −0.01 %). See
288
+ [the measured-results tables](#实测数据rtx-3060-12g--sm_86--生产参数) in the Chinese section.
289
+
290
+ **One advertised claim remains unverified:** "~40 % fewer thinking tokens" — this round's items were
291
+ too easy to resolve at that magnitude. See the verification status table above.
292
+
293
+ > **"Created using Bonsai by Prism ML."**
README_en.md ADDED
@@ -0,0 +1,254 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Swift-Bonsai-2 27B — NInfer ternary artifacts (PQ2 / PTQ1)
2
+
3
+ **中文说明见 [README.md](README.md)**
4
+
5
+ ---
6
+
7
+ ## ⚠️ Engine compatibility — read this first
8
+
9
+ These artifacts use the **`PQ2_0_G128` / `PTQ1_0_G128`** dialect and require an engine from the
10
+ **`Ambolio/ninfer-4090-windows` lineage** (v1.0.6 / v1.0.8, the "极速档 / speed tier" line).
11
+
12
+ **They are NOT interchangeable with other Bonsai `.ninfer` artifacts on the Hub:**
13
+
14
+ | Artifact | Ternary format name | Engine required |
15
+ |---|---|---|
16
+ | **this repo** | **`PQ2_0_G128` / `PTQ1_0_G128`** | **Ambolio lineage (v1.0.6 / v1.0.8)** |
17
+ | `WaveCut/Ternary-Bonsai-2-27B-NInfer-v3` | `t2_g128_fp16` | [`iamwavecut/ninfer-all`](https://github.com/iamwavecut/ninfer-all) |
18
+ | `neroued/Qwen3.8-27B-NInfer` | NVFP4 / groupwise-int | [`Neroued/ninfer`](https://github.com/Neroued/ninfer) (upstream) |
19
+
20
+ Loading the wrong one is **refused at startup** (`refuse this file`) — not merely slow.
21
+
22
+ To tell which dialect a `.ninfer` uses, read the `"format"` fields in its first 1 MiB. No need to
23
+ scan the whole file.
24
+
25
+ ---
26
+
27
+ ## What this is
28
+
29
+ Two tiers of `Swift-Bonsai-2` packaged for NInfer:
30
+
31
+ | File | Tier | Size | Source GGUF |
32
+ |---|---|---|---|
33
+ | `bonsai2_27b_swift_pq2.ninfer` | `PQ2_0_G128` (2-bit) | 8,306,927,628 B / 7.736 GiB | `Swift-Bonsai-2-PQ2_0.gguf` |
34
+ | `bonsai2_27b_swift_ptq1.ninfer` | `PTQ1_0_G128` (1.75-bit) | 7,047,407,628 B / 6.563 GiB | `Swift-Bonsai-2-PTQ1_0.gguf` |
35
+
36
+ **Components: text + vision + mtp (no dflash2).**
37
+
38
+ **This is a format conversion, not a training run.** The ternary codes are moved byte-for-byte;
39
+ nothing is dequantized and requantized.
40
+
41
+ ---
42
+
43
+ ## Structure (verified)
44
+
45
+ ```
46
+ identity {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
47
+ magic NINFER\x00\x02 (version-2 container)
48
+ objects 1126 = text 775 + vision 333 + mtp 12 + frontend 6
49
+
50
+ PQ2 tier BF16 582 | PQ2_0_G128 322 | FP32 97 | Q4G64_F16S 55
51
+ | Q5G64_F16S 54 | W8G32_F16S 7 | I32 2 | Q6G64_F16S 1
52
+ PTQ1 tier identical except PQ2_0_G128 -> PTQ1_0_G128
53
+
54
+ layouts row-split-k128-v1 439 | contiguous-le-v1 681
55
+ added text/hadamard_signs, text/hadamard_widths (required by ternary)
56
+ borrowed vision 333 + mtp 12 + frontend 6 + draft_head 2 = 1,116,856,537 B (1.040 GiB)
57
+ ```
58
+
59
+ The borrowed set are the official, family-wide Qwen3.8-27B components. From the packer:
60
+
61
+ > vision …**the template's tower is the same official one**
62
+ > mtp …**The template's head is the same official Qwen3.8-27B head**: all twelve tensors matched
63
+ > at cos ≥ 0.99966 with the seven norms **BIT-IDENTICAL**
64
+
65
+ ---
66
+
67
+ ## Usage
68
+
69
+ ```powershell
70
+ ninfer-serve.exe bonsai2_27b_swift_pq2.ninfer ^
71
+ --host 127.0.0.1 --port 8087 ^
72
+ --max-context 131072 --kv-capacity 131072 ^
73
+ --kv-dtype int8 ^
74
+ --spec mtp --draft-tokens 3
75
+ ```
76
+
77
+ **KV dtype depends on the GPU architecture:**
78
+
79
+ | GPU | Available KV |
80
+ |---|---|
81
+ | RTX 30 series (sm_86) | `bf16` / `int8` |
82
+ | RTX 40 series (sm_89) | `bf16` / `int8` / `fp8` / `rk4v4` / `rk4v4-e8` |
83
+ | RTX 50 series (sm_120) | `bf16` / `int8` / `fp8` / `nvfp4` / `k8v4` (`rk4v4` is rejected) |
84
+
85
+ **On a 12 GB card (e.g. RTX 3060),** `--max-context 32768~131072` with `--kv-dtype int8` is what was
86
+ **actually measured end-to-end** — see the verification status and measured-results tables below.
87
+
88
+ **Start MTP window scanning at N=2 — but do not copy that number.**
89
+
90
+ Upstream measured **N=2 optimal on a 4080 SUPER** (N=3/4/5 dropping to 40.9 / 33.2 / 22.9 % acceptance).
91
+ **That is not universal.** On a 12 GB RTX 3060, **`d3` is the best tier** and beats the production
92
+ `d4+lm` by 2–5 % on average. Scan 1–4 on your own card, starting both reference models alternately.
93
+
94
+ ---
95
+
96
+ ## Verification status
97
+
98
+ | Item | Status |
99
+ |---|---|
100
+ | Source GGUF geometry + round-trip | ✅ `pad=0` for all 8 shape combinations; `bytes_equal` + `decode_equal` true on every sampled tensor; `zero_share` 0.3277–0.3279 (the `PQ2_0` fingerprint is 0.3278) |
101
+ | Output structure | ✅ format distribution matches the reference artifact on all 8 buckets; **total size is byte-identical too** (both 8,306,927,628 B) |
102
+ | Sizes | ✅ both tiers match the upstream conversion table (PQ2 8.31 GB / PTQ1 7.05 GB) |
103
+ | **End-to-end (RTX 3060 12G / sm_86)** | ✅ **all four layers pass** — startup, output, prefix reuse, vision, tool calling, sampling, reasoning tiers |
104
+ | **PPL (same corpus, vs base)** | ✅ **slightly better than base**: −0.07 % / −0.04 % / −0.01 % across three window/stride settings |
105
+ | **Speed / MTP acceptance** | ✅ **within ±2 % of base** (measurement-noise level) |
106
+ | "Shorter thinking" on long-reasoning tasks | ❓ **not verified** |
107
+
108
+ > **End-to-end validation was completed by the publisher on one RTX 3060 12G** (2026-09-28).
109
+ > Raw records (`l1.txt`, `l23.txt`, `ppl-*.json`, `quiz-*.json`) are kept locally by the
110
+ > tester and are **not distributed with this repository**.
111
+ >
112
+ > **One claim remains unverified:** the fine-tune's advertised "~40 % fewer thinking tokens".
113
+ > This round's question set was too easy (100–500 thinking tokens per item) — **no conclusion is
114
+ > possible at that magnitude, in either direction**. Verifying it needs AIME / competition-level
115
+ > long-reasoning items, 2k–10k thinking tokens each, ≥20 items × 2 rounds.
116
+
117
+ ---
118
+
119
+ ## Measured results (RTX 3060 12G / sm_86 / production parameters)
120
+
121
+ ### PPL — same `perplexity` binary, same `pplab-text` corpus, `int8` KV
122
+
123
+ | Window / stride | Swift | base | Δ |
124
+ |---|---:|---:|---:|
125
+ | 512 / 256 | **8.073669** | 8.079208 | **−0.07 %** |
126
+ | 32 / 16 | **26.962015** | 26.972171 | **−0.04 %** |
127
+ | 8 / 4 | **148.3656** | 148.3863 | **−0.01 %** |
128
+
129
+ **⚠️ Do not compare these absolute values against PPL figures from elsewhere.** PPL is corpus-dependent —
130
+ the reference values in the upstream docs (6.448742 / 26.049634 / 121.157720) come from a **different
131
+ corpus** and are **not directly comparable**. Only the Swift-vs-base delta within one corpus means anything.
132
+
133
+ ### Speed and MTP acceptance
134
+
135
+ Greedy, 400 tokens per run, 1 warmup + 2 runs (median), **both models started alternately**.
136
+ Format: `t/s (acceptance)`.
137
+
138
+ | Config | Model | Chinese | English | Code | Thinking | Mean |
139
+ |---|---|---:|---:|---:|---:|---:|
140
+ | d1 | Swift | 42.2 (58%) | 44.4 (68%) | 46.0 (82%) | 45.8 (84%) | 44.6 |
141
+ | d1 | base | 43.1 (62%) | 44.9 (71%) | 46.6 (85%) | 45.4 (83%) | 45.0 |
142
+ | d2 | Swift | 46.0 (46%) | 49.5 (56%) | 58.4 (80%) | 55.9 (76%) | 52.5 |
143
+ | d2 | base | 43.7 (42%) | 46.9 (51%) | 59.6 (82%) | 55.7 (76%) | 51.5 |
144
+ | **d3** | Swift | 44.0 (33%) | 49.7 (44%) | 64.2 (70%) | 62.2 (68%) | **55.0** |
145
+ | **d3** | base | 43.4 (33%) | 52.4 (48%) | 64.9 (71%) | 59.9 (65%) | **55.1** |
146
+ | d4+lm (production) | Swift | 40.4 (28%) | 48.0 (40%) | 65.5 (67%) | 58.6 (59%) | 53.1 |
147
+ | d4+lm (production) | base | 39.2 (27%) | 48.2 (41%) | 64.6 (66%) | 57.1 (56%) | 52.3 |
148
+
149
+ **Three readings:**
150
+
151
+ 1. **Swift and base are within ±2 % in every configuration** — measurement noise. **The borrowed MTP
152
+ head works no worse with the Swift backbone than with base.**
153
+ 2. **`d3` is best, not the N=2 upstream reported for the 4080 SUPER.** d2 only leads on Chinese.
154
+ **The optimal MTP window is a property of the card, not of the model.**
155
+ 3. **`d3` (without `lm-head`) averages 2–5 % faster than the production `d4+lm`, on both models.**
156
+ This is an **observation not yet folded into the production config** — re-check with the official
157
+ bench before switching.
158
+
159
+ ### Reasoning quiz (12 unique-answer arithmetic/logic items, production params, engine default sampling)
160
+
161
+ | | Round 1 | Round 2 | Total output tokens |
162
+ |---|---|---|---:|
163
+ | Swift | 12/12 | 12/12 | 4,501 |
164
+ | base | 12/12 | 12/12 | 5,144 |
165
+
166
+ **Same accuracy; Swift emitted 12.5 % fewer tokens — but the sample is too small.** Per-item
167
+ distributions overlap heavily and the same model varies up to 2× between rounds, so no significant
168
+ conclusion follows. All replies ended with `finish=stop`; no loops, no truncation.
169
+
170
+ ### Functional checks (production params: ctx 65536 / `int8` / MTP d4+lm / vision)
171
+
172
+ | Item | Result |
173
+ |---|---|
174
+ | Startup | ✅ `/v1/models` returns `qwen3.8-27b`, `max_model_len` = 65536 |
175
+ | English / Chinese | ✅ coherent |
176
+ | Multi-turn, same prefix ×3 | ✅ `200/200/200/200`; 2nd and 3rd hits a 40-token cache with identical output |
177
+ | 10 consecutive distinct requests | ✅ all 200 |
178
+ | Sampling | ✅ two runs differ |
179
+ | Reasoning tiers | ✅ `none/low/medium/xhigh` normal; `high` returns 400 as expected (same as base) |
180
+ | Prefix reuse (~2.4k tokens, 3 rounds) | ✅ prompt time **2848 ms → 258 / 268 ms** |
181
+ | Vision | ✅ correctly identified a red square, `CAT`, and a blue `42` |
182
+ | Tool calling | ✅ returned `get_weather({"city":"Tokyo"})` |
183
+ | No tool call at `reasoning_effort=none` | ⚠️ Same known behaviour as base — not an artifact issue |
184
+
185
+ > **On the prefix-reuse crash:** the upstream-documented failure (a second multi-turn/same-prefix
186
+ > request returning 500, then the whole port going 503) **did not reproduce** in this environment —
187
+ > that engine build already fixes it. On an older build, the workaround is `--no-prefix-reuse`.
188
+
189
+ ---
190
+
191
+ ## Reproduction
192
+
193
+ ```bash
194
+ # 1. source (ukisai's Swift fine-tune, already ternary-quantized)
195
+ Swift-Bonsai-2-PQ2_0.gguf
196
+ sha256 5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5
197
+
198
+ # 2. template — only its skeleton/object list and vision/MTP/frontend payloads are used.
199
+ # Its text weights are discarded entirely.
200
+ Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
201
+ sha256 8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03
202
+
203
+ # 3. pack
204
+ python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
205
+ --gguf Swift-Bonsai-2-PQ2_0.gguf \
206
+ --template qwen3_8_27b_huihui_abliterated.ninfer
207
+ # ~4 minutes, CPU only, no GPU required
208
+ ```
209
+
210
+ Full method: [`REPRODUCE.md`](REPRODUCE.md). The packer and the per-tensor mapping table ship in
211
+ [`tools/`](tools/) under Apache-2.0.
212
+
213
+ ---
214
+
215
+ ## Attribution (all three layers required)
216
+
217
+ ```
218
+ Bonsai 2 27B (original weights) © Prism ML, Inc. Apache-2.0
219
+ huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
220
+ Qwen3.8-27B (geometry base) © Alibaba Cloud Apache-2.0
221
+ Swift fine-tune (weight source) ukisai Apache-2.0
222
+ huggingface.co/ukisai/Swift-Bonsai-2-GGUF
223
+ Packer pack.py — shensanshu/ninfer-ada-ternary (Apache-2.0)
224
+ Container format / engine github.com/Neroued/ninfer (Apache-2.0)
225
+ ```
226
+
227
+ Required upstream attribution string:
228
+
229
+ > **"Created using Bonsai by Prism ML."**
230
+
231
+ See [`NOTICE`](NOTICE) — it also discloses one non-Apache link upstream in the fine-tune family.
232
+
233
+ ---
234
+
235
+ ## Known pitfalls (inherited, not specific to this artifact)
236
+
237
+ 1. **MTP window:** N=2 is optimal; N=3/4/5 lose acceptance (measured on 4080 SUPER; retest on yours)
238
+ 2. **Prefix-reuse crash:** a second multi-turn/same-prefix request returns 500, then the whole port
239
+ goes 503. Workaround: `--no-prefix-reuse`
240
+ 3. **After editing any `.h`, touch all `.cpp`** — MSVC's header dependency scan is broken here, so
241
+ changes silently do not take effect (`/GS` crash `0xC0000409`)
242
+ 4. **Non-ASCII prompts via the command line fail NFC normalization** — use `--messages <json>` (UTF-8, no BOM)
243
+ 5. **`ninfer-perplexity.exe` must be built in the same batch as `ninfer-serve`** — rebuilding only
244
+ the server leaves a stale tool and makes you misjudge a change as ineffective
245
+
246
+ ---
247
+
248
+ ## License
249
+
250
+ **Apache-2.0.** See [`LICENSE`](LICENSE) and [`NOTICE`](NOTICE).
251
+
252
+ **Trademarks:** "Qwen" is a trademark of Alibaba Cloud; "Bonsai" and "Prism ML" belong to
253
+ Prism ML, Inc. This is an unofficial community-produced derivative and is not endorsed by or
254
+ affiliated with Alibaba Cloud, Prism ML, ukisai, shensanshu, or the NInfer project.
REPRODUCE.md ADDED
@@ -0,0 +1,316 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # 复现步骤 · REPRODUCE
2
+
3
+ 本仓的两个 `.ninfer` 制品是怎么造出来的 —— 从零到产出,可逐步复跑。
4
+
5
+ **全程 CPU,不需要 GPU。** 实测约 4 分钟一份。
6
+
7
+ ---
8
+
9
+ ## 0. 依赖
10
+
11
+ ```
12
+ Python 3.11+
13
+ numpy
14
+ torch ← CPU 版即可(打包器不用 GPU,路径上写死了 torch.device("cpu"))
15
+ ```
16
+
17
+ ```bash
18
+ pip install numpy
19
+ pip install torch --index-url https://download.pytorch.org/whl/cpu
20
+ ```
21
+
22
+ **如果你的网络在国内**,Hugging Face 走 `hf-mirror.com` 比走代理快约 50%(实测 7.65 MB/s vs 5.02 MB/s):
23
+
24
+ ```bash
25
+ export HF_ENDPOINT=https://hf-mirror.com
26
+ ```
27
+
28
+ ---
29
+
30
+ ## 1. 下载源与模板
31
+
32
+ ### 1.1 权重源(ukisai 的 Swift 微调)
33
+
34
+ ```
35
+ https://huggingface.co/ukisai/Swift-Bonsai-2-GGUF
36
+ ```
37
+
38
+ | 文件 | 大小 | sha256 |
39
+ |---|---|---|
40
+ | `Swift-Bonsai-2-PQ2_0.gguf` | 7,206,168,928 B | `5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5` |
41
+ | `Swift-Bonsai-2-PTQ1_0.gguf` | 5,946,648,960 B | `de33620b60eaf63e96449b478eb87abe9538507e3ed939da944c1ae15fe1ffc6` |
42
+
43
+ ### 1.2 模板
44
+
45
+ ```
46
+ https://huggingface.co/Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
47
+ ```
48
+
49
+ | 文件 | 大小 | sha256 |
50
+ |---|---|---|
51
+ | `qwen3_8_27b_huihui_abliterated.ninfer` | 18,210,531,328 B | `8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03` |
52
+
53
+ **模板的身份必须满足:**
54
+ ```json
55
+ "identity": {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
56
+ ```
57
+
58
+ **并且对象名必须是"未融合"的那套**(`attention/query_key` + `attention/gate_value` 分开)。
59
+ `nvfp4` 打包的制品**用不了** —— 它把投影融合成了 `attention/query_key_gate_value`,
60
+ 名字不在映射表里,打包器会直接中止。
61
+
62
+ ### 1.3 打包器
63
+
64
+ `tools/` 目录随本仓附带(Apache-2.0,原样取自 `shensanshu/ninfer-ada-ternary`)。
65
+
66
+ 它需要一个 `NINFER_ROOT` 指向的源码 checkout —— 那里只用来 import `tools.artifact` 这一组模块:
67
+
68
+ ```
69
+ <NINFER_ROOT>/
70
+ tools/
71
+ __init__.py
72
+ artifact/
73
+ __init__.py
74
+ container.py
75
+ layouts.py
76
+ numeric.py
77
+ ```
78
+
79
+ 来源是 `Ambolio/ninfer-4090-windows` 血统的树(或任何含这套 `tools/artifact` 的 checkout)。
80
+ **`pack.py` 源码顶部有一个 `NINFER_ROOT` 默认值**,指到你的实际路径上即可。
81
+
82
+ ---
83
+
84
+ ## 2. 先验(不写文件)
85
+
86
+ ```bash
87
+ python -u tools/pack.py check \
88
+ --gguf Swift-Bonsai-2-PQ2_0.gguf \
89
+ --template qwen3_8_27b_huihui_abliterated.ninfer
90
+ ```
91
+
92
+ **期望输出:**
93
+
94
+ ```
95
+ CHECK 1 geometry over every ternary shape present in the GGUF
96
+ PQ2_0_G128 n=1024 k=5120 groups/row=40 src= 1,392,640 payload= 1,392,640 pad= 0 (32 tensors)
97
+ PQ2_0_G128 n=5120 k=6144 groups/row=48 src= 8,355,840 payload= 8,355,840 pad= 0 (64 tensors)
98
+ PQ2_0_G128 n=5120 k=17408 groups/row=136 src= 23,674,880 payload= 23,674,880 pad= 0 (64 tensors)
99
+ PQ2_0_G128 n=6144 k=5120 groups/row=40 src= 8,355,840 payload= 8,355,840 pad= 0 (48 tensors)
100
+ PQ2_0_G128 n=10240 k=5120 groups/row=40 src= 13,926,400 payload= 13,926,400 pad= 0 (48 tensors)
101
+ PQ2_0_G128 n=12288 k=5120 groups/row=40 src= 16,711,680 payload= 16,711,680 pad= 0 (16 tensors)
102
+ PQ2_0_G128 n=17408 k=5120 groups/row=40 src= 23,674,880 payload= 23,674,880 pad= 0 (128 tensors)
103
+ PQ2_0_G128 n=248320 k=5120 groups/row=40 src= 337,715,200 payload= 337,715,200 pad= 0 (2 tensors)
104
+ distinct (format,shape) combos: 8
105
+
106
+ CHECK 2 byte round trip + decode equality on real tensors
107
+ blk.3.attn_q.weight bytes_equal=True decode_equal=True zero_share=0.3279 … PLAUSIBLE
108
+ …
109
+ ```
110
+
111
+ **判据:`pad` 全为 0;`bytes_equal` 与 `decode_equal` 全 True;`zero_share` 落在 0.3277~0.3280。**
112
+
113
+ ### ⚠️ 关于 `check` 的一个已知失败
114
+
115
+ 在内存较小的机器上,`CHECK 2` 会在**超大张量**(`token_embd` / `output`,12.7 亿权重)上抛:
116
+
117
+ ```
118
+ numpy.core._exceptions._ArrayMemoryError: Unable to allocate 9.47 GiB
119
+ for an array with shape (9932800, 128) and data type float64
120
+ ```
121
+
122
+ **这不是转换错误** —— 那是 `mode_check` 为了做 decode 比对而做的**全量反量化**(`dq_pq2_0` 返回
123
+ float64 数组)。`mode_build` 不用这条路径(它是流式的),所以**内存不够时可以直接进第 3 步**。
124
+
125
+ 若确实想跑完 `check`:需要 ≥16 GiB 空闲内存。
126
+
127
+ ---
128
+
129
+ ## 3. 打包
130
+
131
+ ```bash
132
+ python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
133
+ --gguf Swift-Bonsai-2-PQ2_0.gguf \
134
+ --template qwen3_8_27b_huihui_abliterated.ninfer
135
+ ```
136
+
137
+ PTQ1 档同理,把两个路径换成 PTQ1 的即可。
138
+
139
+ **期望输出:**
140
+
141
+ ```
142
+ wrote bonsai2_27b_swift_pq2.ninfer
143
+ total file : 8,306,927,628 B = 7.736 GiB
144
+ text part produced: 7,189,880,844 B = 6.696 GiB
145
+ borrowed payloads : 1,116,856,537 B = 1.040 GiB {'frontend': 6, 'text': 2, 'mtp': 12, 'vision': 333}
146
+ objects : 1126
147
+ ```
148
+
149
+ PTQ1 档应��得到:
150
+
151
+ ```
152
+ total file : 7,047,407,628 B = 6.563 GiB
153
+ text part produced: 5,930,360,844 B = 5.523 GiB
154
+ ```
155
+
156
+ **这两个数字是硬判据** —— 任何偏差都说明源或模板拿错了。
157
+
158
+ ---
159
+
160
+ ## 4. 核对产出
161
+
162
+ ```bash
163
+ python - <<'PY'
164
+ import json, collections, os
165
+ p = "bonsai2_27b_swift_pq2.ninfer"
166
+ raw = open(p,"rb").read(16<<20)
167
+ obj,_ = json.JSONDecoder().raw_decode(raw[16:].decode("utf-8","replace"))
168
+ objs = obj["objects"]
169
+ print("identity:", obj["identity"])
170
+ print("objects :", len(objs))
171
+ print("formats :", dict(sorted(collections.Counter(
172
+ o.get("format") for o in objs if o.get("kind")=="tensor").items())))
173
+ print("layouts :", dict(collections.Counter(
174
+ o.get("layout") for o in objs if o.get("kind")=="tensor")))
175
+ names = {o["name"] for o in objs}
176
+ print("hadamard_signs :", "text/hadamard_signs" in names)
177
+ print("hadamard_widths:", "text/hadamard_widths" in names)
178
+ PY
179
+ ```
180
+
181
+ **期望(PQ2 档):**
182
+
183
+ ```
184
+ identity: {'model_id': 'qwen3.8-27b', 'weights_id': 'groupwise-int'}
185
+ objects : 1126
186
+ formats : {'BF16': 582, 'PQ2_0_G128': 322, 'FP32': 97, 'Q4G64_F16S': 55,
187
+ 'Q5G64_F16S': 54, 'W8G32_F16S': 7, 'I32': 2, 'Q6G64_F16S': 1}
188
+ layouts : {'row-split-k128-v1': 439, 'contiguous-le-v1': 681}
189
+ hadamard_signs : True
190
+ hadamard_widths: True
191
+ ```
192
+
193
+ `objects` 比模板多 2 —— 就是那两个 hadamard 对象。**模板里没有它们,产出里有**,
194
+ 这正是三元制品需要的。
195
+
196
+ ---
197
+
198
+ ## 5. 打包器在做什么(逐步)
199
+
200
+ 理解这一步,才能在出错时判断问题在哪。
201
+
202
+ ### 5.1 三元码原样搬运
203
+
204
+ GGML 的三元 block 有两个成员(`PQ2_0` 是 `{qs}` base + fp16 scale;`PTQ1_0` 多一个
205
+ `{qh}` high 平面),`.ninfer` 的 `row-split-k128-v1` 布局恰好是**同样三个平面**。
206
+ 所以码字**逐字节搬**,不经过浮点。
207
+
208
+ **这条极其重要**:解量化再重量化会引入二次损失,而且体积红利会被吃掉。
209
+
210
+ ### 5.2 撤销 llama.cpp exporter 的约定
211
+
212
+ 源 GGUF 是 llama.cpp 生态导出的,带着三处约定,产物必须还原:
213
+
214
+ ```
215
+ GDN value heads tiled 顺序 -> grouped 顺序
216
+ 按头粒度施加:perm48(head)*128 + inner
217
+ ⚠️ 直接套 perm48(row) 是非双射,会静默产生重复行 + 丢失行
218
+ (这类 bug 守恒一切可数之物:尺寸/行数/字节/往返无损全绿)
219
+
220
+ 零中心 norms 1 + w -> w
221
+ ⚠️ 唯独 gdn/norm(ssm_norm)原样 —— 它的 raw 已经 ≈ +1
222
+
223
+ ssm_a -exp(A_log) -> A_log
224
+ ```
225
+
226
+ ### 5.3 补两个 hadamard 对象
227
+
228
+ 源 GGUF 用 `prism.hadamard.sign_values` / `sign_widths` 承载旋转基的符号向量;
229
+ `.ninfer` 用 `text/hadamard_signs`(28,672 个 fp32)/ `text/hadamard_widths`(3 个 int32)。
230
+ 打包器把前者翻成后者,并给每个被旋转的投影挂上 `hadamard_signs` Use 辅助。
231
+
232
+ ### 5.4 借 payload
233
+
234
+ ```
235
+ vision 333 个 —— 源 GGUF 里没有(Bonsai 的视觉塔是单独的 mmproj.gguf)
236
+ mtp 12 个 —— 源 GGUF 里没有(851 个张量里零个 blk.64.*)
237
+ frontend 6 个 —— tokenizer 等
238
+ draft_head 2 个 —— 频次短名单,同一 tokenizer 即同一名单
239
+ 共 1,116,856,537 B
240
+ ```
241
+
242
+ **这几块是全生态共用的官方件**,从模板借是安全的(MTP 头与目标点积 ≥0.99966、七个 norm 逐字节相同)。
243
+
244
+ ---
245
+
246
+ ## 6. 验证方法学(**这一节比上面的步骤更值钱**)
247
+
248
+ 「搬运无损」类判据(源字节 == payload、往返解码一致)**对偏移错误完全盲** ——
249
+ 读错的同一批字节原样进原样出,照样全绿。
250
+
251
+ 真正能证伪的判据:
252
+
253
+ | # | 判据 | 能抓什么 |
254
+ |---|---|---|
255
+ | 1 | **分布特征自证**(scale 中位数 / 全正 / `zero_share`) | 载荷整体偏移、平面错位 |
256
+ | 2 | **跨实现互验**(两个独立解码器解同一份数据) | 单一实现的系统性误读 |
257
+ | 3 | **负控必须存在**(假格式名必须被拒、错尺寸必须被拒) | "判据太松"导致的假通过 |
258
+ | 4 | **行级指纹比「多重集」+ 直接比对** | 行置换 / 重复 / 丢失(尺寸守恒那类) |
259
+ | 5 | **T>1 用例 + 引擎侧验证** | 布局 / 步长类错误(T=1 两种排布重合,测不出任何东西) |
260
+ | 6 | **端到端数值口径用 PPL** | 采样温度 / 模板带来的错觉 |
261
+
262
+ ### ⚠️ PPL 的一个常见误用:拿别处的绝对值来比
263
+
264
+ **PPL 跟语料强相关。** 上游文档给过一组参照值(`512/256 → 6.448742`、`32/16 → 26.049634`、
265
+ `8/4 → 121.157720`),**那是另一份语料上的数**,和你自己量到的值**不可直接比较**。
266
+
267
+ **只有同一语料下两个模型的差值才有意义。** 本仓的实测(`pplab-text`,`int8` KV,RTX 3060 12G):
268
+
269
+ | 窗口/步长 | Swift | base | 差 |
270
+ |---|---:|---:|---:|
271
+ | 512 / 256 | 8.073669 | 8.079208 | −0.07% |
272
+ | 32 / 16 | 26.962015 | 26.972171 | −0.04% |
273
+ | 8 / 4 | 148.3656 | 148.3863 | −0.01% |
274
+
275
+ **你复现时应该看到同样的"差值方向",而不是同样的绝对值。**
276
+
277
+ **`zero_share` 精确落到 0.3278 是 `PQ2_0` 的格式指纹**(理论零码占比 32.776%),不是"大概的数"。
278
+ 本仓的两个制品都过了这一条。
279
+
280
+ ### 性能测量的坑(会直接误导判断)
281
+
282
+ ```
283
+ micro-benchmark 不 flush L2 会高估;flush 过头会报出超物理上限的数
284
+ nsys 默认不追踪 CUDA graph replay 内的 kernel ⇒ --cuda-graph-trace=node
285
+ 内核"少干一半活"会伪装成提速 ⇒ 必须先用 rel_l2 校验
286
+ kernel launch 失败伪装成"快得离谱 + 全零" ⇒ launch 后查 cudaGetLastError()
287
+ idle 时钟让短基准严重失真 ⇒ 短基准不可信
288
+ 报数不带任务和生成长度 ⇒ 轮耗时才是任务无关量
289
+ ```
290
+
291
+ ---
292
+
293
+ ## 7. 已知的失败与对策
294
+
295
+ | 症状 | 真因 | 对策 |
296
+ |---|---|---|
297
+ | `template has no object X to borrow` | 模板与映射表 schema 不一致 | 换 `groupwise-int` 的模板,别用 `nvfp4` |
298
+ | `unmapped gdn object` / 直接中止 | 模板是 `nvfp4` 打包(投影被融合) | 同上 |
299
+ | `Unable to allocate 9.47 GiB`(`check` 模式) | 全量反量化的 float64 数组 | 内存 ≥16 GiB,或直接跑 `build` |
300
+ | 产出尺寸与期望不符 | 源或模板拿错 | 对 sha256 |
301
+ | 引擎拒绝装载产物 | 方言不对(`t2_g128_fp16` vs `PQ2_0_G128`) | 见根目录 README 的兼容性表 |
302
+
303
+ ---
304
+
305
+ ## 8. 本仓实际使用的命令与结果
306
+
307
+ ```
308
+ packer tools/pack.py(NINFER_ROOT 常量重定向到本地 checkout,无功能修改)
309
+ Python 3.11.5
310
+ GPU 未使用(纯 CPU)
311
+ PQ2 档 4 分钟 → 8,306,927,628 B / 7.736 GiB
312
+ PTQ1 档 4 分钟 → 7,047,407,628 B / 6.563 GiB
313
+ ```
314
+
315
+ **没有任何验证、校验、几何检查或往返证明被禁用、放宽或绕过。**
316
+ `check` 模式在源 GGUF 上完整跑过(见第 2 节)。
SHA256SUMS ADDED
@@ -0,0 +1,2 @@
 
 
 
1
+ cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0 bonsai2_27b_swift_pq2.ninfer
2
+ cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae bonsai2_27b_swift_ptq1.ninfer
artifact-manifest.json ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema": "ninfer.artifact-manifest",
3
+ "schema_version": 1,
4
+ "release": "Swift-Bonsai-2 27B — NInfer ternary artifacts",
5
+ "generated": "2026-09-28T03:12:45",
6
+ "license": "Apache-2.0",
7
+ "provenance": {
8
+ "base_ternary": {
9
+ "repo": "prism-ml/Ternary-Bonsai-2-27B-gguf",
10
+ "license": "apache-2.0"
11
+ },
12
+ "weight_source": {
13
+ "repo": "ukisai/Swift-Bonsai-2-GGUF",
14
+ "license": "apache-2.0"
15
+ },
16
+ "geometry_base": {
17
+ "repo": "Qwen/Qwen3.8-27B",
18
+ "license": "apache-2.0"
19
+ },
20
+ "packer": {
21
+ "repo": "shensanshu/ninfer-ada-ternary",
22
+ "license": "Apache-2.0",
23
+ "path": "tools/pack.py"
24
+ },
25
+ "template": {
26
+ "repo": "Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer",
27
+ "license": "Apache-2.0",
28
+ "used_for": "object skeleton + vision/mtp/frontend/draft_head payloads only"
29
+ }
30
+ },
31
+ "artifacts": [
32
+ {
33
+ "filename": "bonsai2_27b_swift_pq2.ninfer",
34
+ "bytes": 8306927628,
35
+ "sha256": "cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0",
36
+ "identity": {
37
+ "model_id": "qwen3.8-27b",
38
+ "weights_id": "groupwise-int"
39
+ },
40
+ "container_version": 2,
41
+ "objects": 1126,
42
+ "tensors": 1120,
43
+ "resources": 6,
44
+ "formats": {
45
+ "BF16": 582,
46
+ "FP32": 97,
47
+ "I32": 2,
48
+ "PQ2_0_G128": 322,
49
+ "Q4G64_F16S": 55,
50
+ "Q5G64_F16S": 54,
51
+ "Q6G64_F16S": 1,
52
+ "W8G32_F16S": 7
53
+ },
54
+ "layouts": {
55
+ "contiguous-le-v1": 681,
56
+ "row-split-k128-v1": 439
57
+ },
58
+ "ternary_format": "PQ2_0_G128",
59
+ "requires_engine_dialect": "Ambolio/ninfer-4090-windows lineage (v1.0.6 / v1.0.8)"
60
+ },
61
+ {
62
+ "filename": "bonsai2_27b_swift_ptq1.ninfer",
63
+ "bytes": 7047407628,
64
+ "sha256": "cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae",
65
+ "identity": {
66
+ "model_id": "qwen3.8-27b",
67
+ "weights_id": "groupwise-int"
68
+ },
69
+ "container_version": 2,
70
+ "objects": 1126,
71
+ "tensors": 1120,
72
+ "resources": 6,
73
+ "formats": {
74
+ "BF16": 582,
75
+ "FP32": 97,
76
+ "I32": 2,
77
+ "PTQ1_0_G128": 322,
78
+ "Q4G64_F16S": 55,
79
+ "Q5G64_F16S": 54,
80
+ "Q6G64_F16S": 1,
81
+ "W8G32_F16S": 7
82
+ },
83
+ "layouts": {
84
+ "contiguous-le-v1": 681,
85
+ "row-split-k128-v1": 439
86
+ },
87
+ "ternary_format": "PTQ1_0_G128",
88
+ "requires_engine_dialect": "Ambolio/ninfer-4090-windows lineage (v1.0.6 / v1.0.8)"
89
+ }
90
+ ]
91
+ }
bonsai2_27b_swift_pq2.ninfer ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0
3
+ size 8306927628
bonsai2_27b_swift_ptq1.ninfer ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae
3
+ size 7047407628
release.conversion.json ADDED
@@ -0,0 +1,53 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "identity": {
3
+ "model_id": "qwen3.8-27b",
4
+ "weights_id": "groupwise-int"
5
+ },
6
+ "purpose": "Swift-Bonsai-2 27B ternary artifact, PQ2 + PTQ1 tiers, NInfer v2 container",
7
+ "packer": {
8
+ "name": "tools/pack.py",
9
+ "origin": "shensanshu/ninfer-ada-ternary (Apache-2.0)",
10
+ "change": "only the NINFER_ROOT default path constant was redirected; no functional change",
11
+ "verification_mode": "check ran to completion: pad=0 for all 8 shape combinations, bytes_equal and decode_equal true on every sampled tensor"
12
+ },
13
+ "sources": {
14
+ "gguf_pq2": {
15
+ "file": "Swift-Bonsai-2-PQ2_0.gguf",
16
+ "bytes": 7206168928,
17
+ "sha256": "5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5",
18
+ "from": "ukisai/Swift-Bonsai-2-GGUF"
19
+ },
20
+ "gguf_ptq1": {
21
+ "file": "Swift-Bonsai-2-PTQ1_0.gguf",
22
+ "bytes": 5946648960,
23
+ "sha256": "de33620b60eaf63e96449b478eb87abe9538507e3ed939da944c1ae15fe1ffc6",
24
+ "from": "ukisai/Swift-Bonsai-2-GGUF"
25
+ },
26
+ "template": {
27
+ "file": "qwen3_8_27b_huihui_abliterated.ninfer",
28
+ "bytes": 18210531328,
29
+ "sha256": "8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03",
30
+ "from": "Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer",
31
+ "note": "used ONLY as skeleton/manifest + donor of vision/mtp/frontend/draft_head payloads; its text weights are discarded"
32
+ }
33
+ },
34
+ "outputs": {
35
+ "bonsai2_27b_swift_pq2.ninfer": {
36
+ "bytes": 8306927628,
37
+ "sha256": "cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0",
38
+ "note": "PQ2 tier, 2-bit, from gguf_pq2"
39
+ },
40
+ "bonsai2_27b_swift_ptq1.ninfer": {
41
+ "bytes": 7047407628,
42
+ "sha256": "cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae",
43
+ "note": "PTQ1 tier, 1.75-bit, from gguf_ptq1"
44
+ }
45
+ },
46
+ "toolchain": {
47
+ "python": "3.11.5",
48
+ "numpy": "installed",
49
+ "torch": "CPU build",
50
+ "gpu_used": false
51
+ },
52
+ "generated": "2026-09-28T03:10:53"
53
+ }
tools/MAPPING.json ADDED
@@ -0,0 +1,126 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_doc": "Bonsai 2 27B GGUF (arch qwen35) -> ninfer artifact object mapping. Every rule below was established EMPIRICALLY against a groupwise-int qwen3.8-27b artifact of the same model family, except where marked INFERRED. Rotated (hadamard-folded) matrices cannot be correlated - only byte accounting applies to them.",
3
+ "_verified_by": "probe_conventions.py, solve_gdn_order.py (see <WORKSPACE>\\ternary-pack\\)",
4
+ "_layers": {
5
+ "total": 64,
6
+ "gdn_linear_attention": [0,1,2,4,5,6,8,9,10,12,13,14,16,17,18,20,21,22,24,25,26,28,29,30,32,33,34,36,37,38,40,41,42,44,45,46,48,49,50,52,53,54,56,57,58,60,61,62],
7
+ "full_attention": [3,7,11,15,19,23,27,31,35,39,43,47,51,55,59,63],
8
+ "note": "derived: a layer is full-attention iff blk.{l}.attn_q.weight exists in the GGUF"
9
+ },
10
+ "_constants": {
11
+ "hidden": 5120,
12
+ "q_heads": 24, "kv_heads": 4, "head_dim": 256,
13
+ "num_k_heads": 16, "num_v_heads": 48, "v_per_k": 3, "head_k_dim": 128, "head_v_dim": 128,
14
+ "gdn_qkv_channels": 10240, "gdn_qk_channels": 4096, "gdn_v_channels": 6144
15
+ },
16
+
17
+ "_rules": [
18
+ {
19
+ "id": "norm_shift",
20
+ "applies_to": "every BF16/FP32 object that came from a GGUF tensor ending in '.norm.weight'",
21
+ "transform": "value = gguf - 1.0",
22
+ "exception": "text/layers/{l}/gdn/norm <- blk.{l}.ssm_norm.weight is copied RAW (no -1)",
23
+ "evidence": "cos(artifact, gguf-1) = +0.99997 .. +0.99991 for input_norm / post_attention_norm / query_norm / key_norm / final_norm; cos(artifact, gguf) = +0.99961 for gdn/norm",
24
+ "source": "llama.cpp conversion/qwen.py:394 -- `name.endswith(\"norm.weight\") and not name.endswith(\"linear_attn.norm.weight\")` gets +1 on the way OUT to GGUF, so ninfer (HF-form) needs -1 back, except for linear_attn.norm"
25
+ },
26
+ {
27
+ "id": "gdn_v_tiled_to_grouped",
28
+ "applies_to": "the 48-row (num_v_heads) axis of GDN tensors",
29
+ "transform": "t = t.reshape(3,16,*rest).transpose(1,0,*range(2,ndim+1)).reshape(shape)",
30
+ "direction": "GGUF stores TILED, ninfer wants GROUPED",
31
+ "evidence": "48x48 row matching gives a bijection with cos 0.990..0.9998; cos(A, tiled_to_grouped(B)) = +0.99587 vs +0.14814 for the other direction",
32
+ "source": "conversion/qwen.py:446 _LinearAttentionVReorderBase._reorder_v_heads (HF grouped -> ggml tiled)"
33
+ },
34
+ {
35
+ "id": "gdn_a_log",
36
+ "applies_to": "text/layers/{l}/gdn/a_log <- blk.{l}.ssm_a",
37
+ "transform": "tiled_to_grouped(log(-gguf))",
38
+ "evidence": "cos(sorted(artifact), sorted(log(-gguf))) = +1.00000 exactly; raw gives +0.796",
39
+ "source": "conversion/qwen.py:389 -- `if name.endswith(\".A_log\"): data_torch = -torch.exp(data_torch)`"
40
+ },
41
+ {
42
+ "id": "gdn_conv1d",
43
+ "applies_to": "text/layers/{l}/gdn/convolution (4,10240) <- blk.{l}.ssm_conv1d.weight ne=(4,10240)",
44
+ "transform": "read as (10240,4); leave channels [0:4096] alone; apply tiled_to_grouped over the 48-head axis of channels [4096:10240]; then transpose -> shape (4,10240)",
45
+ "evidence": "cos = +0.99504; naively keeping the order gives only +0.72253",
46
+ "source": "conversion/qwen.py:608-615 (.conv1d branch reorders only the V channels)"
47
+ },
48
+ {
49
+ "id": "attn_q_per_head_interleave",
50
+ "applies_to": "blk.{l}.attn_q.weight (N=12288) for full-attention layers",
51
+ "transform": "view as 48 chunks of 256 rows; chunks 0,2,...,46 are the 24 query heads in order; chunks 1,3,...,47 are the 24 output-gate heads in order",
52
+ "evidence": "INFERRED for main layers from the MTP layer, where it was VERIFIED against the artifact at cos +0.99982 per chunk (24/24 and 24/24 chunks exact). The MTP layer is a full-attention layer with identical shapes.",
53
+ "target_use": "attention/query_key = concat(query(6144), attn_k(1024)); attention/gate_value = concat(gate(6144), attn_v(1024))"
54
+ },
55
+ {
56
+ "id": "gdn_value_z",
57
+ "transform": "value_z = concat(tiled_to_grouped_heads(attn_qkv[4096:10240]), tiled_to_grouped_heads(attn_gate[0:6144])). The 48-head permutation MUST be applied at HEAD granularity, because this tensor's row axis is 48 heads x 128 rows: src_row = perm48(row // 128) * 128 + row % 128",
58
+ "evidence": "RESOLVED 2026-09-19 by engine test (was INFERRED). ssm_conv1d's V channels are MEASURED as tiled and its rule already permutes at HEAD granularity (cos +0.99504); alpha/beta (48 rows = 48 heads, so row granularity IS head granularity) use the same permutation and measure +1.00000 / +0.99587.",
59
+ "bug_history": "The first pack applied perm48() straight to the 6144-row index. That is NOT a bijection over rows (perm48(1) == perm48(48) == 16), so 4064 of 6144 source rows were dropped and 2048 row fingerprints repeated -- while every size, row count and byte total stayed EXACTLY right, which is why the size reverse-check, the src==payload round trip, the byte-for-byte reconcile and the 12288-row cross-check all stayed green over it. Engine symptom: fluent-looking garbage, PPL 1,128,420 (worse than uniform over a 248,320-token vocab). Fixed by the perm_row() helper in pack.py; the v2 artifact verifies 6144/6144 distinct rows, 0 duplicated, 0 missing.",
60
+ "acceptance_check": "<BUILD_ROOT>\\check_row_order.py <artifact.ninfer> <Ternary-Bonsai-2-27B-PQ2_0.gguf> -- must report 'rows DUPLICATED: 0' and 'source rows MISSING: 0' AND 'documented rule holds row-by-row: True'. NOTE: a set-membership test is NOT sufficient (every duplicated row still finds a source); compare MULTISETS and compare artifact[a] directly against the rule's named source a, never by searching for a matching fingerprint."
61
+ },
62
+ {
63
+ "id": "gdn_output_raw",
64
+ "applies_to": "text/layers/{l}/gdn/output (5120,6144) <- blk.{l}.ssm_out.weight",
65
+ "transform": "none",
66
+ "evidence": "conversion/qwen.py:617-626 -- a hadamard-FOLDED out_proj keeps the training (grouped) column order and the runtime permutes the activation instead; Bonsai's metadata carries prism.hadamard.gdn_v_grouped = 1, confirming the folded path"
67
+ },
68
+ {
69
+ "id": "ternary_bytes_verbatim",
70
+ "applies_to": "all 402 PQ2_0 / PTQ1_0 matrices",
71
+ "transform": "move the quantized codes byte-for-byte; do NOT dequantize and requantize",
72
+ "note": "these weights live in the rotated basis W'; the activation-side transform is M2's job, not the packer's"
73
+ }
74
+ ],
75
+
76
+ "text_globals": {
77
+ "text/token_embedding": {"src": "token_embd.weight", "shape": [248320, 5120], "fmt": "PQ2_0_G128", "transform": "none (stored as W'; runtime inverse-applies after lookup)"},
78
+ "text/output_head": {"src": "output.weight", "shape": [248320, 5120], "fmt": "PQ2_0_G128", "transform": "none"},
79
+ "text/final_norm": {"src": "output_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
80
+ "text/draft_head": {"src": "GENERATED", "shape": [131072, 5120], "fmt": "Q4G64_F16S", "note": "frequency shortlist, NOT a model weight: tools/convert/qwen3_6/common/draft_head.py materializes it from tools/freq_corpus/fixtures/ranking/ranking.train.counts.i64 + the tokenizer. Regenerate, do not copy."},
81
+ "text/draft_head_token_ids": {"src": "GENERATED", "shape": [131072], "fmt": "I32", "note": "same generator"}
82
+ },
83
+
84
+ "per_layer_gdn": {
85
+ "text/layers/{l}/input_norm": {"src": "blk.{l}.attn_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
86
+ "text/layers/{l}/post_attention_norm": {"src": "blk.{l}.post_attention_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
87
+ "text/layers/{l}/mlp/gate_up": {"src": "concat(blk.{l}.ffn_gate.weight, blk.{l}.ffn_up.weight) along rows", "shape": [34816, 5120], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim; concat order VERIFIED at cos +0.99982 on the MTP analogue"},
88
+ "text/layers/{l}/mlp/down": {"src": "blk.{l}.ffn_down.weight", "shape": [5120, 17408], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
89
+ "text/layers/{l}/gdn/query_key": {"src": "blk.{l}.attn_qkv.weight rows 0:4096", "shape": [4096, 5120], "fmt": "TERNARY", "transform": "none (q 2048 + k 2048, no V tiling on the k-head axis)"},
90
+ "text/layers/{l}/gdn/value_z": {"src": "rows 4096:10240 of blk.{l}.attn_qkv.weight + blk.{l}.attn_gate.weight", "shape": [12288, 5120], "fmt": "TERNARY", "transform": "gdn_value_z"},
91
+ "text/layers/{l}/gdn/output": {"src": "blk.{l}.ssm_out.weight", "shape": [5120, 6144], "fmt": "TERNARY", "transform": "none"},
92
+ "text/layers/{l}/gdn/convolution": {"src": "blk.{l}.ssm_conv1d.weight", "shape": [4, 10240], "fmt": "BF16", "transform": "gdn_conv1d"},
93
+ "text/layers/{l}/gdn/norm": {"src": "blk.{l}.ssm_norm.weight", "shape": [128], "fmt": "BF16", "transform": "RAW (no -1!)"},
94
+ "text/layers/{l}/gdn/a_projection": {"src": "blk.{l}.ssm_alpha.weight", "shape": [48, 5120], "fmt": "BF16", "transform": "gdn_v_tiled_to_grouped"},
95
+ "text/layers/{l}/gdn/b_projection": {"src": "blk.{l}.ssm_beta.weight", "shape": [48, 5120], "fmt": "BF16", "transform": "gdn_v_tiled_to_grouped"},
96
+ "text/layers/{l}/gdn/a_log": {"src": "blk.{l}.ssm_a", "shape": [48], "fmt": "FP32", "transform": "gdn_a_log"},
97
+ "text/layers/{l}/gdn/dt_bias": {"src": "blk.{l}.ssm_dt.bias", "shape": [48], "fmt": "FP32", "transform": "gdn_v_tiled_to_grouped (pure permutation)"}
98
+ },
99
+
100
+ "per_layer_attention": {
101
+ "text/layers/{l}/input_norm": {"src": "blk.{l}.attn_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
102
+ "text/layers/{l}/post_attention_norm": {"src": "blk.{l}.post_attention_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
103
+ "text/layers/{l}/mlp/gate_up": {"src": "concat(blk.{l}.ffn_gate.weight, blk.{l}.ffn_up.weight)", "shape": [34816, 5120], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
104
+ "text/layers/{l}/mlp/down": {"src": "blk.{l}.ffn_down.weight", "shape": [5120, 17408], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
105
+ "text/layers/{l}/attention/query_key": {"src": "attn_q (de-interleaved) 6144 rows + blk.{l}.attn_k.weight 1024 rows", "shape": [7168, 5120], "fmt": "TERNARY", "transform": "attn_q_per_head_interleave"},
106
+ "text/layers/{l}/attention/gate_value": {"src": "attn_q (odd chunks) 6144 rows + blk.{l}.attn_v.weight 1024 rows", "shape": [7168, 5120], "fmt": "TERNARY", "transform": "attn_q_per_head_interleave"},
107
+ "text/layers/{l}/attention/output": {"src": "blk.{l}.attn_output.weight", "shape": [5120, 6144], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
108
+ "text/layers/{l}/attention/query_norm": {"src": "blk.{l}.attn_q_norm.weight", "shape": [256], "fmt": "BF16", "transform": "raw - 1"},
109
+ "text/layers/{l}/attention/key_norm": {"src": "blk.{l}.attn_k_norm.weight", "shape": [256], "fmt": "BF16", "transform": "raw - 1"}
110
+ },
111
+
112
+ "_counts": {
113
+ "artifact_objects_total": 1124,
114
+ "text": 773, "mtp": 12, "vision": 333, "frontend_resources": 6,
115
+ "gguf_tensors_total": 851,
116
+ "gguf_breakdown": "48 GDN layers x 14 + 16 attention layers x 11 + 3 globals (token_embd, output, output_norm) = 851",
117
+ "note": "text/draft_head + draft_head_token_ids have NO GGUF source (generated); vision has no source in this GGUF either (Bonsai ships a separate mmproj)"
118
+ },
119
+
120
+ "_unresolved": [
121
+ "vision/* (333 objects): Bonsai's vision tower lives in Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf (Q8_0), while the template expects Q4G64/Q5G64/Q6G64/BF16. Either convert the mmproj or borrow the template's vision payloads (same official vision tower) - needs a decision.",
122
+ "text/draft_head + draft_head_token_ids: run the frequency-shortlist generator (needs tools/freq_corpus fixture present in the fork).",
123
+ "all 402 ternary matrices are in the rotated basis, so NONE of them can be verified by correlation; only byte accounting, decode round-trip equality and the engine's load test apply.",
124
+ "gdn/value_z's tiled->grouped choice is inferred (see rule) - single line to flip if the engine disagrees."
125
+ ]
126
+ }
tools/README.md ADDED
@@ -0,0 +1,77 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # tools/ — 打包器与验证脚本
2
+
3
+ **许可:Apache-2.0。** 来源见根目录 `NOTICE`。
4
+
5
+ 这些文件原样取自 [`shensanshu/ninfer-ada-ternary`](https://www.modelscope.cn/models/shensanshu/ninfer-ada-ternary)
6
+ (Apache-2.0),**未做功能修改**。
7
+
8
+ 唯一的编辑是**脱敏**:源码顶部的两个默认路径常量(`NINFER_ROOT` 指向作者的开发机、
9
+ `TEMPLATE` 指向他的模板文件),以及 `MAPPING.json` 的 `_verified_by` 说明,
10
+ 一律替换成中性占位符 **`<NINFER_ROOT>`** 与 **`<TEMPLATE>`**;
11
+ `verify/` 下 4 个脚本(`check_assembly` / `check_embedding` / `check_row_order` / `gemm_oracle`)的 `sys.path.insert(...)` 行做了同样处理;`oracle_rot.py` 不 import 那套模块,本来就没有这行。
12
+ **没有任何逻辑被改动。**
13
+
14
+ ## ★ 用之前先处理这两个占位符
15
+
16
+ `pack.py` 顶部:
17
+
18
+ ```python
19
+ NINFER_ROOT = r"<NINFER_ROOT>" # 指到含 tools/artifact 的源码 checkout
20
+ TEMPLATE = r"<TEMPLATE>" # 指到 groupwise-int 的模板 .ninfer
21
+ GGUF = r"<WORKSPACE>\Ternary-Bonsai-2-27B-PQ2_0.gguf" # 作者留的,同样要覆盖
22
+ ```
23
+
24
+ - **`GGUF` 与 `TEMPLATE` 可以用命令行覆盖,不必改源码:**
25
+ ```bash
26
+ python -u pack.py build out.ninfer --gguf 你的.gguf --template 你的模板.ninfer
27
+ # 或环境变量 NINFER_TERNARY_GGUF / NINFER_TERNARY_TEMPLATE
28
+ ```
29
+ - **`NINFER_ROOT` 只能改源码常量** —— 它只用来 `import tools.artifact` 那一组模块。
30
+ 需要的最小结构:
31
+ ```
32
+ <NINFER_ROOT>/tools/artifact/{__init__.py, container.py, layouts.py, numeric.py}
33
+ ```
34
+ 来源是 `Ambolio/ninfer-4090-windows` 血统的树(或任何含这套 `tools/artifact` 的 checkout)。
35
+
36
+ ---
37
+ | 文件 | 作用 |
38
+ |---|---|
39
+ | `pack.py` | GGUF → `.ninfer` 的三元打包器。自写 GGUF 读取器;把 402 个三元矩阵的码字**逐字节搬运**,绝不反量化再量化 |
40
+ | `MAPPING.json` | 逐张量映射表(对象名 / 形状 / 格式 / 规则),每条都注明证据来源 |
41
+ | `verify/check_row_order.py` | 行级指纹的**多重集**比对 —— 能抓行置换/重复/丢失(这类 bug 守恒一切可数之物) |
42
+ | `verify/check_assembly.py` | 全量装配审计 |
43
+ | `verify/oracle_rot.py` | Hadamard 旋转的独立 oracle |
44
+ | `verify/gemm_oracle.py` | GEMM 的独立 oracle |
45
+ | `verify/check_embedding.py` | embedding 路径核对 |
46
+
47
+ ## 怎么用
48
+
49
+ ```bash
50
+ python -u pack.py check --gguf <源.gguf> --template <模板.ninfer>
51
+ # 只验证,不写文件。期望:8 个形状组合 pad=0,抽样张量 bytes_equal + decode_equal 全 True
52
+
53
+ python -u pack.py build <out.ninfer> --gguf <源.gguf> --template <模板.ninfer>
54
+ # 真打包
55
+ ```
56
+
57
+ **`--template` 是必需的。** 它不只是载荷来源 —— 打包器会**遍历模板自己的对象名表**逐个映射,
58
+ 所以模板必须与目标制品同 schema(`identity.weights_id == "groupwise-int"`)。
59
+ `nvfp4` 打包的模板用不了(它把投影融合了,名字不在映射表里,会直接中止)。
60
+
61
+ **依赖:** Python 3.11+、numpy、torch(**CPU 版即可,不需要 GPU**)。
62
+
63
+ ## 一个必须记住的判据陷阱
64
+
65
+ 「搬运无损」类判据(源字节 == payload、往返解码一致)对**偏移错误完全盲** ——
66
+ 读错的同一批字节原样进原样出,照样全绿。
67
+
68
+ 真正有用的旁证是**分布特征**:
69
+
70
+ | 指标 | 正确读取 | 偏移错误时 |
71
+ |---|---|---|
72
+ | scale 高位字节种数 | 11~32 / 256(紧致) | 接近 250 / 256(近似均匀) |
73
+ | scale 是否全正 | 负 0 个 | 出现负值 |
74
+ | 非有限(NaN/Inf) | 0 | 出现 NaN |
75
+ | `zero_share` | **0.3277~0.3280** | 0.318~0.323 散乱 |
76
+
77
+ **`zero_share` 精确落到 0.3278 是 `PQ2_0` 的格式指纹**,不是"大概的数"。
tools/pack.py ADDED
@@ -0,0 +1,958 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Bonsai 2 27B (arch qwen35) GGUF -> ninfer `.ninfer` ternary artifact (M1-D packer).
3
+
4
+ Mapping rules are NOT derived here; they come from MAPPING.json (which records the evidence
5
+ for each). This script owns: a seek-based GGUF reader; an INDEPENDENT reference decoder for
6
+ the two Prism ternary types ported from ggml/src/ggml-quants.c; byte-for-byte ternary plane
7
+ assembly plus its exact inverse (used as the losslessness proof); the inventory built from
8
+ the template artifact's own object list; and a streaming writer with borrowed payloads.
9
+
10
+ Borrowed payloads (deliberate, reported in the run summary):
11
+ vision/* 333 -- Bonsai ships its vision tower as a separate mmproj GGUF; the template's
12
+ tower is the same official one and is not read during text decode.
13
+ mtp/* 12 -- Bonsai's GGUF has NO MTP head (851 tensors, zero `blk.64.*`). The
14
+ template's head is the same official Qwen3.8-27B head: all twelve
15
+ tensors matched at cosine >= 0.99966 with the seven norms BIT-IDENTICAL.
16
+ frontend/* 6 -- tokenizer and friends.
17
+ text/draft_head(+_token_ids) -- frequency shortlist; same tokenizer => same shortlist.
18
+
19
+ Modes
20
+ check geometry + decode + byte-round-trip proofs (no writes)
21
+ layer3 <out> minimal artifact: frontend + sign table + one full-attention layer
22
+ build <out> full text model
23
+
24
+ Paths (both may be overridden; the constants in this file are only defaults)
25
+ --gguf <p> the Bonsai 2 27B PQ2_0 GGUF (env NINFER_TERNARY_GGUF)
26
+ --template <p> a **groupwise-int** qwen3.8-27b artifact
27
+ (env NINFER_TERNARY_TEMPLATE) -- see README FAQ
28
+ The template is the skeleton/manifest this packer walks, not just a donor of the
29
+ vision/MTP payloads: its object NAMES must match the mapping table, so the `nvfp4`
30
+ packing (fused gdn/a_b_projection, gdn/query_key_value_z, attention/query_key_gate_value)
31
+ is rejected up front with instructions on how to produce the right one.
32
+
33
+ Interpreter: <PYTHON>\\python.exe
34
+ """
35
+ from __future__ import annotations
36
+
37
+ import json
38
+ import os
39
+ import struct
40
+ import sys
41
+ from collections import Counter
42
+ from pathlib import Path
43
+
44
+ import numpy as np
45
+
46
+ NINFER_ROOT = r"<NINFER_ROOT>"
47
+ if NINFER_ROOT not in sys.path:
48
+ sys.path.insert(0, NINFER_ROOT)
49
+
50
+ from tools.artifact import ( # noqa: E402
51
+ Artifact,
52
+ ArtifactIdentity,
53
+ ArtifactWriter,
54
+ ResourceSpec,
55
+ TensorSpec,
56
+ encode_direct,
57
+ row_split_geometry,
58
+ )
59
+
60
+ TEMPLATE = r"<TEMPLATE>"
61
+ GGUF = r"<WORKSPACE>\Ternary-Bonsai-2-27B-PQ2_0.gguf"
62
+
63
+ # The template must be the **groupwise-int** packing. It is not merely a donor of the
64
+ # vision/MTP payloads: this packer walks the template's OWN object list and maps every name
65
+ # through a closed table, so a template with different (fused) names cannot be consumed.
66
+ # The sibling packing `nvfp4` fuses the projections (gdn/a_b_projection,
67
+ # gdn/query_key_value_z, attention/query_key_gate_value) and therefore aborts with
68
+ # "unmapped gdn object ...". Both packings are produced from the same model by different
69
+ # converters in tools/convert/qwen3_8_27b/.
70
+ TEMPLATE_SCHEMA = "groupwise-int"
71
+
72
+ _TEMPLATE_HELP = """\
73
+ 模板 schema 不对:pack.py 需要 **groupwise-int** 的 qwen3.8-27b 制品。
74
+ (the template must be the groupwise-int packing, not nvfp4)
75
+
76
+ 你给的模板 : {path}
77
+ 它的 schema : weights_id={got!r}
78
+ 需要的 : weights_id={want!r}
79
+
80
+ 怎么拿到正确的模板(二选一):
81
+ A) 用基树自带的转换器自己产一份 —— groupwise-int 路径,**不是** convert_nvfp4:
82
+ python3 -m tools.convert.qwen3_8_27b.convert \\
83
+ --model <Qwen3.8-27B 权重目录> \\
84
+ --dflash2-model <Qwen3.8-27B-DFlash2 目录> \\
85
+ --out <out.ninfer>
86
+ (convert_nvfp4 产出 nvfp4 schema,其 GDN/attention 为**融合命名**:
87
+ text/layers/N/gdn/a_b_projection、gdn/query_key_value_z、attention/query_key_gate_value
88
+ —— 这些名字不在本脚本的映射表里,必然中止。)
89
+ B) 任何 weights_id=groupwise-int 的 qwen3.8-27b 制品都可以当模板。
90
+
91
+ 自检(满足任一条即为正确):
92
+ 1) 该制品 identity.weights_id == 'groupwise-int'
93
+ 2) text/layers/3/ 下是 attention/query_key 与 attention/gate_value **两个**对象
94
+ (若只有单个 attention/query_key_gate_value ⇒ nvfp4 版,用不了)
95
+ """
96
+
97
+ _SCHEMA_HINT = ("\n hint: 模板 schema 不匹配。本脚本只认 weights_id=groupwise-int 的模板;"
98
+ "nvfp4 模板的融合命名(gdn/a_b_projection、gdn/query_key_value_z、"
99
+ "attention/query_key_gate_value)不在映射表里。见 README FAQ / --template。")
100
+
101
+ T_PQ2_0, T_PTQ1_0, T_F32, T_BF16 = 142, 143, 0, 30
102
+ FMT = {T_PQ2_0: "PQ2_0_G128", T_PTQ1_0: "PTQ1_0_G128"}
103
+ HIDDEN, V_HEADS, V_HEAD_DIM = 5120, 48, 128
104
+ QK_ROWS, V_ROWS = 4096, 6144
105
+ SIGN_WIDTHS = [5120, 6144, 17408]
106
+ FIXED = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
107
+
108
+
109
+ # ---------------------------------------------------------------------------
110
+ class Gguf:
111
+ def __init__(self, path: str):
112
+ self.f = open(path, "rb")
113
+ self.f.seek(0, 2)
114
+ self.size = self.f.tell()
115
+ self.f.seek(0)
116
+ if self._raw(4) != b"GGUF":
117
+ raise SystemExit("not a GGUF file")
118
+ self.version = self._u32()
119
+ self.n_tensors = self._u64()
120
+ self.n_kv = self._u64()
121
+ self.kv: dict[str, object] = {}
122
+ for _ in range(self.n_kv):
123
+ k = self._str()
124
+ self.kv[k] = self._value(self._u32())
125
+ self.kv_end = self.f.tell()
126
+ self.tensor: dict[str, tuple[list[int], int, int]] = {}
127
+ order = []
128
+ for _ in range(self.n_tensors):
129
+ name = self._str()
130
+ ne = [self._u64() for _ in range(self._u32())]
131
+ tt = self._u32()
132
+ off = self._u64()
133
+ self.tensor[name] = (ne, tt, off)
134
+ order.append((off, name))
135
+ # header_end is the end of the TENSOR INFO LIST, not the end of the KV block.
136
+ # Recording it before this loop (as an earlier revision did) shifts data_start down
137
+ # by the size of the list -- ~50 KB in this file -- so every payload read comes from
138
+ # the wrong offset, while names/shapes/types still parse perfectly and a
139
+ # self-consistent byte round trip still passes. That is a silent, maximally
140
+ # misleading failure: probe_gguf_types.py puts this file's header end at
141
+ # 11,120,982 and the reader must agree.
142
+ self.header_end = self.f.tell()
143
+ self.alignment = int(self.kv.get("general.alignment", 32))
144
+ self.data_start = -(-self.header_end // self.alignment) * self.alignment
145
+ order.sort()
146
+ self.tbytes: dict[str, int] = {}
147
+ for i, (off, name) in enumerate(order):
148
+ nxt = order[i + 1][0] if i + 1 < len(order) else (self.size - self.data_start)
149
+ self.tbytes[name] = nxt - off
150
+ self._cache: dict[str, bytes] = {}
151
+
152
+ def _raw(self, n):
153
+ b = self.f.read(n)
154
+ if len(b) != n:
155
+ raise EOFError(f"short read of {n}")
156
+ return b
157
+
158
+ def _u32(self):
159
+ return struct.unpack("<I", self._raw(4))[0]
160
+
161
+ def _u64(self):
162
+ return struct.unpack("<Q", self._raw(8))[0]
163
+
164
+ def _str(self):
165
+ return self._raw(self._u64()).decode("utf-8", "replace")
166
+
167
+ def _value(self, t):
168
+ if t in (0, 1, 7):
169
+ return self._raw(1)[0]
170
+ if t in (2, 3):
171
+ return struct.unpack("<h", self._raw(2))[0]
172
+ if t == 4:
173
+ return struct.unpack("<I", self._raw(4))[0]
174
+ if t == 5:
175
+ return struct.unpack("<i", self._raw(4))[0]
176
+ if t == 6:
177
+ return struct.unpack("<f", self._raw(4))[0]
178
+ if t == 8:
179
+ return self._str()
180
+ if t in (10, 11):
181
+ return struct.unpack("<q", self._raw(8))[0]
182
+ if t == 12:
183
+ return struct.unpack("<d", self._raw(8))[0]
184
+ if t == 9:
185
+ et, n = self._u32(), self._u64()
186
+ if et == 8:
187
+ return [self._str() for _ in range(n)]
188
+ if et == 6:
189
+ return list(struct.unpack("<%df" % n, self._raw(4 * n)))
190
+ if et in (0, 1, 7):
191
+ return list(self._raw(n))
192
+ if et in (2, 3):
193
+ return list(struct.unpack("<%dh" % n, self._raw(2 * n)))
194
+ if et == 4:
195
+ return list(struct.unpack("<%dI" % n, self._raw(4 * n)))
196
+ if et == 5:
197
+ return list(struct.unpack("<%di" % n, self._raw(4 * n)))
198
+ self.f.seek(FIXED[et] * n, 1)
199
+ return f"<{n} values>"
200
+ raise ValueError(f"gguf value type {t}")
201
+
202
+ def payload(self, name: str) -> bytes:
203
+ if name not in self._cache:
204
+ _ne, _tt, off = self.tensor[name]
205
+ self.f.seek(self.data_start + off)
206
+ self._cache[name] = self._raw(self.tbytes[name])
207
+ return self._cache[name]
208
+
209
+ def tinfo(self, name):
210
+ ne, tt, _ = self.tensor[name]
211
+ return ne, tt
212
+
213
+ def row_shape(self, name) -> tuple[int, int]:
214
+ """(n_rows, k) in GGUF row order, plus the number of weights."""
215
+ ne, _tt = self.tinfo(name)
216
+ if len(ne) != 2:
217
+ raise SystemExit(f"{name}: expected rank 2, got {ne}")
218
+ return ne[1], ne[0]
219
+
220
+ def blocks(self, name):
221
+ """(n_rows, groups_per_row, block_bytes, row_bytes, raw)."""
222
+ n, k = self.row_shape(name)
223
+ ne, tt = self.tinfo(name)
224
+ block = 28 if tt == T_PTQ1_0 else 34
225
+ gpr = k // 128
226
+ raw = self.payload(name)
227
+ if len(raw) != n * gpr * block:
228
+ raise SystemExit(f"{name}: {len(raw)} != {n}*{gpr}*{block}")
229
+ return n, gpr, block, gpr * block, raw
230
+
231
+ def row_fn(self, name):
232
+ n, gpr, block, rb, raw = self.blocks(name)
233
+ return lambda i: raw[i * rb:(i + 1) * rb]
234
+
235
+
236
+ # ---------------------------------------------------------------------------
237
+ # Reference decoders -- independent ports of ggml/src/ggml-quants.c
238
+ # ---------------------------------------------------------------------------
239
+ def dq_pq2_0(raw: bytes) -> np.ndarray:
240
+ qs = np.frombuffer(raw, dtype=np.uint8).reshape(-1, 34)
241
+ d = qs[:, 0:2].copy().view(np.float16).astype(np.float32).reshape(-1, 1)
242
+ j = np.arange(128)
243
+ q = (qs[:, 2:34][:, j // 4] >> (2 * (j % 4))) & 0x03
244
+ return ((q.astype(np.int32) - 1) * d).reshape(-1)
245
+
246
+
247
+ def dq_ptq1_0(raw: bytes) -> np.ndarray:
248
+ blk = np.frombuffer(raw, dtype=np.uint8).reshape(-1, 28)
249
+ pow3 = (1, 3, 9, 27, 81, 243)
250
+ g = blk.shape[0]
251
+ d = blk[:, 26:28].copy().view(np.float16).astype(np.float32).reshape(-1)
252
+ qs, qh = blk[:, 0:24], blk[:, 24:26]
253
+ vals = np.empty((g, 120), dtype=np.float32)
254
+ col, j = 0, 0
255
+ for c in (32, 16, 8): # c=32 emits nothing: 0 + 32 > 24
256
+ while j + c <= 24:
257
+ for n in range(5):
258
+ prod = (qs[:, j:j + c].astype(np.uint16) * pow3[n]) & 0xFF
259
+ vals[:, col:col + c] = ((prod.astype(np.uint16) * 3) >> 8).astype(np.float32)
260
+ col += c
261
+ j += c
262
+ if col != 120:
263
+ raise SystemExit(f"ptq1_0 qs walk emitted {col}, expected 120")
264
+ tail = np.empty((g, 8), dtype=np.float32)
265
+ col = 0
266
+ for n in range(4):
267
+ prod = (qh.astype(np.uint16) * pow3[n]) & 0xFF
268
+ for h in range(2):
269
+ tail[:, col] = ((prod[:, h].astype(np.uint16) * 3) >> 8).astype(np.float32)
270
+ col += 1
271
+ out = np.concatenate([vals, tail], axis=1)
272
+ return ((out - 1.0) * d[:, None]).reshape(-1)
273
+
274
+
275
+ DEQUANT = {T_PQ2_0: dq_pq2_0, T_PTQ1_0: dq_ptq1_0}
276
+
277
+
278
+ def read_direct(g: Gguf, name: str) -> np.ndarray:
279
+ ne, tt = g.tinfo(name)
280
+ raw = g.payload(name)
281
+ if tt == T_F32:
282
+ a = np.frombuffer(raw, dtype="<f4").astype(np.float32)
283
+ elif tt == T_BF16:
284
+ a = (np.frombuffer(raw, dtype="<u2").astype(np.uint32) << 16).view(np.float32)
285
+ else:
286
+ raise SystemExit(f"{name}: type {tt} is not F32/BF16")
287
+ return a.reshape(tuple(reversed(ne))) if len(ne) > 1 else a.astype(np.float32)
288
+
289
+
290
+ # ---------------------------------------------------------------------------
291
+ # Ternary plane assembly, byte for byte, and its inverse
292
+ # ---------------------------------------------------------------------------
293
+ def block_to_planes(fmt, blk):
294
+ if fmt == "PTQ1_0_G128":
295
+ return blk[0:24], blk[24:26], blk[26:28]
296
+ if fmt == "PQ2_0_G128":
297
+ return blk[2:34], b"", blk[0:2]
298
+ raise ValueError(fmt)
299
+
300
+
301
+ def planes_to_block(fmt, base, high, scale):
302
+ return base + high + scale if fmt == "PTQ1_0_G128" else scale + base
303
+
304
+
305
+ def assemble_ternary(fmt, shape, row_fn) -> bytes:
306
+ """row_fn(i) -> the raw GGML block bytes for destination row i."""
307
+ n, k = shape
308
+ geo = row_split_geometry(fmt, shape)
309
+ gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
310
+ block = bb + hb + 2
311
+ out = bytearray(geo.payload_bytes)
312
+ for i in range(n):
313
+ row = row_fn(i)
314
+ if len(row) != gpr * block:
315
+ raise SystemExit(f"row {i}: {len(row)} != {gpr * block}")
316
+ for sel in range(gpr):
317
+ base, high, scale = block_to_planes(fmt, row[sel * block:(sel + 1) * block])
318
+ o = geo.base_offset + i * geo.base_row_bytes + sel * bb
319
+ out[o:o + bb] = base
320
+ if hb:
321
+ o = geo.high_offset + i * geo.high_row_bytes + sel * hb
322
+ out[o:o + hb] = high
323
+ o = geo.scale_offset + i * geo.scale_row_bytes + sel * 2
324
+ out[o:o + 2] = scale
325
+ return bytes(out)
326
+
327
+
328
+ def disassemble_ternary(fmt, shape, payload: bytes) -> bytes:
329
+ """Inverse of assemble_ternary: rebuild the contiguous GGML block stream."""
330
+ n, k = shape
331
+ geo = row_split_geometry(fmt, shape)
332
+ gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
333
+ out = bytearray()
334
+ for i in range(n):
335
+ for sel in range(gpr):
336
+ o = geo.base_offset + i * geo.base_row_bytes + sel * bb
337
+ base = payload[o:o + bb]
338
+ high = b""
339
+ if hb:
340
+ o = geo.high_offset + i * geo.high_row_bytes + sel * hb
341
+ high = payload[o:o + hb]
342
+ o = geo.scale_offset + i * geo.scale_row_bytes + sel * 2
343
+ out += planes_to_block(fmt, base, high, payload[o:o + 2])
344
+ return bytes(out)
345
+
346
+
347
+ # ---------------------------------------------------------------------------
348
+ # Non-ternary transforms
349
+ # ---------------------------------------------------------------------------
350
+ def tiled_to_grouped(t: np.ndarray, groups: int = 3) -> np.ndarray:
351
+ """GGUF TILED (3,16) -> ninfer GROUPED (16,3) on the leading 48-head axis."""
352
+ shape = t.shape
353
+ return (t.reshape(groups, shape[0] // groups, *shape[1:])
354
+ .transpose(1, 0, *range(2, len(shape) + 1))
355
+ .reshape(shape))
356
+
357
+
358
+ def bf16_payload(a: np.ndarray) -> bytes:
359
+ import torch
360
+ return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.float32))
361
+ .to(torch.bfloat16), "BF16")
362
+
363
+
364
+ def fp32_payload(a) -> bytes:
365
+ import torch
366
+ return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.float32)), "FP32")
367
+
368
+
369
+ def i32_payload(a) -> bytes:
370
+ import torch
371
+ return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.int32)), "I32")
372
+
373
+
374
+ # ---------------------------------------------------------------------------
375
+ # Inventory (built from the template, never assumed)
376
+ # ---------------------------------------------------------------------------
377
+ def _weights_id(identity):
378
+ """Pull weights_id out of the template manifest's identity block."""
379
+ if isinstance(identity, dict):
380
+ for k in ("weights_id", "weightsId", "weights"):
381
+ v = identity.get(k)
382
+ if isinstance(v, str):
383
+ return v
384
+ return None
385
+
386
+
387
+ def load_template():
388
+ with open(TEMPLATE, "rb") as f:
389
+ f.seek(16)
390
+ blob = f.read(8 << 20)
391
+ obj, _ = json.JSONDecoder().raw_decode(blob.decode("utf-8", "replace"))
392
+ identity, objects = obj["identity"], obj["objects"]
393
+ got = _weights_id(identity)
394
+ if got != TEMPLATE_SCHEMA:
395
+ # Fail HERE, with an actionable message, instead of 200 lines later on an
396
+ # "unmapped gdn object" that does not say what to do about it.
397
+ raise SystemExit(_TEMPLATE_HELP.format(path=TEMPLATE, got=got, want=TEMPLATE_SCHEMA))
398
+ return identity, objects
399
+
400
+
401
+ def layer_kind(objs) -> dict[int, str]:
402
+ kind: dict[int, str] = {}
403
+ for o in objs:
404
+ p = o["name"].split("/")
405
+ if len(p) > 3 and p[0] == "text" and p[1] == "layers":
406
+ l = int(p[2])
407
+ kind.setdefault(l, "attn" if p[3] == "attention" else "gdn")
408
+ return kind
409
+
410
+
411
+ # Target formats for the NEW artifact. The template's own formats describe the
412
+ # groupwise-int artifact and must NOT be reused: the packer replaces all 402 ternary
413
+ # matrices and re-encodes the norms, so the writer would otherwise validate every produced
414
+ # payload against the wrong geometry (CHECK 3 caught exactly this).
415
+ TERNARY_SUFFIXES = frozenset((
416
+ "mlp/gate_up", "mlp/down",
417
+ "attention/query_key", "attention/gate_value", "attention/output",
418
+ "gdn/query_key", "gdn/value_z", "gdn/output",
419
+ ))
420
+ BF16_SUFFIXES = frozenset((
421
+ "input_norm", "post_attention_norm",
422
+ "attention/query_norm", "attention/key_norm",
423
+ "gdn/norm", "gdn/convolution", "gdn/a_projection", "gdn/b_projection",
424
+ ))
425
+ FP32_SUFFIXES = frozenset(("gdn/a_log", "gdn/dt_bias"))
426
+
427
+ SIGN_OBJECTS = (
428
+ ("text/hadamard_signs", (28672,), "FP32"),
429
+ ("text/hadamard_widths", (3,), "I32"),
430
+ )
431
+
432
+
433
+ def build_specs(p):
434
+ """Ordered specs for the NEW artifact.
435
+
436
+ Template objects keep their order, shapes and borrowed formats, but every produced
437
+ tensor takes its TARGET format; the two Hadamard sign-table objects are appended.
438
+ """
439
+ specs = []
440
+ for o in p.objects:
441
+ if o["kind"] == "tensor":
442
+ fmt, layout = p.target_format(o["name"])
443
+ specs.append(TensorSpec(o["name"], tuple(o["shape"]), fmt, layout))
444
+ else:
445
+ specs.append(ResourceSpec(o["name"], o["encoding"], o["bytes"]))
446
+ for name, shape, fmt in SIGN_OBJECTS:
447
+ specs.append(TensorSpec(name, shape, fmt, "contiguous-le-v1"))
448
+ return specs
449
+
450
+
451
+ # ---------------------------------------------------------------------------
452
+ class Packer:
453
+ def __init__(self, g: Gguf):
454
+ self.g = g
455
+ raw_identity, self.objects = load_template()
456
+ self.identity = ArtifactIdentity(raw_identity["model_id"], raw_identity["weights_id"])
457
+ self._by_name = {o["name"]: o for o in self.objects}
458
+ self.kind = layer_kind(self.objects)
459
+ self._signs = None
460
+
461
+ # -- helpers ---------------------------------------------------------
462
+ def rel(self, name: str, l: int) -> str:
463
+ return name.replace("{l}", str(l))
464
+
465
+ def fmt_of(self, gguf_name: str) -> str:
466
+ _ne, tt = self.g.tinfo(gguf_name)
467
+ if tt not in FMT:
468
+ raise SystemExit(f"{gguf_name}: type {tt} is not ternary")
469
+ return FMT[tt]
470
+
471
+ def target_format(self, name: str) -> tuple[str, str]:
472
+ """(format, layout) this object carries in the NEW artifact."""
473
+ if name == "text/hadamard_signs":
474
+ return "FP32", "contiguous-le-v1"
475
+ if name == "text/hadamard_widths":
476
+ return "I32", "contiguous-le-v1"
477
+ if name == "text/token_embedding":
478
+ return self.fmt_of("token_embd.weight"), "row-split-k128-v1"
479
+ if name == "text/output_head":
480
+ return self.fmt_of("output.weight"), "row-split-k128-v1"
481
+ if name == "text/final_norm":
482
+ return "BF16", "contiguous-le-v1"
483
+
484
+ parts = name.split("/")
485
+ if len(parts) > 3 and parts[0] == "text" and parts[1] == "layers":
486
+ suffix = "/".join(parts[3:])
487
+ pre = f"blk.{int(parts[2])}."
488
+ if suffix in TERNARY_SUFFIXES:
489
+ if suffix == "mlp/gate_up":
490
+ return self.fmt_of(pre + "ffn_gate.weight"), "row-split-k128-v1"
491
+ if suffix == "mlp/down":
492
+ return self.fmt_of(pre + "ffn_down.weight"), "row-split-k128-v1"
493
+ if suffix in ("attention/query_key", "attention/gate_value"):
494
+ return self.fmt_of(pre + "attn_q.weight"), "row-split-k128-v1"
495
+ if suffix == "attention/output":
496
+ return self.fmt_of(pre + "attn_output.weight"), "row-split-k128-v1"
497
+ if suffix in ("gdn/query_key", "gdn/value_z"):
498
+ return self.fmt_of(pre + "attn_qkv.weight"), "row-split-k128-v1"
499
+ if suffix == "gdn/output":
500
+ return self.fmt_of(pre + "ssm_out.weight"), "row-split-k128-v1"
501
+ if suffix in BF16_SUFFIXES:
502
+ return "BF16", "contiguous-le-v1"
503
+ if suffix in FP32_SUFFIXES:
504
+ return "FP32", "contiguous-le-v1"
505
+
506
+ # everything else (mtp/*, vision/*, draft_head*) is BORROWED: keep the template's
507
+ o = self._by_name.get(name)
508
+ if o is None:
509
+ raise SystemExit(f"no template object and no producer for {name}")
510
+ return o["format"], o["layout"]
511
+
512
+ def fused(self, entries, shape) -> bytes:
513
+ """entries: list of (gguf_name, row_index); all sources must share one format."""
514
+ fmts = {self.fmt_of(n) for n, _ in entries}
515
+ if len(fmts) != 1:
516
+ raise SystemExit(f"fused tensor mixes formats: {fmts}")
517
+ fmt = fmts.pop()
518
+ srcs = {}
519
+ for n, _ in entries:
520
+ if n not in srcs:
521
+ nrows, _gpr, _blk, rb, raw = self.g.blocks(n)
522
+ srcs[n] = (rb, raw)
523
+
524
+ def row_fn(i):
525
+ n, r = entries[i]
526
+ rb, raw = srcs[n]
527
+ return raw[r * rb:(r + 1) * rb]
528
+
529
+ return assemble_ternary(fmt, shape, row_fn)
530
+
531
+ def direct_ternary(self, gguf_name: str, shape) -> bytes:
532
+ return assemble_ternary(self.fmt_of(gguf_name), shape, self.g.row_fn(gguf_name))
533
+
534
+ def signs(self):
535
+ if self._signs is None:
536
+ vals = self.g.kv["prism.hadamard.sign_values"]
537
+ widths = list(self.g.kv["prism.hadamard.sign_widths"])
538
+ arr = np.asarray(vals, dtype=np.float32)
539
+ if arr.size != 28672 or not np.all(np.abs(arr) == 1.0):
540
+ raise SystemExit("sign table is not 28672 strictly-+-1 values")
541
+ if widths != SIGN_WIDTHS or sum(widths) != arr.size:
542
+ raise SystemExit(f"sign_widths mismatch: {widths}")
543
+ self._signs = (arr, widths)
544
+ return self._signs
545
+
546
+ @staticmethod
547
+ def payload_bytes(spec) -> int | None:
548
+ """Expected payload size for a template object spec, or None when not computable."""
549
+ if spec["kind"] != "tensor":
550
+ return None
551
+ layout = spec.get("layout")
552
+ if layout == "row-split-k128-v1":
553
+ return row_split_geometry(spec["format"], tuple(spec["shape"])).payload_bytes
554
+ if layout == "contiguous-le-v1":
555
+ wb = {"BF16": 2, "FP32": 4, "I32": 4}[spec["format"]]
556
+ return int(np.prod(spec["shape"])) * wb
557
+ return None
558
+
559
+ # -- producers -------------------------------------------------------
560
+ def produce(self, name: str):
561
+ """Return the payload bytes for a text/* object, or None to borrow it."""
562
+ p = name.split("/")
563
+
564
+ if name == "text/hadamard_signs":
565
+ return fp32_payload(self.signs()[0])
566
+ if name == "text/hadamard_widths":
567
+ return i32_payload(self.signs()[1])
568
+
569
+ if name == "text/token_embedding":
570
+ return self.direct_ternary("token_embd.weight", (248320, HIDDEN))
571
+ if name == "text/output_head":
572
+ return self.direct_ternary("output.weight", (248320, HIDDEN))
573
+ if name == "text/final_norm":
574
+ return bf16_payload(read_direct(self.g, "output_norm.weight") - 1.0)
575
+
576
+ if len(p) > 3 and p[0] == "text" and p[1] == "layers":
577
+ l = int(p[2])
578
+ suffix = "/".join(p[3:])
579
+ g = self.g
580
+ pre = f"blk.{l}."
581
+
582
+ if suffix == "input_norm":
583
+ return bf16_payload(read_direct(g, pre + "attn_norm.weight") - 1.0)
584
+ if suffix == "post_attention_norm":
585
+ return bf16_payload(read_direct(g, pre + "post_attention_norm.weight") - 1.0)
586
+ if suffix == "mlp/gate_up":
587
+ ng, _ = g.row_shape(pre + "ffn_gate.weight")
588
+ nu, _ = g.row_shape(pre + "ffn_up.weight")
589
+ ent = [(pre + "ffn_gate.weight", i) for i in range(ng)]
590
+ ent += [(pre + "ffn_up.weight", i) for i in range(nu)]
591
+ return self.fused(ent, (ng + nu, HIDDEN))
592
+ if suffix == "mlp/down":
593
+ return self.direct_ternary(pre + "ffn_down.weight", (5120, 17408))
594
+
595
+ if suffix.startswith("attention/"):
596
+ sub = suffix.split("/", 1)[1]
597
+ if sub == "query_key":
598
+ q = self.deinterleave(pre + "attn_q.weight", 0, (6144, HIDDEN))
599
+ nk, _ = g.row_shape(pre + "attn_k.weight")
600
+ return self.concat_streams(q, 6144, pre + "attn_k.weight", nk,
601
+ (6144 + nk, HIDDEN))
602
+ if sub == "gate_value":
603
+ gt = self.deinterleave(pre + "attn_q.weight", 1, (6144, HIDDEN))
604
+ nv, _ = g.row_shape(pre + "attn_v.weight")
605
+ return self.concat_streams(gt, 6144, pre + "attn_v.weight", nv,
606
+ (6144 + nv, HIDDEN))
607
+ if sub == "output":
608
+ return self.direct_ternary(pre + "attn_output.weight", (5120, 6144))
609
+ if sub == "query_norm":
610
+ return bf16_payload(read_direct(g, pre + "attn_q_norm.weight") - 1.0)
611
+ if sub == "key_norm":
612
+ return bf16_payload(read_direct(g, pre + "attn_k_norm.weight") - 1.0)
613
+ raise SystemExit(f"unmapped attention object {name}" + _SCHEMA_HINT)
614
+
615
+ if suffix.startswith("gdn/"):
616
+ sub = suffix.split("/", 1)[1]
617
+ if sub == "query_key":
618
+ ent = [(pre + "attn_qkv.weight", i) for i in range(QK_ROWS)]
619
+ return self.fused(ent, (QK_ROWS, HIDDEN))
620
+ if sub == "value_z":
621
+ return self.gdn_value_z(l)
622
+ if sub == "output":
623
+ return self.direct_ternary(pre + "ssm_out.weight", (5120, V_ROWS))
624
+ if sub == "convolution":
625
+ return bf16_payload(self.conv1d(l))
626
+ if sub == "norm":
627
+ return bf16_payload(read_direct(g, pre + "ssm_norm.weight")) # RAW
628
+ if sub == "a_projection":
629
+ return bf16_payload(tiled_to_grouped(read_direct(g, pre + "ssm_alpha.weight")))
630
+ if sub == "b_projection":
631
+ return bf16_payload(tiled_to_grouped(read_direct(g, pre + "ssm_beta.weight")))
632
+ if sub == "a_log":
633
+ a = read_direct(g, pre + "ssm_a")
634
+ return fp32_payload(tiled_to_grouped(np.log(-a.astype(np.float64))
635
+ .astype(np.float32)))
636
+ if sub == "dt_bias":
637
+ return fp32_payload(tiled_to_grouped(read_direct(g, pre + "ssm_dt.bias")))
638
+ raise SystemExit(f"unmapped gdn object {name}" + _SCHEMA_HINT)
639
+
640
+ raise SystemExit(f"unmapped layer object {name}" + _SCHEMA_HINT)
641
+
642
+ return None # borrow
643
+
644
+ # -- GDN / attention fusion helpers ----------------------------------
645
+ def deinterleave(self, gguf_name: str, parity: int, shape):
646
+ """query (parity 0) or output-gate (parity 1): every other 256-row chunk."""
647
+ n, k = self.g.row_shape(gguf_name)
648
+ if n % 512 != 0:
649
+ raise SystemExit(f"{gguf_name}: {n} rows is not a multiple of 512")
650
+ heads = n // 512
651
+ fmt = self.fmt_of(gguf_name)
652
+ nrows, gpr, block, rb, raw = self.g.blocks(gguf_name)
653
+ ent = [(gguf_name, (2 * h + parity) * 256 + r)
654
+ for h in range(heads) for r in range(256)]
655
+ return assemble_ternary(fmt, (len(ent), k),
656
+ lambda i: raw[ent[i][1] * rb:(ent[i][1] + 1) * rb])
657
+
658
+ def concat_streams(self, first: bytes, first_rows: int, gname: str, gn: int, shape):
659
+ """Concatenate a pre-assembled ternary payload with another tensor's rows."""
660
+ n, k = shape
661
+ fmt = self.fmt_of(gname)
662
+ geo = row_split_geometry(fmt, shape)
663
+ gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
664
+ block = bb + hb + 2
665
+ _n2, gpr2, block2, rb, raw = self.g.blocks(gname)
666
+ if gpr2 != gpr or block2 != block:
667
+ raise SystemExit("concat_streams: block geometry differs")
668
+ geo_a = row_split_geometry(fmt, (first_rows, k))
669
+
670
+ def row_fn(i):
671
+ if i < first_rows:
672
+ out = bytearray()
673
+ for sel in range(gpr):
674
+ o = geo_a.base_offset + i * geo_a.base_row_bytes + sel * bb
675
+ base = first[o:o + bb]
676
+ high = b""
677
+ if hb:
678
+ o = geo_a.high_offset + i * geo_a.high_row_bytes + sel * hb
679
+ high = first[o:o + hb]
680
+ o = geo_a.scale_offset + i * geo_a.scale_row_bytes + sel * 2
681
+ out += planes_to_block(fmt, base, high, first[o:o + 2])
682
+ return bytes(out)
683
+ j = i - first_rows
684
+ return raw[j * rb:(j + 1) * rb]
685
+
686
+ return assemble_ternary(fmt, shape, row_fn)
687
+
688
+ def gdn_value_z(self, l: int) -> bytes:
689
+ """concat(tiled_to_grouped(attn_qkv[4096:10240]), tiled_to_grouped(attn_gate[0:6144]))."""
690
+ pre = f"blk.{l}."
691
+ qkv, gate = pre + "attn_qkv.weight", pre + "attn_gate.weight"
692
+ f1, f2 = self.fmt_of(qkv), self.fmt_of(gate)
693
+ if f1 != f2:
694
+ raise SystemExit("value_z: mixed formats")
695
+ fmt = f1
696
+
697
+ def perm48(i):
698
+ # tiled (3,16) -> grouped (16,3) for a 48-entry HEAD axis:
699
+ # destination head g*3+t comes from source head t*16+g
700
+ return (i % 3) * 16 + (i // 3)
701
+
702
+ def perm_row(i):
703
+ # perm48() permutes a 48-entry axis, but this tensor's V/z rows are 48 heads of
704
+ # V_HEAD_DIM rows each, so the permutation has to be applied at HEAD granularity.
705
+ # Applying perm48() straight to the 6144-row index is not a bijection: perm48(48) and
706
+ # perm48(1) are both 16, so rows repeat and source rows are dropped while every size,
707
+ # row count and byte total stays exactly right -- which is why byte accounting, the
708
+ # row-count cross-check and the payload round-trip all stayed green over it.
709
+ head, inner = divmod(i, V_HEAD_DIM)
710
+ return perm48(head) * V_HEAD_DIM + inner
711
+
712
+ shape = (12288, HIDDEN)
713
+ geo = row_split_geometry(fmt, shape)
714
+ gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
715
+ block = bb + hb + 2
716
+ gpr1 = self.g.blocks(qkv)[1]
717
+ gpr2_ = self.g.blocks(gate)[1]
718
+ if gpr1 != gpr or gpr2_ != gpr:
719
+ raise SystemExit("value_z: groups_per_row mismatch")
720
+ rb1 = self.g.blocks(qkv)[3]
721
+ rb2 = self.g.blocks(gate)[3]
722
+ raw1, raw2 = self.g.payload(qkv), self.g.payload(gate)
723
+
724
+ def row_fn(i):
725
+ if i < V_ROWS: # attn_qkv rows 4096..10240, tiled->grouped
726
+ src = QK_ROWS + perm_row(i)
727
+ return raw1[src * rb1:(src + 1) * rb1]
728
+ src = perm_row(i - V_ROWS) # attn_gate rows 0..6144, tiled->grouped
729
+ return raw2[src * rb2:(src + 1) * rb2]
730
+
731
+ return assemble_ternary(fmt, shape, row_fn)
732
+
733
+ def conv1d(self, l: int) -> np.ndarray:
734
+ """(4,10240) <- ssm_conv1d: keep channels 0:4096, reorder the V channels, transpose."""
735
+ pre = f"blk.{l}."
736
+ t = read_direct(self.g, pre + "ssm_conv1d.weight") # (10240, 4)
737
+ if t.shape != (10240, 4):
738
+ raise SystemExit(f"conv1d shape {t.shape} != (10240, 4)")
739
+ head = t[0:QK_ROWS]
740
+ v = t[QK_ROWS:QK_ROWS + V_ROWS].reshape(V_HEADS, V_HEAD_DIM, t.shape[1])
741
+ v = (v.reshape(3, V_HEADS // 3, V_HEAD_DIM, t.shape[1])
742
+ .transpose(1, 0, 2, 3)
743
+ .reshape(V_ROWS, t.shape[1]))
744
+ return np.concatenate([head, v], axis=0).T # (4, 10240)
745
+
746
+
747
+ # ---------------------------------------------------------------------------
748
+ def mode_check(g: Gguf) -> int:
749
+ p = Packer(g)
750
+ print("=" * 78)
751
+ print("CHECK 1 geometry over every ternary shape present in the GGUF")
752
+ print("=" * 78)
753
+ combos = {}
754
+ for name, (ne, tt, _off) in g.tensor.items():
755
+ if tt in FMT:
756
+ combos.setdefault((FMT[tt], ne[1], ne[0]), []).append(name)
757
+ for (fmt, n, k), names in sorted(combos.items()):
758
+ geo = row_split_geometry(fmt, (n, k))
759
+ src = n * (k // 128) * (28 if fmt == "PTQ1_0_G128" else 34)
760
+ pad = geo.payload_bytes - src
761
+ print(f" {fmt:<14} n={n:<7} k={k:<6} groups/row={geo.groups_per_row:<4} "
762
+ f"src={src:>13,} payload={geo.payload_bytes:>13,} pad={pad:>5} "
763
+ f"({len(names)} tensors)")
764
+ print(f" distinct (format,shape) combos: {len(combos)}")
765
+
766
+ print("\n" + "=" * 78)
767
+ print("CHECK 2 byte round trip + decode equality on real tensors")
768
+ print("=" * 78)
769
+ sample = ["blk.3.attn_q.weight", "blk.3.attn_output.weight", "blk.3.ffn_down.weight",
770
+ "blk.0.attn_qkv.weight", "blk.0.ssm_out.weight", "token_embd.weight"]
771
+ ok = True
772
+ for name in sample:
773
+ if name not in g.tensor:
774
+ print(f" {name}: absent, skipped")
775
+ continue
776
+ ne, tt = g.tinfo(name)
777
+ n, k = ne[1], ne[0]
778
+ fmt = FMT[tt]
779
+ payload = p.direct_ternary(name, (n, k))
780
+ back = disassemble_ternary(fmt, (n, k), payload)
781
+ src = g.payload(name)
782
+ byte_ok = back == src
783
+ a, b = DEQUANT[tt](src), DEQUANT[tt](back)
784
+ dec_ok = bool(np.array_equal(a, b))
785
+ zeros = float(np.mean(a == 0.0))
786
+
787
+ # OFFSET SELF-CHECK -- the decisive one. bytes_equal only proves the assembly is
788
+ # self-consistent; reading a wrong file offset would ALSO give bytes_equal, because
789
+ # the same wrong bytes go in and come out. So validate the CONTENT: a genuine group
790
+ # scale is a tight positive cluster (few distinct high bytes, all finite, none
791
+ # negative), whereas a mis-placed read draws scale words from arbitrary positions and
792
+ # looks like a uniform uint16 sample with many NaN/inf and negative values.
793
+ arr = np.frombuffer(src, dtype=np.uint8).reshape(-1, 34)
794
+ words = arr[:, 0:2].copy().view(np.uint16).reshape(-1)
795
+ hi_distinct = int(np.unique((words >> 8).astype(np.uint8)).size)
796
+ d = arr[:, 0:2].copy().view(np.float16).astype(np.float32).reshape(-1)
797
+ fin = np.isfinite(d)
798
+ n_nonfinite = int(np.sum(~fin))
799
+ n_neg = int(np.sum(fin & (d < 0)))
800
+ med = float(np.median(d[fin])) if fin.any() else float("nan")
801
+ plaus = (n_nonfinite == 0 and n_neg == 0 and hi_distinct <= 64 and 0.0 < med < 1.0)
802
+
803
+ print(f" {name:<28} {fmt:<14} n={n:<6} k={k:<6} bytes_equal={byte_ok} "
804
+ f"decode_equal={dec_ok} zero_share={zeros:.4f}")
805
+ print(f" {'':<28} scale: hi_distinct={hi_distinct:<4} nonfinite={n_nonfinite} "
806
+ f"negative={n_neg} median={med:.5f} "
807
+ f"{'PLAUSIBLE' if plaus else 'IMPLAUSIBLE -- offset bug?'}")
808
+ ok &= byte_ok and dec_ok and plaus
809
+
810
+ print("\n" + "=" * 78)
811
+ print("CHECK 3 producer smoke test (shapes + sizes only, no artifact written)")
812
+ print("=" * 78)
813
+ for name in ["text/layers/3/attention/query_key", "text/layers/3/attention/gate_value",
814
+ "text/layers/3/mlp/gate_up", "text/layers/3/mlp/down",
815
+ "text/layers/3/attention/output", "text/layers/0/gdn/value_z",
816
+ "text/layers/0/gdn/query_key", "text/layers/0/gdn/output",
817
+ "text/token_embedding", "text/output_head", "text/final_norm",
818
+ "text/hadamard_signs", "text/hadamard_widths"]:
819
+ base = p._by_name.get(name)
820
+ if base is not None:
821
+ shape = base["shape"]
822
+ else:
823
+ shape = dict((n, list(s)) for n, s, _f in SIGN_OBJECTS).get(name)
824
+ if shape is None:
825
+ print(f" {name}: neither in template nor a sign object, skipped")
826
+ continue
827
+ fmt, layout = p.target_format(name)
828
+ spec = {"kind": "tensor", "shape": shape, "format": fmt, "layout": layout}
829
+ data = p.produce(name)
830
+ if data is None:
831
+ print(f" {name:<44} {fmt:<14} {'(borrowed)':>13} SKIP")
832
+ continue
833
+ need = p.payload_bytes(spec)
834
+ got = len(data)
835
+ flag = "OK" if (need is None or got == need) else f"MISMATCH want {need}"
836
+ print(f" {name:<44} {fmt:<14} {got:>13,} {flag}")
837
+ if need is not None and got != need:
838
+ ok = False
839
+
840
+ print("\nRESULT:", "OK" if ok else "FAILED")
841
+ return 0 if ok else 1
842
+
843
+
844
+ def mode_build(g: Gguf, out: str, only_layer: int | None = None) -> int:
845
+ p = Packer(g)
846
+ keep = None
847
+ if only_layer is not None:
848
+ names = {n for n, _s, _f in SIGN_OBJECTS} # the sign objects are appended later
849
+ for o in p.objects:
850
+ n = o["name"]
851
+ if n.startswith("frontend/") or n.startswith("text/hadamard_"):
852
+ names.add(n)
853
+ elif n.startswith(f"text/layers/{only_layer}/"):
854
+ names.add(n)
855
+ keep = names
856
+
857
+ specs = []
858
+ chosen = []
859
+ for o in p.objects + [
860
+ {"name": "text/hadamard_signs", "kind": "tensor", "shape": [28672],
861
+ "format": "FP32", "layout": "contiguous-le-v1"},
862
+ {"name": "text/hadamard_widths", "kind": "tensor", "shape": [3],
863
+ "format": "I32", "layout": "contiguous-le-v1"},
864
+ ]:
865
+ if keep is not None and o["name"] not in keep:
866
+ continue
867
+ if o["kind"] == "tensor":
868
+ fmt, layout = p.target_format(o["name"])
869
+ specs.append(TensorSpec(o["name"], tuple(o["shape"]), fmt, layout))
870
+ else:
871
+ specs.append(ResourceSpec(o["name"], o["encoding"], o["bytes"]))
872
+ chosen.append(o)
873
+
874
+ out_path = Path(out)
875
+ if out_path.exists():
876
+ raise SystemExit(f"refusing to overwrite {out_path}")
877
+
878
+ borrowed = Counter()
879
+ borrowed_bytes = 0
880
+ produced_bytes = 0
881
+ with Artifact.open(TEMPLATE) as tpl, \
882
+ ArtifactWriter(out_path, p.identity, specs) as w:
883
+ for o in chosen:
884
+ name = o["name"]
885
+ if name.startswith("text/") and not name.startswith("text/vision"):
886
+ data = None if name.startswith(("text/draft_head",)) else p.produce(name)
887
+ if data is not None:
888
+ w.write(name, data)
889
+ produced_bytes += len(data)
890
+ continue
891
+ obj = tpl.find(name)
892
+ if obj is None:
893
+ raise SystemExit(f"template has no object {name} to borrow")
894
+ mv = tpl.payload(obj)
895
+ borrowed_bytes += len(mv)
896
+ w.write(name, mv)
897
+ del mv # release the mmap view, else Artifact.close() raises BufferError
898
+ borrowed[name.split("/")[0]] += 1
899
+
900
+ size = out_path.stat().st_size
901
+ print(f"\nwrote {out_path}")
902
+ print(f" total file : {size:>15,} B = {size / 2**30:.3f} GiB")
903
+ print(f" text part produced: {produced_bytes:>15,} B = {produced_bytes / 2**30:.3f} GiB")
904
+ print(f" borrowed payloads : {borrowed_bytes:>15,} B = {borrowed_bytes / 2**30:.3f} GiB "
905
+ f"{dict(borrowed)}")
906
+ print(f" objects : {len(chosen)}")
907
+ return 0
908
+
909
+
910
+ def _pop_opt(args, name):
911
+ """Remove `--name VALUE` / `--name=VALUE` from args; return (value|None, rest)."""
912
+ rest, val, i = [], None, 0
913
+ while i < len(args):
914
+ a = args[i]
915
+ if a == name and i + 1 < len(args):
916
+ val, i = args[i + 1], i + 2
917
+ elif a.startswith(name + "="):
918
+ val, i = a.split("=", 1)[1], i + 1
919
+ else:
920
+ rest.append(a)
921
+ i += 1
922
+ return val, rest
923
+
924
+
925
+ def main() -> int:
926
+ global TEMPLATE, GGUF
927
+ args = sys.argv[1:]
928
+ tpl_opt, args = _pop_opt(args, "--template")
929
+ gguf_opt, args = _pop_opt(args, "--gguf")
930
+ TEMPLATE = tpl_opt or os.environ.get("NINFER_TERNARY_TEMPLATE") or TEMPLATE
931
+ GGUF = gguf_opt or os.environ.get("NINFER_TERNARY_GGUF") or GGUF
932
+ if not args:
933
+ print(__doc__)
934
+ return 2
935
+ if not Path(TEMPLATE).exists():
936
+ raise SystemExit(
937
+ f"模板不存在: {TEMPLATE}\n"
938
+ f" 用 --template <path> 或环境变量 NINFER_TERNARY_TEMPLATE 指定。\n"
939
+ f" 模板必须与目标制品同 schema:weights_id={TEMPLATE_SCHEMA}"
940
+ f"(不是 nvfp4,见 README FAQ)。")
941
+ if not Path(GGUF).exists():
942
+ raise SystemExit(
943
+ f"GGUF 不存在: {GGUF}\n"
944
+ f" 用 --gguf <path> 或环境变量 NINFER_TERNARY_GGUF 指定。\n"
945
+ f" 源码里的默认值是占位符 <WORKSPACE>,必须覆盖。")
946
+ g = Gguf(GGUF)
947
+ if args[0] == "check":
948
+ return mode_check(g)
949
+ if args[0] == "layer3":
950
+ return mode_build(g, args[1], only_layer=3)
951
+ if args[0] == "build":
952
+ return mode_build(g, args[1])
953
+ print(f"unknown mode {args[0]}")
954
+ return 2
955
+
956
+
957
+ if __name__ == "__main__":
958
+ raise SystemExit(main())
tools/verify/check_assembly.py ADDED
@@ -0,0 +1,155 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Assembly audit: is EVERY ternary object a faithful, correctly ordered copy of its source?
2
+
3
+ Byte accounting cannot see a row permutation (sizes, row counts and totals are conserved), and a
4
+ set-membership test cannot see it either (every duplicated row still finds a source). This audit
5
+ therefore compares MULTISETS and, more importantly, checks artifact row a DIRECTLY against the
6
+ source row the mapping rule names for a.
7
+
8
+ Sources are de-interleaved GGUF PQ2_0 blocks; artifacts are row-split planes. Both are reduced to
9
+ the same per-row fingerprint (codes bytes + scale bytes), so the comparison is exact and
10
+ independent of the rotated basis (rows are permuted, never transformed).
11
+
12
+ Usage: check_assembly.py <artifact.ninfer> <pq2.gguf>
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import hashlib
17
+ import sys
18
+ from collections import Counter
19
+ from pathlib import Path
20
+
21
+ import numpy as np
22
+
23
+ sys.path.insert(0, r"<NINFER_ROOT>")
24
+ sys.path.insert(0, r"<WORKSPACE>\tools")
25
+ from _ternary_ref import Gguf # noqa: E402
26
+ from tools.artifact import container # noqa: E402
27
+
28
+ GROUPS = 40
29
+ CODE_BYTES = 32
30
+ SCALE_BYTES = 2
31
+ V_HEAD_DIM = 128
32
+
33
+
34
+ def artifact_rows(art, name: str) -> list[bytes]:
35
+ obj = art.find(name)
36
+ payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8)
37
+ rows = int(obj.shape[0])
38
+ groups = int(obj.shape[1]) // 128
39
+ codes = payload[: rows * groups * CODE_BYTES]
40
+ off = rows * groups * CODE_BYTES
41
+ scales = payload[off: off + rows * groups * SCALE_BYTES]
42
+ return [
43
+ hashlib.sha1(codes[r * groups * CODE_BYTES:(r + 1) * groups * CODE_BYTES].tobytes()
44
+ + scales[r * groups * SCALE_BYTES:(r + 1) * groups * SCALE_BYTES].tobytes()
45
+ ).digest()
46
+ for r in range(rows)
47
+ ]
48
+
49
+
50
+ def gguf_rows(gguf: Gguf, name: str) -> list[bytes]:
51
+ ne, tt, _ = gguf.tensors[name]
52
+ if tt != 142:
53
+ raise SystemExit(f"{name}: expected PQ2_0 (142), got {tt}")
54
+ rows = int(ne[1])
55
+ groups = int(ne[0]) // 128 # ggml ne[0] is the contiguous K extent
56
+ raw = gguf.raw(name, rows)
57
+ block = np.frombuffer(raw, dtype=np.uint8).reshape(rows, groups, 34)
58
+ codes = np.ascontiguousarray(block[:, :, 2:34]).reshape(rows, groups * CODE_BYTES)
59
+ scales = np.ascontiguousarray(block[:, :, 0:2]).reshape(rows, groups * SCALE_BYTES)
60
+ return [hashlib.sha1(codes[r].tobytes() + scales[r].tobytes()).digest() for r in range(rows)]
61
+
62
+
63
+ def perm48(i: int) -> int:
64
+ return (i % 3) * 16 + (i // 3)
65
+
66
+
67
+ def head_perm(rows: int) -> list[int]:
68
+ """48 heads x 128 rows: the permutation must be applied at head granularity."""
69
+ out = []
70
+ for r in range(rows):
71
+ head, inner = divmod(r, V_HEAD_DIM)
72
+ out.append(perm48(head) * V_HEAD_DIM + inner)
73
+ return out
74
+
75
+
76
+ def audit(label: str, art_keys: list[bytes], src_keys: list[bytes]) -> bool:
77
+ art_counts, src_counts = Counter(art_keys), Counter(src_keys)
78
+ repeated = sum(v - 1 for v in art_counts.values() if v > 1)
79
+ missing = sum(1 for k in src_counts if k not in art_counts)
80
+ order_bad = [a for a in range(len(art_keys)) if art_keys[a] != src_keys[a]]
81
+ ok = (repeated == 0) and (missing == 0) and not order_bad
82
+ print(f"{'OK ' if ok else 'FAIL'} {label}")
83
+ print(f" rows art={len(art_keys)} src={len(src_keys)} | "
84
+ f"distinct art={len(art_counts)} src={len(src_counts)} | "
85
+ f"duplicated={repeated} missing={missing} | "
86
+ f"order mismatches={len(order_bad)}/{len(art_keys)}")
87
+ if order_bad:
88
+ print(f" first bad rows: {order_bad[:8]}")
89
+ return ok
90
+
91
+
92
+ def main() -> int:
93
+ art_path, gguf_path = sys.argv[1], sys.argv[2]
94
+ gguf = Gguf(Path(gguf_path))
95
+ results = []
96
+ with container.Artifact.open(art_path) as art:
97
+ def A(name):
98
+ return artifact_rows(art, name)
99
+ def G(name):
100
+ return gguf_rows(gguf, name)
101
+
102
+ # --- globals ---------------------------------------------------------
103
+ results.append(audit("text/token_embedding <- token_embd.weight",
104
+ A("text/token_embedding"), G("token_embd.weight")))
105
+ results.append(audit("text/output_head <- output.weight",
106
+ A("text/output_head"), G("output.weight")))
107
+
108
+ for layer, kind in ((0, "gdn"), (3, "attention")):
109
+ pre = f"blk.{layer}."
110
+ la = f"text/layers/{layer}/"
111
+ print(f"--- layer {layer} ({kind}) ---")
112
+ results.append(audit(f"{la}mlp/down <- ffn_down",
113
+ A(la + "mlp/down"), G(pre + "ffn_down.weight")))
114
+ results.append(audit(f"{la}mlp/gate_up <- concat(ffn_gate, ffn_up)",
115
+ A(la + "mlp/gate_up"),
116
+ G(pre + "ffn_gate.weight") + G(pre + "ffn_up.weight")))
117
+ if kind == "gdn":
118
+ results.append(audit(f"{la}gdn/query_key <- attn_qkv[0:4096]",
119
+ A(la + "gdn/query_key"),
120
+ G(pre + "attn_qkv.weight")[:4096]))
121
+ results.append(audit(f"{la}gdn/output <- ssm_out",
122
+ A(la + "gdn/output"), G(pre + "ssm_out.weight")))
123
+ # value/z halves: compare against the source reordered by the head permutation
124
+ vz = A(la + "gdn/value_z")
125
+ v_src = G(pre + "attn_qkv.weight")[4096:10240]
126
+ z_src = G(pre + "attn_gate.weight")[:6144]
127
+ perm = head_perm(6144)
128
+ results.append(audit(f"{la}gdn/value_z <- attn_qkv[4096:10240] (head perm)",
129
+ vz[:6144], [v_src[p] for p in perm]))
130
+ results.append(audit(f"{la}gdn/value_z <- attn_gate[0:6144] (head perm)",
131
+ vz[6144:], [z_src[p] for p in perm]))
132
+ else:
133
+ aq = G(pre + "attn_q.weight")
134
+ even = [c * 256 + r for c in range(0, 48, 2) for r in range(256)]
135
+ odd = [c * 256 + r for c in range(1, 48, 2) for r in range(256)]
136
+ qk = A(la + "attention/query_key")
137
+ gv = A(la + "attention/gate_value")
138
+ results.append(audit(f"{la}attention/query_key <- attn_q even chunks",
139
+ qk[:6144], [aq[r] for r in even]))
140
+ results.append(audit(f"{la}attention/query_key <- attn_k",
141
+ qk[6144:], G(pre + "attn_k.weight")[:1024]))
142
+ results.append(audit(f"{la}attention/gate_value <- attn_q odd chunks",
143
+ gv[:6144], [aq[r] for r in odd]))
144
+ results.append(audit(f"{la}attention/gate_value <- attn_v",
145
+ gv[6144:], G(pre + "attn_v.weight")[:1024]))
146
+ results.append(audit(f"{la}attention/output <- attn_output",
147
+ A(la + "attention/output"), G(pre + "attn_output.weight")))
148
+
149
+ passed = sum(1 for r in results if r)
150
+ print(f"\nRESULT: {passed}/{len(results)} assembly rules OK")
151
+ return 0 if passed == len(results) else 1
152
+
153
+
154
+ if __name__ == "__main__":
155
+ raise SystemExit(main())
tools/verify/check_embedding.py ADDED
@@ -0,0 +1,135 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Oracle for the ternary embedding path: gather (PQ2_0 decode) + inverse folded-basis mapping.
2
+
3
+ The residual stream enters layer 0 at the embedding output, so a break here makes every downstream
4
+ number meaningless -- which is exactly what a perplexity close to uniform looks like. This compares
5
+ the engine's dumped embedding activation against an independent numpy rebuild of
6
+
7
+ h = s * (H * z), z = decode(token_embedding_row[id])
8
+
9
+ where z comes straight out of the artifact payload (decoded independently of the engine).
10
+
11
+ Negative controls are mandatory: a near-uniform or all-zeros dump would otherwise "match" nothing,
12
+ and the unrotated variant would match if the engine simply skipped the mapping.
13
+
14
+ Usage: check_embedding.py <artifact.ninfer> <dump-file.T<N>>
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import os
19
+ import sys
20
+
21
+ import numpy as np
22
+
23
+ sys.path.insert(0, r"<NINFER_ROOT>")
24
+ from tools.artifact import container # noqa: E402
25
+
26
+ BLOCK = 1024
27
+ GROUPS = 40
28
+ POP = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
29
+
30
+
31
+ def popcount32(a: np.ndarray) -> np.ndarray:
32
+ v = a.astype(np.uint32)
33
+ return (POP[v & 0xFF].astype(np.uint16) + POP[(v >> 8) & 0xFF].astype(np.uint16)
34
+ + POP[(v >> 16) & 0xFF].astype(np.uint16) + POP[(v >> 24) & 0xFF].astype(np.uint16))
35
+
36
+
37
+ def hadamard(n: int) -> np.ndarray:
38
+ idx = np.arange(n, dtype=np.uint32)
39
+ parity = popcount32(idx[:, None] & idx[None, :]) & 1
40
+ return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
41
+
42
+
43
+ def main() -> int:
44
+ art_path, dump_path = sys.argv[1], sys.argv[2]
45
+ raw = np.fromfile(dump_path, dtype=np.uint8)
46
+ tokens, hidden, id_count = np.frombuffer(raw[:12], dtype=np.int32)
47
+ ids = np.frombuffer(raw[12:12 + id_count * 4], dtype=np.int32)
48
+ bf16 = np.frombuffer(raw[12 + id_count * 4:], dtype=np.uint16).reshape(tokens, hidden)
49
+ # The op writes out as [hidden, T] row-major, so element (k, t) sits at k*T + t. Reshaping the
50
+ # flat buffer as (T, hidden) would reinterpret that as t*hidden + k -- a silent transpose that
51
+ # decorrelates everything for T > 1 and looks like a total engine failure.
52
+ got = (bf16.astype(np.uint32) << 16).view(np.float32).astype(np.float64).reshape(hidden, tokens)
53
+ print(f"dump: tokens={tokens} hidden={hidden} ids={id_count}")
54
+ print(f"ids[:12]={ids[:12].tolist()}")
55
+
56
+ with container.Artifact.open(art_path) as art:
57
+ obj = art.find("text/token_embedding")
58
+ payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8).copy()
59
+ signs_all = np.frombuffer(bytes(art.payload("text/hadamard_signs")), dtype=np.float32)
60
+ widths = np.frombuffer(bytes(art.payload("text/hadamard_widths")), dtype=np.int32)
61
+ rows = int(obj.shape[0])
62
+ off = 0
63
+ sign_offsets = {}
64
+ for w in widths:
65
+ sign_offsets[int(w)] = off
66
+ off += int(w)
67
+ signs = signs_all[sign_offsets[hidden]:sign_offsets[hidden] + hidden].astype(np.float64)
68
+
69
+ codes = payload[: rows * GROUPS * 32].reshape(rows, GROUPS, 32)
70
+ scale_off = rows * GROUPS * 32
71
+ scales = payload[scale_off: scale_off + rows * GROUPS * 2].view(np.float16).astype(
72
+ np.float64).reshape(rows, GROUPS)
73
+
74
+ def decode_row(index: int) -> np.ndarray:
75
+ q = codes[index] # (groups, 32 bytes)
76
+ code = (q[:, :, None] >> (2 * np.arange(4, dtype=np.uint8))[None, None, :]) & 3
77
+ flat = code.reshape(-1).astype(np.float64) # 32 bytes x 4 codes x groups
78
+ return (flat - 1.0) * np.repeat(scales[index], 128)
79
+
80
+ H = hadamard(BLOCK)
81
+ n_block = hidden // BLOCK
82
+
83
+ def inverse(vec: np.ndarray) -> np.ndarray:
84
+ out = np.empty_like(vec)
85
+ for b in range(n_block):
86
+ lo = b * BLOCK
87
+ out[lo:lo + BLOCK] = signs[lo:lo + BLOCK] * (H @ vec[lo:lo + BLOCK])
88
+ return out
89
+
90
+ def forward(vec: np.ndarray) -> np.ndarray:
91
+ out = np.empty_like(vec)
92
+ for b in range(n_block):
93
+ lo = b * BLOCK
94
+ out[lo:lo + BLOCK] = H @ (signs[lo:lo + BLOCK] * vec[lo:lo + BLOCK])
95
+ return out
96
+
97
+ checked = 0
98
+ worst = 0.0
99
+ for t in range(min(tokens, 8)):
100
+ z = decode_row(int(ids[t]))
101
+ ref = inverse(z)
102
+ col = got[:, t]
103
+ if t < 3:
104
+ def stats(name, v):
105
+ print(f" {name}: ||v||={np.linalg.norm(v):.6g} mean={v.mean():+.4g} "
106
+ f"std={v.std():.4g} min={v.min():+.4g} max={v.max():+.4g} "
107
+ f"nan={int(np.isnan(v).sum())}")
108
+ print(f" token {t} id={int(ids[t])}")
109
+ stats("engine dump", col)
110
+ stats("decoded row z", z)
111
+ stats("expected s*(H*z)", ref)
112
+ # is the engine maybe returning a DIFFERENT token's row, or a mis-strided gather?
113
+ for shift in (-2, -1, 1, 2):
114
+ if 0 <= t + shift < tokens:
115
+ other = inverse(decode_row(int(ids[t + shift])))
116
+ c = float(np.dot(col, other) /
117
+ max(np.linalg.norm(col) * np.linalg.norm(other), 1e-30))
118
+ print(f" cos vs token{shift:+d} row = {c:+.4f}")
119
+ rel = float(np.linalg.norm(col - ref) / np.linalg.norm(ref))
120
+ cos = float(np.dot(col, ref) / (np.linalg.norm(col) * np.linalg.norm(ref)))
121
+ ctrl_raw = float(np.dot(col, z) / (np.linalg.norm(col) * np.linalg.norm(z)))
122
+ ctrl_fwd = forward(z)
123
+ ctrl_fwd_cos = float(np.dot(col, ctrl_fwd) / (np.linalg.norm(col) * np.linalg.norm(ctrl_fwd)))
124
+ if t < 4:
125
+ print(f" rel_l2={rel:.4e} cos={cos:+.6f} | "
126
+ f"control cos(unrotated)={ctrl_raw:+.4f} cos(signs-first)={ctrl_fwd_cos:+.4f}")
127
+ worst = max(worst, rel)
128
+ checked += 1
129
+ print(f"\nchecked {checked} tokens, worst rel_l2 = {worst:.4e}")
130
+ print(f"RESULT: {'PASS' if worst <= 0.02 else 'FAIL'} (engine embedding == independent rebuild)")
131
+ return 0 if worst <= 0.02 else 1
132
+
133
+
134
+ if __name__ == "__main__":
135
+ raise SystemExit(main())
tools/verify/check_row_order.py ADDED
@@ -0,0 +1,161 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Verify the two INFERRED row-order rules by exact row fingerprints.
2
+
3
+ MAPPING.json flags exactly two rules as inferred rather than measured, and both are row
4
+ permutations of whole quantized rows -- which means each row's bytes survive untouched and can be
5
+ matched exactly between the GGUF and the artifact:
6
+
7
+ gdn_value_z "risk: if the engine test disagrees, flip this single permutation"
8
+ attn_q_per_head_interleave "INFERRED for main layers from the MTP layer"
9
+
10
+ Nothing about the rotated basis matters here: rows are permuted, not transformed, so a byte-level
11
+ row fingerprint is an exact test of the rule (this is the "structural self-proof" the handoff asks
12
+ for instead of value correlation, which is meaningless across bases).
13
+
14
+ Usage: check_row_order.py <artifact.ninfer> <pq2.gguf>
15
+ """
16
+ from __future__ import annotations
17
+
18
+ import hashlib
19
+ import sys
20
+ from pathlib import Path
21
+
22
+ import numpy as np
23
+
24
+ sys.path.insert(0, r"<NINFER_ROOT>")
25
+ sys.path.insert(0, r"<WORKSPACE>\tools")
26
+ from _ternary_ref import Gguf # noqa: E402
27
+ from tools.artifact import container # noqa: E402
28
+
29
+ GROUPS = 40
30
+ CODE_BYTES = 32
31
+ SCALE_BYTES = 2
32
+
33
+
34
+ def artifact_row_keys(art, name: str) -> list[bytes]:
35
+ """Fingerprint every row of a PQ2_0 row-split object as (codes, scales) digests."""
36
+ obj = art.find(name)
37
+ payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8)
38
+ rows = int(obj.shape[0])
39
+ assert int(obj.shape[1]) == GROUPS * 128, obj.shape
40
+ codes = payload[: rows * GROUPS * CODE_BYTES]
41
+ scale_start = rows * GROUPS * CODE_BYTES
42
+ scales = payload[scale_start: scale_start + rows * GROUPS * SCALE_BYTES]
43
+ keys: list[bytes] = []
44
+ for r in range(rows):
45
+ c = codes[r * GROUPS * CODE_BYTES:(r + 1) * GROUPS * CODE_BYTES]
46
+ s = scales[r * GROUPS * SCALE_BYTES:(r + 1) * GROUPS * SCALE_BYTES]
47
+ keys.append(hashlib.sha1(bytes(c) + bytes(s)).digest())
48
+ return keys
49
+
50
+
51
+ def gguf_row_keys(gguf: Gguf, name: str) -> list[bytes]:
52
+ """De-interleave GGUF {fp16 d; uint8 qs[32]} blocks into the same (codes, scales) fingerprint."""
53
+ ne, tt, _ = gguf.tensors[name]
54
+ assert tt == 142, (name, tt) # PQ2_0
55
+ rows = int(ne[1])
56
+ raw = gguf.raw(name, rows)
57
+ block = np.frombuffer(raw, dtype=np.uint8).reshape(rows, GROUPS, 34)
58
+ codes = np.ascontiguousarray(block[:, :, 2:34]).reshape(rows, GROUPS * CODE_BYTES)
59
+ scales = np.ascontiguousarray(block[:, :, 0:2]).reshape(rows, GROUPS * SCALE_BYTES)
60
+ return [hashlib.sha1(codes[r].tobytes() + scales[r].tobytes()).digest() for r in range(rows)]
61
+
62
+
63
+ def report(label: str, art_keys: list[bytes], gguf_subset: list[bytes],
64
+ src_index: list[int], expected) -> bool:
65
+ """Check the artifact rows two ways.
66
+
67
+ The multiset test proves the packer took the right source rows. The permutation test is done by
68
+ comparing the artifact row at index a directly against the source row the rule NAMES for a --
69
+ never by searching for a matching fingerprint first, because identical quantized rows do occur
70
+ and a first-hit search would then report a harmless false mismatch.
71
+ """
72
+ table: dict[bytes, list[int]] = {}
73
+ for r, key in enumerate(gguf_subset):
74
+ table.setdefault(key, []).append(r)
75
+ matched = sum(1 for key in art_keys if key in table)
76
+ ok_set = matched == len(art_keys)
77
+ unique_art = len(set(art_keys))
78
+ unique_src = len(set(gguf_subset))
79
+ print(f"{label}: artifact rows={len(art_keys)} source rows={len(gguf_subset)} "
80
+ f"matched={matched} -> {'row SET matches' if ok_set else 'ROW SET MISMATCH'}")
81
+ print(f" distinct fingerprints: artifact={unique_art} source={unique_src} "
82
+ f"(duplicates make fingerprint-first search unreliable, hence the direct check)")
83
+
84
+ direct = [a for a, src in enumerate(src_index)
85
+ if art_keys[a] != gguf_subset[src]]
86
+ perm_ok = not direct
87
+ print(f" documented rule holds row-by-row: {perm_ok}"
88
+ + (f" ({len(direct)} of {len(art_keys)} rows disagree)" if not perm_ok else ""))
89
+
90
+ # Is the artifact a genuine PERMUTATION of the source rows? A repack that reshaped the wrong
91
+ # axis repeats some rows and drops others, which a "does every row exist somewhere" test cannot
92
+ # see: every row still matches something. Compare the multisets explicitly.
93
+ from collections import Counter
94
+ art_counts = Counter(art_keys)
95
+ src_counts = Counter(gguf_subset)
96
+ repeated = {k: v for k, v in art_counts.items() if v > 1}
97
+ missing = [k for k in src_counts if k not in art_counts]
98
+ extra_rows = sum(v - 1 for v in repeated.values())
99
+ print(f" multiset: artifact {len(art_keys)} rows / {len(art_counts)} distinct; "
100
+ f"source {len(gguf_subset)} rows / {len(src_counts)} distinct")
101
+ print(f" rows DUPLICATED in the artifact: {len(repeated)} fingerprints covering "
102
+ f"{extra_rows} redundant rows; source rows MISSING from the artifact: {len(missing)}")
103
+ if missing or repeated:
104
+ print(" -> NOT a permutation: the packing permutation duplicates and drops rows")
105
+ if not perm_ok:
106
+ print(" first divergent rows (artifact row -> named source vs what it actually equals):")
107
+ shown = 0
108
+ for a in direct[:6]:
109
+ hits = table.get(art_keys[a], [])
110
+ print(f" {a} -> named {src_index[a]}, actually equals source rows {hits[:4]}")
111
+ shown += 1
112
+ print(f" artifact first 24 -> named sources: {src_index[:24]}")
113
+ return ok_set and perm_ok
114
+
115
+
116
+ def tiled_to_grouped_index(grouped_row: int, heads: int = 48, per: int = 3) -> int:
117
+ """grouped row (nk, rep, hd) -> tiled source row (rep, nk, hd), with 128 values per head."""
118
+ hd = grouped_row % 128
119
+ rest = grouped_row // 128
120
+ nk, rep = rest // per, rest % per
121
+ return rep * (heads // per) * 128 + nk * 128 + hd
122
+
123
+
124
+ def main() -> int:
125
+ art_path, gguf_path = sys.argv[1], sys.argv[2]
126
+ gguf = Gguf(Path(gguf_path))
127
+ ok = True
128
+ with container.Artifact.open(art_path) as art:
129
+ # ---- rule: gdn_value_z (48 layers) ----------------------------------
130
+ art_keys = artifact_row_keys(art, "text/layers/0/gdn/value_z")
131
+ value_src = gguf_row_keys(gguf, "blk.0.attn_qkv.weight")[4096:10240] # 6144 V rows
132
+ gate_src = gguf_row_keys(gguf, "blk.0.attn_gate.weight")[0:6144] # 6144 z rows
133
+ print("== gdn_value_z: artifact rows [0:6144] vs gguf attn_qkv[4096:10240]")
134
+ ok &= report(" value part", art_keys[0:6144], value_src,
135
+ [tiled_to_grouped_index(r) for r in range(6144)], None)
136
+ print("== gdn_value_z: artifact rows [6144:12288] vs gguf attn_gate[0:6144]")
137
+ ok &= report(" z part", art_keys[6144:12288], gate_src,
138
+ [tiled_to_grouped_index(r) for r in range(6144)], None)
139
+
140
+ # ---- rule: attn_q_per_head_interleave (16 layers) -------------------
141
+ qk_keys = artifact_row_keys(art, "text/layers/3/attention/query_key")
142
+ attn_q = gguf_row_keys(gguf, "blk.3.attn_q.weight") # 12288 rows: 48 chunks of 256
143
+ even = [r for chunk in range(0, 48, 2) for r in range(chunk * 256, chunk * 256 + 256)]
144
+ print("== attn_q_per_head_interleave: artifact query rows [0:6144] vs even chunks of attn_q")
145
+ ok &= report(" query part", qk_keys[0:6144], [attn_q[r] for r in even],
146
+ list(range(6144)), None)
147
+ attn_k = gguf_row_keys(gguf, "blk.3.attn_k.weight")
148
+ print("== attn key part: artifact rows [6144:7168] vs gguf attn_k[0:1024]")
149
+ ok &= report(" key part", qk_keys[6144:7168], attn_k[0:1024], list(range(1024)), None)
150
+
151
+ gv_keys = artifact_row_keys(art, "text/layers/3/attention/gate_value")
152
+ odd = [r for chunk in range(1, 48, 2) for r in range(chunk * 256, chunk * 256 + 256)]
153
+ print("== attn_q_per_head_interleave: artifact gate rows [0:6144] vs odd chunks of attn_q")
154
+ ok &= report(" gate part", gv_keys[0:6144], [attn_q[r] for r in odd],
155
+ list(range(6144)), None)
156
+ print(f"\nRESULT: {'all row-order rules hold' if ok else 'ROW-ORDER RULE MISMATCH FOUND'}")
157
+ return 0 if ok else 1
158
+
159
+
160
+ if __name__ == "__main__":
161
+ raise SystemExit(main())
tools/verify/gemm_oracle.py ADDED
@@ -0,0 +1,140 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """Dump / check the ternary GEMM + rotation integration against an independent reference.
2
+
3
+ Two modes:
4
+
5
+ dump <artifact> <object> <dump-dir>
6
+ Writes the object's row-split payload, the sign block for its input width, and a
7
+ deterministic activation, so a standalone nvcc harness can run the REAL kernels.
8
+
9
+ check <artifact> <object> <dump-dir>
10
+ Decodes the payload in pure numpy (the arrangement was already proven by
11
+ check_payload_order.py), rebuilds y = W' * (H * (s * (P * x))) explicitly in float64 and
12
+ compares against the kernel dump.
13
+
14
+ The point is to cover what no byte/size/oracle check so far covers: that the plane pointers, the
15
+ 2-bit decode, the sign block for THIS width, and the rotation compose into the documented math.
16
+
17
+ Usage: gemm_oracle.py dump|check <artifact.ninfer> <object-name> <dump-dir>
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import os
22
+ import sys
23
+
24
+ import numpy as np
25
+
26
+ sys.path.insert(0, r"<NINFER_ROOT>")
27
+ from tools.artifact import container # noqa: E402
28
+
29
+ BLOCK = 1024
30
+ POPCOUNT = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
31
+
32
+
33
+ def popcount32(a: np.ndarray) -> np.ndarray:
34
+ v = a.astype(np.uint32)
35
+ return (
36
+ POPCOUNT[v & 0xFF].astype(np.uint16)
37
+ + POPCOUNT[(v >> 8) & 0xFF].astype(np.uint16)
38
+ + POPCOUNT[(v >> 16) & 0xFF].astype(np.uint16)
39
+ + POPCOUNT[(v >> 24) & 0xFF].astype(np.uint16)
40
+ )
41
+
42
+
43
+ def hadamard(n: int) -> np.ndarray:
44
+ idx = np.arange(n, dtype=np.uint32)
45
+ parity = popcount32(idx[:, None] & idx[None, :]) & 1
46
+ return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
47
+
48
+
49
+ def make_x(k: int, tokens: int, seed: int = 12345) -> np.ndarray:
50
+ state = np.uint32(seed)
51
+ out = np.empty(k * tokens, dtype=np.float32)
52
+ for i in range(out.size):
53
+ state = np.uint32(state * np.uint32(1664525) + np.uint32(1013904223))
54
+ out[i] = np.float32((int(state >> 8) & 0xFFFF) / 32768.0 - 1.0)
55
+ return out.reshape(k, tokens)
56
+
57
+
58
+ def load(artifact_path: str, name: str):
59
+ with container.Artifact.open(artifact_path) as art:
60
+ obj = art.find(name)
61
+ payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8).copy()
62
+ signs_all = np.frombuffer(bytes(art.payload("text/hadamard_signs")), dtype=np.float32)
63
+ widths = np.frombuffer(bytes(art.payload("text/hadamard_widths")), dtype=np.int32)
64
+ rows, cols = int(obj.shape[0]), int(obj.shape[1])
65
+ offsets, acc = {}, 0
66
+ for w in widths:
67
+ offsets[int(w)] = acc
68
+ acc += int(w)
69
+ signs = signs_all[offsets[cols]:offsets[cols] + cols].copy()
70
+ return payload, signs, rows, cols, obj.format
71
+
72
+
73
+ def decode_pq2(payload: np.ndarray, rows: int, groups: int) -> np.ndarray:
74
+ """Same arrangement the row-split layout declares, decoded exactly like ggml's PQ2_0."""
75
+ codes = payload[: rows * groups * 32].reshape(rows, groups, 32)
76
+ scales = payload[rows * groups * 32: rows * groups * 32 + rows * groups * 2].view(
77
+ np.float16).astype(np.float64).reshape(rows, groups)
78
+ codes = codes.reshape(rows, groups, 32, 1)
79
+ codes = (codes >> (2 * np.arange(4, dtype=np.uint8)).reshape(1, 1, 1, 4)) & 3 # [rows,groups,32,4]
80
+ codes = codes.reshape(rows, groups, 128).astype(np.float64)
81
+ weights = (codes - 1.0) * scales[:, :, None]
82
+ return weights.reshape(rows, groups * 128)
83
+
84
+
85
+ def main() -> int:
86
+ mode, artifact_path, name, dump_dir = sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4]
87
+ payload, signs, rows, cols, fmt = load(artifact_path, name)
88
+ groups = cols // 128
89
+ if fmt != "PQ2_0_G128":
90
+ print(f"this probe only implements PQ2_0; got {fmt}")
91
+ return 2
92
+ x = make_x(cols, 1)
93
+ os.makedirs(dump_dir, exist_ok=True)
94
+
95
+ if mode == "dump":
96
+ payload.tofile(os.path.join(dump_dir, "gemm.payload.bin"))
97
+ signs.astype(np.float32).tofile(os.path.join(dump_dir, "gemm.signs.f32"))
98
+ x.astype(np.float32).tofile(os.path.join(dump_dir, "gemm.x.f32"))
99
+ print(f"dumped {name}: payload={payload.size} B rows={rows} cols={cols} groups={groups} "
100
+ f"scale_plane_off={rows * groups * 32}")
101
+ return 0
102
+
103
+ weights = decode_pq2(payload, rows, groups)
104
+ print(f"decoded W' {weights.shape} finite={np.isfinite(weights).all()} "
105
+ f"absmax={np.abs(weights).max():.6g}")
106
+
107
+ H = hadamard(BLOCK)
108
+ xr = np.empty_like(x, dtype=np.float64)
109
+ for b in range(cols // BLOCK):
110
+ lo = b * BLOCK
111
+ xr[lo:lo + BLOCK, 0] = H @ (signs[lo:lo + BLOCK].astype(np.float64) * x[lo:lo + BLOCK, 0])
112
+ y_ref = weights @ xr[:, 0]
113
+
114
+ y_got = np.fromfile(os.path.join(dump_dir, "gemm.y.f32"), dtype=np.float32).astype(np.float64)
115
+ if y_got.size != rows:
116
+ print(f"kernel dump has {y_got.size} rows, expected {rows}")
117
+ return 1
118
+
119
+ rel = float(np.linalg.norm(y_got - y_ref) / np.linalg.norm(y_ref))
120
+ diff = float(np.abs(y_got - y_ref).max())
121
+ # negative controls: if the kernel disagreed with the documented math in any of these ways the
122
+ # match would collapse, so report them to prove the comparison has teeth
123
+ y_norot = weights @ x[:, 0]
124
+ y_nosign = np.empty_like(x, dtype=np.float64)
125
+ for b in range(cols // BLOCK):
126
+ lo = b * BLOCK
127
+ y_nosign[lo:lo + BLOCK, 0] = H @ x[lo:lo + BLOCK, 0]
128
+ y_nosign = weights @ y_nosign[:, 0]
129
+ y_unnorm = weights @ (xr[:, 0] * np.sqrt(BLOCK))
130
+ print(f" rel_l2 = {rel:.4e} max|diff| = {diff:.4e} {'PASS' if rel <= 0.02 else 'FAIL'}")
131
+ for label, ref in (("no rotation", y_norot), ("no signs", y_nosign),
132
+ ("unnormalized H", y_unnorm)):
133
+ r = float(np.linalg.norm(y_got - ref) / np.linalg.norm(ref))
134
+ print(f" control {label:16s} rel_l2={r:.4f} "
135
+ f"{'(differs OK)' if r > 0.05 else '(TOO CLOSE)'}")
136
+ return 0 if rel <= 0.02 else 1
137
+
138
+
139
+ if __name__ == "__main__":
140
+ raise SystemExit(main())
tools/verify/oracle_rot.py ADDED
@@ -0,0 +1,168 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """numpy oracle for the folded-basis rotation kernels (forward and inverse).
2
+
3
+ Reads the raw f32 dumps the standalone nvcc harness wrote, rebuilds the documented transform
4
+ explicitly, and compares against the REAL kernel output element for element.
5
+
6
+ The pass criterion is deliberately two-sided: the documented variant must match, AND a family of
7
+ plausible-but-wrong variants must all fail. A one-sided check cannot tell "the kernel is right"
8
+ from "the oracle is too loose" -- which is exactly the trap the handoff notes call a
9
+ self-consistent false green.
10
+
11
+ Usage: oracle_rot.py <dump-dir>
12
+ """
13
+ from __future__ import annotations
14
+
15
+ import os
16
+ import sys
17
+
18
+ import numpy as np
19
+
20
+ BLOCK = 1024
21
+ CASES = [
22
+ # tag, k, tokens, perm_hd, perm_nk, perm_rep, inverse
23
+ ("plain_t1", 5120, 1, 0, 0, 1, False),
24
+ ("plain_t3", 5120, 3, 0, 0, 1, False),
25
+ ("perm_t1", 6144, 1, 128, 16, 3, False),
26
+ ("perm_t2", 6144, 2, 128, 16, 3, False),
27
+ ("wide_t1", 17408, 1, 0, 0, 1, False),
28
+ ("inv_t2", 5120, 2, 0, 0, 1, True),
29
+ ]
30
+
31
+
32
+ _POPCOUNT = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
33
+
34
+
35
+ def _popcount32(a: np.ndarray) -> np.ndarray:
36
+ """numpy 1.26 has no bitwise_count, so unroll the byte lookup explicitly."""
37
+ v = a.astype(np.uint32)
38
+ return (
39
+ _POPCOUNT[v & 0xFF].astype(np.uint16)
40
+ + _POPCOUNT[(v >> 8) & 0xFF].astype(np.uint16)
41
+ + _POPCOUNT[(v >> 16) & 0xFF].astype(np.uint16)
42
+ + _POPCOUNT[(v >> 24) & 0xFF].astype(np.uint16)
43
+ )
44
+
45
+
46
+ def hadamard(n: int) -> np.ndarray:
47
+ idx = np.arange(n, dtype=np.uint32)
48
+ parity = _popcount32(idx[:, None] & idx[None, :]) & 1
49
+ return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
50
+
51
+
52
+ def forward_source_index(j: int, hd_n: int, nk_n: int, rep_n: int) -> int:
53
+ """inverse of the reshape/permute llama.cpp applies: post-P column j -> source column."""
54
+ hd = j % hd_n
55
+ q = j // hd_n
56
+ nk = q // rep_n
57
+ rep = q % rep_n
58
+ return hd + hd_n * nk + hd_n * nk_n * rep
59
+
60
+
61
+ def forward_dest_index(i: int, hd_n: int, nk_n: int, rep_n: int) -> int:
62
+ """the opposite direction: source column i -> post-P column (a negative control)."""
63
+ hd = i % hd_n
64
+ nk = (i // hd_n) % nk_n
65
+ rep = i // (hd_n * nk_n)
66
+ return hd + hd_n * rep + hd_n * rep_n * nk
67
+
68
+
69
+ def rel_l2(got: np.ndarray, ref: np.ndarray) -> float:
70
+ denom = float(np.linalg.norm(ref))
71
+ if denom == 0.0:
72
+ return float(np.linalg.norm(got - ref))
73
+ return float(np.linalg.norm(got - ref) / denom)
74
+
75
+
76
+ def evaluate(tag: str, k: int, tokens: int, hd_n: int, nk_n: int, rep_n: int, inverse: bool,
77
+ dump_dir: str, H: np.ndarray) -> bool:
78
+ x = np.fromfile(os.path.join(dump_dir, tag + ".in.f32"), dtype=np.float32).astype(np.float64)
79
+ s = np.fromfile(os.path.join(dump_dir, tag + ".signs.f32"), dtype=np.float32).astype(np.float64)
80
+ got = np.fromfile(os.path.join(dump_dir, tag + ".out.f32"), dtype=np.float32).astype(np.float64)
81
+ x = x.reshape(k, tokens)
82
+ got = got.reshape(k, tokens)
83
+
84
+ blocks = k // BLOCK
85
+ permuted = (not inverse) and rep_n > 1
86
+
87
+ def blockwise_hadamard(vec: np.ndarray) -> np.ndarray:
88
+ out = np.empty_like(vec)
89
+ for b in range(blocks):
90
+ lo = b * BLOCK
91
+ out[lo:lo + BLOCK] = H @ vec[lo:lo + BLOCK]
92
+ return out
93
+
94
+ def permute(vec: np.ndarray, src) -> np.ndarray:
95
+ return np.array([vec[src(j)] for j in range(k)], dtype=np.float64)
96
+
97
+ # Build every variant as a full [k, tokens] array so the comparison never broadcasts.
98
+ columns: dict[str, list[np.ndarray]] = {}
99
+ for t in range(tokens):
100
+ col = x[:, t]
101
+ if inverse:
102
+ current = {
103
+ "CORRECT s*(H*z)": s * blockwise_hadamard(col),
104
+ "WRONG H*(s*z)": blockwise_hadamard(s * col),
105
+ "WRONG H*z": blockwise_hadamard(col),
106
+ "WRONG s*z": s * col,
107
+ }
108
+ else:
109
+ base = col if not permuted else permute(
110
+ col, lambda j: forward_source_index(j, hd_n, nk_n, rep_n))
111
+ # one sign row off: block b uses sign row b-1
112
+ shifted = np.empty_like(s)
113
+ for b in range(blocks):
114
+ src = ((b - 1) % blocks) * BLOCK
115
+ dst = b * BLOCK
116
+ shifted[dst:dst + BLOCK] = s[src:src + BLOCK]
117
+ current = {
118
+ "CORRECT H*(s*(P*x))": blockwise_hadamard(s * base),
119
+ "WRONG s*(H*(P*x))": s * blockwise_hadamard(base),
120
+ "WRONG unnormalized H": blockwise_hadamard(s * base) * np.sqrt(BLOCK),
121
+ "WRONG sign row shifted": blockwise_hadamard(shifted * base),
122
+ }
123
+ if permuted:
124
+ current["WRONG H*(s*x) no P"] = blockwise_hadamard(s * col)
125
+ current["WRONG P the other way"] = blockwise_hadamard(
126
+ s * permute(col, lambda j: forward_dest_index(j, hd_n, nk_n, rep_n)))
127
+ for name, value in current.items():
128
+ columns.setdefault(name, []).append(value)
129
+
130
+ variants = {name: np.stack(cols, axis=1) for name, cols in columns.items()}
131
+ correct_name = "CORRECT s*(H*z)" if inverse else "CORRECT H*(s*(P*x))"
132
+ ref = variants[correct_name]
133
+
134
+ correct = rel_l2(got, ref)
135
+ print(f"case {tag:9s} k={k:<6d} tokens={tokens} perm={permuted}")
136
+ print(f" norms: ||x||={np.linalg.norm(x):.6g} ||ref||={np.linalg.norm(ref):.6g} "
137
+ f"||got||={np.linalg.norm(got):.6g} ratio={np.linalg.norm(got) / np.linalg.norm(ref):.6g}")
138
+ ok = correct <= 0.01
139
+ print(f" {'PASS' if ok else 'FAIL'} CORRECT variant rel_l2={correct:.3e}")
140
+ separation = True
141
+ for name, value in variants.items():
142
+ if name.startswith("CORRECT"):
143
+ continue
144
+ r = rel_l2(got, value)
145
+ good = r > 10.0 * max(correct, 1e-6)
146
+ separation = separation and good
147
+ print(f" {name:26s} rel_l2={r:.4f} {'(differs OK)' if good else '(TOO CLOSE)'}")
148
+ return ok and separation
149
+
150
+
151
+ def main() -> int:
152
+ dump_dir = sys.argv[1]
153
+ H = hadamard(BLOCK)
154
+ print(f"oracle: explicit normalized Sylvester-Hadamard {BLOCK}x{BLOCK} "
155
+ f"(H[i][j] = popcount(i&j)&1 ? -1 : +1, /sqrt({BLOCK}))")
156
+ print(f" symmetric={np.allclose(H, H.T)} orthogonal={np.allclose(H @ H, np.eye(BLOCK), atol=1e-9)}")
157
+ results = []
158
+ for tag, k, tokens, hd_n, nk_n, rep_n, inverse in CASES:
159
+ results.append(evaluate(tag, k, tokens, hd_n, nk_n, rep_n, inverse, dump_dir, H))
160
+ print()
161
+ passed = sum(1 for r in results if r)
162
+ print(f"RESULT: {passed}/{len(results)} cases pass (correct variant matches AND all "
163
+ f"negative controls separate)")
164
+ return 0 if passed == len(results) else 1
165
+
166
+
167
+ if __name__ == "__main__":
168
+ raise SystemExit(main())