Upload folder using huggingface_hub
Browse files- .gitattributes +36 -35
- LICENSE +201 -0
- NOTICE +129 -0
- README.md +293 -0
- README_en.md +254 -0
- REPRODUCE.md +316 -0
- SHA256SUMS +2 -0
- artifact-manifest.json +91 -0
- bonsai2_27b_swift_pq2.ninfer +3 -0
- bonsai2_27b_swift_ptq1.ninfer +3 -0
- release.conversion.json +53 -0
- tools/MAPPING.json +126 -0
- tools/README.md +77 -0
- tools/pack.py +958 -0
- tools/verify/check_assembly.py +155 -0
- tools/verify/check_embedding.py +135 -0
- tools/verify/check_row_order.py +161 -0
- tools/verify/gemm_oracle.py +140 -0
- tools/verify/oracle_rot.py +168 -0
.gitattributes
CHANGED
|
@@ -1,35 +1,36 @@
|
|
| 1 |
-
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
-
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
-
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
-
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
-
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
-
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
-
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
-
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
-
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
-
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
-
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
-
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
-
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
-
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
-
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
-
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
-
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
-
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
-
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
-
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
-
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
-
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
-
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
-
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
-
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
-
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
-
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
-
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
-
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
-
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
-
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
-
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
-
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
-
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
-
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 1 |
+
*.7z filter=lfs diff=lfs merge=lfs -text
|
| 2 |
+
*.arrow filter=lfs diff=lfs merge=lfs -text
|
| 3 |
+
*.bin filter=lfs diff=lfs merge=lfs -text
|
| 4 |
+
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
| 5 |
+
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
| 6 |
+
*.ftz filter=lfs diff=lfs merge=lfs -text
|
| 7 |
+
*.gz filter=lfs diff=lfs merge=lfs -text
|
| 8 |
+
*.h5 filter=lfs diff=lfs merge=lfs -text
|
| 9 |
+
*.joblib filter=lfs diff=lfs merge=lfs -text
|
| 10 |
+
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
| 11 |
+
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
| 12 |
+
*.model filter=lfs diff=lfs merge=lfs -text
|
| 13 |
+
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
| 14 |
+
*.npy filter=lfs diff=lfs merge=lfs -text
|
| 15 |
+
*.npz filter=lfs diff=lfs merge=lfs -text
|
| 16 |
+
*.onnx filter=lfs diff=lfs merge=lfs -text
|
| 17 |
+
*.ot filter=lfs diff=lfs merge=lfs -text
|
| 18 |
+
*.parquet filter=lfs diff=lfs merge=lfs -text
|
| 19 |
+
*.pb filter=lfs diff=lfs merge=lfs -text
|
| 20 |
+
*.pickle filter=lfs diff=lfs merge=lfs -text
|
| 21 |
+
*.pkl filter=lfs diff=lfs merge=lfs -text
|
| 22 |
+
*.pt filter=lfs diff=lfs merge=lfs -text
|
| 23 |
+
*.pth filter=lfs diff=lfs merge=lfs -text
|
| 24 |
+
*.rar filter=lfs diff=lfs merge=lfs -text
|
| 25 |
+
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
| 26 |
+
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
| 27 |
+
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
| 28 |
+
*.tar filter=lfs diff=lfs merge=lfs -text
|
| 29 |
+
*.tflite filter=lfs diff=lfs merge=lfs -text
|
| 30 |
+
*.tgz filter=lfs diff=lfs merge=lfs -text
|
| 31 |
+
*.wasm filter=lfs diff=lfs merge=lfs -text
|
| 32 |
+
*.xz filter=lfs diff=lfs merge=lfs -text
|
| 33 |
+
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
+
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
+
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
*.ninfer filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
NOTICE
ADDED
|
@@ -0,0 +1,129 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Swift-Bonsai-2 27B — NInfer ternary artifacts (PQ2_0_G128 / PTQ1_0_G128)
|
| 2 |
+
=======================================================================
|
| 3 |
+
|
| 4 |
+
This work is licensed under the Apache License, Version 2.0. See the LICENSE file.
|
| 5 |
+
|
| 6 |
+
Every component in the provenance chain is Apache-2.0 licensed **except one link, which is
|
| 7 |
+
disclosed explicitly in section 2 below**. We do not hide it.
|
| 8 |
+
|
| 9 |
+
|
| 10 |
+
1. PROVENANCE CHAIN
|
| 11 |
+
-------------------
|
| 12 |
+
|
| 13 |
+
Qwen/Qwen3.8-27B Apache-2.0
|
| 14 |
+
Copyright 2026 Alibaba Cloud
|
| 15 |
+
The dense 27B base model. Geometry, tokenizer and the six frontend resources derive here.
|
| 16 |
+
|
| 17 |
+
prism-ml/Ternary-Bonsai-2-27B-gguf Apache-2.0
|
| 18 |
+
Copyright Prism ML, Inc.
|
| 19 |
+
The ternary (1.58-bit) + Hadamard-rotated-basis release of Bonsai 2 27B.
|
| 20 |
+
Defines the `prism.hadamard.*` metadata contract this artifact carries.
|
| 21 |
+
|
| 22 |
+
ukisai/Swift-Bonsai-2-GGUF Apache-2.0
|
| 23 |
+
ukisai — "Swift Bonsai 2", a reasoning-efficient fine-tune.
|
| 24 |
+
>>> THIS IS THE WEIGHT SOURCE OF THE ARTIFACTS IN THIS REPOSITORY. <<<
|
| 25 |
+
|
| 26 |
+
github.com/Neroued/ninfer Apache-2.0
|
| 27 |
+
Upstream engine, converter and the version-2 artifact container format.
|
| 28 |
+
|
| 29 |
+
github.com/Ambolio/ninfer-4090-windows Apache-2.0
|
| 30 |
+
The Ada/Windows source lineage (v1.0.6 / v1.0.8) whose format dialect this artifact uses.
|
| 31 |
+
(The HF copy that existed under the same name is now 404; the lineage persists through forks.)
|
| 32 |
+
|
| 33 |
+
shensanshu/ninfer-ada-ternary Apache-2.0
|
| 34 |
+
The packer `tools/pack.py`, the per-tensor mapping table `tools/MAPPING.json`,
|
| 35 |
+
and the five verification scripts, all redistributed here under Apache-2.0
|
| 36 |
+
with the notice below.
|
| 37 |
+
|
| 38 |
+
Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer Apache-2.0
|
| 39 |
+
Used ONLY as a template: its object list (the skeleton this packer walks) and the
|
| 40 |
+
payloads it donates for vision / mtp / frontend / draft_head.
|
| 41 |
+
**Its text weights are discarded entirely and do not enter this artifact.**
|
| 42 |
+
|
| 43 |
+
Upstream sibling — ukisai/Swift-Qwen3.8-27b swift-open-license-1.0
|
| 44 |
+
This is the BF16 fine-tune checkpoint from which ukisai's Swift derivative family descends.
|
| 45 |
+
It is **not** the direct source of this repository (we pack from ukisai's GGUF release,
|
| 46 |
+
which is itself marked Apache-2.0), but it sits in the same family, and its license is
|
| 47 |
+
**not** Apache. We flag it here rather than omit it so downstream users can make their own
|
| 48 |
+
judgement. If you need a conservative position, treat this chain as carrying that
|
| 49 |
+
upstream license.
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
2. SCOPE OF WHAT THIS REPOSITORY DISTRIBUTES
|
| 53 |
+
--------------------------------------------
|
| 54 |
+
|
| 55 |
+
This repository distributes **weight-derived artifacts** (the two `.ninfer` files), not code
|
| 56 |
+
weights per se. Per the packer project's own NOTICE:
|
| 57 |
+
|
| 58 |
+
> 由本项目产出的 `.ninfer` 制品属于权重派生品,其再分发义务以权重原许可为准,与代码许可无关。
|
| 59 |
+
|
| 60 |
+
The direct source (`ukisai/Swift-Bonsai-2-GGUF`) and the base (`prism-ml/Ternary-Bonsai-2-27B-gguf`,
|
| 61 |
+
`Qwen/Qwen3.8-27B`) are all published as **Apache-2.0**, which permits redistribution.
|
| 62 |
+
The un-Apache link noted in section 1 is upstream of the direct source and is disclosed there.
|
| 63 |
+
|
| 64 |
+
**No model weights were retrained, fine-tuned, abliterated or otherwise behaviourally modified
|
| 65 |
+
by us.** Model behaviour — including all refusal characteristics inherited from the upstream
|
| 66 |
+
fine-tune — derives entirely from `ukisai/Swift-Bonsai-2-GGUF`.
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
STATEMENT OF CHANGES (Apache-2.0 section 4(b))
|
| 70 |
+
----------------------------------------------
|
| 71 |
+
|
| 72 |
+
Relative to `ukisai/Swift-Bonsai-2-GGUF` (the weight source):
|
| 73 |
+
|
| 74 |
+
* The 402 ternary matrices were **moved byte-for-byte**. The 2-bit codes and their per-128
|
| 75 |
+
FP16 scales were copied verbatim from the GGUF block stream into the artifact's
|
| 76 |
+
`row-split-k128-v1` three-plane layout. **No dequantization and no requantization occurred.**
|
| 77 |
+
* llama.cpp's exporter conventions were undone, as the upstream recipe requires:
|
| 78 |
+
- GDN value heads: tiled order -> grouped order (per-head, `perm48(head)*128 + inner`)
|
| 79 |
+
- zero-centred norms: `1 + w` -> `w` (except `gdn/norm` / `ssm_norm`, whose raw value
|
| 80 |
+
is already ~1)
|
| 81 |
+
- `ssm_a` (`-exp(A_log)`) -> `A_log`
|
| 82 |
+
* `text/hadamard_signs` (28,672 fp32 words) and `text/hadamard_widths` (3 int32) were
|
| 83 |
+
**added**: they restate the GGUF's `prism.hadamard.sign_values` / `sign_widths` in the
|
| 84 |
+
artifact's own object vocabulary. The source GGUF already carries these values; the
|
| 85 |
+
template does not, which is why the object count is 1126 rather than 1124.
|
| 86 |
+
* Non-ternary tensors (norms, GDN A/B projections, convolution, `A_log`, `dt_bias`,
|
| 87 |
+
`token_embedding`'s direct companions) were re-encoded from the same GGUF.
|
| 88 |
+
|
| 89 |
+
Relative to `Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer` (the template):
|
| 90 |
+
|
| 91 |
+
* **Nothing of its text tower is used.** Only its object list and the vision (333), mtp (12),
|
| 92 |
+
frontend (6) and draft_head (2) payloads — 1,116,856,537 bytes — were carried over.
|
| 93 |
+
* Those payloads are the official Qwen3.8-27B components shared across the whole family.
|
| 94 |
+
The packer verified the MTP head against the target at cosine >= 0.99966 with seven norms
|
| 95 |
+
**bit-identical**.
|
| 96 |
+
|
| 97 |
+
Relative to `shensanshu/ninfer-ada-ternary` `tools/pack.py`:
|
| 98 |
+
|
| 99 |
+
* **No functional change.** The only edit was to the default path constants at the top of the
|
| 100 |
+
file — `NINFER_ROOT` (which pointed at the author's own development machine) and `TEMPLATE`
|
| 101 |
+
— both replaced with neutral placeholders (`<NINFER_ROOT>`, `<TEMPLATE>`). The same
|
| 102 |
+
substitution was applied to the five scripts under `tools/verify/` and to `MAPPING.json`'s
|
| 103 |
+
`_verified_by` note. **No logic was altered.**
|
| 104 |
+
* `MAPPING.json` is redistributed verbatim.
|
| 105 |
+
* **No validation, checksum, geometry check or byte-round-trip proof was disabled, relaxed
|
| 106 |
+
or bypassed.** The `check` mode was run to completion against the source GGUF and reported
|
| 107 |
+
`pad = 0` for all eight shape combinations, with `bytes_equal` and `decode_equal` true on
|
| 108 |
+
every sampled tensor.
|
| 109 |
+
|
| 110 |
+
Relative to `github.com/Neroued/ninfer`:
|
| 111 |
+
|
| 112 |
+
* **NO CHANGES.** Upstream was not patched, modified or circumvented by this work.
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
Trademarks
|
| 116 |
+
----------
|
| 117 |
+
|
| 118 |
+
"Qwen" is a trademark of Alibaba Cloud. "Bonsai" and "Prism ML" are marks of Prism ML, Inc.
|
| 119 |
+
|
| 120 |
+
This is an unofficial, community-produced derivative. It is **not** endorsed by or affiliated
|
| 121 |
+
with Alibaba Cloud, Prism ML, ukisai, shensanshu, or the NInfer project.
|
| 122 |
+
|
| 123 |
+
|
| 124 |
+
Disclaimer
|
| 125 |
+
----------
|
| 126 |
+
|
| 127 |
+
Provided **AS IS**, without warranty of any kind. All measurements cited in README.md were taken
|
| 128 |
+
in specific hardware/software environments and will differ across GPU model, driver, CUDA version
|
| 129 |
+
and memory bandwidth.
|
README.md
ADDED
|
@@ -0,0 +1,293 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
language:
|
| 4 |
+
- en
|
| 5 |
+
- zh
|
| 6 |
+
library_name: ninfer
|
| 7 |
+
pipeline_tag: text-generation
|
| 8 |
+
base_model:
|
| 9 |
+
- prism-ml/Ternary-Bonsai-2-27B-gguf
|
| 10 |
+
- ukisai/Swift-Bonsai-2-GGUF
|
| 11 |
+
- Qwen/Qwen3.8-27B
|
| 12 |
+
base_model_relation: quantized
|
| 13 |
+
tags:
|
| 14 |
+
- ninfer
|
| 15 |
+
- ternary
|
| 16 |
+
- 1.58-bit
|
| 17 |
+
- bonsai
|
| 18 |
+
- qwen3.8
|
| 19 |
+
- hadamard
|
| 20 |
+
- mtp
|
| 21 |
+
- swift
|
| 22 |
+
- text-generation
|
| 23 |
+
- image-text-to-text
|
| 24 |
+
---
|
| 25 |
+
|
| 26 |
+
# Swift-Bonsai-2 27B — NInfer 三元制品(PQ2 / PTQ1)
|
| 27 |
+
|
| 28 |
+
**English summary at the end.** | [English README](README_en.md)
|
| 29 |
+
|
| 30 |
+
---
|
| 31 |
+
|
| 32 |
+
## ⚠️ 先说引擎兼容性(最重要的一节)
|
| 33 |
+
|
| 34 |
+
本制品是 **`PQ2_0_G128` / `PTQ1_0_G128`** 方言,需要 **`Ambolio/ninfer-4090-windows` 血统**的引擎
|
| 35 |
+
(v1.0.6 / v1.0.8,即"极速档"那条线)。
|
| 36 |
+
|
| 37 |
+
**跟 HF 上其它 Bonsai `.ninfer` 不通用:**
|
| 38 |
+
|
| 39 |
+
| 制品 | 三元格式名 | 需要的引擎 |
|
| 40 |
+
|---|---|---|
|
| 41 |
+
| **本制品** | **`PQ2_0_G128` / `PTQ1_0_G128`** | **Ambolio 血统(v1.0.6 / v1.0.8)** |
|
| 42 |
+
| `WaveCut/Ternary-Bonsai-2-27B-NInfer-v3` | `t2_g128_fp16` | [`iamwavecut/ninfer-all`](https://github.com/iamwavecut/ninfer-all) |
|
| 43 |
+
| `neroued/Qwen3.8-27B-NInfer` | NVFP4 / groupwise-int | [`Neroued/ninfer`](https://github.com/Neroued/ninfer)(上游) |
|
| 44 |
+
|
| 45 |
+
**装错的表现是启动直接被拒(`refuse this file`),不是"慢一点"。**
|
| 46 |
+
|
| 47 |
+
判方言的方法 —— 读 `.ninfer` 前 1 MB 里的 `"format"` 字段即可,不用解压整个文件。
|
| 48 |
+
|
| 49 |
+
---
|
| 50 |
+
|
| 51 |
+
## 这是什么
|
| 52 |
+
|
| 53 |
+
`Swift-Bonsai-2` 的三元 NInfer 制品,两档:
|
| 54 |
+
|
| 55 |
+
| 文件 | 档位 | 大小 | 源 GGUF |
|
| 56 |
+
|---|---|---|---|
|
| 57 |
+
| `bonsai2_27b_swift_pq2.ninfer` | `PQ2_0_G128`(2 bit) | 8,306,927,628 B / 7.736 GiB | `Swift-Bonsai-2-PQ2_0.gguf` |
|
| 58 |
+
| `bonsai2_27b_swift_ptq1.ninfer` | `PTQ1_0_G128`(1.75 bit) | 7,047,407,628 B / 6.563 GiB | `Swift-Bonsai-2-PTQ1_0.gguf` |
|
| 59 |
+
|
| 60 |
+
**组件:text + vision + mtp(不含 dflash2)。**
|
| 61 |
+
|
| 62 |
+
**这是格式转换,不是训练。** 三元码逐字节搬运,没有解量化再重量化。
|
| 63 |
+
|
| 64 |
+
---
|
| 65 |
+
|
| 66 |
+
## 产物结构(已核对)
|
| 67 |
+
|
| 68 |
+
```
|
| 69 |
+
identity {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
|
| 70 |
+
magic NINFER\x00\x02 (version-2 容器)
|
| 71 |
+
objects 1126 = text 775 + vision 333 + mtp 12 + frontend 6
|
| 72 |
+
|
| 73 |
+
PQ2 档 formats BF16 582 | PQ2_0_G128 322 | FP32 97 | Q4G64_F16S 55
|
| 74 |
+
| Q5G64_F16S 54 | W8G32_F16S 7 | I32 2 | Q6G64_F16S 1
|
| 75 |
+
PTQ1 档 formats 同上,仅把 PQ2_0_G128 换成 PTQ1_0_G128
|
| 76 |
+
|
| 77 |
+
layouts row-split-k128-v1 439 | contiguous-le-v1 681
|
| 78 |
+
新增 text/hadamard_signs、text/hadamard_widths(三元必需)
|
| 79 |
+
借来的 vision 333 + mtp 12 + frontend 6 + draft_head 2 = 1,116,856,537 B(1.040 GiB)
|
| 80 |
+
```
|
| 81 |
+
|
| 82 |
+
**借来的那几块是全生态共用的官方件。** 打包器的原话:
|
| 83 |
+
|
| 84 |
+
> vision …**the template's tower is the same official one**
|
| 85 |
+
> mtp …**The template's head is the same official Qwen3.8-27B head**: all twelve tensors
|
| 86 |
+
> matched at cos ≥ 0.99966 with the seven norms **BIT-IDENTICAL**
|
| 87 |
+
|
| 88 |
+
---
|
| 89 |
+
|
| 90 |
+
## 怎么用
|
| 91 |
+
|
| 92 |
+
```powershell
|
| 93 |
+
ninfer-serve.exe bonsai2_27b_swift_pq2.ninfer ^
|
| 94 |
+
--host 127.0.0.1 --port 8087 ^
|
| 95 |
+
--max-context 131072 --kv-capacity 131072 ^
|
| 96 |
+
--kv-dtype int8 ^
|
| 97 |
+
--spec mtp --draft-tokens 3
|
| 98 |
+
```
|
| 99 |
+
|
| 100 |
+
**KV dtype 按卡的架构选:**
|
| 101 |
+
|
| 102 |
+
| 卡 | 可用 KV |
|
| 103 |
+
|---|---|
|
| 104 |
+
| 30 系(sm_86) | `bf16` / `int8` |
|
| 105 |
+
| 40 系(sm_89) | `bf16` / `int8` / `fp8` / `rk4v4` / `rk4v4-e8` |
|
| 106 |
+
| 50 系(sm_120) | `bf16` / `int8` / `fp8` / `nvfp4` / `k8v4`(`rk4v4` 会被拒) |
|
| 107 |
+
|
| 108 |
+
### ★ MTP 窗口:先扫,别照抄
|
| 109 |
+
|
| 110 |
+
上游在 4080S 上测出 **N=2 最优**(N=3/4/5 的接受率落到 40.9 / 33.2 / 22.9%)。
|
| 111 |
+
**但那不是普适值 —— 在 12 GB 的 3060 上,`d3` 才是最好的,且比生产用的 `d4+lm` 平均快 2~5%。**
|
| 112 |
+
|
| 113 |
+
详见下方「实测数据」的 MTP 表。**建议在你自己卡上扫 1~4 档,两个对照模型交替起服。**
|
| 114 |
+
|
| 115 |
+
---
|
| 116 |
+
|
| 117 |
+
## 验收状态
|
| 118 |
+
|
| 119 |
+
| 项 | 状态 |
|
| 120 |
+
|---|---|
|
| 121 |
+
| 源 GGUF 几何 + round-trip | ✅ 8 个形状组合 `pad=0`;5 个真实张量 `bytes_equal` + `decode_equal` 全 True;`zero_share` 0.3277~0.3279(`PQ2_0` 理论指纹 0.3278) |
|
| 122 |
+
| 产出结构对撞 | ✅ 与参考制品 `bonsai2_27b_ternary_v2.ninfer` 的格式分布 **8 项全等**;**总长也逐字节相同**(均 8,306,927,628 B) |
|
| 123 |
+
| 尺寸 | ✅ 两档与上游文档的换算表精确吻合(PQ2 8.31 GB / PTQ1 7.05 GB) |
|
| 124 |
+
| **端到端(RTX 3060 12G / sm_86)** | ✅ **四层全过** —— 启动、输出、前缀复用、视觉、工具调用、采样、思考档位全部正常 |
|
| 125 |
+
| **PPL(同语料 vs base)** | ✅ **略优于 base**,三口径差 −0.07% / −0.04% / −0.01% |
|
| 126 |
+
| **速度 / MTP 接受率** | ✅ **与 base 在 ±2% 内**(测量噪声量级) |
|
| 127 |
+
| 长思考任务上的「思考更短」 | ❓ **未验证** —— 见下 |
|
| 128 |
+
|
| 129 |
+
> **已由发布者在一台 RTX 3060 12G 上完成端到端验收**(2026-09-28)。
|
| 130 |
+
> 原始记录(`l1.txt` / `l23.txt` / `ppl-*.json` / `quiz-*.json`)由测试者保存在本地,**不随本仓分发**。
|
| 131 |
+
>
|
| 132 |
+
> **仍未验证的一项**:Swift 微调宣��的"思考 token 少 ~40%"。本次题集太简单(每题思考仅
|
| 133 |
+
> 100~500 token),**在该量级上得不出显著结论,也无法否证**。要验证需用 AIME / 竞赛级长思考题,
|
| 134 |
+
> 每题 2k~10k 思考 token、≥20 题 × 2 轮。
|
| 135 |
+
|
| 136 |
+
---
|
| 137 |
+
|
| 138 |
+
## 实测数据(RTX 3060 12G / sm_86 / 生产参数)
|
| 139 |
+
|
| 140 |
+
### PPL(同一 `perplexity` 程序、同一份 `pplab-text`、`int8` KV)
|
| 141 |
+
|
| 142 |
+
| 窗口/步长 | Swift | base | 差 |
|
| 143 |
+
|---|---:|---:|---:|
|
| 144 |
+
| 512 / 256 | **8.073669** | 8.079208 | **−0.07%** |
|
| 145 |
+
| 32 / 16 | **26.962015** | 26.972171 | **−0.04%** |
|
| 146 |
+
| 8 / 4 | **148.3656** | 148.3863 | **−0.01%** |
|
| 147 |
+
|
| 148 |
+
**⚠️ 不要拿别处的 PPL 绝对值来比。** PPL 跟语料强相关 —— 上游文档给的参照值(6.448742 /
|
| 149 |
+
26.049634 / 121.157720)用的是**另一份语料**,与上表**不可直接比较**。上表用的是 `pplab-text`,
|
| 150 |
+
只有同一语料下的 Swift vs base 差值才有意义。
|
| 151 |
+
|
| 152 |
+
### 速度与 MTP 接受率
|
| 153 |
+
|
| 154 |
+
条件:贪心,每组 400 token,1 次热身 + 2 次取中位,**两个模型交替起服**。
|
| 155 |
+
格式:`t/s(接受率)`。
|
| 156 |
+
|
| 157 |
+
| 配置 | 模型 | 中文 | 英文 | 代码 | 思考 | 平均 |
|
| 158 |
+
|---|---|---:|---:|---:|---:|---:|
|
| 159 |
+
| d1 | Swift | 42.2 (58%) | 44.4 (68%) | 46.0 (82%) | 45.8 (84%) | 44.6 |
|
| 160 |
+
| d1 | base | 43.1 (62%) | 44.9 (71%) | 46.6 (85%) | 45.4 (83%) | 45.0 |
|
| 161 |
+
| d2 | Swift | 46.0 (46%) | 49.5 (56%) | 58.4 (80%) | 55.9 (76%) | 52.5 |
|
| 162 |
+
| d2 | base | 43.7 (42%) | 46.9 (51%) | 59.6 (82%) | 55.7 (76%) | 51.5 |
|
| 163 |
+
| **d3** | Swift | 44.0 (33%) | 49.7 (44%) | 64.2 (70%) | 62.2 (68%) | **55.0** |
|
| 164 |
+
| **d3** | base | 43.4 (33%) | 52.4 (48%) | 64.9 (71%) | 59.9 (65%) | **55.1** |
|
| 165 |
+
| d4+lm(生产) | Swift | 40.4 (28%) | 48.0 (40%) | 65.5 (67%) | 58.6 (59%) | 53.1 |
|
| 166 |
+
| d4+lm(生产) | base | 39.2 (27%) | 48.2 (41%) | 64.6 (66%) | 57.1 (56%) | 52.3 |
|
| 167 |
+
|
| 168 |
+
**三条读法:**
|
| 169 |
+
|
| 170 |
+
1. **Swift 与 base 在每个配置下都相差 ±2% 以内** —— 这是测量噪声。**借来的 MTP 头与 Swift 主干配合得不比 base 差。**
|
| 171 |
+
2. **d3 最好,不是上游在 4080S 上说的 N=2。** d2 只在中文上占优。**MTP 窗口的最优值是卡的属性,不是模型的。**
|
| 172 |
+
3. **d3(不加 `lm-head`)平均比生产用的 `d4+lm` 高 2~5%,两个对照模型都是这个方向。**
|
| 173 |
+
这是一个**尚未纳入生产配置**的观察 —— 换之前请先用官方 bench 复核。
|
| 174 |
+
|
| 175 |
+
### 推理题(12 道唯一答案的中英文算术/逻辑题,生产参数,引擎默认采样)
|
| 176 |
+
|
| 177 |
+
| | 第 1 轮 | 第 2 轮 | 合计输出 token |
|
| 178 |
+
|---|---|---|---:|
|
| 179 |
+
| Swift | 12/12 | 12/12 | 4,501 |
|
| 180 |
+
| base | 12/12 | 12/12 | 5,144 |
|
| 181 |
+
|
| 182 |
+
**准确率相同;Swift 输出 token 少 12.5%。但样本太小 —— 逐题分布大量重叠,同一模型两轮之间最多能差 2 倍,得不出显著结论。**
|
| 183 |
+
所有回答均以 `finish=stop` 正常结束,无死循环或截断。
|
| 184 |
+
|
| 185 |
+
### 功能面(生产参数:ctx 65536 / `int8` / MTP d4+lm / vision)
|
| 186 |
+
|
| 187 |
+
| 项 | 结果 |
|
| 188 |
+
|---|---|
|
| 189 |
+
| 启动 | ✅ `/v1/models` 返回 `qwen3.8-27b`,`max_model_len` = 65536 |
|
| 190 |
+
| 英文 / 中文 | ✅ 通顺 |
|
| 191 |
+
| 同前缀多轮 ×3 | ✅ `200/200/200/200`,第 2、3 次命中 40 token 缓存,输出一致 |
|
| 192 |
+
| 连续 10 个不同请求 | ✅ 全部 200 |
|
| 193 |
+
| 采样 | ✅ 两次输出不同 |
|
| 194 |
+
| 思考档位 | ✅ `none/low/medium/xhigh` 正常;`high` 按预期返回 400(与 base 一致) |
|
| 195 |
+
| 前缀复用(约 2.4k token,3 轮) | ✅ prompt 耗时 **2848 ms → 258 / 268 ms** |
|
| 196 |
+
| 视觉 | ✅ 正确识别红色方块、`CAT`、蓝色 `42` |
|
| 197 |
+
| 工具调用 | ✅ 返回 `get_weather({"city":"Tokyo"})` |
|
| 198 |
+
| `reasoning_effort=none` 时不调工具 | ⚠️ 与 base 的已知现象相同,非本制品问题 |
|
| 199 |
+
|
| 200 |
+
> **关于前缀复用崩溃**:上游文档记录的"多轮/同前缀第二次请求 500 后全端口 503"在本测试环境**未复现**
|
| 201 |
+
> —— 该引擎已修正此问题。若你的引擎版本较旧且撞上,临时规避是 `--no-prefix-reuse`。
|
| 202 |
+
|
| 203 |
+
---
|
| 204 |
+
|
| 205 |
+
## 怎么造出来的(可复现)
|
| 206 |
+
|
| 207 |
+
```bash
|
| 208 |
+
# 1. 源(ukisai 的 Swift 微调,已三元量化)
|
| 209 |
+
Swift-Bonsai-2-PQ2_0.gguf
|
| 210 |
+
sha256 5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5
|
| 211 |
+
|
| 212 |
+
# 2. 模板 —— 只借它的骨架/对象清单 + vision/MTP/frontend 载荷
|
| 213 |
+
# 它的文本权重被整个丢弃,不进入本制品
|
| 214 |
+
Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
|
| 215 |
+
sha256 8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03
|
| 216 |
+
|
| 217 |
+
# 3. 打包
|
| 218 |
+
python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
|
| 219 |
+
--gguf Swift-Bonsai-2-PQ2_0.gguf \
|
| 220 |
+
--template qwen3_8_27b_huihui_abliterated.ninfer
|
| 221 |
+
# 约 4 分钟(纯 CPU,不需要 GPU)
|
| 222 |
+
```
|
| 223 |
+
|
| 224 |
+
**完整方法见 `REPRODUCE.md`。** 打包器与逐张量映射表随仓附带(`tools/`)。
|
| 225 |
+
|
| 226 |
+
---
|
| 227 |
+
|
| 228 |
+
## 署名(三层,全部必需)
|
| 229 |
+
|
| 230 |
+
```
|
| 231 |
+
Bonsai 2 27B(原始权重) © Prism ML, Inc. Apache-2.0
|
| 232 |
+
huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
|
| 233 |
+
Qwen3.8-27B(底座) © Alibaba Cloud Apache-2.0
|
| 234 |
+
Swift 微调(���制品权重来源) ukisai Apache-2.0
|
| 235 |
+
huggingface.co/ukisai/Swift-Bonsai-2-GGUF
|
| 236 |
+
打包工具 pack.py —— shensanshu/ninfer-ada-ternary(Apache-2.0)
|
| 237 |
+
容器格式 / 引擎 github.com/Neroued/ninfer(Apache-2.0)
|
| 238 |
+
```
|
| 239 |
+
|
| 240 |
+
官方请求的原句署名:
|
| 241 |
+
|
| 242 |
+
> **"Created using Bonsai by Prism ML."**
|
| 243 |
+
|
| 244 |
+
**详见 [`NOTICE`](NOTICE)。**
|
| 245 |
+
|
| 246 |
+
---
|
| 247 |
+
|
| 248 |
+
## 已知的坑(来自上游文档,非本制品独有)
|
| 249 |
+
|
| 250 |
+
1. **MTP 窗口 N=2 最优**,N=3/4/5 接受率崩(4080S 实测;不同卡需自测)
|
| 251 |
+
2. **前缀复用崩溃** —— 多轮/同前缀第二次请求 500 然后全端口 503;临时规避 `--no-prefix-reuse`
|
| 252 |
+
3. **改 `.h` 后必须 touch 所有 `.cpp`** —— MSVC 依赖扫描坏,改动会静默不生效(`/GS` 崩溃 `0xC0000409`)
|
| 253 |
+
4. **中文 prompt 走命令行会报 NFC 错** —— 必须用 `--messages <json文件>`(UTF-8 无 BOM)
|
| 254 |
+
5. **`ninfer-perplexity.exe` 必须跟 `ninfer-serve` 同一批编出来** —— 只重编 serve 会留下旧件,导致误判
|
| 255 |
+
|
| 256 |
+
---
|
| 257 |
+
|
| 258 |
+
## 许可
|
| 259 |
+
|
| 260 |
+
**Apache-2.0。** 见 [`LICENSE`](LICENSE) 与 [`NOTICE`](NOTICE)。
|
| 261 |
+
|
| 262 |
+
**商标**:"Qwen" 是 Alibaba Cloud 的商标;"Bonsai" 属 Prism ML。本制品是社区制作的衍生品,
|
| 263 |
+
与 Alibaba Cloud、Prism ML、ukisai 及 NInfer 项目**无隶属或背书关系**。
|
| 264 |
+
|
| 265 |
+
---
|
| 266 |
+
|
| 267 |
+
## English summary
|
| 268 |
+
|
| 269 |
+
**Swift-Bonsai-2 27B, NInfer ternary artifacts** — `PQ2_0_G128` (2-bit) and `PTQ1_0_G128` (1.75-bit).
|
| 270 |
+
|
| 271 |
+
**Engine compatibility:** this artifact uses the **`PQ2_0_G128` / `PTQ1_0_G128`** dialect and requires an
|
| 272 |
+
engine from the **`Ambolio/ninfer-4090-windows` lineage** (v1.0.6 / v1.0.8).
|
| 273 |
+
It is **NOT** interchangeable with `t2_g128_fp16` artifacts (those need `iamwavecut/ninfer-all`).
|
| 274 |
+
Loading the wrong one is refused at startup, not merely slow.
|
| 275 |
+
|
| 276 |
+
**What this is:** a **format conversion**, not a training run. Ternary codes are moved byte-for-byte
|
| 277 |
+
from the source GGUF; nothing is dequantized and requantized. Components: text + vision + mtp
|
| 278 |
+
(**no dflash2**).
|
| 279 |
+
|
| 280 |
+
**Provenance:** Bonsai 2 27B (Prism ML, Apache-2.0) → Swift fine-tune (ukisai, Apache-2.0)
|
| 281 |
+
→ packed with `pack.py` from `shensanshu/ninfer-ada-ternary` (Apache-2.0).
|
| 282 |
+
|
| 283 |
+
**Verification status:** structural and format-level checks are green (geometry `pad=0` across all 8
|
| 284 |
+
shape combinations; byte round-trip and decode equality on sampled tensors; `zero_share` 0.3278 —
|
| 285 |
+
the `PQ2_0` fingerprint). **End-to-end validation was completed on an RTX 3060 12G**: the server
|
| 286 |
+
starts, output is coherent, prefix reuse, vision and tool calling all work, and PPL is marginally
|
| 287 |
+
**better** than base on the same corpus (−0.07 % / −0.04 % / −0.01 %). See
|
| 288 |
+
[the measured-results tables](#实测数据rtx-3060-12g--sm_86--生产参数) in the Chinese section.
|
| 289 |
+
|
| 290 |
+
**One advertised claim remains unverified:** "~40 % fewer thinking tokens" — this round's items were
|
| 291 |
+
too easy to resolve at that magnitude. See the verification status table above.
|
| 292 |
+
|
| 293 |
+
> **"Created using Bonsai by Prism ML."**
|
README_en.md
ADDED
|
@@ -0,0 +1,254 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Swift-Bonsai-2 27B — NInfer ternary artifacts (PQ2 / PTQ1)
|
| 2 |
+
|
| 3 |
+
**中文说明见 [README.md](README.md)**
|
| 4 |
+
|
| 5 |
+
---
|
| 6 |
+
|
| 7 |
+
## ⚠️ Engine compatibility — read this first
|
| 8 |
+
|
| 9 |
+
These artifacts use the **`PQ2_0_G128` / `PTQ1_0_G128`** dialect and require an engine from the
|
| 10 |
+
**`Ambolio/ninfer-4090-windows` lineage** (v1.0.6 / v1.0.8, the "极速档 / speed tier" line).
|
| 11 |
+
|
| 12 |
+
**They are NOT interchangeable with other Bonsai `.ninfer` artifacts on the Hub:**
|
| 13 |
+
|
| 14 |
+
| Artifact | Ternary format name | Engine required |
|
| 15 |
+
|---|---|---|
|
| 16 |
+
| **this repo** | **`PQ2_0_G128` / `PTQ1_0_G128`** | **Ambolio lineage (v1.0.6 / v1.0.8)** |
|
| 17 |
+
| `WaveCut/Ternary-Bonsai-2-27B-NInfer-v3` | `t2_g128_fp16` | [`iamwavecut/ninfer-all`](https://github.com/iamwavecut/ninfer-all) |
|
| 18 |
+
| `neroued/Qwen3.8-27B-NInfer` | NVFP4 / groupwise-int | [`Neroued/ninfer`](https://github.com/Neroued/ninfer) (upstream) |
|
| 19 |
+
|
| 20 |
+
Loading the wrong one is **refused at startup** (`refuse this file`) — not merely slow.
|
| 21 |
+
|
| 22 |
+
To tell which dialect a `.ninfer` uses, read the `"format"` fields in its first 1 MiB. No need to
|
| 23 |
+
scan the whole file.
|
| 24 |
+
|
| 25 |
+
---
|
| 26 |
+
|
| 27 |
+
## What this is
|
| 28 |
+
|
| 29 |
+
Two tiers of `Swift-Bonsai-2` packaged for NInfer:
|
| 30 |
+
|
| 31 |
+
| File | Tier | Size | Source GGUF |
|
| 32 |
+
|---|---|---|---|
|
| 33 |
+
| `bonsai2_27b_swift_pq2.ninfer` | `PQ2_0_G128` (2-bit) | 8,306,927,628 B / 7.736 GiB | `Swift-Bonsai-2-PQ2_0.gguf` |
|
| 34 |
+
| `bonsai2_27b_swift_ptq1.ninfer` | `PTQ1_0_G128` (1.75-bit) | 7,047,407,628 B / 6.563 GiB | `Swift-Bonsai-2-PTQ1_0.gguf` |
|
| 35 |
+
|
| 36 |
+
**Components: text + vision + mtp (no dflash2).**
|
| 37 |
+
|
| 38 |
+
**This is a format conversion, not a training run.** The ternary codes are moved byte-for-byte;
|
| 39 |
+
nothing is dequantized and requantized.
|
| 40 |
+
|
| 41 |
+
---
|
| 42 |
+
|
| 43 |
+
## Structure (verified)
|
| 44 |
+
|
| 45 |
+
```
|
| 46 |
+
identity {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
|
| 47 |
+
magic NINFER\x00\x02 (version-2 container)
|
| 48 |
+
objects 1126 = text 775 + vision 333 + mtp 12 + frontend 6
|
| 49 |
+
|
| 50 |
+
PQ2 tier BF16 582 | PQ2_0_G128 322 | FP32 97 | Q4G64_F16S 55
|
| 51 |
+
| Q5G64_F16S 54 | W8G32_F16S 7 | I32 2 | Q6G64_F16S 1
|
| 52 |
+
PTQ1 tier identical except PQ2_0_G128 -> PTQ1_0_G128
|
| 53 |
+
|
| 54 |
+
layouts row-split-k128-v1 439 | contiguous-le-v1 681
|
| 55 |
+
added text/hadamard_signs, text/hadamard_widths (required by ternary)
|
| 56 |
+
borrowed vision 333 + mtp 12 + frontend 6 + draft_head 2 = 1,116,856,537 B (1.040 GiB)
|
| 57 |
+
```
|
| 58 |
+
|
| 59 |
+
The borrowed set are the official, family-wide Qwen3.8-27B components. From the packer:
|
| 60 |
+
|
| 61 |
+
> vision …**the template's tower is the same official one**
|
| 62 |
+
> mtp …**The template's head is the same official Qwen3.8-27B head**: all twelve tensors matched
|
| 63 |
+
> at cos ≥ 0.99966 with the seven norms **BIT-IDENTICAL**
|
| 64 |
+
|
| 65 |
+
---
|
| 66 |
+
|
| 67 |
+
## Usage
|
| 68 |
+
|
| 69 |
+
```powershell
|
| 70 |
+
ninfer-serve.exe bonsai2_27b_swift_pq2.ninfer ^
|
| 71 |
+
--host 127.0.0.1 --port 8087 ^
|
| 72 |
+
--max-context 131072 --kv-capacity 131072 ^
|
| 73 |
+
--kv-dtype int8 ^
|
| 74 |
+
--spec mtp --draft-tokens 3
|
| 75 |
+
```
|
| 76 |
+
|
| 77 |
+
**KV dtype depends on the GPU architecture:**
|
| 78 |
+
|
| 79 |
+
| GPU | Available KV |
|
| 80 |
+
|---|---|
|
| 81 |
+
| RTX 30 series (sm_86) | `bf16` / `int8` |
|
| 82 |
+
| RTX 40 series (sm_89) | `bf16` / `int8` / `fp8` / `rk4v4` / `rk4v4-e8` |
|
| 83 |
+
| RTX 50 series (sm_120) | `bf16` / `int8` / `fp8` / `nvfp4` / `k8v4` (`rk4v4` is rejected) |
|
| 84 |
+
|
| 85 |
+
**On a 12 GB card (e.g. RTX 3060),** `--max-context 32768~131072` with `--kv-dtype int8` is what was
|
| 86 |
+
**actually measured end-to-end** — see the verification status and measured-results tables below.
|
| 87 |
+
|
| 88 |
+
**Start MTP window scanning at N=2 — but do not copy that number.**
|
| 89 |
+
|
| 90 |
+
Upstream measured **N=2 optimal on a 4080 SUPER** (N=3/4/5 dropping to 40.9 / 33.2 / 22.9 % acceptance).
|
| 91 |
+
**That is not universal.** On a 12 GB RTX 3060, **`d3` is the best tier** and beats the production
|
| 92 |
+
`d4+lm` by 2–5 % on average. Scan 1–4 on your own card, starting both reference models alternately.
|
| 93 |
+
|
| 94 |
+
---
|
| 95 |
+
|
| 96 |
+
## Verification status
|
| 97 |
+
|
| 98 |
+
| Item | Status |
|
| 99 |
+
|---|---|
|
| 100 |
+
| Source GGUF geometry + round-trip | ✅ `pad=0` for all 8 shape combinations; `bytes_equal` + `decode_equal` true on every sampled tensor; `zero_share` 0.3277–0.3279 (the `PQ2_0` fingerprint is 0.3278) |
|
| 101 |
+
| Output structure | ✅ format distribution matches the reference artifact on all 8 buckets; **total size is byte-identical too** (both 8,306,927,628 B) |
|
| 102 |
+
| Sizes | ✅ both tiers match the upstream conversion table (PQ2 8.31 GB / PTQ1 7.05 GB) |
|
| 103 |
+
| **End-to-end (RTX 3060 12G / sm_86)** | ✅ **all four layers pass** — startup, output, prefix reuse, vision, tool calling, sampling, reasoning tiers |
|
| 104 |
+
| **PPL (same corpus, vs base)** | ✅ **slightly better than base**: −0.07 % / −0.04 % / −0.01 % across three window/stride settings |
|
| 105 |
+
| **Speed / MTP acceptance** | ✅ **within ±2 % of base** (measurement-noise level) |
|
| 106 |
+
| "Shorter thinking" on long-reasoning tasks | ❓ **not verified** |
|
| 107 |
+
|
| 108 |
+
> **End-to-end validation was completed by the publisher on one RTX 3060 12G** (2026-09-28).
|
| 109 |
+
> Raw records (`l1.txt`, `l23.txt`, `ppl-*.json`, `quiz-*.json`) are kept locally by the
|
| 110 |
+
> tester and are **not distributed with this repository**.
|
| 111 |
+
>
|
| 112 |
+
> **One claim remains unverified:** the fine-tune's advertised "~40 % fewer thinking tokens".
|
| 113 |
+
> This round's question set was too easy (100–500 thinking tokens per item) — **no conclusion is
|
| 114 |
+
> possible at that magnitude, in either direction**. Verifying it needs AIME / competition-level
|
| 115 |
+
> long-reasoning items, 2k–10k thinking tokens each, ≥20 items × 2 rounds.
|
| 116 |
+
|
| 117 |
+
---
|
| 118 |
+
|
| 119 |
+
## Measured results (RTX 3060 12G / sm_86 / production parameters)
|
| 120 |
+
|
| 121 |
+
### PPL — same `perplexity` binary, same `pplab-text` corpus, `int8` KV
|
| 122 |
+
|
| 123 |
+
| Window / stride | Swift | base | Δ |
|
| 124 |
+
|---|---:|---:|---:|
|
| 125 |
+
| 512 / 256 | **8.073669** | 8.079208 | **−0.07 %** |
|
| 126 |
+
| 32 / 16 | **26.962015** | 26.972171 | **−0.04 %** |
|
| 127 |
+
| 8 / 4 | **148.3656** | 148.3863 | **−0.01 %** |
|
| 128 |
+
|
| 129 |
+
**⚠️ Do not compare these absolute values against PPL figures from elsewhere.** PPL is corpus-dependent —
|
| 130 |
+
the reference values in the upstream docs (6.448742 / 26.049634 / 121.157720) come from a **different
|
| 131 |
+
corpus** and are **not directly comparable**. Only the Swift-vs-base delta within one corpus means anything.
|
| 132 |
+
|
| 133 |
+
### Speed and MTP acceptance
|
| 134 |
+
|
| 135 |
+
Greedy, 400 tokens per run, 1 warmup + 2 runs (median), **both models started alternately**.
|
| 136 |
+
Format: `t/s (acceptance)`.
|
| 137 |
+
|
| 138 |
+
| Config | Model | Chinese | English | Code | Thinking | Mean |
|
| 139 |
+
|---|---|---:|---:|---:|---:|---:|
|
| 140 |
+
| d1 | Swift | 42.2 (58%) | 44.4 (68%) | 46.0 (82%) | 45.8 (84%) | 44.6 |
|
| 141 |
+
| d1 | base | 43.1 (62%) | 44.9 (71%) | 46.6 (85%) | 45.4 (83%) | 45.0 |
|
| 142 |
+
| d2 | Swift | 46.0 (46%) | 49.5 (56%) | 58.4 (80%) | 55.9 (76%) | 52.5 |
|
| 143 |
+
| d2 | base | 43.7 (42%) | 46.9 (51%) | 59.6 (82%) | 55.7 (76%) | 51.5 |
|
| 144 |
+
| **d3** | Swift | 44.0 (33%) | 49.7 (44%) | 64.2 (70%) | 62.2 (68%) | **55.0** |
|
| 145 |
+
| **d3** | base | 43.4 (33%) | 52.4 (48%) | 64.9 (71%) | 59.9 (65%) | **55.1** |
|
| 146 |
+
| d4+lm (production) | Swift | 40.4 (28%) | 48.0 (40%) | 65.5 (67%) | 58.6 (59%) | 53.1 |
|
| 147 |
+
| d4+lm (production) | base | 39.2 (27%) | 48.2 (41%) | 64.6 (66%) | 57.1 (56%) | 52.3 |
|
| 148 |
+
|
| 149 |
+
**Three readings:**
|
| 150 |
+
|
| 151 |
+
1. **Swift and base are within ±2 % in every configuration** — measurement noise. **The borrowed MTP
|
| 152 |
+
head works no worse with the Swift backbone than with base.**
|
| 153 |
+
2. **`d3` is best, not the N=2 upstream reported for the 4080 SUPER.** d2 only leads on Chinese.
|
| 154 |
+
**The optimal MTP window is a property of the card, not of the model.**
|
| 155 |
+
3. **`d3` (without `lm-head`) averages 2–5 % faster than the production `d4+lm`, on both models.**
|
| 156 |
+
This is an **observation not yet folded into the production config** — re-check with the official
|
| 157 |
+
bench before switching.
|
| 158 |
+
|
| 159 |
+
### Reasoning quiz (12 unique-answer arithmetic/logic items, production params, engine default sampling)
|
| 160 |
+
|
| 161 |
+
| | Round 1 | Round 2 | Total output tokens |
|
| 162 |
+
|---|---|---|---:|
|
| 163 |
+
| Swift | 12/12 | 12/12 | 4,501 |
|
| 164 |
+
| base | 12/12 | 12/12 | 5,144 |
|
| 165 |
+
|
| 166 |
+
**Same accuracy; Swift emitted 12.5 % fewer tokens — but the sample is too small.** Per-item
|
| 167 |
+
distributions overlap heavily and the same model varies up to 2× between rounds, so no significant
|
| 168 |
+
conclusion follows. All replies ended with `finish=stop`; no loops, no truncation.
|
| 169 |
+
|
| 170 |
+
### Functional checks (production params: ctx 65536 / `int8` / MTP d4+lm / vision)
|
| 171 |
+
|
| 172 |
+
| Item | Result |
|
| 173 |
+
|---|---|
|
| 174 |
+
| Startup | ✅ `/v1/models` returns `qwen3.8-27b`, `max_model_len` = 65536 |
|
| 175 |
+
| English / Chinese | ✅ coherent |
|
| 176 |
+
| Multi-turn, same prefix ×3 | ✅ `200/200/200/200`; 2nd and 3rd hits a 40-token cache with identical output |
|
| 177 |
+
| 10 consecutive distinct requests | ✅ all 200 |
|
| 178 |
+
| Sampling | ✅ two runs differ |
|
| 179 |
+
| Reasoning tiers | ✅ `none/low/medium/xhigh` normal; `high` returns 400 as expected (same as base) |
|
| 180 |
+
| Prefix reuse (~2.4k tokens, 3 rounds) | ✅ prompt time **2848 ms → 258 / 268 ms** |
|
| 181 |
+
| Vision | ✅ correctly identified a red square, `CAT`, and a blue `42` |
|
| 182 |
+
| Tool calling | ✅ returned `get_weather({"city":"Tokyo"})` |
|
| 183 |
+
| No tool call at `reasoning_effort=none` | ⚠️ Same known behaviour as base — not an artifact issue |
|
| 184 |
+
|
| 185 |
+
> **On the prefix-reuse crash:** the upstream-documented failure (a second multi-turn/same-prefix
|
| 186 |
+
> request returning 500, then the whole port going 503) **did not reproduce** in this environment —
|
| 187 |
+
> that engine build already fixes it. On an older build, the workaround is `--no-prefix-reuse`.
|
| 188 |
+
|
| 189 |
+
---
|
| 190 |
+
|
| 191 |
+
## Reproduction
|
| 192 |
+
|
| 193 |
+
```bash
|
| 194 |
+
# 1. source (ukisai's Swift fine-tune, already ternary-quantized)
|
| 195 |
+
Swift-Bonsai-2-PQ2_0.gguf
|
| 196 |
+
sha256 5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5
|
| 197 |
+
|
| 198 |
+
# 2. template — only its skeleton/object list and vision/MTP/frontend payloads are used.
|
| 199 |
+
# Its text weights are discarded entirely.
|
| 200 |
+
Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
|
| 201 |
+
sha256 8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03
|
| 202 |
+
|
| 203 |
+
# 3. pack
|
| 204 |
+
python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
|
| 205 |
+
--gguf Swift-Bonsai-2-PQ2_0.gguf \
|
| 206 |
+
--template qwen3_8_27b_huihui_abliterated.ninfer
|
| 207 |
+
# ~4 minutes, CPU only, no GPU required
|
| 208 |
+
```
|
| 209 |
+
|
| 210 |
+
Full method: [`REPRODUCE.md`](REPRODUCE.md). The packer and the per-tensor mapping table ship in
|
| 211 |
+
[`tools/`](tools/) under Apache-2.0.
|
| 212 |
+
|
| 213 |
+
---
|
| 214 |
+
|
| 215 |
+
## Attribution (all three layers required)
|
| 216 |
+
|
| 217 |
+
```
|
| 218 |
+
Bonsai 2 27B (original weights) © Prism ML, Inc. Apache-2.0
|
| 219 |
+
huggingface.co/prism-ml/Ternary-Bonsai-2-27B-gguf
|
| 220 |
+
Qwen3.8-27B (geometry base) © Alibaba Cloud Apache-2.0
|
| 221 |
+
Swift fine-tune (weight source) ukisai Apache-2.0
|
| 222 |
+
huggingface.co/ukisai/Swift-Bonsai-2-GGUF
|
| 223 |
+
Packer pack.py — shensanshu/ninfer-ada-ternary (Apache-2.0)
|
| 224 |
+
Container format / engine github.com/Neroued/ninfer (Apache-2.0)
|
| 225 |
+
```
|
| 226 |
+
|
| 227 |
+
Required upstream attribution string:
|
| 228 |
+
|
| 229 |
+
> **"Created using Bonsai by Prism ML."**
|
| 230 |
+
|
| 231 |
+
See [`NOTICE`](NOTICE) — it also discloses one non-Apache link upstream in the fine-tune family.
|
| 232 |
+
|
| 233 |
+
---
|
| 234 |
+
|
| 235 |
+
## Known pitfalls (inherited, not specific to this artifact)
|
| 236 |
+
|
| 237 |
+
1. **MTP window:** N=2 is optimal; N=3/4/5 lose acceptance (measured on 4080 SUPER; retest on yours)
|
| 238 |
+
2. **Prefix-reuse crash:** a second multi-turn/same-prefix request returns 500, then the whole port
|
| 239 |
+
goes 503. Workaround: `--no-prefix-reuse`
|
| 240 |
+
3. **After editing any `.h`, touch all `.cpp`** — MSVC's header dependency scan is broken here, so
|
| 241 |
+
changes silently do not take effect (`/GS` crash `0xC0000409`)
|
| 242 |
+
4. **Non-ASCII prompts via the command line fail NFC normalization** — use `--messages <json>` (UTF-8, no BOM)
|
| 243 |
+
5. **`ninfer-perplexity.exe` must be built in the same batch as `ninfer-serve`** — rebuilding only
|
| 244 |
+
the server leaves a stale tool and makes you misjudge a change as ineffective
|
| 245 |
+
|
| 246 |
+
---
|
| 247 |
+
|
| 248 |
+
## License
|
| 249 |
+
|
| 250 |
+
**Apache-2.0.** See [`LICENSE`](LICENSE) and [`NOTICE`](NOTICE).
|
| 251 |
+
|
| 252 |
+
**Trademarks:** "Qwen" is a trademark of Alibaba Cloud; "Bonsai" and "Prism ML" belong to
|
| 253 |
+
Prism ML, Inc. This is an unofficial community-produced derivative and is not endorsed by or
|
| 254 |
+
affiliated with Alibaba Cloud, Prism ML, ukisai, shensanshu, or the NInfer project.
|
REPRODUCE.md
ADDED
|
@@ -0,0 +1,316 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# 复现步骤 · REPRODUCE
|
| 2 |
+
|
| 3 |
+
本仓的两个 `.ninfer` 制品是怎么造出来的 —— 从零到产出,可逐步复跑。
|
| 4 |
+
|
| 5 |
+
**全程 CPU,不需要 GPU。** 实测约 4 分钟一份。
|
| 6 |
+
|
| 7 |
+
---
|
| 8 |
+
|
| 9 |
+
## 0. 依赖
|
| 10 |
+
|
| 11 |
+
```
|
| 12 |
+
Python 3.11+
|
| 13 |
+
numpy
|
| 14 |
+
torch ← CPU 版即可(打包器不用 GPU,路径上写死了 torch.device("cpu"))
|
| 15 |
+
```
|
| 16 |
+
|
| 17 |
+
```bash
|
| 18 |
+
pip install numpy
|
| 19 |
+
pip install torch --index-url https://download.pytorch.org/whl/cpu
|
| 20 |
+
```
|
| 21 |
+
|
| 22 |
+
**如果你的网络在国内**,Hugging Face 走 `hf-mirror.com` 比走代理快约 50%(实测 7.65 MB/s vs 5.02 MB/s):
|
| 23 |
+
|
| 24 |
+
```bash
|
| 25 |
+
export HF_ENDPOINT=https://hf-mirror.com
|
| 26 |
+
```
|
| 27 |
+
|
| 28 |
+
---
|
| 29 |
+
|
| 30 |
+
## 1. 下载源与模板
|
| 31 |
+
|
| 32 |
+
### 1.1 权重源(ukisai 的 Swift 微调)
|
| 33 |
+
|
| 34 |
+
```
|
| 35 |
+
https://huggingface.co/ukisai/Swift-Bonsai-2-GGUF
|
| 36 |
+
```
|
| 37 |
+
|
| 38 |
+
| 文件 | 大小 | sha256 |
|
| 39 |
+
|---|---|---|
|
| 40 |
+
| `Swift-Bonsai-2-PQ2_0.gguf` | 7,206,168,928 B | `5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5` |
|
| 41 |
+
| `Swift-Bonsai-2-PTQ1_0.gguf` | 5,946,648,960 B | `de33620b60eaf63e96449b478eb87abe9538507e3ed939da944c1ae15fe1ffc6` |
|
| 42 |
+
|
| 43 |
+
### 1.2 模板
|
| 44 |
+
|
| 45 |
+
```
|
| 46 |
+
https://huggingface.co/Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer
|
| 47 |
+
```
|
| 48 |
+
|
| 49 |
+
| 文件 | 大小 | sha256 |
|
| 50 |
+
|---|---|---|
|
| 51 |
+
| `qwen3_8_27b_huihui_abliterated.ninfer` | 18,210,531,328 B | `8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03` |
|
| 52 |
+
|
| 53 |
+
**模板的身份必须满足:**
|
| 54 |
+
```json
|
| 55 |
+
"identity": {"model_id": "qwen3.8-27b", "weights_id": "groupwise-int"}
|
| 56 |
+
```
|
| 57 |
+
|
| 58 |
+
**并且对象名必须是"未融合"的那套**(`attention/query_key` + `attention/gate_value` 分开)。
|
| 59 |
+
`nvfp4` 打包的制品**用不了** —— 它把投影融合成了 `attention/query_key_gate_value`,
|
| 60 |
+
名字不在映射表里,打包器会直接中止。
|
| 61 |
+
|
| 62 |
+
### 1.3 打包器
|
| 63 |
+
|
| 64 |
+
`tools/` 目录随本仓附带(Apache-2.0,原样取自 `shensanshu/ninfer-ada-ternary`)。
|
| 65 |
+
|
| 66 |
+
它需要一个 `NINFER_ROOT` 指向的源码 checkout —— 那里只用来 import `tools.artifact` 这一组模块:
|
| 67 |
+
|
| 68 |
+
```
|
| 69 |
+
<NINFER_ROOT>/
|
| 70 |
+
tools/
|
| 71 |
+
__init__.py
|
| 72 |
+
artifact/
|
| 73 |
+
__init__.py
|
| 74 |
+
container.py
|
| 75 |
+
layouts.py
|
| 76 |
+
numeric.py
|
| 77 |
+
```
|
| 78 |
+
|
| 79 |
+
来源是 `Ambolio/ninfer-4090-windows` 血统的树(或任何含这套 `tools/artifact` 的 checkout)。
|
| 80 |
+
**`pack.py` 源码顶部有一个 `NINFER_ROOT` 默认值**,指到你的实际路径上即可。
|
| 81 |
+
|
| 82 |
+
---
|
| 83 |
+
|
| 84 |
+
## 2. 先验(不写文件)
|
| 85 |
+
|
| 86 |
+
```bash
|
| 87 |
+
python -u tools/pack.py check \
|
| 88 |
+
--gguf Swift-Bonsai-2-PQ2_0.gguf \
|
| 89 |
+
--template qwen3_8_27b_huihui_abliterated.ninfer
|
| 90 |
+
```
|
| 91 |
+
|
| 92 |
+
**期望输出:**
|
| 93 |
+
|
| 94 |
+
```
|
| 95 |
+
CHECK 1 geometry over every ternary shape present in the GGUF
|
| 96 |
+
PQ2_0_G128 n=1024 k=5120 groups/row=40 src= 1,392,640 payload= 1,392,640 pad= 0 (32 tensors)
|
| 97 |
+
PQ2_0_G128 n=5120 k=6144 groups/row=48 src= 8,355,840 payload= 8,355,840 pad= 0 (64 tensors)
|
| 98 |
+
PQ2_0_G128 n=5120 k=17408 groups/row=136 src= 23,674,880 payload= 23,674,880 pad= 0 (64 tensors)
|
| 99 |
+
PQ2_0_G128 n=6144 k=5120 groups/row=40 src= 8,355,840 payload= 8,355,840 pad= 0 (48 tensors)
|
| 100 |
+
PQ2_0_G128 n=10240 k=5120 groups/row=40 src= 13,926,400 payload= 13,926,400 pad= 0 (48 tensors)
|
| 101 |
+
PQ2_0_G128 n=12288 k=5120 groups/row=40 src= 16,711,680 payload= 16,711,680 pad= 0 (16 tensors)
|
| 102 |
+
PQ2_0_G128 n=17408 k=5120 groups/row=40 src= 23,674,880 payload= 23,674,880 pad= 0 (128 tensors)
|
| 103 |
+
PQ2_0_G128 n=248320 k=5120 groups/row=40 src= 337,715,200 payload= 337,715,200 pad= 0 (2 tensors)
|
| 104 |
+
distinct (format,shape) combos: 8
|
| 105 |
+
|
| 106 |
+
CHECK 2 byte round trip + decode equality on real tensors
|
| 107 |
+
blk.3.attn_q.weight bytes_equal=True decode_equal=True zero_share=0.3279 … PLAUSIBLE
|
| 108 |
+
…
|
| 109 |
+
```
|
| 110 |
+
|
| 111 |
+
**判据:`pad` 全为 0;`bytes_equal` 与 `decode_equal` 全 True;`zero_share` 落在 0.3277~0.3280。**
|
| 112 |
+
|
| 113 |
+
### ⚠️ 关于 `check` 的一个已知失败
|
| 114 |
+
|
| 115 |
+
在内存较小的机器上,`CHECK 2` 会在**超大张量**(`token_embd` / `output`,12.7 亿权重)上抛:
|
| 116 |
+
|
| 117 |
+
```
|
| 118 |
+
numpy.core._exceptions._ArrayMemoryError: Unable to allocate 9.47 GiB
|
| 119 |
+
for an array with shape (9932800, 128) and data type float64
|
| 120 |
+
```
|
| 121 |
+
|
| 122 |
+
**这不是转换错误** —— 那是 `mode_check` 为了做 decode 比对而做的**全量反量化**(`dq_pq2_0` 返回
|
| 123 |
+
float64 数组)。`mode_build` 不用这条路径(它是流式的),所以**内存不够时可以直接进第 3 步**。
|
| 124 |
+
|
| 125 |
+
若确实想跑完 `check`:需要 ≥16 GiB 空闲内存。
|
| 126 |
+
|
| 127 |
+
---
|
| 128 |
+
|
| 129 |
+
## 3. 打包
|
| 130 |
+
|
| 131 |
+
```bash
|
| 132 |
+
python -u tools/pack.py build bonsai2_27b_swift_pq2.ninfer \
|
| 133 |
+
--gguf Swift-Bonsai-2-PQ2_0.gguf \
|
| 134 |
+
--template qwen3_8_27b_huihui_abliterated.ninfer
|
| 135 |
+
```
|
| 136 |
+
|
| 137 |
+
PTQ1 档同理,把两个路径换成 PTQ1 的即可。
|
| 138 |
+
|
| 139 |
+
**期望输出:**
|
| 140 |
+
|
| 141 |
+
```
|
| 142 |
+
wrote bonsai2_27b_swift_pq2.ninfer
|
| 143 |
+
total file : 8,306,927,628 B = 7.736 GiB
|
| 144 |
+
text part produced: 7,189,880,844 B = 6.696 GiB
|
| 145 |
+
borrowed payloads : 1,116,856,537 B = 1.040 GiB {'frontend': 6, 'text': 2, 'mtp': 12, 'vision': 333}
|
| 146 |
+
objects : 1126
|
| 147 |
+
```
|
| 148 |
+
|
| 149 |
+
PTQ1 档应��得到:
|
| 150 |
+
|
| 151 |
+
```
|
| 152 |
+
total file : 7,047,407,628 B = 6.563 GiB
|
| 153 |
+
text part produced: 5,930,360,844 B = 5.523 GiB
|
| 154 |
+
```
|
| 155 |
+
|
| 156 |
+
**这两个数字是硬判据** —— 任何偏差都说明源或模板拿错了。
|
| 157 |
+
|
| 158 |
+
---
|
| 159 |
+
|
| 160 |
+
## 4. 核对产出
|
| 161 |
+
|
| 162 |
+
```bash
|
| 163 |
+
python - <<'PY'
|
| 164 |
+
import json, collections, os
|
| 165 |
+
p = "bonsai2_27b_swift_pq2.ninfer"
|
| 166 |
+
raw = open(p,"rb").read(16<<20)
|
| 167 |
+
obj,_ = json.JSONDecoder().raw_decode(raw[16:].decode("utf-8","replace"))
|
| 168 |
+
objs = obj["objects"]
|
| 169 |
+
print("identity:", obj["identity"])
|
| 170 |
+
print("objects :", len(objs))
|
| 171 |
+
print("formats :", dict(sorted(collections.Counter(
|
| 172 |
+
o.get("format") for o in objs if o.get("kind")=="tensor").items())))
|
| 173 |
+
print("layouts :", dict(collections.Counter(
|
| 174 |
+
o.get("layout") for o in objs if o.get("kind")=="tensor")))
|
| 175 |
+
names = {o["name"] for o in objs}
|
| 176 |
+
print("hadamard_signs :", "text/hadamard_signs" in names)
|
| 177 |
+
print("hadamard_widths:", "text/hadamard_widths" in names)
|
| 178 |
+
PY
|
| 179 |
+
```
|
| 180 |
+
|
| 181 |
+
**期望(PQ2 档):**
|
| 182 |
+
|
| 183 |
+
```
|
| 184 |
+
identity: {'model_id': 'qwen3.8-27b', 'weights_id': 'groupwise-int'}
|
| 185 |
+
objects : 1126
|
| 186 |
+
formats : {'BF16': 582, 'PQ2_0_G128': 322, 'FP32': 97, 'Q4G64_F16S': 55,
|
| 187 |
+
'Q5G64_F16S': 54, 'W8G32_F16S': 7, 'I32': 2, 'Q6G64_F16S': 1}
|
| 188 |
+
layouts : {'row-split-k128-v1': 439, 'contiguous-le-v1': 681}
|
| 189 |
+
hadamard_signs : True
|
| 190 |
+
hadamard_widths: True
|
| 191 |
+
```
|
| 192 |
+
|
| 193 |
+
`objects` 比模板多 2 —— 就是那两个 hadamard 对象。**模板里没有它们,产出里有**,
|
| 194 |
+
这正是三元制品需要的。
|
| 195 |
+
|
| 196 |
+
---
|
| 197 |
+
|
| 198 |
+
## 5. 打包器在做什么(逐步)
|
| 199 |
+
|
| 200 |
+
理解这一步,才能在出错时判断问题在哪。
|
| 201 |
+
|
| 202 |
+
### 5.1 三元码原样搬运
|
| 203 |
+
|
| 204 |
+
GGML 的三元 block 有两个成员(`PQ2_0` 是 `{qs}` base + fp16 scale;`PTQ1_0` 多一个
|
| 205 |
+
`{qh}` high 平面),`.ninfer` 的 `row-split-k128-v1` 布局恰好是**同样三个平面**。
|
| 206 |
+
所以码字**逐字节搬**,不经过浮点。
|
| 207 |
+
|
| 208 |
+
**这条极其重要**:解量化再重量化会引入二次损失,而且体积红利会被吃掉。
|
| 209 |
+
|
| 210 |
+
### 5.2 撤销 llama.cpp exporter 的约定
|
| 211 |
+
|
| 212 |
+
源 GGUF 是 llama.cpp 生态导出的,带着三处约定,产物必须还原:
|
| 213 |
+
|
| 214 |
+
```
|
| 215 |
+
GDN value heads tiled 顺序 -> grouped 顺序
|
| 216 |
+
按头粒度施加:perm48(head)*128 + inner
|
| 217 |
+
⚠️ 直接套 perm48(row) 是非双射,会静默产生重复行 + 丢失行
|
| 218 |
+
(这类 bug 守恒一切可数之物:尺寸/行数/字节/往返无损全绿)
|
| 219 |
+
|
| 220 |
+
零中心 norms 1 + w -> w
|
| 221 |
+
⚠️ 唯独 gdn/norm(ssm_norm)原样 —— 它的 raw 已经 ≈ +1
|
| 222 |
+
|
| 223 |
+
ssm_a -exp(A_log) -> A_log
|
| 224 |
+
```
|
| 225 |
+
|
| 226 |
+
### 5.3 补两个 hadamard 对象
|
| 227 |
+
|
| 228 |
+
源 GGUF 用 `prism.hadamard.sign_values` / `sign_widths` 承载旋转基的符号向量;
|
| 229 |
+
`.ninfer` 用 `text/hadamard_signs`(28,672 个 fp32)/ `text/hadamard_widths`(3 个 int32)。
|
| 230 |
+
打包器把前者翻成后者,并给每个被旋转的投影挂上 `hadamard_signs` Use 辅助。
|
| 231 |
+
|
| 232 |
+
### 5.4 借 payload
|
| 233 |
+
|
| 234 |
+
```
|
| 235 |
+
vision 333 个 —— 源 GGUF 里没有(Bonsai 的视觉塔是单独的 mmproj.gguf)
|
| 236 |
+
mtp 12 个 —— 源 GGUF 里没有(851 个张量里零个 blk.64.*)
|
| 237 |
+
frontend 6 个 —— tokenizer 等
|
| 238 |
+
draft_head 2 个 —— 频次短名单,同一 tokenizer 即同一名单
|
| 239 |
+
共 1,116,856,537 B
|
| 240 |
+
```
|
| 241 |
+
|
| 242 |
+
**这几块是全生态共用的官方件**,从模板借是安全的(MTP 头与目标点积 ≥0.99966、七个 norm 逐字节相同)。
|
| 243 |
+
|
| 244 |
+
---
|
| 245 |
+
|
| 246 |
+
## 6. 验证方法学(**这一节比上面的步骤更值钱**)
|
| 247 |
+
|
| 248 |
+
「搬运无损」类判据(源字节 == payload、往返解码一致)**对偏移错误完全盲** ——
|
| 249 |
+
读错的同一批字节原样进原样出,照样全绿。
|
| 250 |
+
|
| 251 |
+
真正能证伪的判据:
|
| 252 |
+
|
| 253 |
+
| # | 判据 | 能抓什么 |
|
| 254 |
+
|---|---|---|
|
| 255 |
+
| 1 | **分布特征自证**(scale 中位数 / 全正 / `zero_share`) | 载荷整体偏移、平面错位 |
|
| 256 |
+
| 2 | **跨实现互验**(两个独立解码器解同一份数据) | 单一实现的系统性误读 |
|
| 257 |
+
| 3 | **负控必须存在**(假格式名必须被拒、错尺寸必须被拒) | "判据太松"导致的假通过 |
|
| 258 |
+
| 4 | **行级指纹比「多重集」+ 直接比对** | 行置换 / 重复 / 丢失(尺寸守恒那类) |
|
| 259 |
+
| 5 | **T>1 用例 + 引擎侧验证** | 布局 / 步长类错误(T=1 两种排布重合,测不出任何东西) |
|
| 260 |
+
| 6 | **端到端数值口径用 PPL** | 采样温度 / 模板带来的错觉 |
|
| 261 |
+
|
| 262 |
+
### ⚠️ PPL 的一个常见误用:拿别处的绝对值来比
|
| 263 |
+
|
| 264 |
+
**PPL 跟语料强相关。** 上游文档给过一组参照值(`512/256 → 6.448742`、`32/16 → 26.049634`、
|
| 265 |
+
`8/4 → 121.157720`),**那是另一份语料上的数**,和你自己量到的值**不可直接比较**。
|
| 266 |
+
|
| 267 |
+
**只有同一语料下两个模型的差值才有意义。** 本仓的实测(`pplab-text`,`int8` KV,RTX 3060 12G):
|
| 268 |
+
|
| 269 |
+
| 窗口/步长 | Swift | base | 差 |
|
| 270 |
+
|---|---:|---:|---:|
|
| 271 |
+
| 512 / 256 | 8.073669 | 8.079208 | −0.07% |
|
| 272 |
+
| 32 / 16 | 26.962015 | 26.972171 | −0.04% |
|
| 273 |
+
| 8 / 4 | 148.3656 | 148.3863 | −0.01% |
|
| 274 |
+
|
| 275 |
+
**你复现时应该看到同样的"差值方向",而不是同样的绝对值。**
|
| 276 |
+
|
| 277 |
+
**`zero_share` 精确落到 0.3278 是 `PQ2_0` 的格式指纹**(理论零码占比 32.776%),不是"大概的数"。
|
| 278 |
+
本仓的两个制品都过了这一条。
|
| 279 |
+
|
| 280 |
+
### 性能测量的坑(会直接误导判断)
|
| 281 |
+
|
| 282 |
+
```
|
| 283 |
+
micro-benchmark 不 flush L2 会高估;flush 过头会报出超物理上限的数
|
| 284 |
+
nsys 默认不追踪 CUDA graph replay 内的 kernel ⇒ --cuda-graph-trace=node
|
| 285 |
+
内核"少干一半活"会伪装成提速 ⇒ 必须先用 rel_l2 校验
|
| 286 |
+
kernel launch 失败伪装成"快得离谱 + 全零" ⇒ launch 后查 cudaGetLastError()
|
| 287 |
+
idle 时钟让短基准严重失真 ⇒ 短基准不可信
|
| 288 |
+
报数不带任务和生成长度 ⇒ 轮耗时才是任务无关量
|
| 289 |
+
```
|
| 290 |
+
|
| 291 |
+
---
|
| 292 |
+
|
| 293 |
+
## 7. 已知的失败与对策
|
| 294 |
+
|
| 295 |
+
| 症状 | 真因 | 对策 |
|
| 296 |
+
|---|---|---|
|
| 297 |
+
| `template has no object X to borrow` | 模板与映射表 schema 不一致 | 换 `groupwise-int` 的模板,别用 `nvfp4` |
|
| 298 |
+
| `unmapped gdn object` / 直接中止 | 模板是 `nvfp4` 打包(投影被融合) | 同上 |
|
| 299 |
+
| `Unable to allocate 9.47 GiB`(`check` 模式) | 全量反量化的 float64 数组 | 内存 ≥16 GiB,或直接跑 `build` |
|
| 300 |
+
| 产出尺寸与期望不符 | 源或模板拿错 | 对 sha256 |
|
| 301 |
+
| 引擎拒绝装载产物 | 方言不对(`t2_g128_fp16` vs `PQ2_0_G128`) | 见根目录 README 的兼容性表 |
|
| 302 |
+
|
| 303 |
+
---
|
| 304 |
+
|
| 305 |
+
## 8. 本仓实际使用的命令与结果
|
| 306 |
+
|
| 307 |
+
```
|
| 308 |
+
packer tools/pack.py(NINFER_ROOT 常量重定向到本地 checkout,无功能修改)
|
| 309 |
+
Python 3.11.5
|
| 310 |
+
GPU 未使用(纯 CPU)
|
| 311 |
+
PQ2 档 4 分钟 → 8,306,927,628 B / 7.736 GiB
|
| 312 |
+
PTQ1 档 4 分钟 → 7,047,407,628 B / 6.563 GiB
|
| 313 |
+
```
|
| 314 |
+
|
| 315 |
+
**没有任何验证、校验、几何检查或往返证明被禁用、放宽或绕过。**
|
| 316 |
+
`check` 模式在源 GGUF 上完整跑过(见第 2 节)。
|
SHA256SUMS
ADDED
|
@@ -0,0 +1,2 @@
|
|
|
|
|
|
|
|
|
|
| 1 |
+
cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0 bonsai2_27b_swift_pq2.ninfer
|
| 2 |
+
cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae bonsai2_27b_swift_ptq1.ninfer
|
artifact-manifest.json
ADDED
|
@@ -0,0 +1,91 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"schema": "ninfer.artifact-manifest",
|
| 3 |
+
"schema_version": 1,
|
| 4 |
+
"release": "Swift-Bonsai-2 27B — NInfer ternary artifacts",
|
| 5 |
+
"generated": "2026-09-28T03:12:45",
|
| 6 |
+
"license": "Apache-2.0",
|
| 7 |
+
"provenance": {
|
| 8 |
+
"base_ternary": {
|
| 9 |
+
"repo": "prism-ml/Ternary-Bonsai-2-27B-gguf",
|
| 10 |
+
"license": "apache-2.0"
|
| 11 |
+
},
|
| 12 |
+
"weight_source": {
|
| 13 |
+
"repo": "ukisai/Swift-Bonsai-2-GGUF",
|
| 14 |
+
"license": "apache-2.0"
|
| 15 |
+
},
|
| 16 |
+
"geometry_base": {
|
| 17 |
+
"repo": "Qwen/Qwen3.8-27B",
|
| 18 |
+
"license": "apache-2.0"
|
| 19 |
+
},
|
| 20 |
+
"packer": {
|
| 21 |
+
"repo": "shensanshu/ninfer-ada-ternary",
|
| 22 |
+
"license": "Apache-2.0",
|
| 23 |
+
"path": "tools/pack.py"
|
| 24 |
+
},
|
| 25 |
+
"template": {
|
| 26 |
+
"repo": "Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer",
|
| 27 |
+
"license": "Apache-2.0",
|
| 28 |
+
"used_for": "object skeleton + vision/mtp/frontend/draft_head payloads only"
|
| 29 |
+
}
|
| 30 |
+
},
|
| 31 |
+
"artifacts": [
|
| 32 |
+
{
|
| 33 |
+
"filename": "bonsai2_27b_swift_pq2.ninfer",
|
| 34 |
+
"bytes": 8306927628,
|
| 35 |
+
"sha256": "cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0",
|
| 36 |
+
"identity": {
|
| 37 |
+
"model_id": "qwen3.8-27b",
|
| 38 |
+
"weights_id": "groupwise-int"
|
| 39 |
+
},
|
| 40 |
+
"container_version": 2,
|
| 41 |
+
"objects": 1126,
|
| 42 |
+
"tensors": 1120,
|
| 43 |
+
"resources": 6,
|
| 44 |
+
"formats": {
|
| 45 |
+
"BF16": 582,
|
| 46 |
+
"FP32": 97,
|
| 47 |
+
"I32": 2,
|
| 48 |
+
"PQ2_0_G128": 322,
|
| 49 |
+
"Q4G64_F16S": 55,
|
| 50 |
+
"Q5G64_F16S": 54,
|
| 51 |
+
"Q6G64_F16S": 1,
|
| 52 |
+
"W8G32_F16S": 7
|
| 53 |
+
},
|
| 54 |
+
"layouts": {
|
| 55 |
+
"contiguous-le-v1": 681,
|
| 56 |
+
"row-split-k128-v1": 439
|
| 57 |
+
},
|
| 58 |
+
"ternary_format": "PQ2_0_G128",
|
| 59 |
+
"requires_engine_dialect": "Ambolio/ninfer-4090-windows lineage (v1.0.6 / v1.0.8)"
|
| 60 |
+
},
|
| 61 |
+
{
|
| 62 |
+
"filename": "bonsai2_27b_swift_ptq1.ninfer",
|
| 63 |
+
"bytes": 7047407628,
|
| 64 |
+
"sha256": "cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae",
|
| 65 |
+
"identity": {
|
| 66 |
+
"model_id": "qwen3.8-27b",
|
| 67 |
+
"weights_id": "groupwise-int"
|
| 68 |
+
},
|
| 69 |
+
"container_version": 2,
|
| 70 |
+
"objects": 1126,
|
| 71 |
+
"tensors": 1120,
|
| 72 |
+
"resources": 6,
|
| 73 |
+
"formats": {
|
| 74 |
+
"BF16": 582,
|
| 75 |
+
"FP32": 97,
|
| 76 |
+
"I32": 2,
|
| 77 |
+
"PTQ1_0_G128": 322,
|
| 78 |
+
"Q4G64_F16S": 55,
|
| 79 |
+
"Q5G64_F16S": 54,
|
| 80 |
+
"Q6G64_F16S": 1,
|
| 81 |
+
"W8G32_F16S": 7
|
| 82 |
+
},
|
| 83 |
+
"layouts": {
|
| 84 |
+
"contiguous-le-v1": 681,
|
| 85 |
+
"row-split-k128-v1": 439
|
| 86 |
+
},
|
| 87 |
+
"ternary_format": "PTQ1_0_G128",
|
| 88 |
+
"requires_engine_dialect": "Ambolio/ninfer-4090-windows lineage (v1.0.6 / v1.0.8)"
|
| 89 |
+
}
|
| 90 |
+
]
|
| 91 |
+
}
|
bonsai2_27b_swift_pq2.ninfer
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0
|
| 3 |
+
size 8306927628
|
bonsai2_27b_swift_ptq1.ninfer
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae
|
| 3 |
+
size 7047407628
|
release.conversion.json
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"identity": {
|
| 3 |
+
"model_id": "qwen3.8-27b",
|
| 4 |
+
"weights_id": "groupwise-int"
|
| 5 |
+
},
|
| 6 |
+
"purpose": "Swift-Bonsai-2 27B ternary artifact, PQ2 + PTQ1 tiers, NInfer v2 container",
|
| 7 |
+
"packer": {
|
| 8 |
+
"name": "tools/pack.py",
|
| 9 |
+
"origin": "shensanshu/ninfer-ada-ternary (Apache-2.0)",
|
| 10 |
+
"change": "only the NINFER_ROOT default path constant was redirected; no functional change",
|
| 11 |
+
"verification_mode": "check ran to completion: pad=0 for all 8 shape combinations, bytes_equal and decode_equal true on every sampled tensor"
|
| 12 |
+
},
|
| 13 |
+
"sources": {
|
| 14 |
+
"gguf_pq2": {
|
| 15 |
+
"file": "Swift-Bonsai-2-PQ2_0.gguf",
|
| 16 |
+
"bytes": 7206168928,
|
| 17 |
+
"sha256": "5912bb739217cf25b4283a098c7b893d9baff36bdf2e72d3dbf4cb999d5633d5",
|
| 18 |
+
"from": "ukisai/Swift-Bonsai-2-GGUF"
|
| 19 |
+
},
|
| 20 |
+
"gguf_ptq1": {
|
| 21 |
+
"file": "Swift-Bonsai-2-PTQ1_0.gguf",
|
| 22 |
+
"bytes": 5946648960,
|
| 23 |
+
"sha256": "de33620b60eaf63e96449b478eb87abe9538507e3ed939da944c1ae15fe1ffc6",
|
| 24 |
+
"from": "ukisai/Swift-Bonsai-2-GGUF"
|
| 25 |
+
},
|
| 26 |
+
"template": {
|
| 27 |
+
"file": "qwen3_8_27b_huihui_abliterated.ninfer",
|
| 28 |
+
"bytes": 18210531328,
|
| 29 |
+
"sha256": "8c9f9d67a07ac97506978f6db6695d8074f78dec0fb80c4a85a8fb6fbedd7f03",
|
| 30 |
+
"from": "Barding-Defense/Qwen3.8-27B-huihui-abliterated-groupwise-int-NInfer",
|
| 31 |
+
"note": "used ONLY as skeleton/manifest + donor of vision/mtp/frontend/draft_head payloads; its text weights are discarded"
|
| 32 |
+
}
|
| 33 |
+
},
|
| 34 |
+
"outputs": {
|
| 35 |
+
"bonsai2_27b_swift_pq2.ninfer": {
|
| 36 |
+
"bytes": 8306927628,
|
| 37 |
+
"sha256": "cc54be3800099ada67165ad352450be83d28e8c4d423be9cae573a7b6e6350a0",
|
| 38 |
+
"note": "PQ2 tier, 2-bit, from gguf_pq2"
|
| 39 |
+
},
|
| 40 |
+
"bonsai2_27b_swift_ptq1.ninfer": {
|
| 41 |
+
"bytes": 7047407628,
|
| 42 |
+
"sha256": "cc9e890728ea7357b1d8a0797a4accdc6ca143031471a63e49c9cf6c8314b6ae",
|
| 43 |
+
"note": "PTQ1 tier, 1.75-bit, from gguf_ptq1"
|
| 44 |
+
}
|
| 45 |
+
},
|
| 46 |
+
"toolchain": {
|
| 47 |
+
"python": "3.11.5",
|
| 48 |
+
"numpy": "installed",
|
| 49 |
+
"torch": "CPU build",
|
| 50 |
+
"gpu_used": false
|
| 51 |
+
},
|
| 52 |
+
"generated": "2026-09-28T03:10:53"
|
| 53 |
+
}
|
tools/MAPPING.json
ADDED
|
@@ -0,0 +1,126 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_doc": "Bonsai 2 27B GGUF (arch qwen35) -> ninfer artifact object mapping. Every rule below was established EMPIRICALLY against a groupwise-int qwen3.8-27b artifact of the same model family, except where marked INFERRED. Rotated (hadamard-folded) matrices cannot be correlated - only byte accounting applies to them.",
|
| 3 |
+
"_verified_by": "probe_conventions.py, solve_gdn_order.py (see <WORKSPACE>\\ternary-pack\\)",
|
| 4 |
+
"_layers": {
|
| 5 |
+
"total": 64,
|
| 6 |
+
"gdn_linear_attention": [0,1,2,4,5,6,8,9,10,12,13,14,16,17,18,20,21,22,24,25,26,28,29,30,32,33,34,36,37,38,40,41,42,44,45,46,48,49,50,52,53,54,56,57,58,60,61,62],
|
| 7 |
+
"full_attention": [3,7,11,15,19,23,27,31,35,39,43,47,51,55,59,63],
|
| 8 |
+
"note": "derived: a layer is full-attention iff blk.{l}.attn_q.weight exists in the GGUF"
|
| 9 |
+
},
|
| 10 |
+
"_constants": {
|
| 11 |
+
"hidden": 5120,
|
| 12 |
+
"q_heads": 24, "kv_heads": 4, "head_dim": 256,
|
| 13 |
+
"num_k_heads": 16, "num_v_heads": 48, "v_per_k": 3, "head_k_dim": 128, "head_v_dim": 128,
|
| 14 |
+
"gdn_qkv_channels": 10240, "gdn_qk_channels": 4096, "gdn_v_channels": 6144
|
| 15 |
+
},
|
| 16 |
+
|
| 17 |
+
"_rules": [
|
| 18 |
+
{
|
| 19 |
+
"id": "norm_shift",
|
| 20 |
+
"applies_to": "every BF16/FP32 object that came from a GGUF tensor ending in '.norm.weight'",
|
| 21 |
+
"transform": "value = gguf - 1.0",
|
| 22 |
+
"exception": "text/layers/{l}/gdn/norm <- blk.{l}.ssm_norm.weight is copied RAW (no -1)",
|
| 23 |
+
"evidence": "cos(artifact, gguf-1) = +0.99997 .. +0.99991 for input_norm / post_attention_norm / query_norm / key_norm / final_norm; cos(artifact, gguf) = +0.99961 for gdn/norm",
|
| 24 |
+
"source": "llama.cpp conversion/qwen.py:394 -- `name.endswith(\"norm.weight\") and not name.endswith(\"linear_attn.norm.weight\")` gets +1 on the way OUT to GGUF, so ninfer (HF-form) needs -1 back, except for linear_attn.norm"
|
| 25 |
+
},
|
| 26 |
+
{
|
| 27 |
+
"id": "gdn_v_tiled_to_grouped",
|
| 28 |
+
"applies_to": "the 48-row (num_v_heads) axis of GDN tensors",
|
| 29 |
+
"transform": "t = t.reshape(3,16,*rest).transpose(1,0,*range(2,ndim+1)).reshape(shape)",
|
| 30 |
+
"direction": "GGUF stores TILED, ninfer wants GROUPED",
|
| 31 |
+
"evidence": "48x48 row matching gives a bijection with cos 0.990..0.9998; cos(A, tiled_to_grouped(B)) = +0.99587 vs +0.14814 for the other direction",
|
| 32 |
+
"source": "conversion/qwen.py:446 _LinearAttentionVReorderBase._reorder_v_heads (HF grouped -> ggml tiled)"
|
| 33 |
+
},
|
| 34 |
+
{
|
| 35 |
+
"id": "gdn_a_log",
|
| 36 |
+
"applies_to": "text/layers/{l}/gdn/a_log <- blk.{l}.ssm_a",
|
| 37 |
+
"transform": "tiled_to_grouped(log(-gguf))",
|
| 38 |
+
"evidence": "cos(sorted(artifact), sorted(log(-gguf))) = +1.00000 exactly; raw gives +0.796",
|
| 39 |
+
"source": "conversion/qwen.py:389 -- `if name.endswith(\".A_log\"): data_torch = -torch.exp(data_torch)`"
|
| 40 |
+
},
|
| 41 |
+
{
|
| 42 |
+
"id": "gdn_conv1d",
|
| 43 |
+
"applies_to": "text/layers/{l}/gdn/convolution (4,10240) <- blk.{l}.ssm_conv1d.weight ne=(4,10240)",
|
| 44 |
+
"transform": "read as (10240,4); leave channels [0:4096] alone; apply tiled_to_grouped over the 48-head axis of channels [4096:10240]; then transpose -> shape (4,10240)",
|
| 45 |
+
"evidence": "cos = +0.99504; naively keeping the order gives only +0.72253",
|
| 46 |
+
"source": "conversion/qwen.py:608-615 (.conv1d branch reorders only the V channels)"
|
| 47 |
+
},
|
| 48 |
+
{
|
| 49 |
+
"id": "attn_q_per_head_interleave",
|
| 50 |
+
"applies_to": "blk.{l}.attn_q.weight (N=12288) for full-attention layers",
|
| 51 |
+
"transform": "view as 48 chunks of 256 rows; chunks 0,2,...,46 are the 24 query heads in order; chunks 1,3,...,47 are the 24 output-gate heads in order",
|
| 52 |
+
"evidence": "INFERRED for main layers from the MTP layer, where it was VERIFIED against the artifact at cos +0.99982 per chunk (24/24 and 24/24 chunks exact). The MTP layer is a full-attention layer with identical shapes.",
|
| 53 |
+
"target_use": "attention/query_key = concat(query(6144), attn_k(1024)); attention/gate_value = concat(gate(6144), attn_v(1024))"
|
| 54 |
+
},
|
| 55 |
+
{
|
| 56 |
+
"id": "gdn_value_z",
|
| 57 |
+
"transform": "value_z = concat(tiled_to_grouped_heads(attn_qkv[4096:10240]), tiled_to_grouped_heads(attn_gate[0:6144])). The 48-head permutation MUST be applied at HEAD granularity, because this tensor's row axis is 48 heads x 128 rows: src_row = perm48(row // 128) * 128 + row % 128",
|
| 58 |
+
"evidence": "RESOLVED 2026-09-19 by engine test (was INFERRED). ssm_conv1d's V channels are MEASURED as tiled and its rule already permutes at HEAD granularity (cos +0.99504); alpha/beta (48 rows = 48 heads, so row granularity IS head granularity) use the same permutation and measure +1.00000 / +0.99587.",
|
| 59 |
+
"bug_history": "The first pack applied perm48() straight to the 6144-row index. That is NOT a bijection over rows (perm48(1) == perm48(48) == 16), so 4064 of 6144 source rows were dropped and 2048 row fingerprints repeated -- while every size, row count and byte total stayed EXACTLY right, which is why the size reverse-check, the src==payload round trip, the byte-for-byte reconcile and the 12288-row cross-check all stayed green over it. Engine symptom: fluent-looking garbage, PPL 1,128,420 (worse than uniform over a 248,320-token vocab). Fixed by the perm_row() helper in pack.py; the v2 artifact verifies 6144/6144 distinct rows, 0 duplicated, 0 missing.",
|
| 60 |
+
"acceptance_check": "<BUILD_ROOT>\\check_row_order.py <artifact.ninfer> <Ternary-Bonsai-2-27B-PQ2_0.gguf> -- must report 'rows DUPLICATED: 0' and 'source rows MISSING: 0' AND 'documented rule holds row-by-row: True'. NOTE: a set-membership test is NOT sufficient (every duplicated row still finds a source); compare MULTISETS and compare artifact[a] directly against the rule's named source a, never by searching for a matching fingerprint."
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"id": "gdn_output_raw",
|
| 64 |
+
"applies_to": "text/layers/{l}/gdn/output (5120,6144) <- blk.{l}.ssm_out.weight",
|
| 65 |
+
"transform": "none",
|
| 66 |
+
"evidence": "conversion/qwen.py:617-626 -- a hadamard-FOLDED out_proj keeps the training (grouped) column order and the runtime permutes the activation instead; Bonsai's metadata carries prism.hadamard.gdn_v_grouped = 1, confirming the folded path"
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"id": "ternary_bytes_verbatim",
|
| 70 |
+
"applies_to": "all 402 PQ2_0 / PTQ1_0 matrices",
|
| 71 |
+
"transform": "move the quantized codes byte-for-byte; do NOT dequantize and requantize",
|
| 72 |
+
"note": "these weights live in the rotated basis W'; the activation-side transform is M2's job, not the packer's"
|
| 73 |
+
}
|
| 74 |
+
],
|
| 75 |
+
|
| 76 |
+
"text_globals": {
|
| 77 |
+
"text/token_embedding": {"src": "token_embd.weight", "shape": [248320, 5120], "fmt": "PQ2_0_G128", "transform": "none (stored as W'; runtime inverse-applies after lookup)"},
|
| 78 |
+
"text/output_head": {"src": "output.weight", "shape": [248320, 5120], "fmt": "PQ2_0_G128", "transform": "none"},
|
| 79 |
+
"text/final_norm": {"src": "output_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
|
| 80 |
+
"text/draft_head": {"src": "GENERATED", "shape": [131072, 5120], "fmt": "Q4G64_F16S", "note": "frequency shortlist, NOT a model weight: tools/convert/qwen3_6/common/draft_head.py materializes it from tools/freq_corpus/fixtures/ranking/ranking.train.counts.i64 + the tokenizer. Regenerate, do not copy."},
|
| 81 |
+
"text/draft_head_token_ids": {"src": "GENERATED", "shape": [131072], "fmt": "I32", "note": "same generator"}
|
| 82 |
+
},
|
| 83 |
+
|
| 84 |
+
"per_layer_gdn": {
|
| 85 |
+
"text/layers/{l}/input_norm": {"src": "blk.{l}.attn_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
|
| 86 |
+
"text/layers/{l}/post_attention_norm": {"src": "blk.{l}.post_attention_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
|
| 87 |
+
"text/layers/{l}/mlp/gate_up": {"src": "concat(blk.{l}.ffn_gate.weight, blk.{l}.ffn_up.weight) along rows", "shape": [34816, 5120], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim; concat order VERIFIED at cos +0.99982 on the MTP analogue"},
|
| 88 |
+
"text/layers/{l}/mlp/down": {"src": "blk.{l}.ffn_down.weight", "shape": [5120, 17408], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
|
| 89 |
+
"text/layers/{l}/gdn/query_key": {"src": "blk.{l}.attn_qkv.weight rows 0:4096", "shape": [4096, 5120], "fmt": "TERNARY", "transform": "none (q 2048 + k 2048, no V tiling on the k-head axis)"},
|
| 90 |
+
"text/layers/{l}/gdn/value_z": {"src": "rows 4096:10240 of blk.{l}.attn_qkv.weight + blk.{l}.attn_gate.weight", "shape": [12288, 5120], "fmt": "TERNARY", "transform": "gdn_value_z"},
|
| 91 |
+
"text/layers/{l}/gdn/output": {"src": "blk.{l}.ssm_out.weight", "shape": [5120, 6144], "fmt": "TERNARY", "transform": "none"},
|
| 92 |
+
"text/layers/{l}/gdn/convolution": {"src": "blk.{l}.ssm_conv1d.weight", "shape": [4, 10240], "fmt": "BF16", "transform": "gdn_conv1d"},
|
| 93 |
+
"text/layers/{l}/gdn/norm": {"src": "blk.{l}.ssm_norm.weight", "shape": [128], "fmt": "BF16", "transform": "RAW (no -1!)"},
|
| 94 |
+
"text/layers/{l}/gdn/a_projection": {"src": "blk.{l}.ssm_alpha.weight", "shape": [48, 5120], "fmt": "BF16", "transform": "gdn_v_tiled_to_grouped"},
|
| 95 |
+
"text/layers/{l}/gdn/b_projection": {"src": "blk.{l}.ssm_beta.weight", "shape": [48, 5120], "fmt": "BF16", "transform": "gdn_v_tiled_to_grouped"},
|
| 96 |
+
"text/layers/{l}/gdn/a_log": {"src": "blk.{l}.ssm_a", "shape": [48], "fmt": "FP32", "transform": "gdn_a_log"},
|
| 97 |
+
"text/layers/{l}/gdn/dt_bias": {"src": "blk.{l}.ssm_dt.bias", "shape": [48], "fmt": "FP32", "transform": "gdn_v_tiled_to_grouped (pure permutation)"}
|
| 98 |
+
},
|
| 99 |
+
|
| 100 |
+
"per_layer_attention": {
|
| 101 |
+
"text/layers/{l}/input_norm": {"src": "blk.{l}.attn_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
|
| 102 |
+
"text/layers/{l}/post_attention_norm": {"src": "blk.{l}.post_attention_norm.weight", "shape": [5120], "fmt": "BF16", "transform": "raw - 1"},
|
| 103 |
+
"text/layers/{l}/mlp/gate_up": {"src": "concat(blk.{l}.ffn_gate.weight, blk.{l}.ffn_up.weight)", "shape": [34816, 5120], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
|
| 104 |
+
"text/layers/{l}/mlp/down": {"src": "blk.{l}.ffn_down.weight", "shape": [5120, 17408], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
|
| 105 |
+
"text/layers/{l}/attention/query_key": {"src": "attn_q (de-interleaved) 6144 rows + blk.{l}.attn_k.weight 1024 rows", "shape": [7168, 5120], "fmt": "TERNARY", "transform": "attn_q_per_head_interleave"},
|
| 106 |
+
"text/layers/{l}/attention/gate_value": {"src": "attn_q (odd chunks) 6144 rows + blk.{l}.attn_v.weight 1024 rows", "shape": [7168, 5120], "fmt": "TERNARY", "transform": "attn_q_per_head_interleave"},
|
| 107 |
+
"text/layers/{l}/attention/output": {"src": "blk.{l}.attn_output.weight", "shape": [5120, 6144], "fmt": "TERNARY", "transform": "ternary_bytes_verbatim"},
|
| 108 |
+
"text/layers/{l}/attention/query_norm": {"src": "blk.{l}.attn_q_norm.weight", "shape": [256], "fmt": "BF16", "transform": "raw - 1"},
|
| 109 |
+
"text/layers/{l}/attention/key_norm": {"src": "blk.{l}.attn_k_norm.weight", "shape": [256], "fmt": "BF16", "transform": "raw - 1"}
|
| 110 |
+
},
|
| 111 |
+
|
| 112 |
+
"_counts": {
|
| 113 |
+
"artifact_objects_total": 1124,
|
| 114 |
+
"text": 773, "mtp": 12, "vision": 333, "frontend_resources": 6,
|
| 115 |
+
"gguf_tensors_total": 851,
|
| 116 |
+
"gguf_breakdown": "48 GDN layers x 14 + 16 attention layers x 11 + 3 globals (token_embd, output, output_norm) = 851",
|
| 117 |
+
"note": "text/draft_head + draft_head_token_ids have NO GGUF source (generated); vision has no source in this GGUF either (Bonsai ships a separate mmproj)"
|
| 118 |
+
},
|
| 119 |
+
|
| 120 |
+
"_unresolved": [
|
| 121 |
+
"vision/* (333 objects): Bonsai's vision tower lives in Ternary-Bonsai-2-27B-mmproj-Q8_0.gguf (Q8_0), while the template expects Q4G64/Q5G64/Q6G64/BF16. Either convert the mmproj or borrow the template's vision payloads (same official vision tower) - needs a decision.",
|
| 122 |
+
"text/draft_head + draft_head_token_ids: run the frequency-shortlist generator (needs tools/freq_corpus fixture present in the fork).",
|
| 123 |
+
"all 402 ternary matrices are in the rotated basis, so NONE of them can be verified by correlation; only byte accounting, decode round-trip equality and the engine's load test apply.",
|
| 124 |
+
"gdn/value_z's tiled->grouped choice is inferred (see rule) - single line to flip if the engine disagrees."
|
| 125 |
+
]
|
| 126 |
+
}
|
tools/README.md
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# tools/ — 打包器与验证脚本
|
| 2 |
+
|
| 3 |
+
**许可:Apache-2.0。** 来源见根目录 `NOTICE`。
|
| 4 |
+
|
| 5 |
+
这些文件原样取自 [`shensanshu/ninfer-ada-ternary`](https://www.modelscope.cn/models/shensanshu/ninfer-ada-ternary)
|
| 6 |
+
(Apache-2.0),**未做功能修改**。
|
| 7 |
+
|
| 8 |
+
唯一的编辑是**脱敏**:源码顶部的两个默认路径常量(`NINFER_ROOT` 指向作者的开发机、
|
| 9 |
+
`TEMPLATE` 指向他的模板文件),以及 `MAPPING.json` 的 `_verified_by` 说明,
|
| 10 |
+
一律替换成中性占位符 **`<NINFER_ROOT>`** 与 **`<TEMPLATE>`**;
|
| 11 |
+
`verify/` 下 4 个脚本(`check_assembly` / `check_embedding` / `check_row_order` / `gemm_oracle`)的 `sys.path.insert(...)` 行做了同样处理;`oracle_rot.py` 不 import 那套模块,本来就没有这行。
|
| 12 |
+
**没有任何逻辑被改动。**
|
| 13 |
+
|
| 14 |
+
## ★ 用之前先处理这两个占位符
|
| 15 |
+
|
| 16 |
+
`pack.py` 顶部:
|
| 17 |
+
|
| 18 |
+
```python
|
| 19 |
+
NINFER_ROOT = r"<NINFER_ROOT>" # 指到含 tools/artifact 的源码 checkout
|
| 20 |
+
TEMPLATE = r"<TEMPLATE>" # 指到 groupwise-int 的模板 .ninfer
|
| 21 |
+
GGUF = r"<WORKSPACE>\Ternary-Bonsai-2-27B-PQ2_0.gguf" # 作者留的,同样要覆盖
|
| 22 |
+
```
|
| 23 |
+
|
| 24 |
+
- **`GGUF` 与 `TEMPLATE` 可以用命令行覆盖,不必改源码:**
|
| 25 |
+
```bash
|
| 26 |
+
python -u pack.py build out.ninfer --gguf 你的.gguf --template 你的模板.ninfer
|
| 27 |
+
# 或环境变量 NINFER_TERNARY_GGUF / NINFER_TERNARY_TEMPLATE
|
| 28 |
+
```
|
| 29 |
+
- **`NINFER_ROOT` 只能改源码常量** —— 它只用来 `import tools.artifact` 那一组模块。
|
| 30 |
+
需要的最小结构:
|
| 31 |
+
```
|
| 32 |
+
<NINFER_ROOT>/tools/artifact/{__init__.py, container.py, layouts.py, numeric.py}
|
| 33 |
+
```
|
| 34 |
+
来源是 `Ambolio/ninfer-4090-windows` 血统的树(或任何含这套 `tools/artifact` 的 checkout)。
|
| 35 |
+
|
| 36 |
+
---
|
| 37 |
+
| 文件 | 作用 |
|
| 38 |
+
|---|---|
|
| 39 |
+
| `pack.py` | GGUF → `.ninfer` 的三元打包器。自写 GGUF 读取器;把 402 个三元矩阵的码字**逐字节搬运**,绝不反量化再量化 |
|
| 40 |
+
| `MAPPING.json` | 逐张量映射表(对象名 / 形状 / 格式 / 规则),每条都注明证据来源 |
|
| 41 |
+
| `verify/check_row_order.py` | 行级指纹的**多重集**比对 —— 能抓行置换/重复/丢失(这类 bug 守恒一切可数之物) |
|
| 42 |
+
| `verify/check_assembly.py` | 全量装配审计 |
|
| 43 |
+
| `verify/oracle_rot.py` | Hadamard 旋转的独立 oracle |
|
| 44 |
+
| `verify/gemm_oracle.py` | GEMM 的独立 oracle |
|
| 45 |
+
| `verify/check_embedding.py` | embedding 路径核对 |
|
| 46 |
+
|
| 47 |
+
## 怎么用
|
| 48 |
+
|
| 49 |
+
```bash
|
| 50 |
+
python -u pack.py check --gguf <源.gguf> --template <模板.ninfer>
|
| 51 |
+
# 只验证,不写文件。期望:8 个形状组合 pad=0,抽样张量 bytes_equal + decode_equal 全 True
|
| 52 |
+
|
| 53 |
+
python -u pack.py build <out.ninfer> --gguf <源.gguf> --template <模板.ninfer>
|
| 54 |
+
# 真打包
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
**`--template` 是必需的。** 它不只是载荷来源 —— 打包器会**遍历模板自己的对象名表**逐个映射,
|
| 58 |
+
所以模板必须与目标制品同 schema(`identity.weights_id == "groupwise-int"`)。
|
| 59 |
+
`nvfp4` 打包的模板用不了(它把投影融合了,名字不在映射表里,会直接中止)。
|
| 60 |
+
|
| 61 |
+
**依赖:** Python 3.11+、numpy、torch(**CPU 版即可,不需要 GPU**)。
|
| 62 |
+
|
| 63 |
+
## 一个必须记住的判据陷阱
|
| 64 |
+
|
| 65 |
+
「搬运无损」类判据(源字节 == payload、往返解码一致)对**偏移错误完全盲** ——
|
| 66 |
+
读错的同一批字节原样进原样出,照样全绿。
|
| 67 |
+
|
| 68 |
+
真正有用的旁证是**分布特征**:
|
| 69 |
+
|
| 70 |
+
| 指标 | 正确读取 | 偏移错误时 |
|
| 71 |
+
|---|---|---|
|
| 72 |
+
| scale 高位字节种数 | 11~32 / 256(紧致) | 接近 250 / 256(近似均匀) |
|
| 73 |
+
| scale 是否全正 | 负 0 个 | 出现负值 |
|
| 74 |
+
| 非有限(NaN/Inf) | 0 | 出现 NaN |
|
| 75 |
+
| `zero_share` | **0.3277~0.3280** | 0.318~0.323 散乱 |
|
| 76 |
+
|
| 77 |
+
**`zero_share` 精确落到 0.3278 是 `PQ2_0` 的格式指纹**,不是"大概的数"。
|
tools/pack.py
ADDED
|
@@ -0,0 +1,958 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Bonsai 2 27B (arch qwen35) GGUF -> ninfer `.ninfer` ternary artifact (M1-D packer).
|
| 3 |
+
|
| 4 |
+
Mapping rules are NOT derived here; they come from MAPPING.json (which records the evidence
|
| 5 |
+
for each). This script owns: a seek-based GGUF reader; an INDEPENDENT reference decoder for
|
| 6 |
+
the two Prism ternary types ported from ggml/src/ggml-quants.c; byte-for-byte ternary plane
|
| 7 |
+
assembly plus its exact inverse (used as the losslessness proof); the inventory built from
|
| 8 |
+
the template artifact's own object list; and a streaming writer with borrowed payloads.
|
| 9 |
+
|
| 10 |
+
Borrowed payloads (deliberate, reported in the run summary):
|
| 11 |
+
vision/* 333 -- Bonsai ships its vision tower as a separate mmproj GGUF; the template's
|
| 12 |
+
tower is the same official one and is not read during text decode.
|
| 13 |
+
mtp/* 12 -- Bonsai's GGUF has NO MTP head (851 tensors, zero `blk.64.*`). The
|
| 14 |
+
template's head is the same official Qwen3.8-27B head: all twelve
|
| 15 |
+
tensors matched at cosine >= 0.99966 with the seven norms BIT-IDENTICAL.
|
| 16 |
+
frontend/* 6 -- tokenizer and friends.
|
| 17 |
+
text/draft_head(+_token_ids) -- frequency shortlist; same tokenizer => same shortlist.
|
| 18 |
+
|
| 19 |
+
Modes
|
| 20 |
+
check geometry + decode + byte-round-trip proofs (no writes)
|
| 21 |
+
layer3 <out> minimal artifact: frontend + sign table + one full-attention layer
|
| 22 |
+
build <out> full text model
|
| 23 |
+
|
| 24 |
+
Paths (both may be overridden; the constants in this file are only defaults)
|
| 25 |
+
--gguf <p> the Bonsai 2 27B PQ2_0 GGUF (env NINFER_TERNARY_GGUF)
|
| 26 |
+
--template <p> a **groupwise-int** qwen3.8-27b artifact
|
| 27 |
+
(env NINFER_TERNARY_TEMPLATE) -- see README FAQ
|
| 28 |
+
The template is the skeleton/manifest this packer walks, not just a donor of the
|
| 29 |
+
vision/MTP payloads: its object NAMES must match the mapping table, so the `nvfp4`
|
| 30 |
+
packing (fused gdn/a_b_projection, gdn/query_key_value_z, attention/query_key_gate_value)
|
| 31 |
+
is rejected up front with instructions on how to produce the right one.
|
| 32 |
+
|
| 33 |
+
Interpreter: <PYTHON>\\python.exe
|
| 34 |
+
"""
|
| 35 |
+
from __future__ import annotations
|
| 36 |
+
|
| 37 |
+
import json
|
| 38 |
+
import os
|
| 39 |
+
import struct
|
| 40 |
+
import sys
|
| 41 |
+
from collections import Counter
|
| 42 |
+
from pathlib import Path
|
| 43 |
+
|
| 44 |
+
import numpy as np
|
| 45 |
+
|
| 46 |
+
NINFER_ROOT = r"<NINFER_ROOT>"
|
| 47 |
+
if NINFER_ROOT not in sys.path:
|
| 48 |
+
sys.path.insert(0, NINFER_ROOT)
|
| 49 |
+
|
| 50 |
+
from tools.artifact import ( # noqa: E402
|
| 51 |
+
Artifact,
|
| 52 |
+
ArtifactIdentity,
|
| 53 |
+
ArtifactWriter,
|
| 54 |
+
ResourceSpec,
|
| 55 |
+
TensorSpec,
|
| 56 |
+
encode_direct,
|
| 57 |
+
row_split_geometry,
|
| 58 |
+
)
|
| 59 |
+
|
| 60 |
+
TEMPLATE = r"<TEMPLATE>"
|
| 61 |
+
GGUF = r"<WORKSPACE>\Ternary-Bonsai-2-27B-PQ2_0.gguf"
|
| 62 |
+
|
| 63 |
+
# The template must be the **groupwise-int** packing. It is not merely a donor of the
|
| 64 |
+
# vision/MTP payloads: this packer walks the template's OWN object list and maps every name
|
| 65 |
+
# through a closed table, so a template with different (fused) names cannot be consumed.
|
| 66 |
+
# The sibling packing `nvfp4` fuses the projections (gdn/a_b_projection,
|
| 67 |
+
# gdn/query_key_value_z, attention/query_key_gate_value) and therefore aborts with
|
| 68 |
+
# "unmapped gdn object ...". Both packings are produced from the same model by different
|
| 69 |
+
# converters in tools/convert/qwen3_8_27b/.
|
| 70 |
+
TEMPLATE_SCHEMA = "groupwise-int"
|
| 71 |
+
|
| 72 |
+
_TEMPLATE_HELP = """\
|
| 73 |
+
模板 schema 不对:pack.py 需要 **groupwise-int** 的 qwen3.8-27b 制品。
|
| 74 |
+
(the template must be the groupwise-int packing, not nvfp4)
|
| 75 |
+
|
| 76 |
+
你给的模板 : {path}
|
| 77 |
+
它的 schema : weights_id={got!r}
|
| 78 |
+
需要的 : weights_id={want!r}
|
| 79 |
+
|
| 80 |
+
怎么拿到正确的模板(二选一):
|
| 81 |
+
A) 用基树自带的转换器自己产一份 —— groupwise-int 路径,**不是** convert_nvfp4:
|
| 82 |
+
python3 -m tools.convert.qwen3_8_27b.convert \\
|
| 83 |
+
--model <Qwen3.8-27B 权重目录> \\
|
| 84 |
+
--dflash2-model <Qwen3.8-27B-DFlash2 目录> \\
|
| 85 |
+
--out <out.ninfer>
|
| 86 |
+
(convert_nvfp4 产出 nvfp4 schema,其 GDN/attention 为**融合命名**:
|
| 87 |
+
text/layers/N/gdn/a_b_projection、gdn/query_key_value_z、attention/query_key_gate_value
|
| 88 |
+
—— 这些名字不在本脚本的映射表里,必然中止。)
|
| 89 |
+
B) 任何 weights_id=groupwise-int 的 qwen3.8-27b 制品都可以当模板。
|
| 90 |
+
|
| 91 |
+
自检(满足任一条即为正确):
|
| 92 |
+
1) 该制品 identity.weights_id == 'groupwise-int'
|
| 93 |
+
2) text/layers/3/ 下是 attention/query_key 与 attention/gate_value **两个**对象
|
| 94 |
+
(若只有单个 attention/query_key_gate_value ⇒ nvfp4 版,用不了)
|
| 95 |
+
"""
|
| 96 |
+
|
| 97 |
+
_SCHEMA_HINT = ("\n hint: 模板 schema 不匹配。本脚本只认 weights_id=groupwise-int 的模板;"
|
| 98 |
+
"nvfp4 模板的融合命名(gdn/a_b_projection、gdn/query_key_value_z、"
|
| 99 |
+
"attention/query_key_gate_value)不在映射表里。见 README FAQ / --template。")
|
| 100 |
+
|
| 101 |
+
T_PQ2_0, T_PTQ1_0, T_F32, T_BF16 = 142, 143, 0, 30
|
| 102 |
+
FMT = {T_PQ2_0: "PQ2_0_G128", T_PTQ1_0: "PTQ1_0_G128"}
|
| 103 |
+
HIDDEN, V_HEADS, V_HEAD_DIM = 5120, 48, 128
|
| 104 |
+
QK_ROWS, V_ROWS = 4096, 6144
|
| 105 |
+
SIGN_WIDTHS = [5120, 6144, 17408]
|
| 106 |
+
FIXED = {0: 1, 1: 1, 2: 2, 3: 2, 4: 4, 5: 4, 6: 4, 7: 1, 10: 8, 11: 8, 12: 8}
|
| 107 |
+
|
| 108 |
+
|
| 109 |
+
# ---------------------------------------------------------------------------
|
| 110 |
+
class Gguf:
|
| 111 |
+
def __init__(self, path: str):
|
| 112 |
+
self.f = open(path, "rb")
|
| 113 |
+
self.f.seek(0, 2)
|
| 114 |
+
self.size = self.f.tell()
|
| 115 |
+
self.f.seek(0)
|
| 116 |
+
if self._raw(4) != b"GGUF":
|
| 117 |
+
raise SystemExit("not a GGUF file")
|
| 118 |
+
self.version = self._u32()
|
| 119 |
+
self.n_tensors = self._u64()
|
| 120 |
+
self.n_kv = self._u64()
|
| 121 |
+
self.kv: dict[str, object] = {}
|
| 122 |
+
for _ in range(self.n_kv):
|
| 123 |
+
k = self._str()
|
| 124 |
+
self.kv[k] = self._value(self._u32())
|
| 125 |
+
self.kv_end = self.f.tell()
|
| 126 |
+
self.tensor: dict[str, tuple[list[int], int, int]] = {}
|
| 127 |
+
order = []
|
| 128 |
+
for _ in range(self.n_tensors):
|
| 129 |
+
name = self._str()
|
| 130 |
+
ne = [self._u64() for _ in range(self._u32())]
|
| 131 |
+
tt = self._u32()
|
| 132 |
+
off = self._u64()
|
| 133 |
+
self.tensor[name] = (ne, tt, off)
|
| 134 |
+
order.append((off, name))
|
| 135 |
+
# header_end is the end of the TENSOR INFO LIST, not the end of the KV block.
|
| 136 |
+
# Recording it before this loop (as an earlier revision did) shifts data_start down
|
| 137 |
+
# by the size of the list -- ~50 KB in this file -- so every payload read comes from
|
| 138 |
+
# the wrong offset, while names/shapes/types still parse perfectly and a
|
| 139 |
+
# self-consistent byte round trip still passes. That is a silent, maximally
|
| 140 |
+
# misleading failure: probe_gguf_types.py puts this file's header end at
|
| 141 |
+
# 11,120,982 and the reader must agree.
|
| 142 |
+
self.header_end = self.f.tell()
|
| 143 |
+
self.alignment = int(self.kv.get("general.alignment", 32))
|
| 144 |
+
self.data_start = -(-self.header_end // self.alignment) * self.alignment
|
| 145 |
+
order.sort()
|
| 146 |
+
self.tbytes: dict[str, int] = {}
|
| 147 |
+
for i, (off, name) in enumerate(order):
|
| 148 |
+
nxt = order[i + 1][0] if i + 1 < len(order) else (self.size - self.data_start)
|
| 149 |
+
self.tbytes[name] = nxt - off
|
| 150 |
+
self._cache: dict[str, bytes] = {}
|
| 151 |
+
|
| 152 |
+
def _raw(self, n):
|
| 153 |
+
b = self.f.read(n)
|
| 154 |
+
if len(b) != n:
|
| 155 |
+
raise EOFError(f"short read of {n}")
|
| 156 |
+
return b
|
| 157 |
+
|
| 158 |
+
def _u32(self):
|
| 159 |
+
return struct.unpack("<I", self._raw(4))[0]
|
| 160 |
+
|
| 161 |
+
def _u64(self):
|
| 162 |
+
return struct.unpack("<Q", self._raw(8))[0]
|
| 163 |
+
|
| 164 |
+
def _str(self):
|
| 165 |
+
return self._raw(self._u64()).decode("utf-8", "replace")
|
| 166 |
+
|
| 167 |
+
def _value(self, t):
|
| 168 |
+
if t in (0, 1, 7):
|
| 169 |
+
return self._raw(1)[0]
|
| 170 |
+
if t in (2, 3):
|
| 171 |
+
return struct.unpack("<h", self._raw(2))[0]
|
| 172 |
+
if t == 4:
|
| 173 |
+
return struct.unpack("<I", self._raw(4))[0]
|
| 174 |
+
if t == 5:
|
| 175 |
+
return struct.unpack("<i", self._raw(4))[0]
|
| 176 |
+
if t == 6:
|
| 177 |
+
return struct.unpack("<f", self._raw(4))[0]
|
| 178 |
+
if t == 8:
|
| 179 |
+
return self._str()
|
| 180 |
+
if t in (10, 11):
|
| 181 |
+
return struct.unpack("<q", self._raw(8))[0]
|
| 182 |
+
if t == 12:
|
| 183 |
+
return struct.unpack("<d", self._raw(8))[0]
|
| 184 |
+
if t == 9:
|
| 185 |
+
et, n = self._u32(), self._u64()
|
| 186 |
+
if et == 8:
|
| 187 |
+
return [self._str() for _ in range(n)]
|
| 188 |
+
if et == 6:
|
| 189 |
+
return list(struct.unpack("<%df" % n, self._raw(4 * n)))
|
| 190 |
+
if et in (0, 1, 7):
|
| 191 |
+
return list(self._raw(n))
|
| 192 |
+
if et in (2, 3):
|
| 193 |
+
return list(struct.unpack("<%dh" % n, self._raw(2 * n)))
|
| 194 |
+
if et == 4:
|
| 195 |
+
return list(struct.unpack("<%dI" % n, self._raw(4 * n)))
|
| 196 |
+
if et == 5:
|
| 197 |
+
return list(struct.unpack("<%di" % n, self._raw(4 * n)))
|
| 198 |
+
self.f.seek(FIXED[et] * n, 1)
|
| 199 |
+
return f"<{n} values>"
|
| 200 |
+
raise ValueError(f"gguf value type {t}")
|
| 201 |
+
|
| 202 |
+
def payload(self, name: str) -> bytes:
|
| 203 |
+
if name not in self._cache:
|
| 204 |
+
_ne, _tt, off = self.tensor[name]
|
| 205 |
+
self.f.seek(self.data_start + off)
|
| 206 |
+
self._cache[name] = self._raw(self.tbytes[name])
|
| 207 |
+
return self._cache[name]
|
| 208 |
+
|
| 209 |
+
def tinfo(self, name):
|
| 210 |
+
ne, tt, _ = self.tensor[name]
|
| 211 |
+
return ne, tt
|
| 212 |
+
|
| 213 |
+
def row_shape(self, name) -> tuple[int, int]:
|
| 214 |
+
"""(n_rows, k) in GGUF row order, plus the number of weights."""
|
| 215 |
+
ne, _tt = self.tinfo(name)
|
| 216 |
+
if len(ne) != 2:
|
| 217 |
+
raise SystemExit(f"{name}: expected rank 2, got {ne}")
|
| 218 |
+
return ne[1], ne[0]
|
| 219 |
+
|
| 220 |
+
def blocks(self, name):
|
| 221 |
+
"""(n_rows, groups_per_row, block_bytes, row_bytes, raw)."""
|
| 222 |
+
n, k = self.row_shape(name)
|
| 223 |
+
ne, tt = self.tinfo(name)
|
| 224 |
+
block = 28 if tt == T_PTQ1_0 else 34
|
| 225 |
+
gpr = k // 128
|
| 226 |
+
raw = self.payload(name)
|
| 227 |
+
if len(raw) != n * gpr * block:
|
| 228 |
+
raise SystemExit(f"{name}: {len(raw)} != {n}*{gpr}*{block}")
|
| 229 |
+
return n, gpr, block, gpr * block, raw
|
| 230 |
+
|
| 231 |
+
def row_fn(self, name):
|
| 232 |
+
n, gpr, block, rb, raw = self.blocks(name)
|
| 233 |
+
return lambda i: raw[i * rb:(i + 1) * rb]
|
| 234 |
+
|
| 235 |
+
|
| 236 |
+
# ---------------------------------------------------------------------------
|
| 237 |
+
# Reference decoders -- independent ports of ggml/src/ggml-quants.c
|
| 238 |
+
# ---------------------------------------------------------------------------
|
| 239 |
+
def dq_pq2_0(raw: bytes) -> np.ndarray:
|
| 240 |
+
qs = np.frombuffer(raw, dtype=np.uint8).reshape(-1, 34)
|
| 241 |
+
d = qs[:, 0:2].copy().view(np.float16).astype(np.float32).reshape(-1, 1)
|
| 242 |
+
j = np.arange(128)
|
| 243 |
+
q = (qs[:, 2:34][:, j // 4] >> (2 * (j % 4))) & 0x03
|
| 244 |
+
return ((q.astype(np.int32) - 1) * d).reshape(-1)
|
| 245 |
+
|
| 246 |
+
|
| 247 |
+
def dq_ptq1_0(raw: bytes) -> np.ndarray:
|
| 248 |
+
blk = np.frombuffer(raw, dtype=np.uint8).reshape(-1, 28)
|
| 249 |
+
pow3 = (1, 3, 9, 27, 81, 243)
|
| 250 |
+
g = blk.shape[0]
|
| 251 |
+
d = blk[:, 26:28].copy().view(np.float16).astype(np.float32).reshape(-1)
|
| 252 |
+
qs, qh = blk[:, 0:24], blk[:, 24:26]
|
| 253 |
+
vals = np.empty((g, 120), dtype=np.float32)
|
| 254 |
+
col, j = 0, 0
|
| 255 |
+
for c in (32, 16, 8): # c=32 emits nothing: 0 + 32 > 24
|
| 256 |
+
while j + c <= 24:
|
| 257 |
+
for n in range(5):
|
| 258 |
+
prod = (qs[:, j:j + c].astype(np.uint16) * pow3[n]) & 0xFF
|
| 259 |
+
vals[:, col:col + c] = ((prod.astype(np.uint16) * 3) >> 8).astype(np.float32)
|
| 260 |
+
col += c
|
| 261 |
+
j += c
|
| 262 |
+
if col != 120:
|
| 263 |
+
raise SystemExit(f"ptq1_0 qs walk emitted {col}, expected 120")
|
| 264 |
+
tail = np.empty((g, 8), dtype=np.float32)
|
| 265 |
+
col = 0
|
| 266 |
+
for n in range(4):
|
| 267 |
+
prod = (qh.astype(np.uint16) * pow3[n]) & 0xFF
|
| 268 |
+
for h in range(2):
|
| 269 |
+
tail[:, col] = ((prod[:, h].astype(np.uint16) * 3) >> 8).astype(np.float32)
|
| 270 |
+
col += 1
|
| 271 |
+
out = np.concatenate([vals, tail], axis=1)
|
| 272 |
+
return ((out - 1.0) * d[:, None]).reshape(-1)
|
| 273 |
+
|
| 274 |
+
|
| 275 |
+
DEQUANT = {T_PQ2_0: dq_pq2_0, T_PTQ1_0: dq_ptq1_0}
|
| 276 |
+
|
| 277 |
+
|
| 278 |
+
def read_direct(g: Gguf, name: str) -> np.ndarray:
|
| 279 |
+
ne, tt = g.tinfo(name)
|
| 280 |
+
raw = g.payload(name)
|
| 281 |
+
if tt == T_F32:
|
| 282 |
+
a = np.frombuffer(raw, dtype="<f4").astype(np.float32)
|
| 283 |
+
elif tt == T_BF16:
|
| 284 |
+
a = (np.frombuffer(raw, dtype="<u2").astype(np.uint32) << 16).view(np.float32)
|
| 285 |
+
else:
|
| 286 |
+
raise SystemExit(f"{name}: type {tt} is not F32/BF16")
|
| 287 |
+
return a.reshape(tuple(reversed(ne))) if len(ne) > 1 else a.astype(np.float32)
|
| 288 |
+
|
| 289 |
+
|
| 290 |
+
# ---------------------------------------------------------------------------
|
| 291 |
+
# Ternary plane assembly, byte for byte, and its inverse
|
| 292 |
+
# ---------------------------------------------------------------------------
|
| 293 |
+
def block_to_planes(fmt, blk):
|
| 294 |
+
if fmt == "PTQ1_0_G128":
|
| 295 |
+
return blk[0:24], blk[24:26], blk[26:28]
|
| 296 |
+
if fmt == "PQ2_0_G128":
|
| 297 |
+
return blk[2:34], b"", blk[0:2]
|
| 298 |
+
raise ValueError(fmt)
|
| 299 |
+
|
| 300 |
+
|
| 301 |
+
def planes_to_block(fmt, base, high, scale):
|
| 302 |
+
return base + high + scale if fmt == "PTQ1_0_G128" else scale + base
|
| 303 |
+
|
| 304 |
+
|
| 305 |
+
def assemble_ternary(fmt, shape, row_fn) -> bytes:
|
| 306 |
+
"""row_fn(i) -> the raw GGML block bytes for destination row i."""
|
| 307 |
+
n, k = shape
|
| 308 |
+
geo = row_split_geometry(fmt, shape)
|
| 309 |
+
gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
|
| 310 |
+
block = bb + hb + 2
|
| 311 |
+
out = bytearray(geo.payload_bytes)
|
| 312 |
+
for i in range(n):
|
| 313 |
+
row = row_fn(i)
|
| 314 |
+
if len(row) != gpr * block:
|
| 315 |
+
raise SystemExit(f"row {i}: {len(row)} != {gpr * block}")
|
| 316 |
+
for sel in range(gpr):
|
| 317 |
+
base, high, scale = block_to_planes(fmt, row[sel * block:(sel + 1) * block])
|
| 318 |
+
o = geo.base_offset + i * geo.base_row_bytes + sel * bb
|
| 319 |
+
out[o:o + bb] = base
|
| 320 |
+
if hb:
|
| 321 |
+
o = geo.high_offset + i * geo.high_row_bytes + sel * hb
|
| 322 |
+
out[o:o + hb] = high
|
| 323 |
+
o = geo.scale_offset + i * geo.scale_row_bytes + sel * 2
|
| 324 |
+
out[o:o + 2] = scale
|
| 325 |
+
return bytes(out)
|
| 326 |
+
|
| 327 |
+
|
| 328 |
+
def disassemble_ternary(fmt, shape, payload: bytes) -> bytes:
|
| 329 |
+
"""Inverse of assemble_ternary: rebuild the contiguous GGML block stream."""
|
| 330 |
+
n, k = shape
|
| 331 |
+
geo = row_split_geometry(fmt, shape)
|
| 332 |
+
gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
|
| 333 |
+
out = bytearray()
|
| 334 |
+
for i in range(n):
|
| 335 |
+
for sel in range(gpr):
|
| 336 |
+
o = geo.base_offset + i * geo.base_row_bytes + sel * bb
|
| 337 |
+
base = payload[o:o + bb]
|
| 338 |
+
high = b""
|
| 339 |
+
if hb:
|
| 340 |
+
o = geo.high_offset + i * geo.high_row_bytes + sel * hb
|
| 341 |
+
high = payload[o:o + hb]
|
| 342 |
+
o = geo.scale_offset + i * geo.scale_row_bytes + sel * 2
|
| 343 |
+
out += planes_to_block(fmt, base, high, payload[o:o + 2])
|
| 344 |
+
return bytes(out)
|
| 345 |
+
|
| 346 |
+
|
| 347 |
+
# ---------------------------------------------------------------------------
|
| 348 |
+
# Non-ternary transforms
|
| 349 |
+
# ---------------------------------------------------------------------------
|
| 350 |
+
def tiled_to_grouped(t: np.ndarray, groups: int = 3) -> np.ndarray:
|
| 351 |
+
"""GGUF TILED (3,16) -> ninfer GROUPED (16,3) on the leading 48-head axis."""
|
| 352 |
+
shape = t.shape
|
| 353 |
+
return (t.reshape(groups, shape[0] // groups, *shape[1:])
|
| 354 |
+
.transpose(1, 0, *range(2, len(shape) + 1))
|
| 355 |
+
.reshape(shape))
|
| 356 |
+
|
| 357 |
+
|
| 358 |
+
def bf16_payload(a: np.ndarray) -> bytes:
|
| 359 |
+
import torch
|
| 360 |
+
return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.float32))
|
| 361 |
+
.to(torch.bfloat16), "BF16")
|
| 362 |
+
|
| 363 |
+
|
| 364 |
+
def fp32_payload(a) -> bytes:
|
| 365 |
+
import torch
|
| 366 |
+
return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.float32)), "FP32")
|
| 367 |
+
|
| 368 |
+
|
| 369 |
+
def i32_payload(a) -> bytes:
|
| 370 |
+
import torch
|
| 371 |
+
return encode_direct(torch.from_numpy(np.ascontiguousarray(a, dtype=np.int32)), "I32")
|
| 372 |
+
|
| 373 |
+
|
| 374 |
+
# ---------------------------------------------------------------------------
|
| 375 |
+
# Inventory (built from the template, never assumed)
|
| 376 |
+
# ---------------------------------------------------------------------------
|
| 377 |
+
def _weights_id(identity):
|
| 378 |
+
"""Pull weights_id out of the template manifest's identity block."""
|
| 379 |
+
if isinstance(identity, dict):
|
| 380 |
+
for k in ("weights_id", "weightsId", "weights"):
|
| 381 |
+
v = identity.get(k)
|
| 382 |
+
if isinstance(v, str):
|
| 383 |
+
return v
|
| 384 |
+
return None
|
| 385 |
+
|
| 386 |
+
|
| 387 |
+
def load_template():
|
| 388 |
+
with open(TEMPLATE, "rb") as f:
|
| 389 |
+
f.seek(16)
|
| 390 |
+
blob = f.read(8 << 20)
|
| 391 |
+
obj, _ = json.JSONDecoder().raw_decode(blob.decode("utf-8", "replace"))
|
| 392 |
+
identity, objects = obj["identity"], obj["objects"]
|
| 393 |
+
got = _weights_id(identity)
|
| 394 |
+
if got != TEMPLATE_SCHEMA:
|
| 395 |
+
# Fail HERE, with an actionable message, instead of 200 lines later on an
|
| 396 |
+
# "unmapped gdn object" that does not say what to do about it.
|
| 397 |
+
raise SystemExit(_TEMPLATE_HELP.format(path=TEMPLATE, got=got, want=TEMPLATE_SCHEMA))
|
| 398 |
+
return identity, objects
|
| 399 |
+
|
| 400 |
+
|
| 401 |
+
def layer_kind(objs) -> dict[int, str]:
|
| 402 |
+
kind: dict[int, str] = {}
|
| 403 |
+
for o in objs:
|
| 404 |
+
p = o["name"].split("/")
|
| 405 |
+
if len(p) > 3 and p[0] == "text" and p[1] == "layers":
|
| 406 |
+
l = int(p[2])
|
| 407 |
+
kind.setdefault(l, "attn" if p[3] == "attention" else "gdn")
|
| 408 |
+
return kind
|
| 409 |
+
|
| 410 |
+
|
| 411 |
+
# Target formats for the NEW artifact. The template's own formats describe the
|
| 412 |
+
# groupwise-int artifact and must NOT be reused: the packer replaces all 402 ternary
|
| 413 |
+
# matrices and re-encodes the norms, so the writer would otherwise validate every produced
|
| 414 |
+
# payload against the wrong geometry (CHECK 3 caught exactly this).
|
| 415 |
+
TERNARY_SUFFIXES = frozenset((
|
| 416 |
+
"mlp/gate_up", "mlp/down",
|
| 417 |
+
"attention/query_key", "attention/gate_value", "attention/output",
|
| 418 |
+
"gdn/query_key", "gdn/value_z", "gdn/output",
|
| 419 |
+
))
|
| 420 |
+
BF16_SUFFIXES = frozenset((
|
| 421 |
+
"input_norm", "post_attention_norm",
|
| 422 |
+
"attention/query_norm", "attention/key_norm",
|
| 423 |
+
"gdn/norm", "gdn/convolution", "gdn/a_projection", "gdn/b_projection",
|
| 424 |
+
))
|
| 425 |
+
FP32_SUFFIXES = frozenset(("gdn/a_log", "gdn/dt_bias"))
|
| 426 |
+
|
| 427 |
+
SIGN_OBJECTS = (
|
| 428 |
+
("text/hadamard_signs", (28672,), "FP32"),
|
| 429 |
+
("text/hadamard_widths", (3,), "I32"),
|
| 430 |
+
)
|
| 431 |
+
|
| 432 |
+
|
| 433 |
+
def build_specs(p):
|
| 434 |
+
"""Ordered specs for the NEW artifact.
|
| 435 |
+
|
| 436 |
+
Template objects keep their order, shapes and borrowed formats, but every produced
|
| 437 |
+
tensor takes its TARGET format; the two Hadamard sign-table objects are appended.
|
| 438 |
+
"""
|
| 439 |
+
specs = []
|
| 440 |
+
for o in p.objects:
|
| 441 |
+
if o["kind"] == "tensor":
|
| 442 |
+
fmt, layout = p.target_format(o["name"])
|
| 443 |
+
specs.append(TensorSpec(o["name"], tuple(o["shape"]), fmt, layout))
|
| 444 |
+
else:
|
| 445 |
+
specs.append(ResourceSpec(o["name"], o["encoding"], o["bytes"]))
|
| 446 |
+
for name, shape, fmt in SIGN_OBJECTS:
|
| 447 |
+
specs.append(TensorSpec(name, shape, fmt, "contiguous-le-v1"))
|
| 448 |
+
return specs
|
| 449 |
+
|
| 450 |
+
|
| 451 |
+
# ---------------------------------------------------------------------------
|
| 452 |
+
class Packer:
|
| 453 |
+
def __init__(self, g: Gguf):
|
| 454 |
+
self.g = g
|
| 455 |
+
raw_identity, self.objects = load_template()
|
| 456 |
+
self.identity = ArtifactIdentity(raw_identity["model_id"], raw_identity["weights_id"])
|
| 457 |
+
self._by_name = {o["name"]: o for o in self.objects}
|
| 458 |
+
self.kind = layer_kind(self.objects)
|
| 459 |
+
self._signs = None
|
| 460 |
+
|
| 461 |
+
# -- helpers ---------------------------------------------------------
|
| 462 |
+
def rel(self, name: str, l: int) -> str:
|
| 463 |
+
return name.replace("{l}", str(l))
|
| 464 |
+
|
| 465 |
+
def fmt_of(self, gguf_name: str) -> str:
|
| 466 |
+
_ne, tt = self.g.tinfo(gguf_name)
|
| 467 |
+
if tt not in FMT:
|
| 468 |
+
raise SystemExit(f"{gguf_name}: type {tt} is not ternary")
|
| 469 |
+
return FMT[tt]
|
| 470 |
+
|
| 471 |
+
def target_format(self, name: str) -> tuple[str, str]:
|
| 472 |
+
"""(format, layout) this object carries in the NEW artifact."""
|
| 473 |
+
if name == "text/hadamard_signs":
|
| 474 |
+
return "FP32", "contiguous-le-v1"
|
| 475 |
+
if name == "text/hadamard_widths":
|
| 476 |
+
return "I32", "contiguous-le-v1"
|
| 477 |
+
if name == "text/token_embedding":
|
| 478 |
+
return self.fmt_of("token_embd.weight"), "row-split-k128-v1"
|
| 479 |
+
if name == "text/output_head":
|
| 480 |
+
return self.fmt_of("output.weight"), "row-split-k128-v1"
|
| 481 |
+
if name == "text/final_norm":
|
| 482 |
+
return "BF16", "contiguous-le-v1"
|
| 483 |
+
|
| 484 |
+
parts = name.split("/")
|
| 485 |
+
if len(parts) > 3 and parts[0] == "text" and parts[1] == "layers":
|
| 486 |
+
suffix = "/".join(parts[3:])
|
| 487 |
+
pre = f"blk.{int(parts[2])}."
|
| 488 |
+
if suffix in TERNARY_SUFFIXES:
|
| 489 |
+
if suffix == "mlp/gate_up":
|
| 490 |
+
return self.fmt_of(pre + "ffn_gate.weight"), "row-split-k128-v1"
|
| 491 |
+
if suffix == "mlp/down":
|
| 492 |
+
return self.fmt_of(pre + "ffn_down.weight"), "row-split-k128-v1"
|
| 493 |
+
if suffix in ("attention/query_key", "attention/gate_value"):
|
| 494 |
+
return self.fmt_of(pre + "attn_q.weight"), "row-split-k128-v1"
|
| 495 |
+
if suffix == "attention/output":
|
| 496 |
+
return self.fmt_of(pre + "attn_output.weight"), "row-split-k128-v1"
|
| 497 |
+
if suffix in ("gdn/query_key", "gdn/value_z"):
|
| 498 |
+
return self.fmt_of(pre + "attn_qkv.weight"), "row-split-k128-v1"
|
| 499 |
+
if suffix == "gdn/output":
|
| 500 |
+
return self.fmt_of(pre + "ssm_out.weight"), "row-split-k128-v1"
|
| 501 |
+
if suffix in BF16_SUFFIXES:
|
| 502 |
+
return "BF16", "contiguous-le-v1"
|
| 503 |
+
if suffix in FP32_SUFFIXES:
|
| 504 |
+
return "FP32", "contiguous-le-v1"
|
| 505 |
+
|
| 506 |
+
# everything else (mtp/*, vision/*, draft_head*) is BORROWED: keep the template's
|
| 507 |
+
o = self._by_name.get(name)
|
| 508 |
+
if o is None:
|
| 509 |
+
raise SystemExit(f"no template object and no producer for {name}")
|
| 510 |
+
return o["format"], o["layout"]
|
| 511 |
+
|
| 512 |
+
def fused(self, entries, shape) -> bytes:
|
| 513 |
+
"""entries: list of (gguf_name, row_index); all sources must share one format."""
|
| 514 |
+
fmts = {self.fmt_of(n) for n, _ in entries}
|
| 515 |
+
if len(fmts) != 1:
|
| 516 |
+
raise SystemExit(f"fused tensor mixes formats: {fmts}")
|
| 517 |
+
fmt = fmts.pop()
|
| 518 |
+
srcs = {}
|
| 519 |
+
for n, _ in entries:
|
| 520 |
+
if n not in srcs:
|
| 521 |
+
nrows, _gpr, _blk, rb, raw = self.g.blocks(n)
|
| 522 |
+
srcs[n] = (rb, raw)
|
| 523 |
+
|
| 524 |
+
def row_fn(i):
|
| 525 |
+
n, r = entries[i]
|
| 526 |
+
rb, raw = srcs[n]
|
| 527 |
+
return raw[r * rb:(r + 1) * rb]
|
| 528 |
+
|
| 529 |
+
return assemble_ternary(fmt, shape, row_fn)
|
| 530 |
+
|
| 531 |
+
def direct_ternary(self, gguf_name: str, shape) -> bytes:
|
| 532 |
+
return assemble_ternary(self.fmt_of(gguf_name), shape, self.g.row_fn(gguf_name))
|
| 533 |
+
|
| 534 |
+
def signs(self):
|
| 535 |
+
if self._signs is None:
|
| 536 |
+
vals = self.g.kv["prism.hadamard.sign_values"]
|
| 537 |
+
widths = list(self.g.kv["prism.hadamard.sign_widths"])
|
| 538 |
+
arr = np.asarray(vals, dtype=np.float32)
|
| 539 |
+
if arr.size != 28672 or not np.all(np.abs(arr) == 1.0):
|
| 540 |
+
raise SystemExit("sign table is not 28672 strictly-+-1 values")
|
| 541 |
+
if widths != SIGN_WIDTHS or sum(widths) != arr.size:
|
| 542 |
+
raise SystemExit(f"sign_widths mismatch: {widths}")
|
| 543 |
+
self._signs = (arr, widths)
|
| 544 |
+
return self._signs
|
| 545 |
+
|
| 546 |
+
@staticmethod
|
| 547 |
+
def payload_bytes(spec) -> int | None:
|
| 548 |
+
"""Expected payload size for a template object spec, or None when not computable."""
|
| 549 |
+
if spec["kind"] != "tensor":
|
| 550 |
+
return None
|
| 551 |
+
layout = spec.get("layout")
|
| 552 |
+
if layout == "row-split-k128-v1":
|
| 553 |
+
return row_split_geometry(spec["format"], tuple(spec["shape"])).payload_bytes
|
| 554 |
+
if layout == "contiguous-le-v1":
|
| 555 |
+
wb = {"BF16": 2, "FP32": 4, "I32": 4}[spec["format"]]
|
| 556 |
+
return int(np.prod(spec["shape"])) * wb
|
| 557 |
+
return None
|
| 558 |
+
|
| 559 |
+
# -- producers -------------------------------------------------------
|
| 560 |
+
def produce(self, name: str):
|
| 561 |
+
"""Return the payload bytes for a text/* object, or None to borrow it."""
|
| 562 |
+
p = name.split("/")
|
| 563 |
+
|
| 564 |
+
if name == "text/hadamard_signs":
|
| 565 |
+
return fp32_payload(self.signs()[0])
|
| 566 |
+
if name == "text/hadamard_widths":
|
| 567 |
+
return i32_payload(self.signs()[1])
|
| 568 |
+
|
| 569 |
+
if name == "text/token_embedding":
|
| 570 |
+
return self.direct_ternary("token_embd.weight", (248320, HIDDEN))
|
| 571 |
+
if name == "text/output_head":
|
| 572 |
+
return self.direct_ternary("output.weight", (248320, HIDDEN))
|
| 573 |
+
if name == "text/final_norm":
|
| 574 |
+
return bf16_payload(read_direct(self.g, "output_norm.weight") - 1.0)
|
| 575 |
+
|
| 576 |
+
if len(p) > 3 and p[0] == "text" and p[1] == "layers":
|
| 577 |
+
l = int(p[2])
|
| 578 |
+
suffix = "/".join(p[3:])
|
| 579 |
+
g = self.g
|
| 580 |
+
pre = f"blk.{l}."
|
| 581 |
+
|
| 582 |
+
if suffix == "input_norm":
|
| 583 |
+
return bf16_payload(read_direct(g, pre + "attn_norm.weight") - 1.0)
|
| 584 |
+
if suffix == "post_attention_norm":
|
| 585 |
+
return bf16_payload(read_direct(g, pre + "post_attention_norm.weight") - 1.0)
|
| 586 |
+
if suffix == "mlp/gate_up":
|
| 587 |
+
ng, _ = g.row_shape(pre + "ffn_gate.weight")
|
| 588 |
+
nu, _ = g.row_shape(pre + "ffn_up.weight")
|
| 589 |
+
ent = [(pre + "ffn_gate.weight", i) for i in range(ng)]
|
| 590 |
+
ent += [(pre + "ffn_up.weight", i) for i in range(nu)]
|
| 591 |
+
return self.fused(ent, (ng + nu, HIDDEN))
|
| 592 |
+
if suffix == "mlp/down":
|
| 593 |
+
return self.direct_ternary(pre + "ffn_down.weight", (5120, 17408))
|
| 594 |
+
|
| 595 |
+
if suffix.startswith("attention/"):
|
| 596 |
+
sub = suffix.split("/", 1)[1]
|
| 597 |
+
if sub == "query_key":
|
| 598 |
+
q = self.deinterleave(pre + "attn_q.weight", 0, (6144, HIDDEN))
|
| 599 |
+
nk, _ = g.row_shape(pre + "attn_k.weight")
|
| 600 |
+
return self.concat_streams(q, 6144, pre + "attn_k.weight", nk,
|
| 601 |
+
(6144 + nk, HIDDEN))
|
| 602 |
+
if sub == "gate_value":
|
| 603 |
+
gt = self.deinterleave(pre + "attn_q.weight", 1, (6144, HIDDEN))
|
| 604 |
+
nv, _ = g.row_shape(pre + "attn_v.weight")
|
| 605 |
+
return self.concat_streams(gt, 6144, pre + "attn_v.weight", nv,
|
| 606 |
+
(6144 + nv, HIDDEN))
|
| 607 |
+
if sub == "output":
|
| 608 |
+
return self.direct_ternary(pre + "attn_output.weight", (5120, 6144))
|
| 609 |
+
if sub == "query_norm":
|
| 610 |
+
return bf16_payload(read_direct(g, pre + "attn_q_norm.weight") - 1.0)
|
| 611 |
+
if sub == "key_norm":
|
| 612 |
+
return bf16_payload(read_direct(g, pre + "attn_k_norm.weight") - 1.0)
|
| 613 |
+
raise SystemExit(f"unmapped attention object {name}" + _SCHEMA_HINT)
|
| 614 |
+
|
| 615 |
+
if suffix.startswith("gdn/"):
|
| 616 |
+
sub = suffix.split("/", 1)[1]
|
| 617 |
+
if sub == "query_key":
|
| 618 |
+
ent = [(pre + "attn_qkv.weight", i) for i in range(QK_ROWS)]
|
| 619 |
+
return self.fused(ent, (QK_ROWS, HIDDEN))
|
| 620 |
+
if sub == "value_z":
|
| 621 |
+
return self.gdn_value_z(l)
|
| 622 |
+
if sub == "output":
|
| 623 |
+
return self.direct_ternary(pre + "ssm_out.weight", (5120, V_ROWS))
|
| 624 |
+
if sub == "convolution":
|
| 625 |
+
return bf16_payload(self.conv1d(l))
|
| 626 |
+
if sub == "norm":
|
| 627 |
+
return bf16_payload(read_direct(g, pre + "ssm_norm.weight")) # RAW
|
| 628 |
+
if sub == "a_projection":
|
| 629 |
+
return bf16_payload(tiled_to_grouped(read_direct(g, pre + "ssm_alpha.weight")))
|
| 630 |
+
if sub == "b_projection":
|
| 631 |
+
return bf16_payload(tiled_to_grouped(read_direct(g, pre + "ssm_beta.weight")))
|
| 632 |
+
if sub == "a_log":
|
| 633 |
+
a = read_direct(g, pre + "ssm_a")
|
| 634 |
+
return fp32_payload(tiled_to_grouped(np.log(-a.astype(np.float64))
|
| 635 |
+
.astype(np.float32)))
|
| 636 |
+
if sub == "dt_bias":
|
| 637 |
+
return fp32_payload(tiled_to_grouped(read_direct(g, pre + "ssm_dt.bias")))
|
| 638 |
+
raise SystemExit(f"unmapped gdn object {name}" + _SCHEMA_HINT)
|
| 639 |
+
|
| 640 |
+
raise SystemExit(f"unmapped layer object {name}" + _SCHEMA_HINT)
|
| 641 |
+
|
| 642 |
+
return None # borrow
|
| 643 |
+
|
| 644 |
+
# -- GDN / attention fusion helpers ----------------------------------
|
| 645 |
+
def deinterleave(self, gguf_name: str, parity: int, shape):
|
| 646 |
+
"""query (parity 0) or output-gate (parity 1): every other 256-row chunk."""
|
| 647 |
+
n, k = self.g.row_shape(gguf_name)
|
| 648 |
+
if n % 512 != 0:
|
| 649 |
+
raise SystemExit(f"{gguf_name}: {n} rows is not a multiple of 512")
|
| 650 |
+
heads = n // 512
|
| 651 |
+
fmt = self.fmt_of(gguf_name)
|
| 652 |
+
nrows, gpr, block, rb, raw = self.g.blocks(gguf_name)
|
| 653 |
+
ent = [(gguf_name, (2 * h + parity) * 256 + r)
|
| 654 |
+
for h in range(heads) for r in range(256)]
|
| 655 |
+
return assemble_ternary(fmt, (len(ent), k),
|
| 656 |
+
lambda i: raw[ent[i][1] * rb:(ent[i][1] + 1) * rb])
|
| 657 |
+
|
| 658 |
+
def concat_streams(self, first: bytes, first_rows: int, gname: str, gn: int, shape):
|
| 659 |
+
"""Concatenate a pre-assembled ternary payload with another tensor's rows."""
|
| 660 |
+
n, k = shape
|
| 661 |
+
fmt = self.fmt_of(gname)
|
| 662 |
+
geo = row_split_geometry(fmt, shape)
|
| 663 |
+
gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
|
| 664 |
+
block = bb + hb + 2
|
| 665 |
+
_n2, gpr2, block2, rb, raw = self.g.blocks(gname)
|
| 666 |
+
if gpr2 != gpr or block2 != block:
|
| 667 |
+
raise SystemExit("concat_streams: block geometry differs")
|
| 668 |
+
geo_a = row_split_geometry(fmt, (first_rows, k))
|
| 669 |
+
|
| 670 |
+
def row_fn(i):
|
| 671 |
+
if i < first_rows:
|
| 672 |
+
out = bytearray()
|
| 673 |
+
for sel in range(gpr):
|
| 674 |
+
o = geo_a.base_offset + i * geo_a.base_row_bytes + sel * bb
|
| 675 |
+
base = first[o:o + bb]
|
| 676 |
+
high = b""
|
| 677 |
+
if hb:
|
| 678 |
+
o = geo_a.high_offset + i * geo_a.high_row_bytes + sel * hb
|
| 679 |
+
high = first[o:o + hb]
|
| 680 |
+
o = geo_a.scale_offset + i * geo_a.scale_row_bytes + sel * 2
|
| 681 |
+
out += planes_to_block(fmt, base, high, first[o:o + 2])
|
| 682 |
+
return bytes(out)
|
| 683 |
+
j = i - first_rows
|
| 684 |
+
return raw[j * rb:(j + 1) * rb]
|
| 685 |
+
|
| 686 |
+
return assemble_ternary(fmt, shape, row_fn)
|
| 687 |
+
|
| 688 |
+
def gdn_value_z(self, l: int) -> bytes:
|
| 689 |
+
"""concat(tiled_to_grouped(attn_qkv[4096:10240]), tiled_to_grouped(attn_gate[0:6144]))."""
|
| 690 |
+
pre = f"blk.{l}."
|
| 691 |
+
qkv, gate = pre + "attn_qkv.weight", pre + "attn_gate.weight"
|
| 692 |
+
f1, f2 = self.fmt_of(qkv), self.fmt_of(gate)
|
| 693 |
+
if f1 != f2:
|
| 694 |
+
raise SystemExit("value_z: mixed formats")
|
| 695 |
+
fmt = f1
|
| 696 |
+
|
| 697 |
+
def perm48(i):
|
| 698 |
+
# tiled (3,16) -> grouped (16,3) for a 48-entry HEAD axis:
|
| 699 |
+
# destination head g*3+t comes from source head t*16+g
|
| 700 |
+
return (i % 3) * 16 + (i // 3)
|
| 701 |
+
|
| 702 |
+
def perm_row(i):
|
| 703 |
+
# perm48() permutes a 48-entry axis, but this tensor's V/z rows are 48 heads of
|
| 704 |
+
# V_HEAD_DIM rows each, so the permutation has to be applied at HEAD granularity.
|
| 705 |
+
# Applying perm48() straight to the 6144-row index is not a bijection: perm48(48) and
|
| 706 |
+
# perm48(1) are both 16, so rows repeat and source rows are dropped while every size,
|
| 707 |
+
# row count and byte total stays exactly right -- which is why byte accounting, the
|
| 708 |
+
# row-count cross-check and the payload round-trip all stayed green over it.
|
| 709 |
+
head, inner = divmod(i, V_HEAD_DIM)
|
| 710 |
+
return perm48(head) * V_HEAD_DIM + inner
|
| 711 |
+
|
| 712 |
+
shape = (12288, HIDDEN)
|
| 713 |
+
geo = row_split_geometry(fmt, shape)
|
| 714 |
+
gpr, bb, hb = geo.groups_per_row, geo.base_bytes_per_group, geo.high_bytes_per_group
|
| 715 |
+
block = bb + hb + 2
|
| 716 |
+
gpr1 = self.g.blocks(qkv)[1]
|
| 717 |
+
gpr2_ = self.g.blocks(gate)[1]
|
| 718 |
+
if gpr1 != gpr or gpr2_ != gpr:
|
| 719 |
+
raise SystemExit("value_z: groups_per_row mismatch")
|
| 720 |
+
rb1 = self.g.blocks(qkv)[3]
|
| 721 |
+
rb2 = self.g.blocks(gate)[3]
|
| 722 |
+
raw1, raw2 = self.g.payload(qkv), self.g.payload(gate)
|
| 723 |
+
|
| 724 |
+
def row_fn(i):
|
| 725 |
+
if i < V_ROWS: # attn_qkv rows 4096..10240, tiled->grouped
|
| 726 |
+
src = QK_ROWS + perm_row(i)
|
| 727 |
+
return raw1[src * rb1:(src + 1) * rb1]
|
| 728 |
+
src = perm_row(i - V_ROWS) # attn_gate rows 0..6144, tiled->grouped
|
| 729 |
+
return raw2[src * rb2:(src + 1) * rb2]
|
| 730 |
+
|
| 731 |
+
return assemble_ternary(fmt, shape, row_fn)
|
| 732 |
+
|
| 733 |
+
def conv1d(self, l: int) -> np.ndarray:
|
| 734 |
+
"""(4,10240) <- ssm_conv1d: keep channels 0:4096, reorder the V channels, transpose."""
|
| 735 |
+
pre = f"blk.{l}."
|
| 736 |
+
t = read_direct(self.g, pre + "ssm_conv1d.weight") # (10240, 4)
|
| 737 |
+
if t.shape != (10240, 4):
|
| 738 |
+
raise SystemExit(f"conv1d shape {t.shape} != (10240, 4)")
|
| 739 |
+
head = t[0:QK_ROWS]
|
| 740 |
+
v = t[QK_ROWS:QK_ROWS + V_ROWS].reshape(V_HEADS, V_HEAD_DIM, t.shape[1])
|
| 741 |
+
v = (v.reshape(3, V_HEADS // 3, V_HEAD_DIM, t.shape[1])
|
| 742 |
+
.transpose(1, 0, 2, 3)
|
| 743 |
+
.reshape(V_ROWS, t.shape[1]))
|
| 744 |
+
return np.concatenate([head, v], axis=0).T # (4, 10240)
|
| 745 |
+
|
| 746 |
+
|
| 747 |
+
# ---------------------------------------------------------------------------
|
| 748 |
+
def mode_check(g: Gguf) -> int:
|
| 749 |
+
p = Packer(g)
|
| 750 |
+
print("=" * 78)
|
| 751 |
+
print("CHECK 1 geometry over every ternary shape present in the GGUF")
|
| 752 |
+
print("=" * 78)
|
| 753 |
+
combos = {}
|
| 754 |
+
for name, (ne, tt, _off) in g.tensor.items():
|
| 755 |
+
if tt in FMT:
|
| 756 |
+
combos.setdefault((FMT[tt], ne[1], ne[0]), []).append(name)
|
| 757 |
+
for (fmt, n, k), names in sorted(combos.items()):
|
| 758 |
+
geo = row_split_geometry(fmt, (n, k))
|
| 759 |
+
src = n * (k // 128) * (28 if fmt == "PTQ1_0_G128" else 34)
|
| 760 |
+
pad = geo.payload_bytes - src
|
| 761 |
+
print(f" {fmt:<14} n={n:<7} k={k:<6} groups/row={geo.groups_per_row:<4} "
|
| 762 |
+
f"src={src:>13,} payload={geo.payload_bytes:>13,} pad={pad:>5} "
|
| 763 |
+
f"({len(names)} tensors)")
|
| 764 |
+
print(f" distinct (format,shape) combos: {len(combos)}")
|
| 765 |
+
|
| 766 |
+
print("\n" + "=" * 78)
|
| 767 |
+
print("CHECK 2 byte round trip + decode equality on real tensors")
|
| 768 |
+
print("=" * 78)
|
| 769 |
+
sample = ["blk.3.attn_q.weight", "blk.3.attn_output.weight", "blk.3.ffn_down.weight",
|
| 770 |
+
"blk.0.attn_qkv.weight", "blk.0.ssm_out.weight", "token_embd.weight"]
|
| 771 |
+
ok = True
|
| 772 |
+
for name in sample:
|
| 773 |
+
if name not in g.tensor:
|
| 774 |
+
print(f" {name}: absent, skipped")
|
| 775 |
+
continue
|
| 776 |
+
ne, tt = g.tinfo(name)
|
| 777 |
+
n, k = ne[1], ne[0]
|
| 778 |
+
fmt = FMT[tt]
|
| 779 |
+
payload = p.direct_ternary(name, (n, k))
|
| 780 |
+
back = disassemble_ternary(fmt, (n, k), payload)
|
| 781 |
+
src = g.payload(name)
|
| 782 |
+
byte_ok = back == src
|
| 783 |
+
a, b = DEQUANT[tt](src), DEQUANT[tt](back)
|
| 784 |
+
dec_ok = bool(np.array_equal(a, b))
|
| 785 |
+
zeros = float(np.mean(a == 0.0))
|
| 786 |
+
|
| 787 |
+
# OFFSET SELF-CHECK -- the decisive one. bytes_equal only proves the assembly is
|
| 788 |
+
# self-consistent; reading a wrong file offset would ALSO give bytes_equal, because
|
| 789 |
+
# the same wrong bytes go in and come out. So validate the CONTENT: a genuine group
|
| 790 |
+
# scale is a tight positive cluster (few distinct high bytes, all finite, none
|
| 791 |
+
# negative), whereas a mis-placed read draws scale words from arbitrary positions and
|
| 792 |
+
# looks like a uniform uint16 sample with many NaN/inf and negative values.
|
| 793 |
+
arr = np.frombuffer(src, dtype=np.uint8).reshape(-1, 34)
|
| 794 |
+
words = arr[:, 0:2].copy().view(np.uint16).reshape(-1)
|
| 795 |
+
hi_distinct = int(np.unique((words >> 8).astype(np.uint8)).size)
|
| 796 |
+
d = arr[:, 0:2].copy().view(np.float16).astype(np.float32).reshape(-1)
|
| 797 |
+
fin = np.isfinite(d)
|
| 798 |
+
n_nonfinite = int(np.sum(~fin))
|
| 799 |
+
n_neg = int(np.sum(fin & (d < 0)))
|
| 800 |
+
med = float(np.median(d[fin])) if fin.any() else float("nan")
|
| 801 |
+
plaus = (n_nonfinite == 0 and n_neg == 0 and hi_distinct <= 64 and 0.0 < med < 1.0)
|
| 802 |
+
|
| 803 |
+
print(f" {name:<28} {fmt:<14} n={n:<6} k={k:<6} bytes_equal={byte_ok} "
|
| 804 |
+
f"decode_equal={dec_ok} zero_share={zeros:.4f}")
|
| 805 |
+
print(f" {'':<28} scale: hi_distinct={hi_distinct:<4} nonfinite={n_nonfinite} "
|
| 806 |
+
f"negative={n_neg} median={med:.5f} "
|
| 807 |
+
f"{'PLAUSIBLE' if plaus else 'IMPLAUSIBLE -- offset bug?'}")
|
| 808 |
+
ok &= byte_ok and dec_ok and plaus
|
| 809 |
+
|
| 810 |
+
print("\n" + "=" * 78)
|
| 811 |
+
print("CHECK 3 producer smoke test (shapes + sizes only, no artifact written)")
|
| 812 |
+
print("=" * 78)
|
| 813 |
+
for name in ["text/layers/3/attention/query_key", "text/layers/3/attention/gate_value",
|
| 814 |
+
"text/layers/3/mlp/gate_up", "text/layers/3/mlp/down",
|
| 815 |
+
"text/layers/3/attention/output", "text/layers/0/gdn/value_z",
|
| 816 |
+
"text/layers/0/gdn/query_key", "text/layers/0/gdn/output",
|
| 817 |
+
"text/token_embedding", "text/output_head", "text/final_norm",
|
| 818 |
+
"text/hadamard_signs", "text/hadamard_widths"]:
|
| 819 |
+
base = p._by_name.get(name)
|
| 820 |
+
if base is not None:
|
| 821 |
+
shape = base["shape"]
|
| 822 |
+
else:
|
| 823 |
+
shape = dict((n, list(s)) for n, s, _f in SIGN_OBJECTS).get(name)
|
| 824 |
+
if shape is None:
|
| 825 |
+
print(f" {name}: neither in template nor a sign object, skipped")
|
| 826 |
+
continue
|
| 827 |
+
fmt, layout = p.target_format(name)
|
| 828 |
+
spec = {"kind": "tensor", "shape": shape, "format": fmt, "layout": layout}
|
| 829 |
+
data = p.produce(name)
|
| 830 |
+
if data is None:
|
| 831 |
+
print(f" {name:<44} {fmt:<14} {'(borrowed)':>13} SKIP")
|
| 832 |
+
continue
|
| 833 |
+
need = p.payload_bytes(spec)
|
| 834 |
+
got = len(data)
|
| 835 |
+
flag = "OK" if (need is None or got == need) else f"MISMATCH want {need}"
|
| 836 |
+
print(f" {name:<44} {fmt:<14} {got:>13,} {flag}")
|
| 837 |
+
if need is not None and got != need:
|
| 838 |
+
ok = False
|
| 839 |
+
|
| 840 |
+
print("\nRESULT:", "OK" if ok else "FAILED")
|
| 841 |
+
return 0 if ok else 1
|
| 842 |
+
|
| 843 |
+
|
| 844 |
+
def mode_build(g: Gguf, out: str, only_layer: int | None = None) -> int:
|
| 845 |
+
p = Packer(g)
|
| 846 |
+
keep = None
|
| 847 |
+
if only_layer is not None:
|
| 848 |
+
names = {n for n, _s, _f in SIGN_OBJECTS} # the sign objects are appended later
|
| 849 |
+
for o in p.objects:
|
| 850 |
+
n = o["name"]
|
| 851 |
+
if n.startswith("frontend/") or n.startswith("text/hadamard_"):
|
| 852 |
+
names.add(n)
|
| 853 |
+
elif n.startswith(f"text/layers/{only_layer}/"):
|
| 854 |
+
names.add(n)
|
| 855 |
+
keep = names
|
| 856 |
+
|
| 857 |
+
specs = []
|
| 858 |
+
chosen = []
|
| 859 |
+
for o in p.objects + [
|
| 860 |
+
{"name": "text/hadamard_signs", "kind": "tensor", "shape": [28672],
|
| 861 |
+
"format": "FP32", "layout": "contiguous-le-v1"},
|
| 862 |
+
{"name": "text/hadamard_widths", "kind": "tensor", "shape": [3],
|
| 863 |
+
"format": "I32", "layout": "contiguous-le-v1"},
|
| 864 |
+
]:
|
| 865 |
+
if keep is not None and o["name"] not in keep:
|
| 866 |
+
continue
|
| 867 |
+
if o["kind"] == "tensor":
|
| 868 |
+
fmt, layout = p.target_format(o["name"])
|
| 869 |
+
specs.append(TensorSpec(o["name"], tuple(o["shape"]), fmt, layout))
|
| 870 |
+
else:
|
| 871 |
+
specs.append(ResourceSpec(o["name"], o["encoding"], o["bytes"]))
|
| 872 |
+
chosen.append(o)
|
| 873 |
+
|
| 874 |
+
out_path = Path(out)
|
| 875 |
+
if out_path.exists():
|
| 876 |
+
raise SystemExit(f"refusing to overwrite {out_path}")
|
| 877 |
+
|
| 878 |
+
borrowed = Counter()
|
| 879 |
+
borrowed_bytes = 0
|
| 880 |
+
produced_bytes = 0
|
| 881 |
+
with Artifact.open(TEMPLATE) as tpl, \
|
| 882 |
+
ArtifactWriter(out_path, p.identity, specs) as w:
|
| 883 |
+
for o in chosen:
|
| 884 |
+
name = o["name"]
|
| 885 |
+
if name.startswith("text/") and not name.startswith("text/vision"):
|
| 886 |
+
data = None if name.startswith(("text/draft_head",)) else p.produce(name)
|
| 887 |
+
if data is not None:
|
| 888 |
+
w.write(name, data)
|
| 889 |
+
produced_bytes += len(data)
|
| 890 |
+
continue
|
| 891 |
+
obj = tpl.find(name)
|
| 892 |
+
if obj is None:
|
| 893 |
+
raise SystemExit(f"template has no object {name} to borrow")
|
| 894 |
+
mv = tpl.payload(obj)
|
| 895 |
+
borrowed_bytes += len(mv)
|
| 896 |
+
w.write(name, mv)
|
| 897 |
+
del mv # release the mmap view, else Artifact.close() raises BufferError
|
| 898 |
+
borrowed[name.split("/")[0]] += 1
|
| 899 |
+
|
| 900 |
+
size = out_path.stat().st_size
|
| 901 |
+
print(f"\nwrote {out_path}")
|
| 902 |
+
print(f" total file : {size:>15,} B = {size / 2**30:.3f} GiB")
|
| 903 |
+
print(f" text part produced: {produced_bytes:>15,} B = {produced_bytes / 2**30:.3f} GiB")
|
| 904 |
+
print(f" borrowed payloads : {borrowed_bytes:>15,} B = {borrowed_bytes / 2**30:.3f} GiB "
|
| 905 |
+
f"{dict(borrowed)}")
|
| 906 |
+
print(f" objects : {len(chosen)}")
|
| 907 |
+
return 0
|
| 908 |
+
|
| 909 |
+
|
| 910 |
+
def _pop_opt(args, name):
|
| 911 |
+
"""Remove `--name VALUE` / `--name=VALUE` from args; return (value|None, rest)."""
|
| 912 |
+
rest, val, i = [], None, 0
|
| 913 |
+
while i < len(args):
|
| 914 |
+
a = args[i]
|
| 915 |
+
if a == name and i + 1 < len(args):
|
| 916 |
+
val, i = args[i + 1], i + 2
|
| 917 |
+
elif a.startswith(name + "="):
|
| 918 |
+
val, i = a.split("=", 1)[1], i + 1
|
| 919 |
+
else:
|
| 920 |
+
rest.append(a)
|
| 921 |
+
i += 1
|
| 922 |
+
return val, rest
|
| 923 |
+
|
| 924 |
+
|
| 925 |
+
def main() -> int:
|
| 926 |
+
global TEMPLATE, GGUF
|
| 927 |
+
args = sys.argv[1:]
|
| 928 |
+
tpl_opt, args = _pop_opt(args, "--template")
|
| 929 |
+
gguf_opt, args = _pop_opt(args, "--gguf")
|
| 930 |
+
TEMPLATE = tpl_opt or os.environ.get("NINFER_TERNARY_TEMPLATE") or TEMPLATE
|
| 931 |
+
GGUF = gguf_opt or os.environ.get("NINFER_TERNARY_GGUF") or GGUF
|
| 932 |
+
if not args:
|
| 933 |
+
print(__doc__)
|
| 934 |
+
return 2
|
| 935 |
+
if not Path(TEMPLATE).exists():
|
| 936 |
+
raise SystemExit(
|
| 937 |
+
f"模板不存在: {TEMPLATE}\n"
|
| 938 |
+
f" 用 --template <path> 或环境变量 NINFER_TERNARY_TEMPLATE 指定。\n"
|
| 939 |
+
f" 模板必须与目标制品同 schema:weights_id={TEMPLATE_SCHEMA}"
|
| 940 |
+
f"(不是 nvfp4,见 README FAQ)。")
|
| 941 |
+
if not Path(GGUF).exists():
|
| 942 |
+
raise SystemExit(
|
| 943 |
+
f"GGUF 不存在: {GGUF}\n"
|
| 944 |
+
f" 用 --gguf <path> 或环境变量 NINFER_TERNARY_GGUF 指定。\n"
|
| 945 |
+
f" 源码里的默认值是占位符 <WORKSPACE>,必须覆盖。")
|
| 946 |
+
g = Gguf(GGUF)
|
| 947 |
+
if args[0] == "check":
|
| 948 |
+
return mode_check(g)
|
| 949 |
+
if args[0] == "layer3":
|
| 950 |
+
return mode_build(g, args[1], only_layer=3)
|
| 951 |
+
if args[0] == "build":
|
| 952 |
+
return mode_build(g, args[1])
|
| 953 |
+
print(f"unknown mode {args[0]}")
|
| 954 |
+
return 2
|
| 955 |
+
|
| 956 |
+
|
| 957 |
+
if __name__ == "__main__":
|
| 958 |
+
raise SystemExit(main())
|
tools/verify/check_assembly.py
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Assembly audit: is EVERY ternary object a faithful, correctly ordered copy of its source?
|
| 2 |
+
|
| 3 |
+
Byte accounting cannot see a row permutation (sizes, row counts and totals are conserved), and a
|
| 4 |
+
set-membership test cannot see it either (every duplicated row still finds a source). This audit
|
| 5 |
+
therefore compares MULTISETS and, more importantly, checks artifact row a DIRECTLY against the
|
| 6 |
+
source row the mapping rule names for a.
|
| 7 |
+
|
| 8 |
+
Sources are de-interleaved GGUF PQ2_0 blocks; artifacts are row-split planes. Both are reduced to
|
| 9 |
+
the same per-row fingerprint (codes bytes + scale bytes), so the comparison is exact and
|
| 10 |
+
independent of the rotated basis (rows are permuted, never transformed).
|
| 11 |
+
|
| 12 |
+
Usage: check_assembly.py <artifact.ninfer> <pq2.gguf>
|
| 13 |
+
"""
|
| 14 |
+
from __future__ import annotations
|
| 15 |
+
|
| 16 |
+
import hashlib
|
| 17 |
+
import sys
|
| 18 |
+
from collections import Counter
|
| 19 |
+
from pathlib import Path
|
| 20 |
+
|
| 21 |
+
import numpy as np
|
| 22 |
+
|
| 23 |
+
sys.path.insert(0, r"<NINFER_ROOT>")
|
| 24 |
+
sys.path.insert(0, r"<WORKSPACE>\tools")
|
| 25 |
+
from _ternary_ref import Gguf # noqa: E402
|
| 26 |
+
from tools.artifact import container # noqa: E402
|
| 27 |
+
|
| 28 |
+
GROUPS = 40
|
| 29 |
+
CODE_BYTES = 32
|
| 30 |
+
SCALE_BYTES = 2
|
| 31 |
+
V_HEAD_DIM = 128
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def artifact_rows(art, name: str) -> list[bytes]:
|
| 35 |
+
obj = art.find(name)
|
| 36 |
+
payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8)
|
| 37 |
+
rows = int(obj.shape[0])
|
| 38 |
+
groups = int(obj.shape[1]) // 128
|
| 39 |
+
codes = payload[: rows * groups * CODE_BYTES]
|
| 40 |
+
off = rows * groups * CODE_BYTES
|
| 41 |
+
scales = payload[off: off + rows * groups * SCALE_BYTES]
|
| 42 |
+
return [
|
| 43 |
+
hashlib.sha1(codes[r * groups * CODE_BYTES:(r + 1) * groups * CODE_BYTES].tobytes()
|
| 44 |
+
+ scales[r * groups * SCALE_BYTES:(r + 1) * groups * SCALE_BYTES].tobytes()
|
| 45 |
+
).digest()
|
| 46 |
+
for r in range(rows)
|
| 47 |
+
]
|
| 48 |
+
|
| 49 |
+
|
| 50 |
+
def gguf_rows(gguf: Gguf, name: str) -> list[bytes]:
|
| 51 |
+
ne, tt, _ = gguf.tensors[name]
|
| 52 |
+
if tt != 142:
|
| 53 |
+
raise SystemExit(f"{name}: expected PQ2_0 (142), got {tt}")
|
| 54 |
+
rows = int(ne[1])
|
| 55 |
+
groups = int(ne[0]) // 128 # ggml ne[0] is the contiguous K extent
|
| 56 |
+
raw = gguf.raw(name, rows)
|
| 57 |
+
block = np.frombuffer(raw, dtype=np.uint8).reshape(rows, groups, 34)
|
| 58 |
+
codes = np.ascontiguousarray(block[:, :, 2:34]).reshape(rows, groups * CODE_BYTES)
|
| 59 |
+
scales = np.ascontiguousarray(block[:, :, 0:2]).reshape(rows, groups * SCALE_BYTES)
|
| 60 |
+
return [hashlib.sha1(codes[r].tobytes() + scales[r].tobytes()).digest() for r in range(rows)]
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def perm48(i: int) -> int:
|
| 64 |
+
return (i % 3) * 16 + (i // 3)
|
| 65 |
+
|
| 66 |
+
|
| 67 |
+
def head_perm(rows: int) -> list[int]:
|
| 68 |
+
"""48 heads x 128 rows: the permutation must be applied at head granularity."""
|
| 69 |
+
out = []
|
| 70 |
+
for r in range(rows):
|
| 71 |
+
head, inner = divmod(r, V_HEAD_DIM)
|
| 72 |
+
out.append(perm48(head) * V_HEAD_DIM + inner)
|
| 73 |
+
return out
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def audit(label: str, art_keys: list[bytes], src_keys: list[bytes]) -> bool:
|
| 77 |
+
art_counts, src_counts = Counter(art_keys), Counter(src_keys)
|
| 78 |
+
repeated = sum(v - 1 for v in art_counts.values() if v > 1)
|
| 79 |
+
missing = sum(1 for k in src_counts if k not in art_counts)
|
| 80 |
+
order_bad = [a for a in range(len(art_keys)) if art_keys[a] != src_keys[a]]
|
| 81 |
+
ok = (repeated == 0) and (missing == 0) and not order_bad
|
| 82 |
+
print(f"{'OK ' if ok else 'FAIL'} {label}")
|
| 83 |
+
print(f" rows art={len(art_keys)} src={len(src_keys)} | "
|
| 84 |
+
f"distinct art={len(art_counts)} src={len(src_counts)} | "
|
| 85 |
+
f"duplicated={repeated} missing={missing} | "
|
| 86 |
+
f"order mismatches={len(order_bad)}/{len(art_keys)}")
|
| 87 |
+
if order_bad:
|
| 88 |
+
print(f" first bad rows: {order_bad[:8]}")
|
| 89 |
+
return ok
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def main() -> int:
|
| 93 |
+
art_path, gguf_path = sys.argv[1], sys.argv[2]
|
| 94 |
+
gguf = Gguf(Path(gguf_path))
|
| 95 |
+
results = []
|
| 96 |
+
with container.Artifact.open(art_path) as art:
|
| 97 |
+
def A(name):
|
| 98 |
+
return artifact_rows(art, name)
|
| 99 |
+
def G(name):
|
| 100 |
+
return gguf_rows(gguf, name)
|
| 101 |
+
|
| 102 |
+
# --- globals ---------------------------------------------------------
|
| 103 |
+
results.append(audit("text/token_embedding <- token_embd.weight",
|
| 104 |
+
A("text/token_embedding"), G("token_embd.weight")))
|
| 105 |
+
results.append(audit("text/output_head <- output.weight",
|
| 106 |
+
A("text/output_head"), G("output.weight")))
|
| 107 |
+
|
| 108 |
+
for layer, kind in ((0, "gdn"), (3, "attention")):
|
| 109 |
+
pre = f"blk.{layer}."
|
| 110 |
+
la = f"text/layers/{layer}/"
|
| 111 |
+
print(f"--- layer {layer} ({kind}) ---")
|
| 112 |
+
results.append(audit(f"{la}mlp/down <- ffn_down",
|
| 113 |
+
A(la + "mlp/down"), G(pre + "ffn_down.weight")))
|
| 114 |
+
results.append(audit(f"{la}mlp/gate_up <- concat(ffn_gate, ffn_up)",
|
| 115 |
+
A(la + "mlp/gate_up"),
|
| 116 |
+
G(pre + "ffn_gate.weight") + G(pre + "ffn_up.weight")))
|
| 117 |
+
if kind == "gdn":
|
| 118 |
+
results.append(audit(f"{la}gdn/query_key <- attn_qkv[0:4096]",
|
| 119 |
+
A(la + "gdn/query_key"),
|
| 120 |
+
G(pre + "attn_qkv.weight")[:4096]))
|
| 121 |
+
results.append(audit(f"{la}gdn/output <- ssm_out",
|
| 122 |
+
A(la + "gdn/output"), G(pre + "ssm_out.weight")))
|
| 123 |
+
# value/z halves: compare against the source reordered by the head permutation
|
| 124 |
+
vz = A(la + "gdn/value_z")
|
| 125 |
+
v_src = G(pre + "attn_qkv.weight")[4096:10240]
|
| 126 |
+
z_src = G(pre + "attn_gate.weight")[:6144]
|
| 127 |
+
perm = head_perm(6144)
|
| 128 |
+
results.append(audit(f"{la}gdn/value_z <- attn_qkv[4096:10240] (head perm)",
|
| 129 |
+
vz[:6144], [v_src[p] for p in perm]))
|
| 130 |
+
results.append(audit(f"{la}gdn/value_z <- attn_gate[0:6144] (head perm)",
|
| 131 |
+
vz[6144:], [z_src[p] for p in perm]))
|
| 132 |
+
else:
|
| 133 |
+
aq = G(pre + "attn_q.weight")
|
| 134 |
+
even = [c * 256 + r for c in range(0, 48, 2) for r in range(256)]
|
| 135 |
+
odd = [c * 256 + r for c in range(1, 48, 2) for r in range(256)]
|
| 136 |
+
qk = A(la + "attention/query_key")
|
| 137 |
+
gv = A(la + "attention/gate_value")
|
| 138 |
+
results.append(audit(f"{la}attention/query_key <- attn_q even chunks",
|
| 139 |
+
qk[:6144], [aq[r] for r in even]))
|
| 140 |
+
results.append(audit(f"{la}attention/query_key <- attn_k",
|
| 141 |
+
qk[6144:], G(pre + "attn_k.weight")[:1024]))
|
| 142 |
+
results.append(audit(f"{la}attention/gate_value <- attn_q odd chunks",
|
| 143 |
+
gv[:6144], [aq[r] for r in odd]))
|
| 144 |
+
results.append(audit(f"{la}attention/gate_value <- attn_v",
|
| 145 |
+
gv[6144:], G(pre + "attn_v.weight")[:1024]))
|
| 146 |
+
results.append(audit(f"{la}attention/output <- attn_output",
|
| 147 |
+
A(la + "attention/output"), G(pre + "attn_output.weight")))
|
| 148 |
+
|
| 149 |
+
passed = sum(1 for r in results if r)
|
| 150 |
+
print(f"\nRESULT: {passed}/{len(results)} assembly rules OK")
|
| 151 |
+
return 0 if passed == len(results) else 1
|
| 152 |
+
|
| 153 |
+
|
| 154 |
+
if __name__ == "__main__":
|
| 155 |
+
raise SystemExit(main())
|
tools/verify/check_embedding.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Oracle for the ternary embedding path: gather (PQ2_0 decode) + inverse folded-basis mapping.
|
| 2 |
+
|
| 3 |
+
The residual stream enters layer 0 at the embedding output, so a break here makes every downstream
|
| 4 |
+
number meaningless -- which is exactly what a perplexity close to uniform looks like. This compares
|
| 5 |
+
the engine's dumped embedding activation against an independent numpy rebuild of
|
| 6 |
+
|
| 7 |
+
h = s * (H * z), z = decode(token_embedding_row[id])
|
| 8 |
+
|
| 9 |
+
where z comes straight out of the artifact payload (decoded independently of the engine).
|
| 10 |
+
|
| 11 |
+
Negative controls are mandatory: a near-uniform or all-zeros dump would otherwise "match" nothing,
|
| 12 |
+
and the unrotated variant would match if the engine simply skipped the mapping.
|
| 13 |
+
|
| 14 |
+
Usage: check_embedding.py <artifact.ninfer> <dump-file.T<N>>
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import os
|
| 19 |
+
import sys
|
| 20 |
+
|
| 21 |
+
import numpy as np
|
| 22 |
+
|
| 23 |
+
sys.path.insert(0, r"<NINFER_ROOT>")
|
| 24 |
+
from tools.artifact import container # noqa: E402
|
| 25 |
+
|
| 26 |
+
BLOCK = 1024
|
| 27 |
+
GROUPS = 40
|
| 28 |
+
POP = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
|
| 29 |
+
|
| 30 |
+
|
| 31 |
+
def popcount32(a: np.ndarray) -> np.ndarray:
|
| 32 |
+
v = a.astype(np.uint32)
|
| 33 |
+
return (POP[v & 0xFF].astype(np.uint16) + POP[(v >> 8) & 0xFF].astype(np.uint16)
|
| 34 |
+
+ POP[(v >> 16) & 0xFF].astype(np.uint16) + POP[(v >> 24) & 0xFF].astype(np.uint16))
|
| 35 |
+
|
| 36 |
+
|
| 37 |
+
def hadamard(n: int) -> np.ndarray:
|
| 38 |
+
idx = np.arange(n, dtype=np.uint32)
|
| 39 |
+
parity = popcount32(idx[:, None] & idx[None, :]) & 1
|
| 40 |
+
return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def main() -> int:
|
| 44 |
+
art_path, dump_path = sys.argv[1], sys.argv[2]
|
| 45 |
+
raw = np.fromfile(dump_path, dtype=np.uint8)
|
| 46 |
+
tokens, hidden, id_count = np.frombuffer(raw[:12], dtype=np.int32)
|
| 47 |
+
ids = np.frombuffer(raw[12:12 + id_count * 4], dtype=np.int32)
|
| 48 |
+
bf16 = np.frombuffer(raw[12 + id_count * 4:], dtype=np.uint16).reshape(tokens, hidden)
|
| 49 |
+
# The op writes out as [hidden, T] row-major, so element (k, t) sits at k*T + t. Reshaping the
|
| 50 |
+
# flat buffer as (T, hidden) would reinterpret that as t*hidden + k -- a silent transpose that
|
| 51 |
+
# decorrelates everything for T > 1 and looks like a total engine failure.
|
| 52 |
+
got = (bf16.astype(np.uint32) << 16).view(np.float32).astype(np.float64).reshape(hidden, tokens)
|
| 53 |
+
print(f"dump: tokens={tokens} hidden={hidden} ids={id_count}")
|
| 54 |
+
print(f"ids[:12]={ids[:12].tolist()}")
|
| 55 |
+
|
| 56 |
+
with container.Artifact.open(art_path) as art:
|
| 57 |
+
obj = art.find("text/token_embedding")
|
| 58 |
+
payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8).copy()
|
| 59 |
+
signs_all = np.frombuffer(bytes(art.payload("text/hadamard_signs")), dtype=np.float32)
|
| 60 |
+
widths = np.frombuffer(bytes(art.payload("text/hadamard_widths")), dtype=np.int32)
|
| 61 |
+
rows = int(obj.shape[0])
|
| 62 |
+
off = 0
|
| 63 |
+
sign_offsets = {}
|
| 64 |
+
for w in widths:
|
| 65 |
+
sign_offsets[int(w)] = off
|
| 66 |
+
off += int(w)
|
| 67 |
+
signs = signs_all[sign_offsets[hidden]:sign_offsets[hidden] + hidden].astype(np.float64)
|
| 68 |
+
|
| 69 |
+
codes = payload[: rows * GROUPS * 32].reshape(rows, GROUPS, 32)
|
| 70 |
+
scale_off = rows * GROUPS * 32
|
| 71 |
+
scales = payload[scale_off: scale_off + rows * GROUPS * 2].view(np.float16).astype(
|
| 72 |
+
np.float64).reshape(rows, GROUPS)
|
| 73 |
+
|
| 74 |
+
def decode_row(index: int) -> np.ndarray:
|
| 75 |
+
q = codes[index] # (groups, 32 bytes)
|
| 76 |
+
code = (q[:, :, None] >> (2 * np.arange(4, dtype=np.uint8))[None, None, :]) & 3
|
| 77 |
+
flat = code.reshape(-1).astype(np.float64) # 32 bytes x 4 codes x groups
|
| 78 |
+
return (flat - 1.0) * np.repeat(scales[index], 128)
|
| 79 |
+
|
| 80 |
+
H = hadamard(BLOCK)
|
| 81 |
+
n_block = hidden // BLOCK
|
| 82 |
+
|
| 83 |
+
def inverse(vec: np.ndarray) -> np.ndarray:
|
| 84 |
+
out = np.empty_like(vec)
|
| 85 |
+
for b in range(n_block):
|
| 86 |
+
lo = b * BLOCK
|
| 87 |
+
out[lo:lo + BLOCK] = signs[lo:lo + BLOCK] * (H @ vec[lo:lo + BLOCK])
|
| 88 |
+
return out
|
| 89 |
+
|
| 90 |
+
def forward(vec: np.ndarray) -> np.ndarray:
|
| 91 |
+
out = np.empty_like(vec)
|
| 92 |
+
for b in range(n_block):
|
| 93 |
+
lo = b * BLOCK
|
| 94 |
+
out[lo:lo + BLOCK] = H @ (signs[lo:lo + BLOCK] * vec[lo:lo + BLOCK])
|
| 95 |
+
return out
|
| 96 |
+
|
| 97 |
+
checked = 0
|
| 98 |
+
worst = 0.0
|
| 99 |
+
for t in range(min(tokens, 8)):
|
| 100 |
+
z = decode_row(int(ids[t]))
|
| 101 |
+
ref = inverse(z)
|
| 102 |
+
col = got[:, t]
|
| 103 |
+
if t < 3:
|
| 104 |
+
def stats(name, v):
|
| 105 |
+
print(f" {name}: ||v||={np.linalg.norm(v):.6g} mean={v.mean():+.4g} "
|
| 106 |
+
f"std={v.std():.4g} min={v.min():+.4g} max={v.max():+.4g} "
|
| 107 |
+
f"nan={int(np.isnan(v).sum())}")
|
| 108 |
+
print(f" token {t} id={int(ids[t])}")
|
| 109 |
+
stats("engine dump", col)
|
| 110 |
+
stats("decoded row z", z)
|
| 111 |
+
stats("expected s*(H*z)", ref)
|
| 112 |
+
# is the engine maybe returning a DIFFERENT token's row, or a mis-strided gather?
|
| 113 |
+
for shift in (-2, -1, 1, 2):
|
| 114 |
+
if 0 <= t + shift < tokens:
|
| 115 |
+
other = inverse(decode_row(int(ids[t + shift])))
|
| 116 |
+
c = float(np.dot(col, other) /
|
| 117 |
+
max(np.linalg.norm(col) * np.linalg.norm(other), 1e-30))
|
| 118 |
+
print(f" cos vs token{shift:+d} row = {c:+.4f}")
|
| 119 |
+
rel = float(np.linalg.norm(col - ref) / np.linalg.norm(ref))
|
| 120 |
+
cos = float(np.dot(col, ref) / (np.linalg.norm(col) * np.linalg.norm(ref)))
|
| 121 |
+
ctrl_raw = float(np.dot(col, z) / (np.linalg.norm(col) * np.linalg.norm(z)))
|
| 122 |
+
ctrl_fwd = forward(z)
|
| 123 |
+
ctrl_fwd_cos = float(np.dot(col, ctrl_fwd) / (np.linalg.norm(col) * np.linalg.norm(ctrl_fwd)))
|
| 124 |
+
if t < 4:
|
| 125 |
+
print(f" rel_l2={rel:.4e} cos={cos:+.6f} | "
|
| 126 |
+
f"control cos(unrotated)={ctrl_raw:+.4f} cos(signs-first)={ctrl_fwd_cos:+.4f}")
|
| 127 |
+
worst = max(worst, rel)
|
| 128 |
+
checked += 1
|
| 129 |
+
print(f"\nchecked {checked} tokens, worst rel_l2 = {worst:.4e}")
|
| 130 |
+
print(f"RESULT: {'PASS' if worst <= 0.02 else 'FAIL'} (engine embedding == independent rebuild)")
|
| 131 |
+
return 0 if worst <= 0.02 else 1
|
| 132 |
+
|
| 133 |
+
|
| 134 |
+
if __name__ == "__main__":
|
| 135 |
+
raise SystemExit(main())
|
tools/verify/check_row_order.py
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Verify the two INFERRED row-order rules by exact row fingerprints.
|
| 2 |
+
|
| 3 |
+
MAPPING.json flags exactly two rules as inferred rather than measured, and both are row
|
| 4 |
+
permutations of whole quantized rows -- which means each row's bytes survive untouched and can be
|
| 5 |
+
matched exactly between the GGUF and the artifact:
|
| 6 |
+
|
| 7 |
+
gdn_value_z "risk: if the engine test disagrees, flip this single permutation"
|
| 8 |
+
attn_q_per_head_interleave "INFERRED for main layers from the MTP layer"
|
| 9 |
+
|
| 10 |
+
Nothing about the rotated basis matters here: rows are permuted, not transformed, so a byte-level
|
| 11 |
+
row fingerprint is an exact test of the rule (this is the "structural self-proof" the handoff asks
|
| 12 |
+
for instead of value correlation, which is meaningless across bases).
|
| 13 |
+
|
| 14 |
+
Usage: check_row_order.py <artifact.ninfer> <pq2.gguf>
|
| 15 |
+
"""
|
| 16 |
+
from __future__ import annotations
|
| 17 |
+
|
| 18 |
+
import hashlib
|
| 19 |
+
import sys
|
| 20 |
+
from pathlib import Path
|
| 21 |
+
|
| 22 |
+
import numpy as np
|
| 23 |
+
|
| 24 |
+
sys.path.insert(0, r"<NINFER_ROOT>")
|
| 25 |
+
sys.path.insert(0, r"<WORKSPACE>\tools")
|
| 26 |
+
from _ternary_ref import Gguf # noqa: E402
|
| 27 |
+
from tools.artifact import container # noqa: E402
|
| 28 |
+
|
| 29 |
+
GROUPS = 40
|
| 30 |
+
CODE_BYTES = 32
|
| 31 |
+
SCALE_BYTES = 2
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
def artifact_row_keys(art, name: str) -> list[bytes]:
|
| 35 |
+
"""Fingerprint every row of a PQ2_0 row-split object as (codes, scales) digests."""
|
| 36 |
+
obj = art.find(name)
|
| 37 |
+
payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8)
|
| 38 |
+
rows = int(obj.shape[0])
|
| 39 |
+
assert int(obj.shape[1]) == GROUPS * 128, obj.shape
|
| 40 |
+
codes = payload[: rows * GROUPS * CODE_BYTES]
|
| 41 |
+
scale_start = rows * GROUPS * CODE_BYTES
|
| 42 |
+
scales = payload[scale_start: scale_start + rows * GROUPS * SCALE_BYTES]
|
| 43 |
+
keys: list[bytes] = []
|
| 44 |
+
for r in range(rows):
|
| 45 |
+
c = codes[r * GROUPS * CODE_BYTES:(r + 1) * GROUPS * CODE_BYTES]
|
| 46 |
+
s = scales[r * GROUPS * SCALE_BYTES:(r + 1) * GROUPS * SCALE_BYTES]
|
| 47 |
+
keys.append(hashlib.sha1(bytes(c) + bytes(s)).digest())
|
| 48 |
+
return keys
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def gguf_row_keys(gguf: Gguf, name: str) -> list[bytes]:
|
| 52 |
+
"""De-interleave GGUF {fp16 d; uint8 qs[32]} blocks into the same (codes, scales) fingerprint."""
|
| 53 |
+
ne, tt, _ = gguf.tensors[name]
|
| 54 |
+
assert tt == 142, (name, tt) # PQ2_0
|
| 55 |
+
rows = int(ne[1])
|
| 56 |
+
raw = gguf.raw(name, rows)
|
| 57 |
+
block = np.frombuffer(raw, dtype=np.uint8).reshape(rows, GROUPS, 34)
|
| 58 |
+
codes = np.ascontiguousarray(block[:, :, 2:34]).reshape(rows, GROUPS * CODE_BYTES)
|
| 59 |
+
scales = np.ascontiguousarray(block[:, :, 0:2]).reshape(rows, GROUPS * SCALE_BYTES)
|
| 60 |
+
return [hashlib.sha1(codes[r].tobytes() + scales[r].tobytes()).digest() for r in range(rows)]
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def report(label: str, art_keys: list[bytes], gguf_subset: list[bytes],
|
| 64 |
+
src_index: list[int], expected) -> bool:
|
| 65 |
+
"""Check the artifact rows two ways.
|
| 66 |
+
|
| 67 |
+
The multiset test proves the packer took the right source rows. The permutation test is done by
|
| 68 |
+
comparing the artifact row at index a directly against the source row the rule NAMES for a --
|
| 69 |
+
never by searching for a matching fingerprint first, because identical quantized rows do occur
|
| 70 |
+
and a first-hit search would then report a harmless false mismatch.
|
| 71 |
+
"""
|
| 72 |
+
table: dict[bytes, list[int]] = {}
|
| 73 |
+
for r, key in enumerate(gguf_subset):
|
| 74 |
+
table.setdefault(key, []).append(r)
|
| 75 |
+
matched = sum(1 for key in art_keys if key in table)
|
| 76 |
+
ok_set = matched == len(art_keys)
|
| 77 |
+
unique_art = len(set(art_keys))
|
| 78 |
+
unique_src = len(set(gguf_subset))
|
| 79 |
+
print(f"{label}: artifact rows={len(art_keys)} source rows={len(gguf_subset)} "
|
| 80 |
+
f"matched={matched} -> {'row SET matches' if ok_set else 'ROW SET MISMATCH'}")
|
| 81 |
+
print(f" distinct fingerprints: artifact={unique_art} source={unique_src} "
|
| 82 |
+
f"(duplicates make fingerprint-first search unreliable, hence the direct check)")
|
| 83 |
+
|
| 84 |
+
direct = [a for a, src in enumerate(src_index)
|
| 85 |
+
if art_keys[a] != gguf_subset[src]]
|
| 86 |
+
perm_ok = not direct
|
| 87 |
+
print(f" documented rule holds row-by-row: {perm_ok}"
|
| 88 |
+
+ (f" ({len(direct)} of {len(art_keys)} rows disagree)" if not perm_ok else ""))
|
| 89 |
+
|
| 90 |
+
# Is the artifact a genuine PERMUTATION of the source rows? A repack that reshaped the wrong
|
| 91 |
+
# axis repeats some rows and drops others, which a "does every row exist somewhere" test cannot
|
| 92 |
+
# see: every row still matches something. Compare the multisets explicitly.
|
| 93 |
+
from collections import Counter
|
| 94 |
+
art_counts = Counter(art_keys)
|
| 95 |
+
src_counts = Counter(gguf_subset)
|
| 96 |
+
repeated = {k: v for k, v in art_counts.items() if v > 1}
|
| 97 |
+
missing = [k for k in src_counts if k not in art_counts]
|
| 98 |
+
extra_rows = sum(v - 1 for v in repeated.values())
|
| 99 |
+
print(f" multiset: artifact {len(art_keys)} rows / {len(art_counts)} distinct; "
|
| 100 |
+
f"source {len(gguf_subset)} rows / {len(src_counts)} distinct")
|
| 101 |
+
print(f" rows DUPLICATED in the artifact: {len(repeated)} fingerprints covering "
|
| 102 |
+
f"{extra_rows} redundant rows; source rows MISSING from the artifact: {len(missing)}")
|
| 103 |
+
if missing or repeated:
|
| 104 |
+
print(" -> NOT a permutation: the packing permutation duplicates and drops rows")
|
| 105 |
+
if not perm_ok:
|
| 106 |
+
print(" first divergent rows (artifact row -> named source vs what it actually equals):")
|
| 107 |
+
shown = 0
|
| 108 |
+
for a in direct[:6]:
|
| 109 |
+
hits = table.get(art_keys[a], [])
|
| 110 |
+
print(f" {a} -> named {src_index[a]}, actually equals source rows {hits[:4]}")
|
| 111 |
+
shown += 1
|
| 112 |
+
print(f" artifact first 24 -> named sources: {src_index[:24]}")
|
| 113 |
+
return ok_set and perm_ok
|
| 114 |
+
|
| 115 |
+
|
| 116 |
+
def tiled_to_grouped_index(grouped_row: int, heads: int = 48, per: int = 3) -> int:
|
| 117 |
+
"""grouped row (nk, rep, hd) -> tiled source row (rep, nk, hd), with 128 values per head."""
|
| 118 |
+
hd = grouped_row % 128
|
| 119 |
+
rest = grouped_row // 128
|
| 120 |
+
nk, rep = rest // per, rest % per
|
| 121 |
+
return rep * (heads // per) * 128 + nk * 128 + hd
|
| 122 |
+
|
| 123 |
+
|
| 124 |
+
def main() -> int:
|
| 125 |
+
art_path, gguf_path = sys.argv[1], sys.argv[2]
|
| 126 |
+
gguf = Gguf(Path(gguf_path))
|
| 127 |
+
ok = True
|
| 128 |
+
with container.Artifact.open(art_path) as art:
|
| 129 |
+
# ---- rule: gdn_value_z (48 layers) ----------------------------------
|
| 130 |
+
art_keys = artifact_row_keys(art, "text/layers/0/gdn/value_z")
|
| 131 |
+
value_src = gguf_row_keys(gguf, "blk.0.attn_qkv.weight")[4096:10240] # 6144 V rows
|
| 132 |
+
gate_src = gguf_row_keys(gguf, "blk.0.attn_gate.weight")[0:6144] # 6144 z rows
|
| 133 |
+
print("== gdn_value_z: artifact rows [0:6144] vs gguf attn_qkv[4096:10240]")
|
| 134 |
+
ok &= report(" value part", art_keys[0:6144], value_src,
|
| 135 |
+
[tiled_to_grouped_index(r) for r in range(6144)], None)
|
| 136 |
+
print("== gdn_value_z: artifact rows [6144:12288] vs gguf attn_gate[0:6144]")
|
| 137 |
+
ok &= report(" z part", art_keys[6144:12288], gate_src,
|
| 138 |
+
[tiled_to_grouped_index(r) for r in range(6144)], None)
|
| 139 |
+
|
| 140 |
+
# ---- rule: attn_q_per_head_interleave (16 layers) -------------------
|
| 141 |
+
qk_keys = artifact_row_keys(art, "text/layers/3/attention/query_key")
|
| 142 |
+
attn_q = gguf_row_keys(gguf, "blk.3.attn_q.weight") # 12288 rows: 48 chunks of 256
|
| 143 |
+
even = [r for chunk in range(0, 48, 2) for r in range(chunk * 256, chunk * 256 + 256)]
|
| 144 |
+
print("== attn_q_per_head_interleave: artifact query rows [0:6144] vs even chunks of attn_q")
|
| 145 |
+
ok &= report(" query part", qk_keys[0:6144], [attn_q[r] for r in even],
|
| 146 |
+
list(range(6144)), None)
|
| 147 |
+
attn_k = gguf_row_keys(gguf, "blk.3.attn_k.weight")
|
| 148 |
+
print("== attn key part: artifact rows [6144:7168] vs gguf attn_k[0:1024]")
|
| 149 |
+
ok &= report(" key part", qk_keys[6144:7168], attn_k[0:1024], list(range(1024)), None)
|
| 150 |
+
|
| 151 |
+
gv_keys = artifact_row_keys(art, "text/layers/3/attention/gate_value")
|
| 152 |
+
odd = [r for chunk in range(1, 48, 2) for r in range(chunk * 256, chunk * 256 + 256)]
|
| 153 |
+
print("== attn_q_per_head_interleave: artifact gate rows [0:6144] vs odd chunks of attn_q")
|
| 154 |
+
ok &= report(" gate part", gv_keys[0:6144], [attn_q[r] for r in odd],
|
| 155 |
+
list(range(6144)), None)
|
| 156 |
+
print(f"\nRESULT: {'all row-order rules hold' if ok else 'ROW-ORDER RULE MISMATCH FOUND'}")
|
| 157 |
+
return 0 if ok else 1
|
| 158 |
+
|
| 159 |
+
|
| 160 |
+
if __name__ == "__main__":
|
| 161 |
+
raise SystemExit(main())
|
tools/verify/gemm_oracle.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""Dump / check the ternary GEMM + rotation integration against an independent reference.
|
| 2 |
+
|
| 3 |
+
Two modes:
|
| 4 |
+
|
| 5 |
+
dump <artifact> <object> <dump-dir>
|
| 6 |
+
Writes the object's row-split payload, the sign block for its input width, and a
|
| 7 |
+
deterministic activation, so a standalone nvcc harness can run the REAL kernels.
|
| 8 |
+
|
| 9 |
+
check <artifact> <object> <dump-dir>
|
| 10 |
+
Decodes the payload in pure numpy (the arrangement was already proven by
|
| 11 |
+
check_payload_order.py), rebuilds y = W' * (H * (s * (P * x))) explicitly in float64 and
|
| 12 |
+
compares against the kernel dump.
|
| 13 |
+
|
| 14 |
+
The point is to cover what no byte/size/oracle check so far covers: that the plane pointers, the
|
| 15 |
+
2-bit decode, the sign block for THIS width, and the rotation compose into the documented math.
|
| 16 |
+
|
| 17 |
+
Usage: gemm_oracle.py dump|check <artifact.ninfer> <object-name> <dump-dir>
|
| 18 |
+
"""
|
| 19 |
+
from __future__ import annotations
|
| 20 |
+
|
| 21 |
+
import os
|
| 22 |
+
import sys
|
| 23 |
+
|
| 24 |
+
import numpy as np
|
| 25 |
+
|
| 26 |
+
sys.path.insert(0, r"<NINFER_ROOT>")
|
| 27 |
+
from tools.artifact import container # noqa: E402
|
| 28 |
+
|
| 29 |
+
BLOCK = 1024
|
| 30 |
+
POPCOUNT = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def popcount32(a: np.ndarray) -> np.ndarray:
|
| 34 |
+
v = a.astype(np.uint32)
|
| 35 |
+
return (
|
| 36 |
+
POPCOUNT[v & 0xFF].astype(np.uint16)
|
| 37 |
+
+ POPCOUNT[(v >> 8) & 0xFF].astype(np.uint16)
|
| 38 |
+
+ POPCOUNT[(v >> 16) & 0xFF].astype(np.uint16)
|
| 39 |
+
+ POPCOUNT[(v >> 24) & 0xFF].astype(np.uint16)
|
| 40 |
+
)
|
| 41 |
+
|
| 42 |
+
|
| 43 |
+
def hadamard(n: int) -> np.ndarray:
|
| 44 |
+
idx = np.arange(n, dtype=np.uint32)
|
| 45 |
+
parity = popcount32(idx[:, None] & idx[None, :]) & 1
|
| 46 |
+
return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
def make_x(k: int, tokens: int, seed: int = 12345) -> np.ndarray:
|
| 50 |
+
state = np.uint32(seed)
|
| 51 |
+
out = np.empty(k * tokens, dtype=np.float32)
|
| 52 |
+
for i in range(out.size):
|
| 53 |
+
state = np.uint32(state * np.uint32(1664525) + np.uint32(1013904223))
|
| 54 |
+
out[i] = np.float32((int(state >> 8) & 0xFFFF) / 32768.0 - 1.0)
|
| 55 |
+
return out.reshape(k, tokens)
|
| 56 |
+
|
| 57 |
+
|
| 58 |
+
def load(artifact_path: str, name: str):
|
| 59 |
+
with container.Artifact.open(artifact_path) as art:
|
| 60 |
+
obj = art.find(name)
|
| 61 |
+
payload = np.frombuffer(bytes(art.payload(obj)), dtype=np.uint8).copy()
|
| 62 |
+
signs_all = np.frombuffer(bytes(art.payload("text/hadamard_signs")), dtype=np.float32)
|
| 63 |
+
widths = np.frombuffer(bytes(art.payload("text/hadamard_widths")), dtype=np.int32)
|
| 64 |
+
rows, cols = int(obj.shape[0]), int(obj.shape[1])
|
| 65 |
+
offsets, acc = {}, 0
|
| 66 |
+
for w in widths:
|
| 67 |
+
offsets[int(w)] = acc
|
| 68 |
+
acc += int(w)
|
| 69 |
+
signs = signs_all[offsets[cols]:offsets[cols] + cols].copy()
|
| 70 |
+
return payload, signs, rows, cols, obj.format
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
def decode_pq2(payload: np.ndarray, rows: int, groups: int) -> np.ndarray:
|
| 74 |
+
"""Same arrangement the row-split layout declares, decoded exactly like ggml's PQ2_0."""
|
| 75 |
+
codes = payload[: rows * groups * 32].reshape(rows, groups, 32)
|
| 76 |
+
scales = payload[rows * groups * 32: rows * groups * 32 + rows * groups * 2].view(
|
| 77 |
+
np.float16).astype(np.float64).reshape(rows, groups)
|
| 78 |
+
codes = codes.reshape(rows, groups, 32, 1)
|
| 79 |
+
codes = (codes >> (2 * np.arange(4, dtype=np.uint8)).reshape(1, 1, 1, 4)) & 3 # [rows,groups,32,4]
|
| 80 |
+
codes = codes.reshape(rows, groups, 128).astype(np.float64)
|
| 81 |
+
weights = (codes - 1.0) * scales[:, :, None]
|
| 82 |
+
return weights.reshape(rows, groups * 128)
|
| 83 |
+
|
| 84 |
+
|
| 85 |
+
def main() -> int:
|
| 86 |
+
mode, artifact_path, name, dump_dir = sys.argv[1], sys.argv[2], sys.argv[3], sys.argv[4]
|
| 87 |
+
payload, signs, rows, cols, fmt = load(artifact_path, name)
|
| 88 |
+
groups = cols // 128
|
| 89 |
+
if fmt != "PQ2_0_G128":
|
| 90 |
+
print(f"this probe only implements PQ2_0; got {fmt}")
|
| 91 |
+
return 2
|
| 92 |
+
x = make_x(cols, 1)
|
| 93 |
+
os.makedirs(dump_dir, exist_ok=True)
|
| 94 |
+
|
| 95 |
+
if mode == "dump":
|
| 96 |
+
payload.tofile(os.path.join(dump_dir, "gemm.payload.bin"))
|
| 97 |
+
signs.astype(np.float32).tofile(os.path.join(dump_dir, "gemm.signs.f32"))
|
| 98 |
+
x.astype(np.float32).tofile(os.path.join(dump_dir, "gemm.x.f32"))
|
| 99 |
+
print(f"dumped {name}: payload={payload.size} B rows={rows} cols={cols} groups={groups} "
|
| 100 |
+
f"scale_plane_off={rows * groups * 32}")
|
| 101 |
+
return 0
|
| 102 |
+
|
| 103 |
+
weights = decode_pq2(payload, rows, groups)
|
| 104 |
+
print(f"decoded W' {weights.shape} finite={np.isfinite(weights).all()} "
|
| 105 |
+
f"absmax={np.abs(weights).max():.6g}")
|
| 106 |
+
|
| 107 |
+
H = hadamard(BLOCK)
|
| 108 |
+
xr = np.empty_like(x, dtype=np.float64)
|
| 109 |
+
for b in range(cols // BLOCK):
|
| 110 |
+
lo = b * BLOCK
|
| 111 |
+
xr[lo:lo + BLOCK, 0] = H @ (signs[lo:lo + BLOCK].astype(np.float64) * x[lo:lo + BLOCK, 0])
|
| 112 |
+
y_ref = weights @ xr[:, 0]
|
| 113 |
+
|
| 114 |
+
y_got = np.fromfile(os.path.join(dump_dir, "gemm.y.f32"), dtype=np.float32).astype(np.float64)
|
| 115 |
+
if y_got.size != rows:
|
| 116 |
+
print(f"kernel dump has {y_got.size} rows, expected {rows}")
|
| 117 |
+
return 1
|
| 118 |
+
|
| 119 |
+
rel = float(np.linalg.norm(y_got - y_ref) / np.linalg.norm(y_ref))
|
| 120 |
+
diff = float(np.abs(y_got - y_ref).max())
|
| 121 |
+
# negative controls: if the kernel disagreed with the documented math in any of these ways the
|
| 122 |
+
# match would collapse, so report them to prove the comparison has teeth
|
| 123 |
+
y_norot = weights @ x[:, 0]
|
| 124 |
+
y_nosign = np.empty_like(x, dtype=np.float64)
|
| 125 |
+
for b in range(cols // BLOCK):
|
| 126 |
+
lo = b * BLOCK
|
| 127 |
+
y_nosign[lo:lo + BLOCK, 0] = H @ x[lo:lo + BLOCK, 0]
|
| 128 |
+
y_nosign = weights @ y_nosign[:, 0]
|
| 129 |
+
y_unnorm = weights @ (xr[:, 0] * np.sqrt(BLOCK))
|
| 130 |
+
print(f" rel_l2 = {rel:.4e} max|diff| = {diff:.4e} {'PASS' if rel <= 0.02 else 'FAIL'}")
|
| 131 |
+
for label, ref in (("no rotation", y_norot), ("no signs", y_nosign),
|
| 132 |
+
("unnormalized H", y_unnorm)):
|
| 133 |
+
r = float(np.linalg.norm(y_got - ref) / np.linalg.norm(ref))
|
| 134 |
+
print(f" control {label:16s} rel_l2={r:.4f} "
|
| 135 |
+
f"{'(differs OK)' if r > 0.05 else '(TOO CLOSE)'}")
|
| 136 |
+
return 0 if rel <= 0.02 else 1
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
if __name__ == "__main__":
|
| 140 |
+
raise SystemExit(main())
|
tools/verify/oracle_rot.py
ADDED
|
@@ -0,0 +1,168 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""numpy oracle for the folded-basis rotation kernels (forward and inverse).
|
| 2 |
+
|
| 3 |
+
Reads the raw f32 dumps the standalone nvcc harness wrote, rebuilds the documented transform
|
| 4 |
+
explicitly, and compares against the REAL kernel output element for element.
|
| 5 |
+
|
| 6 |
+
The pass criterion is deliberately two-sided: the documented variant must match, AND a family of
|
| 7 |
+
plausible-but-wrong variants must all fail. A one-sided check cannot tell "the kernel is right"
|
| 8 |
+
from "the oracle is too loose" -- which is exactly the trap the handoff notes call a
|
| 9 |
+
self-consistent false green.
|
| 10 |
+
|
| 11 |
+
Usage: oracle_rot.py <dump-dir>
|
| 12 |
+
"""
|
| 13 |
+
from __future__ import annotations
|
| 14 |
+
|
| 15 |
+
import os
|
| 16 |
+
import sys
|
| 17 |
+
|
| 18 |
+
import numpy as np
|
| 19 |
+
|
| 20 |
+
BLOCK = 1024
|
| 21 |
+
CASES = [
|
| 22 |
+
# tag, k, tokens, perm_hd, perm_nk, perm_rep, inverse
|
| 23 |
+
("plain_t1", 5120, 1, 0, 0, 1, False),
|
| 24 |
+
("plain_t3", 5120, 3, 0, 0, 1, False),
|
| 25 |
+
("perm_t1", 6144, 1, 128, 16, 3, False),
|
| 26 |
+
("perm_t2", 6144, 2, 128, 16, 3, False),
|
| 27 |
+
("wide_t1", 17408, 1, 0, 0, 1, False),
|
| 28 |
+
("inv_t2", 5120, 2, 0, 0, 1, True),
|
| 29 |
+
]
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
_POPCOUNT = np.array([bin(i).count("1") for i in range(256)], dtype=np.uint8)
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def _popcount32(a: np.ndarray) -> np.ndarray:
|
| 36 |
+
"""numpy 1.26 has no bitwise_count, so unroll the byte lookup explicitly."""
|
| 37 |
+
v = a.astype(np.uint32)
|
| 38 |
+
return (
|
| 39 |
+
_POPCOUNT[v & 0xFF].astype(np.uint16)
|
| 40 |
+
+ _POPCOUNT[(v >> 8) & 0xFF].astype(np.uint16)
|
| 41 |
+
+ _POPCOUNT[(v >> 16) & 0xFF].astype(np.uint16)
|
| 42 |
+
+ _POPCOUNT[(v >> 24) & 0xFF].astype(np.uint16)
|
| 43 |
+
)
|
| 44 |
+
|
| 45 |
+
|
| 46 |
+
def hadamard(n: int) -> np.ndarray:
|
| 47 |
+
idx = np.arange(n, dtype=np.uint32)
|
| 48 |
+
parity = _popcount32(idx[:, None] & idx[None, :]) & 1
|
| 49 |
+
return np.where(parity == 1, -1.0, 1.0) / np.sqrt(float(n))
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def forward_source_index(j: int, hd_n: int, nk_n: int, rep_n: int) -> int:
|
| 53 |
+
"""inverse of the reshape/permute llama.cpp applies: post-P column j -> source column."""
|
| 54 |
+
hd = j % hd_n
|
| 55 |
+
q = j // hd_n
|
| 56 |
+
nk = q // rep_n
|
| 57 |
+
rep = q % rep_n
|
| 58 |
+
return hd + hd_n * nk + hd_n * nk_n * rep
|
| 59 |
+
|
| 60 |
+
|
| 61 |
+
def forward_dest_index(i: int, hd_n: int, nk_n: int, rep_n: int) -> int:
|
| 62 |
+
"""the opposite direction: source column i -> post-P column (a negative control)."""
|
| 63 |
+
hd = i % hd_n
|
| 64 |
+
nk = (i // hd_n) % nk_n
|
| 65 |
+
rep = i // (hd_n * nk_n)
|
| 66 |
+
return hd + hd_n * rep + hd_n * rep_n * nk
|
| 67 |
+
|
| 68 |
+
|
| 69 |
+
def rel_l2(got: np.ndarray, ref: np.ndarray) -> float:
|
| 70 |
+
denom = float(np.linalg.norm(ref))
|
| 71 |
+
if denom == 0.0:
|
| 72 |
+
return float(np.linalg.norm(got - ref))
|
| 73 |
+
return float(np.linalg.norm(got - ref) / denom)
|
| 74 |
+
|
| 75 |
+
|
| 76 |
+
def evaluate(tag: str, k: int, tokens: int, hd_n: int, nk_n: int, rep_n: int, inverse: bool,
|
| 77 |
+
dump_dir: str, H: np.ndarray) -> bool:
|
| 78 |
+
x = np.fromfile(os.path.join(dump_dir, tag + ".in.f32"), dtype=np.float32).astype(np.float64)
|
| 79 |
+
s = np.fromfile(os.path.join(dump_dir, tag + ".signs.f32"), dtype=np.float32).astype(np.float64)
|
| 80 |
+
got = np.fromfile(os.path.join(dump_dir, tag + ".out.f32"), dtype=np.float32).astype(np.float64)
|
| 81 |
+
x = x.reshape(k, tokens)
|
| 82 |
+
got = got.reshape(k, tokens)
|
| 83 |
+
|
| 84 |
+
blocks = k // BLOCK
|
| 85 |
+
permuted = (not inverse) and rep_n > 1
|
| 86 |
+
|
| 87 |
+
def blockwise_hadamard(vec: np.ndarray) -> np.ndarray:
|
| 88 |
+
out = np.empty_like(vec)
|
| 89 |
+
for b in range(blocks):
|
| 90 |
+
lo = b * BLOCK
|
| 91 |
+
out[lo:lo + BLOCK] = H @ vec[lo:lo + BLOCK]
|
| 92 |
+
return out
|
| 93 |
+
|
| 94 |
+
def permute(vec: np.ndarray, src) -> np.ndarray:
|
| 95 |
+
return np.array([vec[src(j)] for j in range(k)], dtype=np.float64)
|
| 96 |
+
|
| 97 |
+
# Build every variant as a full [k, tokens] array so the comparison never broadcasts.
|
| 98 |
+
columns: dict[str, list[np.ndarray]] = {}
|
| 99 |
+
for t in range(tokens):
|
| 100 |
+
col = x[:, t]
|
| 101 |
+
if inverse:
|
| 102 |
+
current = {
|
| 103 |
+
"CORRECT s*(H*z)": s * blockwise_hadamard(col),
|
| 104 |
+
"WRONG H*(s*z)": blockwise_hadamard(s * col),
|
| 105 |
+
"WRONG H*z": blockwise_hadamard(col),
|
| 106 |
+
"WRONG s*z": s * col,
|
| 107 |
+
}
|
| 108 |
+
else:
|
| 109 |
+
base = col if not permuted else permute(
|
| 110 |
+
col, lambda j: forward_source_index(j, hd_n, nk_n, rep_n))
|
| 111 |
+
# one sign row off: block b uses sign row b-1
|
| 112 |
+
shifted = np.empty_like(s)
|
| 113 |
+
for b in range(blocks):
|
| 114 |
+
src = ((b - 1) % blocks) * BLOCK
|
| 115 |
+
dst = b * BLOCK
|
| 116 |
+
shifted[dst:dst + BLOCK] = s[src:src + BLOCK]
|
| 117 |
+
current = {
|
| 118 |
+
"CORRECT H*(s*(P*x))": blockwise_hadamard(s * base),
|
| 119 |
+
"WRONG s*(H*(P*x))": s * blockwise_hadamard(base),
|
| 120 |
+
"WRONG unnormalized H": blockwise_hadamard(s * base) * np.sqrt(BLOCK),
|
| 121 |
+
"WRONG sign row shifted": blockwise_hadamard(shifted * base),
|
| 122 |
+
}
|
| 123 |
+
if permuted:
|
| 124 |
+
current["WRONG H*(s*x) no P"] = blockwise_hadamard(s * col)
|
| 125 |
+
current["WRONG P the other way"] = blockwise_hadamard(
|
| 126 |
+
s * permute(col, lambda j: forward_dest_index(j, hd_n, nk_n, rep_n)))
|
| 127 |
+
for name, value in current.items():
|
| 128 |
+
columns.setdefault(name, []).append(value)
|
| 129 |
+
|
| 130 |
+
variants = {name: np.stack(cols, axis=1) for name, cols in columns.items()}
|
| 131 |
+
correct_name = "CORRECT s*(H*z)" if inverse else "CORRECT H*(s*(P*x))"
|
| 132 |
+
ref = variants[correct_name]
|
| 133 |
+
|
| 134 |
+
correct = rel_l2(got, ref)
|
| 135 |
+
print(f"case {tag:9s} k={k:<6d} tokens={tokens} perm={permuted}")
|
| 136 |
+
print(f" norms: ||x||={np.linalg.norm(x):.6g} ||ref||={np.linalg.norm(ref):.6g} "
|
| 137 |
+
f"||got||={np.linalg.norm(got):.6g} ratio={np.linalg.norm(got) / np.linalg.norm(ref):.6g}")
|
| 138 |
+
ok = correct <= 0.01
|
| 139 |
+
print(f" {'PASS' if ok else 'FAIL'} CORRECT variant rel_l2={correct:.3e}")
|
| 140 |
+
separation = True
|
| 141 |
+
for name, value in variants.items():
|
| 142 |
+
if name.startswith("CORRECT"):
|
| 143 |
+
continue
|
| 144 |
+
r = rel_l2(got, value)
|
| 145 |
+
good = r > 10.0 * max(correct, 1e-6)
|
| 146 |
+
separation = separation and good
|
| 147 |
+
print(f" {name:26s} rel_l2={r:.4f} {'(differs OK)' if good else '(TOO CLOSE)'}")
|
| 148 |
+
return ok and separation
|
| 149 |
+
|
| 150 |
+
|
| 151 |
+
def main() -> int:
|
| 152 |
+
dump_dir = sys.argv[1]
|
| 153 |
+
H = hadamard(BLOCK)
|
| 154 |
+
print(f"oracle: explicit normalized Sylvester-Hadamard {BLOCK}x{BLOCK} "
|
| 155 |
+
f"(H[i][j] = popcount(i&j)&1 ? -1 : +1, /sqrt({BLOCK}))")
|
| 156 |
+
print(f" symmetric={np.allclose(H, H.T)} orthogonal={np.allclose(H @ H, np.eye(BLOCK), atol=1e-9)}")
|
| 157 |
+
results = []
|
| 158 |
+
for tag, k, tokens, hd_n, nk_n, rep_n, inverse in CASES:
|
| 159 |
+
results.append(evaluate(tag, k, tokens, hd_n, nk_n, rep_n, inverse, dump_dir, H))
|
| 160 |
+
print()
|
| 161 |
+
passed = sum(1 for r in results if r)
|
| 162 |
+
print(f"RESULT: {passed}/{len(results)} cases pass (correct variant matches AND all "
|
| 163 |
+
f"negative controls separate)")
|
| 164 |
+
return 0 if passed == len(results) else 1
|
| 165 |
+
|
| 166 |
+
|
| 167 |
+
if __name__ == "__main__":
|
| 168 |
+
raise SystemExit(main())
|