1337Hero commited on
Commit
5ad91c7
·
verified ·
1 Parent(s): ee813e7

Add model card and release metadata

Browse files
Files changed (6) hide show
  1. ARTIFACT.json +43 -0
  2. LICENSE +202 -0
  3. NOTICE +13 -0
  4. README.md +245 -0
  5. SHA256SUMS +6 -0
  6. claims.yaml +145 -0
ARTIFACT.json ADDED
@@ -0,0 +1,43 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "schema_version": 1,
3
+ "artifact": {
4
+ "filename": "Qwen3.6-27B-Q8_0_ROCMFPX.gguf",
5
+ "bytes": 27755345216,
6
+ "sha256": "ff4dbc9093c1df6fd1242294d15eb94c7bfe42ed67f98e9d58b9789ac6912c1b",
7
+ "quantization": "Q8_0_ROCMFPX",
8
+ "bits_per_weight": 8.25,
9
+ "importance_matrix": false,
10
+ "quantized_by": "1337Hero",
11
+ "quantization_command": [
12
+ "./build-r9700/bin/llama-quantize",
13
+ "Qwen3.6-27B-BF16.gguf",
14
+ "Qwen3.6-27B-Q8_0_ROCMFPX.gguf",
15
+ "Q8_0_ROCMFPX",
16
+ "16"
17
+ ]
18
+ },
19
+ "base_model": {
20
+ "repo_id": "Qwen/Qwen3.6-27B",
21
+ "license": "apache-2.0",
22
+ "source_gguf_sha256": "0438be1f5bc861ffa84e1d2d4036920f6f3d9759f3cdedbc40e554a321d1c9c5"
23
+ },
24
+ "runtime": {
25
+ "repository": "https://github.com/1337hero/ROCmFPX",
26
+ "base_commit": "6bf20cd688ba0af882d1f68ba50b292edf646ab4",
27
+ "implementation_commits": [
28
+ "eb38c6f67701ff9c74e8597f573eedf9ccecf774",
29
+ "45bcff509c4b1cff137e2cc1ea84671c61ceddea"
30
+ ],
31
+ "llama_server_sha256": "1471753a94ba007b094842474cc3b3ffd48106f15a7c71934899dd832ae4cbb3",
32
+ "llama_bench_sha256": "38bc30e53badc5ae5efb5e3d449d989a4672c52fe5801eb1ce74e70233283749",
33
+ "llama_quantize_sha256": "0aa545af25e8235613349987fb29323745b70980aa2a4fb45c169cdf561181c6"
34
+ },
35
+ "validated_platform": {
36
+ "gpu": "AMD Radeon AI PRO R9700",
37
+ "architecture": "gfx1201",
38
+ "rocm": "7.2.4",
39
+ "host_os": "Arch Linux",
40
+ "power_profile": "250 W, -75 mV"
41
+ },
42
+ "claims": "claims.yaml"
43
+ }
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
NOTICE ADDED
@@ -0,0 +1,13 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ Qwen3.6-27B Q8_0_ROCMFPX GGUF
2
+
3
+ This artifact is derived from Qwen/Qwen3.6-27B.
4
+ Copyright 2026 Alibaba Cloud.
5
+
6
+ Modification notice:
7
+ The original model was converted to GGUF and quantized to the experimental
8
+ Q8_0_ROCMFPX representation in July 2026 by the 1337Hero community account.
9
+ The derivative requires a matching experimental ROCmFPX/llama.cpp runtime.
10
+
11
+ The Apache License, Version 2.0, governs redistribution. Qwen and related
12
+ marks belong to their owners. This community artifact is not affiliated with
13
+ or endorsed by Qwen, Alibaba Cloud, AMD, ROCmFPX, or llama.cpp.
README.md ADDED
@@ -0,0 +1,245 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ license_link: https://huggingface.co/Qwen/Qwen3.6-27B/blob/main/LICENSE
4
+ base_model: Qwen/Qwen3.6-27B
5
+ base_model_relation: quantized
6
+ pipeline_tag: text-generation
7
+ quantized_by: 1337Hero
8
+ tags:
9
+ - gguf
10
+ - qwen3.6
11
+ - quantized
12
+ - rocm
13
+ - amd
14
+ - rdna4
15
+ - gfx1201
16
+ - experimental
17
+ ---
18
+
19
+ # Qwen3.6-27B — Q8_0_ROCMFPX GGUF (experimental, AMD gfx1201)
20
+
21
+ An experimental Q8 quantization of
22
+ [Qwen/Qwen3.6-27B](https://huggingface.co/Qwen/Qwen3.6-27B), tuned and
23
+ validated for the AMD Radeon AI PRO R9700 (`gfx1201`).
24
+
25
+ > [!IMPORTANT]
26
+ > This file does **not** run on upstream llama.cpp, Ollama, LM Studio, or
27
+ > vLLM. It uses the custom `Q8_0_ROCMFPX` tensor type and requires the pinned
28
+ > ROCmFPX fork build described below. Unsupported runtimes should reject the
29
+ > file; if a tool appears to load it anyway, do not trust the output. Hugging
30
+ > Face's GGUF metadata viewer may also mislabel the custom tensor type or fail
31
+ > to parse the file.
32
+
33
+ ## Should you use this?
34
+
35
+ Use this model if all of the following are true:
36
+
37
+ - you run a Radeon AI PRO R9700 (`gfx1201`) under ROCm;
38
+ - you are willing to build the pinned fork from source;
39
+ - you want a Q8 that is 2.94% smaller than upstream `Q8_0` and performs equal
40
+ observed model work about 6% faster on this hardware.
41
+
42
+ Otherwise, use an ordinary `Q8_0` GGUF of Qwen3.6-27B. It has broad support in
43
+ the standard GGUF runtime ecosystem, and its quality is equivalent within the
44
+ tolerances measured here.
45
+
46
+ ## TL;DR — measured against upstream `Q8_0`
47
+
48
+ | Measure | Result |
49
+ | --- | ---: |
50
+ | Model size | 2.94% smaller |
51
+ | HumanEval | 140/164 — tied |
52
+ | MBPP base / MBPP+ | 348/378 and 291/378 vs 350/378 and 293/378 — within the predeclared tolerance |
53
+ | Proxy — full-model decode | 5.77–5.87% faster |
54
+ | Proxy — `pp4096+tg512` combined | 4.75–4.96% faster |
55
+ | Equal-work — agent-derived evaluation (18 matched pairs) | 5.82% faster median, 17/18 pairs, exact one-sided `p=0.000965` |
56
+ | End-to-end — raw live-agent wall time | 2.21% faster — **failed** the predeclared 3% threshold |
57
+
58
+ The supported conclusion is deliberately narrow: this format performs equal
59
+ observed model work faster than upstream `Q8_0` on the measured deployment.
60
+ The campaign did **not** confirm that complete live-agent jobs finish at
61
+ least 3% faster, so no such claim is made. All thresholds were fixed before
62
+ the runs; failed gates are reported, not reinterpreted.
63
+
64
+ The public [claim registry](claims.yaml) records the scope, result class, and
65
+ sealed aggregate-evidence hash behind each release claim.
66
+
67
+ ## What `Q8_0_ROCMFPX` actually is
68
+
69
+ Each block stores 32 signed eight-bit codes and one finite UE4M3 scale byte:
70
+ 33 bytes per 32 weights, or 8.25 bits per weight. The HIP path decodes the
71
+ scale and performs integer MMVQ/MMQ dot products with float accumulation.
72
+ This is **not** native FP8 or FP4 matrix arithmetic.
73
+
74
+ ```text
75
+ GGUF block
76
+ 32 signed int8 codes + UE4M3 scale
77
+ |
78
+ v
79
+ activation quantization to Q8_1
80
+ |
81
+ +--> decode: gfx1201 MMVQ, VDR8 + measured wave policy
82
+ |
83
+ `--> prefill: integer MMQ path
84
+ |
85
+ v
86
+ float accumulation / model output
87
+ ```
88
+
89
+ The experimental format and execution path come from the upstream
90
+ [ROCmFPX](https://github.com/charlie12345/ROCmFPX) project. This repository's
91
+ contribution is the `gfx1201` decode tuning (VDR8 vector-dot width and a
92
+ measured wave policy) and the validation below; other GPU targets retain
93
+ ROCmFPX defaults and are unmeasured.
94
+
95
+ The quantization is uniform Q8 with no importance matrix and no per-tensor
96
+ routing.
97
+
98
+ ## Required runtime
99
+
100
+ Build the pinned fork branch:
101
+
102
+ ```bash
103
+ git clone https://github.com/1337hero/ROCmFPX.git
104
+ cd ROCmFPX
105
+ git checkout 45bcff509c4b1cff137e2cc1ea84671c61ceddea
106
+ env JOBS=16 scripts/build-r9700.sh
107
+ ```
108
+
109
+ The wrapper is the simplest supported build. The sealed performance runners
110
+ used this HIP-only configuration:
111
+
112
+ ```bash
113
+ cmake -S . -B build-r9700 -G Ninja \
114
+ -DCMAKE_BUILD_TYPE=Release \
115
+ -DGGML_HIP=ON \
116
+ -DGGML_HIP_FORCE_MMQ=ON \
117
+ -DGGML_VULKAN=OFF \
118
+ -DGGML_CUDA=OFF \
119
+ -DCMAKE_HIP_ARCHITECTURES=gfx1201 \
120
+ -DGPU_TARGETS=gfx1201 \
121
+ -DCMAKE_HIP_FLAGS= \
122
+ -DLLAMA_BUILD_SERVER=ON \
123
+ -DLLAMA_BUILD_WEBUI=OFF \
124
+ -DLLAMA_USE_PREBUILT_WEBUI=OFF \
125
+ -DLLAMA_BUILD_TESTS=ON \
126
+ -DGGML_BUILD_TESTS=OFF
127
+ cmake --build build-r9700 -j 16 --target \
128
+ llama-server llama-bench llama-quantize
129
+ ```
130
+
131
+ The validated implementation is ROCmFPX base commit
132
+ `6bf20cd688ba0af882d1f68ba50b292edf646ab4` plus commits
133
+ `eb38c6f67701ff9c74e8597f573eedf9ccecf774` and
134
+ `45bcff509c4b1cff137e2cc1ea84671c61ceddea`.
135
+
136
+ Binaries validated by the lab:
137
+
138
+ | Binary | SHA-256 |
139
+ | --- | --- |
140
+ | `llama-server` | `1471753a94ba007b094842474cc3b3ffd48106f15a7c71934899dd832ae4cbb3` |
141
+ | `llama-bench` | `38bc30e53badc5ae5efb5e3d449d989a4672c52fe5801eb1ce74e70233283749` |
142
+ | `llama-quantize` | `0aa545af25e8235613349987fb29323745b70980aa2a4fb45c169cdf561181c6` |
143
+
144
+ ## Example deployment
145
+
146
+ This shape matches the validated two-card, 262,144-token envelope. Device
147
+ names depend on the host; check `--list-devices` first.
148
+
149
+ ```bash
150
+ ./build-r9700/bin/llama-server \
151
+ -m Qwen3.6-27B-Q8_0_ROCMFPX.gguf \
152
+ --no-mmap \
153
+ -c 262144 \
154
+ -b 2048 \
155
+ -ub 512 \
156
+ -t 16 \
157
+ -ngl 99 \
158
+ -sm layer \
159
+ -ts 1,1 \
160
+ -dev ROCm0,ROCm2 \
161
+ --fit off \
162
+ -ctk f16 \
163
+ -ctv f16 \
164
+ -fa on \
165
+ -np 1
166
+ ```
167
+
168
+ Both this model and the upstream `Q8_0` control allocated one 262,144-token
169
+ slot across two R9700s with F16 K/V cache and generated bounded valid output;
170
+ this model used 804,319,232 fewer resident bytes in the captured snapshots.
171
+ That test proved allocation and bounded generation, not a timed full-context
172
+ prompt.
173
+
174
+ ## Measurement conditions
175
+
176
+ All performance numbers were measured on:
177
+
178
+ - Qwen3.6-27B, this GGUF vs an upstream `Q8_0` control of the same
179
+ conversion, same host, same workload;
180
+ - two non-display Radeon AI PRO R9700s at a pre-existing 250 W, −75 mV tune —
181
+ **not** stock 300 W;
182
+ - ROCm 7.2.4 on Arch Linux, outside the official Radeon Linux support matrix;
183
+ - pinned source revisions, forced model residency, reversed candidate order,
184
+ and predeclared promotion thresholds.
185
+
186
+ The equal-work result comes from 18 common-seed matched agent-workload pairs
187
+ with ABBA/BAAB counterbalancing, warmup before every measured run, full
188
+ request accounting (1,899 logical requests, zero retries), and exact paired
189
+ Wilcoxon signed-rank tests.
190
+
191
+ ## Artifact
192
+
193
+ | Field | Value |
194
+ | --- | --- |
195
+ | File | `Qwen3.6-27B-Q8_0_ROCMFPX.gguf` |
196
+ | Size | 27,755,345,216 bytes |
197
+ | SHA-256 | `ff4dbc9093c1df6fd1242294d15eb94c7bfe42ed67f98e9d58b9789ac6912c1b` |
198
+ | Source GGUF | BF16, SHA-256 `0438be1f5bc861ffa84e1d2d4036920f6f3d9759f3cdedbc40e554a321d1c9c5` |
199
+ | Quantization | uniform `Q8_0_ROCMFPX`, no importance matrix |
200
+ | Content | main text model only; no MTP artifact or multimodal projector |
201
+
202
+ Quantization command (the qualifying run wrote to a temporary filename and
203
+ then promoted the verified hash without overwriting an existing artifact):
204
+
205
+ ```bash
206
+ ./build-r9700/bin/llama-quantize \
207
+ Qwen3.6-27B-BF16.gguf \
208
+ Qwen3.6-27B-Q8_0_ROCMFPX.gguf \
209
+ Q8_0_ROCMFPX \
210
+ 16
211
+ ```
212
+
213
+ Verify after download:
214
+
215
+ ```bash
216
+ sha256sum -c SHA256SUMS
217
+ ```
218
+
219
+ ## Limitations
220
+
221
+ - Requires an experimental fork; upstream llama.cpp compatibility awaits a
222
+ separate code contribution.
223
+ - Validated on exactly one model conversion, one Arch Linux host, ROCm 7.2.4,
224
+ and `gfx1201` GPUs at a tuned power profile. Other GPUs, hosts, and stock
225
+ power are unmeasured.
226
+ - Aggregate quality stays within the declared tolerances, but this model and
227
+ upstream `Q8_0` do not produce identical per-task outcomes.
228
+ - The raw private lab archive (prompts, outputs, paths, environment) is not
229
+ published; it awaits a separate sanitization review.
230
+
231
+ ## License and attribution
232
+
233
+ - **Base model:** Qwen3.6-27B, Copyright 2026 Alibaba Cloud, Apache-2.0. This
234
+ repository redistributes a converted and quantized derivative under the same
235
+ license. See `LICENSE` and `NOTICE`.
236
+ - **Format and execution path:** the experimental `Q8_0_ROCMFPX`
237
+ representation and HIP kernels are the work of the
238
+ [ROCmFPX](https://github.com/charlie12345/ROCmFPX) project, which builds on
239
+ [llama.cpp](https://github.com/ggml-org/llama.cpp).
240
+ - **This repository:** the `gfx1201` decode tuning, quantized artifact, and
241
+ validation evidence.
242
+
243
+ Qwen and related marks belong to their owners. This community quantization is
244
+ not affiliated with or endorsed by Qwen, Alibaba Cloud, AMD, ROCmFPX, or
245
+ llama.cpp.
SHA256SUMS ADDED
@@ -0,0 +1,6 @@
 
 
 
 
 
 
 
1
+ 34f4a2d2879ae6fb8b90af386c3ae78fd52e9d9e3c3b745d7b0c9241ea1001cf ARTIFACT.json
2
+ 39b34cdcb38146a1eff65fb3b604291c0ea2a136f35334efbf61c8d7fe697772 claims.yaml
3
+ 689f0c220c9b4e857f35ca7d0dca51397e5a545dc4ad3987438c023e24ff0ab6 LICENSE
4
+ 0e93cce70f17bc22164a0a27596b6bc41073b9f42f0b95edcd03ed08cf4fc545 NOTICE
5
+ 985037f872776846bfd039479a7648c3cc6b5e7a99eb2eb531b545cd29e53ef3 README.md
6
+ ff4dbc9093c1df6fd1242294d15eb94c7bfe42ed67f98e9d58b9789ac6912c1b Qwen3.6-27B-Q8_0_ROCMFPX.gguf
claims.yaml ADDED
@@ -0,0 +1,145 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ schema_version: 1
2
+ release: Qwen3.6-27B-Q8_0_ROCMFPX-GGUF
3
+ artifact_sha256: ff4dbc9093c1df6fd1242294d15eb94c7bfe42ed67f98e9d58b9789ac6912c1b
4
+
5
+ evidence_archive:
6
+ publication_status: private_pending_sanitization
7
+ gate_manifest_sha256: 4d9953b3698684fcf1f795d2303cdb80ee4e90bd9d8afba9cf8a7b3bab240198
8
+ note: >-
9
+ Raw prompts, outputs, telemetry, environment captures, and host paths are
10
+ intentionally excluded from this model release. The paths below identify
11
+ records in the sealed research archive; their hashes bind the aggregate
12
+ statements published here.
13
+
14
+ claims:
15
+ - id: C01
16
+ class: compatibility
17
+ status: supported
18
+ claim: >-
19
+ The artifact requires the pinned ROCmFPX fork and is not compatible with
20
+ stock llama.cpp, Ollama, LM Studio, or vLLM.
21
+ scope: custom Q8_0_ROCMFPX GGUF tensor type
22
+ evidence:
23
+ runtime_commit: 45bcff509c4b1cff137e2cc1ea84671c61ceddea
24
+ llama_server_sha256: 1471753a94ba007b094842474cc3b3ffd48106f15a7c71934899dd832ae4cbb3
25
+
26
+ - id: C02
27
+ class: format
28
+ status: supported
29
+ claim: >-
30
+ Each weight block stores 32 signed int8 codes and one finite UE4M3 scale
31
+ byte (8.25 bits per weight); HIP uses integer MMVQ/MMQ dot products with
32
+ float accumulation. This is not native FP8 or FP4 matrix arithmetic.
33
+ scope: Q8_0_ROCMFPX representation and measured HIP execution path
34
+ evidence:
35
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
36
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9
37
+
38
+ - id: C03
39
+ class: artifact_size
40
+ status: supported
41
+ claim: >-
42
+ The custom GGUF is 27,755,345,216 bytes, 840,418,208 bytes or 2.94%
43
+ smaller than the matched upstream Q8_0 control.
44
+ scope: the two named Qwen3.6-27B conversions only
45
+ evidence:
46
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
47
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9
48
+
49
+ - id: C04
50
+ class: quality
51
+ status: supported_aggregate_only
52
+ claim: >-
53
+ Custom and upstream Q8_0 each score 140/164 on deterministic HumanEval.
54
+ scope: aggregate pass@1; per-task outcomes differ
55
+ evidence:
56
+ custom_score_sha256: bae8408c545a7738917d143a36b1b725beecb9ad59d8a0e6caf14fb18a7bb5a7
57
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
58
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9
59
+
60
+ - id: C05
61
+ class: quality
62
+ status: pass_within_predeclared_tolerance
63
+ claim: >-
64
+ Custom scores 348/378 on MBPP and 291/378 on MBPP+, versus 350/378 and
65
+ 293/378 for upstream Q8_0; both two-task deficits are within the fixed
66
+ seven-task tolerance.
67
+ scope: aggregate pass@1; not behavioral identity
68
+ evidence:
69
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/quality/mbppplus/SUMMARY.md
70
+ summary_sha256: 036add79e0b9f1b0c8e28c7d48ecb3edefcc14b1dc216486ad05ecf7d9c1e965
71
+
72
+ - id: C06
73
+ class: performance_proxy
74
+ status: pass
75
+ claim: >-
76
+ Full-model decode is 5.77% to 5.87% faster than the matched upstream
77
+ Q8_0 control.
78
+ scope: >-
79
+ Two non-display R9700s, ROCm 7.2.4 on Arch Linux, 250 W/-75 mV profile,
80
+ fixed runner and model hashes
81
+ evidence:
82
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
83
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9
84
+
85
+ - id: C07
86
+ class: performance_proxy
87
+ status: pass
88
+ claim: >-
89
+ The pp4096+tg512 combined proxy is 4.75% to 4.96% faster than the
90
+ matched upstream Q8_0 control.
91
+ scope: same measured deployment as C06; not a live-agent wall-time claim
92
+ evidence:
93
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
94
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9
95
+
96
+ - id: C08
97
+ class: equal_work_performance
98
+ status: pass
99
+ claim: >-
100
+ In 18 predeclared matched pairs, standardized evaluation of equal
101
+ within-pair prompt and completion work is 5.82% faster at the median,
102
+ favors custom in 17/18 pairs, and has exact one-sided p=0.000965.
103
+ scope: fixed six-scenario agent-derived pack and measured deployment
104
+ evidence:
105
+ protocol_sha256: 6b1ea93a91c288700086f593ac69ad39bb65d71eb8eebdd6e225936ef54d023f
106
+ analysis_sha256: 775ffe638a0df718e7c8d72eacb8471b4fe03b7495cc670c211cd46ace8e9ce3
107
+ summary_sha256: 6c86e5a2b19688436eb68ae80e221926bf0bde94aefea611ef6433193feb12c4
108
+ nested_manifest_sha256: 18adf876740d265182b97cbc9f95a4a4e14626d68e59843813c11343e6adc156
109
+
110
+ - id: C09
111
+ class: end_to_end_performance
112
+ status: rejected_below_predeclared_magnitude
113
+ claim: >-
114
+ Raw live-agent wall time improves 2.21% at the median and favors custom
115
+ in 13/18 pairs, but fails the fixed 3% minimum; this release does not
116
+ claim that complete live-agent jobs finish faster.
117
+ scope: same V4 campaign as C08
118
+ evidence:
119
+ analysis_sha256: 775ffe638a0df718e7c8d72eacb8471b4fe03b7495cc670c211cd46ace8e9ce3
120
+ summary_sha256: 6c86e5a2b19688436eb68ae80e221926bf0bde94aefea611ef6433193feb12c4
121
+
122
+ - id: C10
123
+ class: deployment_capacity
124
+ status: pass
125
+ claim: >-
126
+ Both models allocate one 262,144-token slot and generate bounded output
127
+ on two R9700s; custom uses 804,319,232 fewer resident bytes in the
128
+ captured snapshots.
129
+ scope: >-
130
+ Capacity allocation and bounded generation only; no timed 262,144-token
131
+ prompt and no peak-memory claim
132
+ evidence:
133
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/deployment/262144-two-card/SUMMARY.md
134
+ summary_sha256: 36c1b025dbb67baf858f1b56c76799ed5e25e65f97b4ea82fa659fdaebfd1b66
135
+
136
+ - id: C11
137
+ class: hardware_scope
138
+ status: measured_single_host
139
+ claim: >-
140
+ Results are validated only on Radeon AI PRO R9700 gfx1201 GPUs under
141
+ ROCm 7.2.4 on Arch Linux at a 250 W/-75 mV tune.
142
+ scope: other GPUs, hosts, ROCm versions, and stock power are unmeasured
143
+ evidence:
144
+ summary: reports/phase2/model-gate/20260713T214155Z/q8-rocmfpx/SUMMARY.md
145
+ summary_sha256: 0a8f780c98fd40560e254121542382d37bc1c2e69a43000bbe4a87303f7e7eb9