Qwen3.5-9B CVE-to-CWE LoRA adapter (run 3, trial 23): weights, tokenizer, model card, predictions

#1
by aaron-psl - opened
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
README.md CHANGED
@@ -1,3 +1,548 @@
1
  ---
2
- license: apache-2.0
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
3
  ---
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
  ---
2
+ # ---- Identity -------------------------------------------------------------
3
+ base_model: Qwen/Qwen3.5-9B
4
+ base_model_relation: adapter
5
+ library_name: peft
6
+ pipeline_tag: text-generation
7
+ language:
8
+ - en
9
+ license: apache-2.0 # verified: inherited from Qwen/Qwen3.5-9B, whose Hub
10
+ # metadata declares license:apache-2.0 and which ships an
11
+ # Apache-2.0 LICENSE file (copied here verbatim).
12
+
13
+ # ---- Discovery ------------------------------------------------------------
14
+ tags:
15
+ - lora
16
+ - qlora
17
+ - sft
18
+ - trl
19
+ - peft
20
+ - text-classification
21
+ - cve
22
+ - cwe
23
+ - vulnerability
24
+ - security
25
+
26
+ datasets:
27
+ - exploitintel/cve-cwe-consensus
28
+
29
+ metrics:
30
+ - f1
31
+ - accuracy
32
+
33
+ # ---- Structured evaluation ------------------------------------------------
34
+ model-index:
35
+ - name: qwen3.5-9b-cve-cwe-lora
36
+ results:
37
+ - task:
38
+ type: text-classification
39
+ name: CVE description to CWE weakness class (117-way, generated as text)
40
+ dataset:
41
+ type: exploitintel/cve-cwe-consensus
42
+ name: exploitintel/cve-cwe-consensus, 300-row natural-distribution sample of the validation split
43
+ split: validation
44
+ revision: 606cef101302fc2e0f69fd298f0de28b20a2aacf
45
+ metrics:
46
+ - type: f1
47
+ name: Macro-F1 over the union of gold and predicted classes (81 classes)
48
+ value: 0.473288
49
+ - type: f1
50
+ name: Macro-F1 over gold classes only (68 classes)
51
+ value: 0.563769
52
+ - type: accuracy
53
+ name: Accuracy (= micro-F1 for single-label output)
54
+ value: 0.71
55
  ---
56
+
57
+ # Qwen3.5-9B CVE-to-CWE Classifier (LoRA)
58
+
59
+ Given the free-text description of a published CVE, emits the single CWE
60
+ identifier that best characterises the root-cause weakness, as one line of
61
+ text (`CWE-79`, `CWE-787`, ...), over the 117-class label space of
62
+ [exploitintel/cve-cwe-consensus](https://huggingface.co/datasets/exploitintel/cve-cwe-consensus).
63
+
64
+ This is a **LoRA adapter for**
65
+ [Qwen/Qwen3.5-9B](https://huggingface.co/Qwen/Qwen3.5-9B), trained with
66
+ **QLoRA (4-bit NF4 base, bf16 compute)** via
67
+ [TRL](https://github.com/huggingface/trl) SFT.
68
+
69
+ > **The headline number is one draw from a noisy measurement.** This checkpoint
70
+ > scored macro-F1 0.473 on 300 held-out rows, but the identical configuration
71
+ > re-run later in the same search scored 0.395. Read
72
+ > [How these values were chosen](#how-these-values-were-chosen) and
73
+ > [Evaluation](#evaluation) before quoting anything, and do not report 0.473 as
74
+ > the model's expected performance.
75
+
76
+ ## Model details
77
+
78
+ | | |
79
+ |---|---|
80
+ | Developed by | SASVA AI Model Cognition Labs (MCL) Team |
81
+ | Base model | [`Qwen/Qwen3.5-9B`](https://huggingface.co/Qwen/Qwen3.5-9B) |
82
+ | Base revision | `c202236235762e1c871ad0ccb60c8ee5ba337b9a` |
83
+ | Base parameters | 9.65B (dense; 9,653,104,368 in the safetensors index, ~19.3 GB bf16) |
84
+ | Architecture family | `qwen3_5` (`Qwen3_5ForConditionalGeneration`, text tower `qwen3_5_text`) |
85
+ | Adaptation | LoRA (`r=128`, `alpha=256`, `dropout=0.15`, rsLoRA off, DoRA off) |
86
+ | Trainable modules | `in_proj_qkv`, `in_proj_z`, `out_proj` (Gated DeltaNet layers); `q_proj`, `k_proj`, `v_proj`, `o_proj` (attention layers); `gate_proj`, `up_proj`, `down_proj` (every layer) |
87
+ | Excluded modules | none (the base is text-only in practice; no adapter tensor touches the vision tower) |
88
+ | Training method | `qlora` (`--load-in-4bit`, run 3 / trial 23) |
89
+ | Refinement | none |
90
+ | Precision | 4-bit NF4 base with double quantisation, bf16 compute; adapter stored in float32 |
91
+ | Language | English |
92
+ | License | Apache-2.0 (inherited from the base model) |
93
+
94
+ Trainable parameters: **320,864,256** across 200 modules, 3.2975% of the
95
+ 9,730,678,000 parameters PEFT counted with the adapter attached. The adapter
96
+ file is 1,283,518,408 bytes (400 tensors, `lora_A` + `lora_B` per module, all
97
+ float32). Every tensor sits under `base_model.model.model.language_model`.
98
+
99
+ **One Qwen 3.5 structural fact shapes the module list.** Confirmed against the
100
+ base model's `config.json`: the 32 layers follow a 3-linear / 1-full pattern
101
+ (`full_attention_interval: 4`), so 24 layers are Gated DeltaNet linear
102
+ attention and expose `in_proj_qkv` / `in_proj_z` / `out_proj`, while the 8
103
+ full-attention layers (3, 7, 11, ..., 31) are grouped-query attention with 16
104
+ heads over 4 KV heads and expose `q_proj` / `k_proj` / `v_proj` / `o_proj`.
105
+ The adapter's tensor counts match exactly: 48 per linear-attention projection
106
+ (24 layers x A/B), 16 per attention projection (8 layers x A/B), 64 per MLP
107
+ projection (32 layers x A/B). There are no MoE experts in this base.
108
+
109
+ ## Intended use
110
+
111
+ **Direct use.** Map a CVE description to one CWE id, for triage and
112
+ labelling pipelines that already consume CWE ids. The 117-class label space is
113
+ the dataset's; ids outside it were never seen in training.
114
+
115
+ The model was trained on a specific prompt shape and that shape is part of the
116
+ contract:
117
+
118
+ - System prompt (verbatim): *"You are a CVE-to-CWE classifier. Given a CVE
119
+ vulnerability description, identify the single root-cause CWE weakness class
120
+ that best characterizes the flaw. Output exactly one CWE identifier (e.g.,
121
+ CWE-79) on a single line with no explanation or additional text."*
122
+ - User turn: the instruction *"Classify the following CVE description into
123
+ exactly one CWE weakness class. Reply with the CWE ID only, for example
124
+ CWE-79."*, a blank line, then the CVE description inside a bare ```` ``` ````
125
+ fence. This is the exact string the evaluator rendered
126
+ (`instruction + "\n\n```\n" + description + "\n```"`).
127
+ - Applied through the tokenizer's chat template (`chat_template.jinja`,
128
+ shipped in this repo) with `add_generation_prompt=True` and
129
+ `enable_thinking=False`. Do not concatenate strings by hand.
130
+ - The output is one line, `CWE-<n>`. In evaluation all 300 generations were a
131
+ single well-formed id; the first line of the output is the prediction.
132
+ - Decode greedily (`do_sample=False`) with a small budget; 64 new tokens is
133
+ what the metric was scored with.
134
+
135
+ ## How to get started
136
+
137
+ ```python
138
+ import torch
139
+ from peft import PeftModel
140
+ from transformers import AutoModelForImageTextToText, AutoTokenizer
141
+
142
+ BASE = "Qwen/Qwen3.5-9B"
143
+ ADAPTER = "SASVAAI/Qwen-3.5-9B-CVE-to-CWE"
144
+
145
+ # NOTE: AutoModelForImageTextToText, not AutoModelForCausalLM. Qwen 3.5's
146
+ # CausalLM mapping raises AttributeError: 'Qwen3_5Config' object has no
147
+ # attribute 'vocab_size' on the transformers build used here; the
148
+ # image-text-to-text mapping resolves to Qwen3_5ForConditionalGeneration and
149
+ # loads the text model correctly. Training used the same class.
150
+ tokenizer = AutoTokenizer.from_pretrained(ADAPTER, trust_remote_code=True)
151
+ model = AutoModelForImageTextToText.from_pretrained(
152
+ BASE, dtype=torch.bfloat16, device_map="auto", trust_remote_code=True
153
+ )
154
+ model = PeftModel.from_pretrained(model, ADAPTER)
155
+ model.eval()
156
+
157
+ # Verbatim from the training invocation. Do not paraphrase.
158
+ SYSTEM = (
159
+ "You are a CVE-to-CWE classifier. Given a CVE vulnerability description, "
160
+ "identify the single root-cause CWE weakness class that best characterizes "
161
+ "the flaw. Output exactly one CWE identifier (e.g., CWE-79) on a single line "
162
+ "with no explanation or additional text."
163
+ )
164
+ INSTRUCTION = (
165
+ "Classify the following CVE description into exactly one CWE weakness class. "
166
+ "Reply with the CWE ID only, for example CWE-79."
167
+ )
168
+ description = (
169
+ "A stored cross-site scripting flaw in the FAQ page lets an attacker inject "
170
+ "script that runs in other users' browsers."
171
+ )
172
+ messages = [
173
+ {"role": "system", "content": SYSTEM},
174
+ {"role": "user", "content": f"{INSTRUCTION}\n\n```\n{description}\n```"},
175
+ ]
176
+ inputs = tokenizer.apply_chat_template(
177
+ messages, add_generation_prompt=True, enable_thinking=False,
178
+ return_tensors="pt", return_dict=True,
179
+ ).to(model.device)
180
+
181
+ out = model.generate(**inputs, max_new_tokens=64, do_sample=False)
182
+ print(tokenizer.decode(out[0][inputs["input_ids"].shape[-1]:], skip_special_tokens=True).strip())
183
+ # -> CWE-79
184
+ ```
185
+
186
+ The base model is ~19 GB in bfloat16; one 24 GB-class GPU is enough for
187
+ inference with the adapter. Loading the base in 4-bit with `bitsandbytes`
188
+ (`load_in_4bit=True`, `bnb_4bit_quant_type="nf4"`,
189
+ `bnb_4bit_use_double_quant=True`, `bnb_4bit_compute_dtype=torch.bfloat16`)
190
+ reproduces the training-time numerics and needs about 7 GB.
191
+
192
+ > Decoding matters. The metric was scored greedily with `max_new_tokens=64`
193
+ > through the chat template with thinking disabled. No sampling setting was
194
+ > validated, and enabling thinking changes the prompt the model sees.
195
+
196
+ ## Training details
197
+
198
+ **Data.** 700 training rows and 300 validation rows built from
199
+ [exploitintel/cve-cwe-consensus](https://huggingface.co/datasets/exploitintel/cve-cwe-consensus)
200
+ at revision `606cef101302fc2e0f69fd298f0de28b20a2aacf` by a deterministic
201
+ script (`autocatalyst.datagen.cve_cwe_consensus`, seed 0). No model generated
202
+ any training content.
203
+
204
+ Row selection, as recorded in the builder's manifest:
205
+
206
+ - Source rows are `{"messages": [system, user, assistant]}`; the user turn is
207
+ the CVE description, the assistant turn the label.
208
+ - Rows whose label is not exactly one `CWE-<n>` id were dropped (about 9% of
209
+ the dataset are comma-separated multi-label lists). Rows whose id is outside
210
+ the dataset's 117-class set, or whose description is empty, were dropped.
211
+ Duplicate descriptions within a split were collapsed to the first.
212
+ - **Train**: from the dataset's `train` split (50,074 raw rows, 44,337 kept
213
+ after dropping 4,390 multi-label or malformed and 1,347 duplicate rows),
214
+ 700 rows were sampled with one row guaranteed per class and the remainder
215
+ weighted by the square root of class frequency, so the head classes do not
216
+ crowd out the tail that macro-F1 scores. All 117 classes are present.
217
+ - **Validation**: from the dataset's own `validation` split (11,052 raw rows,
218
+ 8,878 kept), 300 rows sampled at the natural class distribution. 68 classes
219
+ appear; every one of them is in the training set; no description text is
220
+ shared with the 700 training rows (0 overlaps measured).
221
+
222
+ | | |
223
+ |---|---|
224
+ | Train samples | 700 rows / 117 classes |
225
+ | Validation samples | 300 rows / 68 classes |
226
+ | Text overlap | 0 rows |
227
+ | Prompt format | chat template + system prompt + instruction/fenced-description user turn (see Intended use) |
228
+ | Loss masking | answer tokens only; prompt tokens set to `-100` |
229
+ | Truncation | prompt left-truncated to fit `max_seq_len` 2048; answer never truncated |
230
+
231
+ The dataset's head is heavy: in the 300 validation rows CWE-79 appears 49
232
+ times, CWE-862 22, CWE-89 17, CWE-200 and CWE-22 13 each, and 68 classes share
233
+ the rest.
234
+
235
+ An LLM (Claude Opus 4.6, `claude-opus-4-6`, via an internal inference gateway)
236
+ proposed the hyperparameters the search tried. It generated no training
237
+ content and computed no metric.
238
+
239
+ ### Method
240
+
241
+ | | |
242
+ |---|---|
243
+ | SFT method | `qlora` |
244
+ | Base quantisation during training | 4-bit NF4, double quantisation, bf16 compute (`bitsandbytes`) |
245
+ | Refinement stage | none |
246
+ | Attention implementation | `flash_attention_2` for the 8 GQA layers; the 24 Gated DeltaNet layers ran the PyTorch fallback because `flash-linear-attention` was not installed |
247
+ | Auto class | `AutoModelForImageTextToText` (see How to get started) |
248
+ | Hardware | 8x NVIDIA H100 80GB HBM3, `torchrun --nproc_per_node=8` |
249
+
250
+ The project allowed two methods for this run (`bf16_lora`, `qlora`) and the
251
+ search tried both; see the trial table below.
252
+
253
+ No refinement stage ran; the published weights are the SFT adapter.
254
+
255
+ ### Final hyperparameters
256
+
257
+ | Hyperparameter | Value | Source |
258
+ |---|---|---|
259
+ | `learning_rate` | 0.0002 | `[TRAIN]` cmdline |
260
+ | `lr_scheduler_type` | cosine | `[TRAIN]` cmdline |
261
+ | `num_train_epochs` | 5 | `[TRAIN]` cmdline, `trainer_state.json` |
262
+ | `per_device_train_batch_size` | 2 | `[TRAIN]` cmdline, `trainer_state.json` |
263
+ | `gradient_accumulation_steps` | 4 | `[TRAIN]` cmdline |
264
+ | `max_seq_length` | 2048 | `[TRAIN]` cmdline |
265
+ | `warmup_ratio` | 0.05 | `[TRAIN]` cmdline |
266
+ | `weight_decay` | 0.05 | `[TRAIN]` cmdline |
267
+ | `loraplus_lr_ratio` | 2.0 (LoRA `B` matrices at 2x the learning rate) | `[TRAIN]` cmdline |
268
+ | `lora_r` / `lora_alpha` / `lora_dropout` | 128 / 256 / 0.15 | `adapter_config.json` |
269
+ | `use_rslora` / `use_dora` | `false` / `false` | `adapter_config.json` |
270
+ | `target_modules` | the 10 listed in Model details | `adapter_config.json` |
271
+ | `load_in_4bit` | `true` (NF4, double quant, bf16 compute) | `[TRAIN]` cmdline |
272
+
273
+ **Effective batch size: 64** (`2 x 4 x 8`). Optimizer steps: 55 (11 per
274
+ epoch).
275
+
276
+ `neftune_noise_alpha` (0.0), `use_liger_kernel`, `use_sample_packing` and
277
+ `lora_init` (`default`) were left at their no-op defaults. KD parameters are
278
+ omitted deliberately: this is a `qlora` run, not a distillation run.
279
+
280
+ > Provenance note: every value above was recovered from the platform database
281
+ > (`runs`, `experiments`, `events` tables for run 3) and cross-checked against
282
+ > the literal `[TRAIN]` command line recorded in the run log and against the
283
+ > shipped `adapter_config.json`.
284
+
285
+ ### How these values were chosen
286
+
287
+ > These hyperparameters were selected by an automated search
288
+ > (`autocatalyst.cli.run_autoresearch`): an agent proposes one change at a
289
+ > time, runs train then eval, and keeps or discards on `f1_macro` (higher is
290
+ > better).
291
+
292
+ Run 3 ran **46 trials in 7 h 17 m** (2026-09-08 20:29 to 2026-09-09 03:46);
293
+ 35 scored and 11 errored. This checkpoint is **trial 23**, the run's best.
294
+
295
+ **Every trial is comparable.** The search's five "data strategies" all
296
+ re-adapted the same fixed 700-row training file (the `train_v1` ... `train_v5`
297
+ files below hold the same 700 rows) and every trial was scored on the same 300
298
+ validation rows, so the whole table is one comparison.
299
+
300
+ | # | Method | rsLoRA | r | Dropout | LR | Epochs | Grad accum | Warmup | Weight decay | LoRA+ ratio | Data | f1_macro | Kept |
301
+ |---|---|---|---|---|---|---|---|---|---|---|---|---|---|
302
+ | 1 | bf16_lora | no | 16 | 0.05 | 2e-4 | 2 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.271028 | yes |
303
+ | 2 | qlora | no | 16 | 0.05 | 2e-4 | 2 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.323589 | yes |
304
+ | 3 | bf16_lora | no | 16 | 0.05 | 2e-4 | 2 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.271161 | no |
305
+ | 4 | qlora | no | 16 | 0.05 | 1e-4 | 2 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.224824 | no |
306
+ | 5 | qlora | no | 16 | 0.05 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.328347 | yes |
307
+ | 6 | qlora | no | 16 | 0.05 | 2e-4 | 8 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.315655 | no |
308
+ | 7 | qlora | no | 32 | 0.05 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.339796 | yes |
309
+ | 8 | qlora | no | 32 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.358649 | yes |
310
+ | 9 | qlora | no | 32 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.379137 | yes |
311
+ | 10 | qlora | no | 64 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.411651 | yes |
312
+ | 11 | qlora | yes | 64 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | 0.385129 | no |
313
+ | 12 | qlora | no | 64 | 0.05 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.405714 | no |
314
+ | 13 | qlora | no | 64 | 0.15 | 2e-4 | 5 | 8 | 0.05 | 0.01 | 2.0 | train_v1 | 0.377237 | no |
315
+ | 14 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.440878 | yes |
316
+ | 15 | qlora | no | 128 | 0.15 | 2e-4 | 8 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.377763 | no |
317
+ | 16 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | error | no |
318
+ | 17 | qlora | no | 128 | 0.15 | 1.5e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.386976 | no |
319
+ | 18 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v1 | error | no |
320
+ | 19 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v1 | 0.396817 | no |
321
+ | 20 | qlora | no | 128 | 0.15 | 2e-4 | 8 | 4 | 0.05 | 0.01 | 2.0 | train_v2 | 0.407741 | no |
322
+ | 21 | bf16_lora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v2 | 0.403981 | no |
323
+ | 22 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 1.0 | train_v2 | error | no |
324
+ | **23** | **qlora** | **no** | **128** | **0.15** | **2e-4** | **5** | **4** | **0.05** | **0.05** | **2.0** | **train_v2** | **0.473288** | **yes** |
325
+ | 24 | qlora | no | 128 | 0.10 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v2 | 0.444999 | no |
326
+ | 25 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v2 | error | no |
327
+ | 26 | qlora | no | 128 | 0.15 | 2.5e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v2 | 0.427152 | no |
328
+ | 27 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v2 | error | no |
329
+ | 28 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v2 | error | no |
330
+ | 29 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v3 | error | no |
331
+ | 30 | qlora | no | 128 | 0.15 | 2e-4 | 8 | 4 | 0.05 | 0.05 | 2.0 | train_v3 | 0.384342 | no |
332
+ | 31 | bf16_lora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v3 | 0.403688 | no |
333
+ | 32 | qlora | no | 128 | 0.05 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v3 | 0.415561 | no |
334
+ | 33 | qlora | no | 128 | 0.15 | 1.75e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v3 | 0.435022 | no |
335
+ | 34 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.03 | 0.05 | 2.0 | train_v3 | 0.416758 | no |
336
+ | 35 | qlora | no | 128 | 0.15 | 2e-4 | 8 | 4 | 0.05 | 0.05 | 2.0 | train_v4 | 0.406259 | no |
337
+ | 36 | bf16_lora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v4 | 0.408250 | no |
338
+ | 37 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v4 | error | no |
339
+ | 38 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v4 | 0.395110 | no |
340
+ | 39 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.01 | 2.0 | train_v4 | 0.437202 | no |
341
+ | 40 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v4 | error | no |
342
+ | 41 | qlora | no | 128 | 0.15 | 2e-4 | 8 | 4 | 0.05 | 0.05 | 2.0 | train_v5 | 0.391961 | no |
343
+ | 42 | bf16_lora | no | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v5 | 0.428970 | no |
344
+ | 43 | qlora | no | 128 | 0.05 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 2.0 | train_v5 | 0.423615 | no |
345
+ | 44 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v5 | error | no |
346
+ | 45 | qlora | yes | 128 | 0.15 | 2e-4 | 5 | 4 | 0.05 | 0.05 | 1.0 | train_v5 | error | no |
347
+ | 46 | qlora | no | 128 | 0.15 | 2e-4 | 5 | 8 | 0.05 | 0.05 | 2.0 | train_v5 | 0.402578 | no |
348
+
349
+ All trials: `lora_alpha = 2 x r`, cosine schedule, batch 2 per device,
350
+ `max_seq_len` 2048.
351
+
352
+ **What the search actually established.**
353
+
354
+ - **Rank dominates, up to a point.** Holding QLoRA, 5 epochs, dropout 0.15,
355
+ LR 2e-4, weight decay 0.01, LoRA+ 2.0: `r` 32 → 64 → 128 moved macro-F1
356
+ 0.379137 → 0.411651 → 0.440878 (#9, #10, #14).
357
+ - **LoRA+ helped.** The one clean A/B at `r=32` (#8 → #9) moved 0.358649 →
358
+ 0.379137 by putting the `B` matrices at 2x the learning rate.
359
+ - **More epochs hurt.** 8 epochs lost to 5 every time it was tried at `r=128`
360
+ (#15, #20, #30, #35, #41), by 0.01 to 0.06.
361
+ - **4-bit versus bf16 base: no measurable difference.** At `r=16` the 4-bit
362
+ base beat bf16 (#2 at 0.323589 against #1 and #3 at 0.271); at `r=128` with
363
+ weight decay 0.05 the three bf16 runs (#31, #36, #42: 0.404 to 0.429) sit
364
+ inside the two 4-bit runs of the same configuration (#38 at 0.395, #23 at
365
+ 0.473). Quantising the base cost nothing this search could detect.
366
+ - **rsLoRA at `r=128` diverges.** All 11 rsLoRA trials at `r=128` (#16, #18,
367
+ #22, #25, #27, #28, #29, #37, #40, #44, #45) were aborted by the training
368
+ watchdog at the third logging step with gradient norm above 10 (the
369
+ configured limit; observed 12.3 in the first). The likely cause is the
370
+ scaling rule: rsLoRA scales updates by `alpha / sqrt(r)` instead of
371
+ `alpha / r`, which with `alpha = 2r` is `2 sqrt(r)`, about 22.6 at `r=128`
372
+ against 2.0 without it, an 11x larger effective update. The one rsLoRA
373
+ trial that finished (#11, `r=64`) scored below its non-rsLoRA sibling (#10).
374
+ The 11 aborted trials are why the search shows 46 trials but 35 scores.
375
+
376
+ **What it did not establish: the winning margin.** The search's decision
377
+ "weight decay 0.01 → 0.05 improved 0.440878 → 0.473288" (#14 → #23) does not
378
+ survive the re-runs. Identical configurations were run more than once because
379
+ each "data strategy" restarted the inner loop on the same data:
380
+
381
+ | Configuration (all QLoRA, r=128, dropout 0.15, LR 2e-4, 5 epochs, grad accum 4, warmup 0.05, LoRA+ 2.0) | Trials | f1_macro | Spread |
382
+ |---|---|---|---|
383
+ | weight decay 0.05 (**this checkpoint's config**) | #23, #38 | 0.473288, 0.395110 | 0.078 |
384
+ | weight decay 0.01 | #14, #19, #39 | 0.440878, 0.396817, 0.437202 | 0.044 |
385
+ | weight decay 0.05, bf16 base instead of 4-bit | #31, #36, #42 | 0.403688, 0.408250, 0.428970 | 0.025 |
386
+ | weight decay 0.05, 8 epochs | #30, #35, #41 | 0.384342, 0.406259, 0.391961 | 0.022 |
387
+ | weight decay 0.05, dropout 0.05 | #32, #43 | 0.415561, 0.423615 | 0.008 |
388
+
389
+ Seeds were not pinned, so these differ only in initialisation and data order.
390
+ The 0.078 gap between #23 and #38 is larger than the 0.032 "improvement" the
391
+ search kept, and larger than most differences in the table. Read 0.473 as the
392
+ high draw of a configuration whose expected score is somewhere in the low-to-mid
393
+ 0.4s, and treat any two trials within about 0.05 of each other as tied.
394
+
395
+ **Search space.** Six knobs were varied (`LORA_R`, `LORA_DROPOUT`,
396
+ `LEARNING_RATE`, `EPOCHS`, `WEIGHT_DECAY`, `LORAPLUS_LR_RATIO`) plus one
397
+ `GRAD_ACCUM` probe (#13, #46), one warmup probe (#34), the bf16-vs-4-bit
398
+ method switch, and the rsLoRA attempts. `LR_SCHEDULER`, `MAX_SEQ_LEN`,
399
+ `BATCH_SIZE`, `NEFTUNE_NOISE_ALPHA`, `USE_DORA`, `LORA_INIT`,
400
+ `USE_LIGER_KERNEL` and `USE_SAMPLE_PACKING` were never moved.
401
+
402
+ **Observed training metrics** (this checkpoint).
403
+
404
+ | | |
405
+ |---|---|
406
+ | Final train loss (mean over the run) | 0.6052066683769226 |
407
+ | Final eval loss (teacher-forced, answer tokens) | 0.9717274904251099 |
408
+ | Eval mean token accuracy | 0.8136 |
409
+ | Train runtime | 387.7625 s |
410
+ | Total FLOPs | 4.311845712166912e+16 |
411
+ | Throughput | 9.026 samples/s, 0.142 steps/s |
412
+
413
+ 55 optimizer steps ran. The logged train loss fell from 1.2962 at step 10 (grad
414
+ norm 1.06) to the run mean of 0.6052; the reported train loss is the mean over
415
+ the run, not a converged value.
416
+
417
+ ## Evaluation
418
+
419
+ **Protocol.** All 300 validation rows, greedy decoding, `max_new_tokens=64`,
420
+ prompts rendered through the chat template with `enable_thinking=False`. The
421
+ project's `classification` evaluator takes the first line of the generation as
422
+ the predicted label and compares it to the gold id. Generation ran through
423
+ Hugging Face `generate` in batches of 4 (`max_input_len` 4096) after vLLM
424
+ 0.19.1 refused the LoRA on this architecture and the evaluator fell back; the
425
+ whole pass took 63 s on 8 GPUs. Every one of the 300 outputs was a single
426
+ well-formed `CWE-<n>` id, so no prediction was lost to formatting, and the
427
+ 64-token budget cut nothing (the longest output is one id).
428
+
429
+ | Metric | Value |
430
+ |---|---|
431
+ | Macro-F1, union of gold and predicted classes (**the search metric**) | 0.473288 |
432
+ | Macro-F1, gold classes only | 0.563769 |
433
+ | Accuracy (= micro-F1) | 0.71 (213 / 300) |
434
+ | Classes in gold / predicted / union | 68 / 72 / 81 |
435
+
436
+ **Two macro-F1 numbers, one convention.** The evaluator averages F1 over the
437
+ *union* of gold and predicted classes, so the 13 classes the model predicted
438
+ that never occur in the 300 gold rows each contribute an F1 of 0 and pull the
439
+ macro average down from 0.5638 to 0.4733. Both are reported; the union
440
+ convention is the one the search optimised and the one the published baselines
441
+ below appear to use, but check before comparing.
442
+
443
+ **Published comparison points.** The dataset authors report, on their own
444
+ evaluation of the same dataset (a much larger split than these 300 rows):
445
+
446
+ | Model | Method | Micro-F1 | Macro-F1 |
447
+ |---|---|---|---|
448
+ | Qwen3-32B ([exploitintel/cve-cwe-qwen3-32b](https://huggingface.co/exploitintel/cve-cwe-qwen3-32b)) | QLoRA r=16 | 0.729 | 0.595 |
449
+ | Qwen3-8B (same author, same recipe) | QLoRA r=16 | 0.702 | 0.511 |
450
+ | **This adapter** (Qwen3.5-9B, QLoRA r=128, 700 training rows) | | **0.71** | **0.473** |
451
+
452
+ Those models trained on the full ~44K-row training split; this adapter saw
453
+ 700 rows. The micro-F1 is in the same range; the macro-F1 is 0.04 to 0.12
454
+ lower, which is the long tail this adapter had one to twelve examples of per
455
+ class to learn from. The evaluation sets also differ in size and composition,
456
+ so read this as context, not a controlled comparison.
457
+
458
+ **Sibling run on the same 700 / 300 split.** A Gemma 4 E4B adapter trained by
459
+ the same platform on this split reached accuracy 0.697 and union macro-F1
460
+ 0.4587, essentially tied with this checkpoint given the spread above.
461
+
462
+ **Baseline for comparison. Not measured.** The untuned `Qwen/Qwen3.5-9B` was
463
+ never scored on these 300 rows, so nothing here quantifies how much of the
464
+ score the fine-tuning is responsible for. This is the most important gap in
465
+ this card.
466
+
467
+ **This is a validation split the search selected against.** 35 trials were
468
+ scored on these same 300 rows and the best was kept, so expect optimistic bias
469
+ on top of the re-run spread already described. The rows are drawn from the
470
+ dataset authors' own validation split, so they are unseen CVEs from the same
471
+ period as training, not future CVEs.
472
+
473
+ **The evaluation set is reproducible.** `predictions.jsonl` in this repo holds
474
+ every one of the 300 rows: instruction, description, prediction and gold. The
475
+ builder's manifest (dataset revision, seed, drop counts, per-class counts) is
476
+ summarised under Training details.
477
+
478
+ ## Limitations and bias
479
+
480
+ **One number, wide error bars.** The same configuration scored 0.473 and 0.395
481
+ in two runs. Anyone deploying this should re-evaluate on their own data rather
482
+ than trust either figure.
483
+
484
+ **No baseline, so no established gain.** See Evaluation.
485
+
486
+ **Head classes dominate what accuracy measures.** CWE-79 alone is 16% of the
487
+ validation rows. A model that got only the top ten classes right would post a
488
+ respectable accuracy and a poor macro-F1; the two numbers here disagree by
489
+ 0.24 for that reason.
490
+
491
+ **Tail classes were barely trained.** 117 classes over 700 rows means many
492
+ classes had a single training example. Expect the model to fall back to a
493
+ frequent neighbour (CWE-20 for input validation issues, CWE-200 for disclosure)
494
+ when the description is ambiguous, and to emit ids it saw rarely with low
495
+ reliability. It also produced 13 ids in evaluation that never occur in the 300
496
+ gold rows; some may be reasonable alternative labels, some are wrong.
497
+
498
+ **Only one label.** Real CVEs often carry two CWE ids (about 9% of the source
499
+ dataset). The training data dropped those rows, so the model always commits to
500
+ one.
501
+
502
+ **Prompt shape is the contract.** Change the system prompt, the instruction
503
+ sentence, the code fence, or enable thinking, and you are evaluating a model
504
+ nobody measured.
505
+
506
+ **Domain narrowness.** English CVE descriptions in NVD / CNA style. Advisories
507
+ in other formats, other languages, source-code inputs and exploit write-ups are
508
+ unmeasured.
509
+
510
+ **Inherits all biases and limitations of the base model.** This adapter changes
511
+ 3.3% of the parameters and was not evaluated for safety or fairness. The base
512
+ model's own card governs those properties.
513
+
514
+ ## Environmental impact
515
+
516
+ | | |
517
+ |---|---|
518
+ | Hardware | 8x NVIDIA H100 80GB HBM3 |
519
+ | Training time | 6.46 minutes (387.7625 s) |
520
+ | Cloud provider / region | on-premise |
521
+
522
+ Covers this trial only. The full 46-trial search that selected it took 7 h 17 m
523
+ on the same hardware.
524
+
525
+ ## Framework versions
526
+
527
+ - PEFT 0.18.1
528
+ - TRL: 1.0.0
529
+ - Transformers: 5.7.0.dev0 (git main)
530
+ - Pytorch: 2.5.1+cu121
531
+ - bitsandbytes: 0.49.2
532
+ - flash-attn: 2.8.3
533
+ - Python: 3.12
534
+
535
+ PEFT's version is the one recorded in `adapter_config.json` at save time; the
536
+ rest are the pinned versions of the training environment. `transformers` is a
537
+ git-main build: the `qwen3_5` architecture is not in the stable PyPI release.
538
+
539
+ ## Citation
540
+
541
+ ```bibtex
542
+ @misc{qwen35_cve_cwe_lora_2026,
543
+ title = {Qwen3.5-9B CVE-to-CWE Classifier LoRA},
544
+ author = {Banerjee, Aaron and Anbuselvan, Pooja and Jodhpurkar, Om},
545
+ year = {2026},
546
+ url = {https://huggingface.co/SASVAAI/Qwen-3.5-9B-CVE-to-CWE}
547
+ }
548
+ ```
adapter_config.json ADDED
@@ -0,0 +1,49 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "alora_invocation_tokens": null,
3
+ "alpha_pattern": {},
4
+ "arrow_config": null,
5
+ "auto_mapping": null,
6
+ "base_model_name_or_path": "Qwen/Qwen3.5-9B",
7
+ "bias": "none",
8
+ "corda_config": null,
9
+ "ensure_weight_tying": false,
10
+ "eva_config": null,
11
+ "exclude_modules": null,
12
+ "fan_in_fan_out": false,
13
+ "inference_mode": true,
14
+ "init_lora_weights": true,
15
+ "layer_replication": null,
16
+ "layers_pattern": null,
17
+ "layers_to_transform": null,
18
+ "loftq_config": {},
19
+ "lora_alpha": 256,
20
+ "lora_bias": false,
21
+ "lora_dropout": 0.15,
22
+ "megatron_config": null,
23
+ "megatron_core": "megatron.core",
24
+ "modules_to_save": null,
25
+ "peft_type": "LORA",
26
+ "peft_version": "0.18.1",
27
+ "qalora_group_size": 16,
28
+ "r": 128,
29
+ "rank_pattern": {},
30
+ "revision": null,
31
+ "target_modules": [
32
+ "gate_proj",
33
+ "in_proj_z",
34
+ "v_proj",
35
+ "o_proj",
36
+ "q_proj",
37
+ "k_proj",
38
+ "down_proj",
39
+ "up_proj",
40
+ "out_proj",
41
+ "in_proj_qkv"
42
+ ],
43
+ "target_parameters": null,
44
+ "task_type": "CAUSAL_LM",
45
+ "trainable_token_indices": null,
46
+ "use_dora": false,
47
+ "use_qalora": false,
48
+ "use_rslora": false
49
+ }
adapter_model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4ab2a7cf8c629725381aa12f056fd3e266f52d8bd8efce6c132711dcbe014663
3
+ size 1283518408
all_results.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "total_flos": 4.311845712166912e+16,
3
+ "train_loss": 0.6052066683769226,
4
+ "train_runtime": 387.7625,
5
+ "train_samples_per_second": 9.026,
6
+ "train_steps_per_second": 0.142
7
+ }
chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is false %}
150
+ {{- '<think>\n\n</think>\n\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
eval_results.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metrics": {
3
+ "accuracy": 0.71,
4
+ "f1_micro": 0.71,
5
+ "f1_macro": 0.47328777778730374
6
+ },
7
+ "num_samples": 300
8
+ }
predictions.jsonl ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:06b9509352d2af50381ab2247e083b80d32d5c0aba91c272ca9ff729b6a0e523
3
+ size 19989325
tokenizer_config.json ADDED
@@ -0,0 +1,32 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "audio_bos_token": "<|audio_start|>",
4
+ "audio_eos_token": "<|audio_end|>",
5
+ "audio_token": "<|audio_pad|>",
6
+ "backend": "tokenizers",
7
+ "bos_token": null,
8
+ "clean_up_tokenization_spaces": false,
9
+ "eos_token": "<|im_end|>",
10
+ "errors": "replace",
11
+ "image_token": "<|image_pad|>",
12
+ "is_local": false,
13
+ "local_files_only": false,
14
+ "model_max_length": 262144,
15
+ "model_specific_special_tokens": {
16
+ "audio_bos_token": "<|audio_start|>",
17
+ "audio_eos_token": "<|audio_end|>",
18
+ "audio_token": "<|audio_pad|>",
19
+ "image_token": "<|image_pad|>",
20
+ "video_token": "<|video_pad|>",
21
+ "vision_bos_token": "<|vision_start|>",
22
+ "vision_eos_token": "<|vision_end|>"
23
+ },
24
+ "pad_token": "<|endoftext|>",
25
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
26
+ "split_special_tokens": false,
27
+ "tokenizer_class": "Qwen2Tokenizer",
28
+ "unk_token": null,
29
+ "video_token": "<|video_pad|>",
30
+ "vision_bos_token": "<|vision_start|>",
31
+ "vision_eos_token": "<|vision_end|>"
32
+ }
train_results.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "total_flos": 4.311845712166912e+16,
3
+ "train_loss": 0.6052066683769226,
4
+ "train_runtime": 387.7625,
5
+ "train_samples_per_second": 9.026,
6
+ "train_steps_per_second": 0.142
7
+ }
trainer_state.json ADDED
@@ -0,0 +1,104 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": 55,
3
+ "best_metric": 0.9717274904251099,
4
+ "best_model_checkpoint": "checkpoint-55",
5
+ "epoch": 5.0,
6
+ "eval_steps": 9999,
7
+ "global_step": 55,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "entropy": 1.1605470314621926,
14
+ "epoch": 0.9090909090909091,
15
+ "grad_norm": 1.0625,
16
+ "learning_rate": 0.0001935016242685415,
17
+ "loss": 1.29622802734375,
18
+ "mean_token_accuracy": 0.7211568385362626,
19
+ "num_tokens": 130008.0,
20
+ "step": 10
21
+ },
22
+ {
23
+ "entropy": 0.7875008955597878,
24
+ "epoch": 1.8181818181818183,
25
+ "grad_norm": 0.328125,
26
+ "learning_rate": 0.00015680647467311557,
27
+ "loss": 0.7808365821838379,
28
+ "mean_token_accuracy": 0.8202625662088394,
29
+ "num_tokens": 260150.0,
30
+ "step": 20
31
+ },
32
+ {
33
+ "entropy": 0.5799194134771823,
34
+ "epoch": 2.7272727272727275,
35
+ "grad_norm": 0.455078125,
36
+ "learning_rate": 0.0001,
37
+ "loss": 0.5477523803710938,
38
+ "mean_token_accuracy": 0.8625506401062012,
39
+ "num_tokens": 391211.0,
40
+ "step": 30
41
+ },
42
+ {
43
+ "entropy": 0.3913830861449242,
44
+ "epoch": 3.6363636363636362,
45
+ "grad_norm": 0.3984375,
46
+ "learning_rate": 4.3193525326884435e-05,
47
+ "loss": 0.35632944107055664,
48
+ "mean_token_accuracy": 0.9054566085338592,
49
+ "num_tokens": 519150.0,
50
+ "step": 40
51
+ },
52
+ {
53
+ "entropy": 0.3066374149173498,
54
+ "epoch": 4.545454545454545,
55
+ "grad_norm": 0.4296875,
56
+ "learning_rate": 6.498375731458528e-06,
57
+ "loss": 0.24866189956665039,
58
+ "mean_token_accuracy": 0.9313088282942772,
59
+ "num_tokens": 649203.0,
60
+ "step": 50
61
+ },
62
+ {
63
+ "epoch": 5.0,
64
+ "eval_entropy": 0.5388994969819721,
65
+ "eval_loss": 0.9717274904251099,
66
+ "eval_mean_token_accuracy": 0.8135968666327628,
67
+ "eval_num_tokens": 714955.0,
68
+ "eval_runtime": 6.6266,
69
+ "eval_samples_per_second": 45.272,
70
+ "eval_steps_per_second": 2.867,
71
+ "step": 55
72
+ },
73
+ {
74
+ "epoch": 5.0,
75
+ "step": 55,
76
+ "total_flos": 4.311845712166912e+16,
77
+ "train_loss": 0.6052066683769226,
78
+ "train_runtime": 387.7625,
79
+ "train_samples_per_second": 9.026,
80
+ "train_steps_per_second": 0.142
81
+ }
82
+ ],
83
+ "logging_steps": 10,
84
+ "max_steps": 55,
85
+ "num_input_tokens_seen": 0,
86
+ "num_train_epochs": 5,
87
+ "save_steps": 9999,
88
+ "stateful_callbacks": {
89
+ "TrainerControl": {
90
+ "args": {
91
+ "should_epoch_stop": false,
92
+ "should_evaluate": false,
93
+ "should_log": false,
94
+ "should_save": true,
95
+ "should_training_stop": true
96
+ },
97
+ "attributes": {}
98
+ }
99
+ },
100
+ "total_flos": 4.311845712166912e+16,
101
+ "train_batch_size": 2,
102
+ "trial_name": null,
103
+ "trial_params": null
104
+ }