sepsy070716 commited on
Commit
40bc205
·
verified ·
1 Parent(s): 6c34305

Release Qwen3.5-4B-A3B Student v2 with evaluation evidence

Browse files
Files changed (49) hide show
  1. .gitattributes +1 -0
  2. LICENSE +202 -0
  3. README.md +155 -0
  4. chat_template.jinja +154 -0
  5. config.json +80 -0
  6. conversion_manifest.json +12 -0
  7. evaluation/chat_gate_60.json +762 -0
  8. evaluation/lm_eval_teacher4b_dev100.json +208 -0
  9. evaluation/lm_eval_v2_dev100.json +208 -0
  10. evaluation/multilingual_lm_loss.json +115 -0
  11. evaluation/openai_service_20.json +165 -0
  12. merges.txt +0 -0
  13. model-common.safetensors +3 -0
  14. model-layer-00.safetensors +3 -0
  15. model-layer-01.safetensors +3 -0
  16. model-layer-02.safetensors +3 -0
  17. model-layer-03.safetensors +3 -0
  18. model-layer-04.safetensors +3 -0
  19. model-layer-05.safetensors +3 -0
  20. model-layer-06.safetensors +3 -0
  21. model-layer-07.safetensors +3 -0
  22. model-layer-08.safetensors +3 -0
  23. model-layer-09.safetensors +3 -0
  24. model-layer-10.safetensors +3 -0
  25. model-layer-11.safetensors +3 -0
  26. model-layer-12.safetensors +3 -0
  27. model-layer-13.safetensors +3 -0
  28. model-layer-14.safetensors +3 -0
  29. model-layer-15.safetensors +3 -0
  30. model-layer-16.safetensors +3 -0
  31. model-layer-17.safetensors +3 -0
  32. model-layer-18.safetensors +3 -0
  33. model-layer-19.safetensors +3 -0
  34. model-layer-20.safetensors +3 -0
  35. model-layer-21.safetensors +3 -0
  36. model-layer-22.safetensors +3 -0
  37. model-layer-23.safetensors +3 -0
  38. model.safetensors.index.json +423 -0
  39. research_code/chat-gate-60.jsonl +60 -0
  40. research_code/compare_lm_loss.py +116 -0
  41. research_code/convert_2b_to_4b_a3b.py +228 -0
  42. research_code/evaluate_chat_gate.py +121 -0
  43. research_code/rescore_chat_gate.py +35 -0
  44. research_code/service/README.md +20 -0
  45. research_code/service/app.py +167 -0
  46. research_code/service/test_openai_service.py +69 -0
  47. tokenizer.json +3 -0
  48. tokenizer_config.json +305 -0
  49. vocab.json +0 -0
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
LICENSE ADDED
@@ -0,0 +1,202 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+
2
+ Apache License
3
+ Version 2.0, January 2004
4
+ http://www.apache.org/licenses/
5
+
6
+ TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
7
+
8
+ 1. Definitions.
9
+
10
+ "License" shall mean the terms and conditions for use, reproduction,
11
+ and distribution as defined by Sections 1 through 9 of this document.
12
+
13
+ "Licensor" shall mean the copyright owner or entity authorized by
14
+ the copyright owner that is granting the License.
15
+
16
+ "Legal Entity" shall mean the union of the acting entity and all
17
+ other entities that control, are controlled by, or are under common
18
+ control with that entity. For the purposes of this definition,
19
+ "control" means (i) the power, direct or indirect, to cause the
20
+ direction or management of such entity, whether by contract or
21
+ otherwise, or (ii) ownership of fifty percent (50%) or more of the
22
+ outstanding shares, or (iii) beneficial ownership of such entity.
23
+
24
+ "You" (or "Your") shall mean an individual or Legal Entity
25
+ exercising permissions granted by this License.
26
+
27
+ "Source" form shall mean the preferred form for making modifications,
28
+ including but not limited to software source code, documentation
29
+ source, and configuration files.
30
+
31
+ "Object" form shall mean any form resulting from mechanical
32
+ transformation or translation of a Source form, including but
33
+ not limited to compiled object code, generated documentation,
34
+ and conversions to other media types.
35
+
36
+ "Work" shall mean the work of authorship, whether in Source or
37
+ Object form, made available under the License, as indicated by a
38
+ copyright notice that is included in or attached to the work
39
+ (an example is provided in the Appendix below).
40
+
41
+ "Derivative Works" shall mean any work, whether in Source or Object
42
+ form, that is based on (or derived from) the Work and for which the
43
+ editorial revisions, annotations, elaborations, or other modifications
44
+ represent, as a whole, an original work of authorship. For the purposes
45
+ of this License, Derivative Works shall not include works that remain
46
+ separable from, or merely link (or bind by name) to the interfaces of,
47
+ the Work and Derivative Works thereof.
48
+
49
+ "Contribution" shall mean any work of authorship, including
50
+ the original version of the Work and any modifications or additions
51
+ to that Work or Derivative Works thereof, that is intentionally
52
+ submitted to Licensor for inclusion in the Work by the copyright owner
53
+ or by an individual or Legal Entity authorized to submit on behalf of
54
+ the copyright owner. For the purposes of this definition, "submitted"
55
+ means any form of electronic, verbal, or written communication sent
56
+ to the Licensor or its representatives, including but not limited to
57
+ communication on electronic mailing lists, source code control systems,
58
+ and issue tracking systems that are managed by, or on behalf of, the
59
+ Licensor for the purpose of discussing and improving the Work, but
60
+ excluding communication that is conspicuously marked or otherwise
61
+ designated in writing by the copyright owner as "Not a Contribution."
62
+
63
+ "Contributor" shall mean Licensor and any individual or Legal Entity
64
+ on behalf of whom a Contribution has been received by Licensor and
65
+ subsequently incorporated within the Work.
66
+
67
+ 2. Grant of Copyright License. Subject to the terms and conditions of
68
+ this License, each Contributor hereby grants to You a perpetual,
69
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
70
+ copyright license to reproduce, prepare Derivative Works of,
71
+ publicly display, publicly perform, sublicense, and distribute the
72
+ Work and such Derivative Works in Source or Object form.
73
+
74
+ 3. Grant of Patent License. Subject to the terms and conditions of
75
+ this License, each Contributor hereby grants to You a perpetual,
76
+ worldwide, non-exclusive, no-charge, royalty-free, irrevocable
77
+ (except as stated in this section) patent license to make, have made,
78
+ use, offer to sell, sell, import, and otherwise transfer the Work,
79
+ where such license applies only to those patent claims licensable
80
+ by such Contributor that are necessarily infringed by their
81
+ Contribution(s) alone or by combination of their Contribution(s)
82
+ with the Work to which such Contribution(s) was submitted. If You
83
+ institute patent litigation against any entity (including a
84
+ cross-claim or counterclaim in a lawsuit) alleging that the Work
85
+ or a Contribution incorporated within the Work constitutes direct
86
+ or contributory patent infringement, then any patent licenses
87
+ granted to You under this License for that Work shall terminate
88
+ as of the date such litigation is filed.
89
+
90
+ 4. Redistribution. You may reproduce and distribute copies of the
91
+ Work or Derivative Works thereof in any medium, with or without
92
+ modifications, and in Source or Object form, provided that You
93
+ meet the following conditions:
94
+
95
+ (a) You must give any other recipients of the Work or
96
+ Derivative Works a copy of this License; and
97
+
98
+ (b) You must cause any modified files to carry prominent notices
99
+ stating that You changed the files; and
100
+
101
+ (c) You must retain, in the Source form of any Derivative Works
102
+ that You distribute, all copyright, patent, trademark, and
103
+ attribution notices from the Source form of the Work,
104
+ excluding those notices that do not pertain to any part of
105
+ the Derivative Works; and
106
+
107
+ (d) If the Work includes a "NOTICE" text file as part of its
108
+ distribution, then any Derivative Works that You distribute must
109
+ include a readable copy of the attribution notices contained
110
+ within such NOTICE file, excluding those notices that do not
111
+ pertain to any part of the Derivative Works, in at least one
112
+ of the following places: within a NOTICE text file distributed
113
+ as part of the Derivative Works; within the Source form or
114
+ documentation, if provided along with the Derivative Works; or,
115
+ within a display generated by the Derivative Works, if and
116
+ wherever such third-party notices normally appear. The contents
117
+ of the NOTICE file are for informational purposes only and
118
+ do not modify the License. You may add Your own attribution
119
+ notices within Derivative Works that You distribute, alongside
120
+ or as an addendum to the NOTICE text from the Work, provided
121
+ that such additional attribution notices cannot be construed
122
+ as modifying the License.
123
+
124
+ You may add Your own copyright statement to Your modifications and
125
+ may provide additional or different license terms and conditions
126
+ for use, reproduction, or distribution of Your modifications, or
127
+ for any such Derivative Works as a whole, provided Your use,
128
+ reproduction, and distribution of the Work otherwise complies with
129
+ the conditions stated in this License.
130
+
131
+ 5. Submission of Contributions. Unless You explicitly state otherwise,
132
+ any Contribution intentionally submitted for inclusion in the Work
133
+ by You to the Licensor shall be under the terms and conditions of
134
+ this License, without any additional terms or conditions.
135
+ Notwithstanding the above, nothing herein shall supersede or modify
136
+ the terms of any separate license agreement you may have executed
137
+ with Licensor regarding such Contributions.
138
+
139
+ 6. Trademarks. This License does not grant permission to use the trade
140
+ names, trademarks, service marks, or product names of the Licensor,
141
+ except as required for reasonable and customary use in describing the
142
+ origin of the Work and reproducing the content of the NOTICE file.
143
+
144
+ 7. Disclaimer of Warranty. Unless required by applicable law or
145
+ agreed to in writing, Licensor provides the Work (and each
146
+ Contributor provides its Contributions) on an "AS IS" BASIS,
147
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
148
+ implied, including, without limitation, any warranties or conditions
149
+ of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
150
+ PARTICULAR PURPOSE. You are solely responsible for determining the
151
+ appropriateness of using or redistributing the Work and assume any
152
+ risks associated with Your exercise of permissions under this License.
153
+
154
+ 8. Limitation of Liability. In no event and under no legal theory,
155
+ whether in tort (including negligence), contract, or otherwise,
156
+ unless required by applicable law (such as deliberate and grossly
157
+ negligent acts) or agreed to in writing, shall any Contributor be
158
+ liable to You for damages, including any direct, indirect, special,
159
+ incidental, or consequential damages of any character arising as a
160
+ result of this License or out of the use or inability to use the
161
+ Work (including but not limited to damages for loss of goodwill,
162
+ work stoppage, computer failure or malfunction, or any and all
163
+ other commercial damages or losses), even if such Contributor
164
+ has been advised of the possibility of such damages.
165
+
166
+ 9. Accepting Warranty or Additional Liability. While redistributing
167
+ the Work or Derivative Works thereof, You may choose to offer,
168
+ and charge a fee for, acceptance of support, warranty, indemnity,
169
+ or other liability obligations and/or rights consistent with this
170
+ License. However, in accepting such obligations, You may act only
171
+ on Your own behalf and on Your sole responsibility, not on behalf
172
+ of any other Contributor, and only if You agree to indemnify,
173
+ defend, and hold each Contributor harmless for any liability
174
+ incurred by, or claims asserted against, such Contributor by reason
175
+ of your accepting any such warranty or additional liability.
176
+
177
+ END OF TERMS AND CONDITIONS
178
+
179
+ APPENDIX: How to apply the Apache License to your work.
180
+
181
+ To apply the Apache License to your work, attach the following
182
+ boilerplate notice, with the fields enclosed by brackets "[]"
183
+ replaced with your own identifying information. (Don't include
184
+ the brackets!) The text should be enclosed in the appropriate
185
+ comment syntax for the file format. We also recommend that a
186
+ file or class name and description of purpose be included on the
187
+ same "printed page" as the copyright notice for easier
188
+ identification within third-party archives.
189
+
190
+ Copyright 2026 Alibaba Cloud
191
+
192
+ Licensed under the Apache License, Version 2.0 (the "License");
193
+ you may not use this file except in compliance with the License.
194
+ You may obtain a copy of the License at
195
+
196
+ http://www.apache.org/licenses/LICENSE-2.0
197
+
198
+ Unless required by applicable law or agreed to in writing, software
199
+ distributed under the License is distributed on an "AS IS" BASIS,
200
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
201
+ See the License for the specific language governing permissions and
202
+ limitations under the License.
README.md ADDED
@@ -0,0 +1,155 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: apache-2.0
3
+ base_model:
4
+ - Qwen/Qwen3.5-2B
5
+ - Qwen/Qwen3.5-4B
6
+ library_name: transformers
7
+ pipeline_tag: text-generation
8
+ tags:
9
+ - qwen3_5_moe
10
+ - moe
11
+ - upcycled
12
+ - text-generation
13
+ - research
14
+ language: [ko, en, zh, ja, es, de]
15
+ ---
16
+
17
+ # Qwen3.5-4B-A3B-Student-v2
18
+
19
+ A text-only sparse-MoE release candidate built as a practical local alternative
20
+ to Qwen3.5-4B. It has 4.0B total parameters and 3.0B active parameters per
21
+ token. The initialization preserves Qwen3.5-2B behavior, while adding
22
+ output-neutral trainable capacity for later Qwen3.5-4B distillation.
23
+
24
+ This is an independently measured research release. It is not an official Qwen
25
+ model and does not include vision.
26
+
27
+ ## Architecture
28
+
29
+ | Property | Value |
30
+ |---|---:|
31
+ | Total parameters | 3,995,901,760 |
32
+ | Active parameters/token | 2,995,560,256 |
33
+ | Transformer layers / hidden size | 24 / 2,048 |
34
+ | Experts / selected per token | 2 / 1 |
35
+ | Shared / routed intermediate width | 6,912 / 6,784 |
36
+ | Vision tower | No |
37
+ | Weight dtype | BF16 |
38
+
39
+ Each MoE layer initially computes one half of the original Qwen3.5-2B dense MLP
40
+ through the shared path and one half through the selected routed expert. The
41
+ two routed experts begin functionally identical. Additional neurons have random
42
+ gate/up projections and zero down projections, making them output-neutral but
43
+ trainable. Exact conversion metadata is in `conversion_manifest.json`.
44
+
45
+ ## Evaluation
46
+
47
+ All reported results were produced locally on an Apple M4 with 32GB unified
48
+ memory. Raw JSON reports are included in `evaluation/`.
49
+
50
+ ### Chat and sentence generation
51
+
52
+ The fixed gate contains 60 prompts: 10 each in Korean, English, Chinese,
53
+ Japanese, Spanish, and German. It covers facts, arithmetic, translation,
54
+ instruction following, and free-form sentence generation.
55
+
56
+ | Result | Score |
57
+ |---|---:|
58
+ | Non-degenerate/correct automatic checks | 60 / 60 |
59
+ | Languages meeting the gate | 6 / 6 |
60
+
61
+ Six chemical-formula answers used the correct Unicode spelling `H₂O`; the
62
+ scorer normalizes Unicode subscripts before comparison.
63
+
64
+ ### Multilingual held-out LM loss
65
+
66
+ Four held-out FineWeb/FineWeb2 documents per language, 128 tokens per document:
67
+
68
+ | Model | Mean loss | Relative to Qwen3.5-4B |
69
+ |---|---:|---:|
70
+ | Qwen3.5-4B | 2.9399 | 1.000x |
71
+ | This model | 3.2085 | 1.091x |
72
+ | Qwen3.5-2B | 3.2085 | 1.091x |
73
+
74
+ ### Standard benchmark development subset
75
+
76
+ EleutherAI `lm-evaluation-harness==0.4.12`, zero-shot, BF16, first 100 examples
77
+ per task. These limited results are development indicators, **not full-task
78
+ benchmark claims**.
79
+
80
+ | Model | ARC-Easy acc_norm | HellaSwag acc_norm | Mean |
81
+ |---|---:|---:|---:|
82
+ | Qwen3.5-4B | 0.81 | 0.68 | 0.745 |
83
+ | This model | 0.73 | 0.62 | 0.675 |
84
+
85
+ The subset mean is 90.6% of the Qwen3.5-4B teacher mean.
86
+
87
+ ### Local service gate
88
+
89
+ The included FastAPI service completed 20/20 consecutive non-streaming
90
+ `POST /v1/chat/completions` requests:
91
+
92
+ | Metric | Value |
93
+ |---|---:|
94
+ | Successful requests | 20 / 20 |
95
+ | Mean latency | 2.60 s |
96
+ | p95 latency | 3.57 s |
97
+ | MPS allocated memory | 7.62 GB |
98
+
99
+ Requests generated up to 16 new tokens. See `evaluation/openai_service_20.json`.
100
+
101
+ ## Transformers usage
102
+
103
+ Use Transformers 5.13.0 or another version that provides
104
+ `Qwen3_5MoeForCausalLM`:
105
+
106
+ ```python
107
+ import torch
108
+ from transformers import AutoTokenizer, Qwen3_5MoeForCausalLM
109
+
110
+ model_id = "sepsy070716/Qwen3.5-4B-A3B-Student-v2"
111
+ tokenizer = AutoTokenizer.from_pretrained(model_id)
112
+ model = Qwen3_5MoeForCausalLM.from_pretrained(
113
+ model_id,
114
+ dtype=torch.bfloat16,
115
+ device_map="auto",
116
+ )
117
+
118
+ messages = [{"role": "user", "content": "대한민국의 수도는 어디인가요?"}]
119
+ inputs = tokenizer.apply_chat_template(
120
+ messages,
121
+ tokenize=True,
122
+ add_generation_prompt=True,
123
+ enable_thinking=False,
124
+ return_tensors="pt",
125
+ return_dict=True,
126
+ ).to(model.device)
127
+ output = model.generate(**inputs, max_new_tokens=64, do_sample=False)
128
+ print(tokenizer.decode(output[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True))
129
+ ```
130
+
131
+ ## Reproduction and service
132
+
133
+ `research_code/` contains the converter, multilingual loss comparison, 60-prompt
134
+ gate and scorer, plus the local OpenAI-compatible service and its 20-request
135
+ test. The service implements `GET /health`, `GET /v1/models`, and non-streaming
136
+ `POST /v1/chat/completions`.
137
+
138
+ ## Limitations
139
+
140
+ - Current quality is inherited primarily from Qwen3.5-2B; the extra capacity has
141
+ not yet received large-scale continued pretraining or teacher distillation.
142
+ - The 100-example ARC-Easy/HellaSwag figures are small development subsets and
143
+ have substantial sampling uncertainty. Run the full tasks before making
144
+ publication or production claims.
145
+ - This model is text-only and cannot replace the original model's vision path.
146
+ - The included server is a single-process local research server. It has no
147
+ authentication, TLS, streaming, tool calling, or multi-worker support.
148
+ - Apply the same safety, bias, privacy, and factuality evaluation required for
149
+ any deployment of the upstream Qwen models.
150
+
151
+ ## License and attribution
152
+
153
+ Released under Apache-2.0, following the included upstream license. Derived from
154
+ Qwen3.5-2B weights and evaluated against Qwen3.5-4B. Qwen model names and
155
+ trademarks belong to their respective owners.
chat_template.jinja ADDED
@@ -0,0 +1,154 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {%- set image_count = namespace(value=0) %}
2
+ {%- set video_count = namespace(value=0) %}
3
+ {%- macro render_content(content, do_vision_count, is_system_content=false) %}
4
+ {%- if content is string %}
5
+ {{- content }}
6
+ {%- elif content is iterable and content is not mapping %}
7
+ {%- for item in content %}
8
+ {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
9
+ {%- if is_system_content %}
10
+ {{- raise_exception('System message cannot contain images.') }}
11
+ {%- endif %}
12
+ {%- if do_vision_count %}
13
+ {%- set image_count.value = image_count.value + 1 %}
14
+ {%- endif %}
15
+ {%- if add_vision_id %}
16
+ {{- 'Picture ' ~ image_count.value ~ ': ' }}
17
+ {%- endif %}
18
+ {{- '<|vision_start|><|image_pad|><|vision_end|>' }}
19
+ {%- elif 'video' in item or item.type == 'video' %}
20
+ {%- if is_system_content %}
21
+ {{- raise_exception('System message cannot contain videos.') }}
22
+ {%- endif %}
23
+ {%- if do_vision_count %}
24
+ {%- set video_count.value = video_count.value + 1 %}
25
+ {%- endif %}
26
+ {%- if add_vision_id %}
27
+ {{- 'Video ' ~ video_count.value ~ ': ' }}
28
+ {%- endif %}
29
+ {{- '<|vision_start|><|video_pad|><|vision_end|>' }}
30
+ {%- elif 'text' in item %}
31
+ {{- item.text }}
32
+ {%- else %}
33
+ {{- raise_exception('Unexpected item type in content.') }}
34
+ {%- endif %}
35
+ {%- endfor %}
36
+ {%- elif content is none or content is undefined %}
37
+ {{- '' }}
38
+ {%- else %}
39
+ {{- raise_exception('Unexpected content type.') }}
40
+ {%- endif %}
41
+ {%- endmacro %}
42
+ {%- if not messages %}
43
+ {{- raise_exception('No messages provided.') }}
44
+ {%- endif %}
45
+ {%- if tools and tools is iterable and tools is not mapping %}
46
+ {{- '<|im_start|>system\n' }}
47
+ {{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
48
+ {%- for tool in tools %}
49
+ {{- "\n" }}
50
+ {{- tool | tojson }}
51
+ {%- endfor %}
52
+ {{- "\n</tools>" }}
53
+ {{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
54
+ {%- if messages[0].role == 'system' %}
55
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
56
+ {%- if content %}
57
+ {{- '\n\n' + content }}
58
+ {%- endif %}
59
+ {%- endif %}
60
+ {{- '<|im_end|>\n' }}
61
+ {%- else %}
62
+ {%- if messages[0].role == 'system' %}
63
+ {%- set content = render_content(messages[0].content, false, true)|trim %}
64
+ {{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
65
+ {%- endif %}
66
+ {%- endif %}
67
+ {%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
68
+ {%- for message in messages[::-1] %}
69
+ {%- set index = (messages|length - 1) - loop.index0 %}
70
+ {%- if ns.multi_step_tool and message.role == "user" %}
71
+ {%- set content = render_content(message.content, false)|trim %}
72
+ {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
73
+ {%- set ns.multi_step_tool = false %}
74
+ {%- set ns.last_query_index = index %}
75
+ {%- endif %}
76
+ {%- endif %}
77
+ {%- endfor %}
78
+ {%- if ns.multi_step_tool %}
79
+ {{- raise_exception('No user query found in messages.') }}
80
+ {%- endif %}
81
+ {%- for message in messages %}
82
+ {%- set content = render_content(message.content, true)|trim %}
83
+ {%- if message.role == "system" %}
84
+ {%- if not loop.first %}
85
+ {{- raise_exception('System message must be at the beginning.') }}
86
+ {%- endif %}
87
+ {%- elif message.role == "user" %}
88
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
89
+ {%- elif message.role == "assistant" %}
90
+ {%- set reasoning_content = '' %}
91
+ {%- if message.reasoning_content is string %}
92
+ {%- set reasoning_content = message.reasoning_content %}
93
+ {%- else %}
94
+ {%- if '</think>' in content %}
95
+ {%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
96
+ {%- set content = content.split('</think>')[-1].lstrip('\n') %}
97
+ {%- endif %}
98
+ {%- endif %}
99
+ {%- set reasoning_content = reasoning_content|trim %}
100
+ {%- if loop.index0 > ns.last_query_index %}
101
+ {{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
102
+ {%- else %}
103
+ {{- '<|im_start|>' + message.role + '\n' + content }}
104
+ {%- endif %}
105
+ {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
106
+ {%- for tool_call in message.tool_calls %}
107
+ {%- if tool_call.function is defined %}
108
+ {%- set tool_call = tool_call.function %}
109
+ {%- endif %}
110
+ {%- if loop.first %}
111
+ {%- if content|trim %}
112
+ {{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
113
+ {%- else %}
114
+ {{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
115
+ {%- endif %}
116
+ {%- else %}
117
+ {{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
118
+ {%- endif %}
119
+ {%- if tool_call.arguments is defined %}
120
+ {%- for args_name, args_value in tool_call.arguments|items %}
121
+ {{- '<parameter=' + args_name + '>\n' }}
122
+ {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
123
+ {{- args_value }}
124
+ {{- '\n</parameter>\n' }}
125
+ {%- endfor %}
126
+ {%- endif %}
127
+ {{- '</function>\n</tool_call>' }}
128
+ {%- endfor %}
129
+ {%- endif %}
130
+ {{- '<|im_end|>\n' }}
131
+ {%- elif message.role == "tool" %}
132
+ {%- if loop.previtem and loop.previtem.role != "tool" %}
133
+ {{- '<|im_start|>user' }}
134
+ {%- endif %}
135
+ {{- '\n<tool_response>\n' }}
136
+ {{- content }}
137
+ {{- '\n</tool_response>' }}
138
+ {%- if not loop.last and loop.nextitem.role != "tool" %}
139
+ {{- '<|im_end|>\n' }}
140
+ {%- elif loop.last %}
141
+ {{- '<|im_end|>\n' }}
142
+ {%- endif %}
143
+ {%- else %}
144
+ {{- raise_exception('Unexpected message role.') }}
145
+ {%- endif %}
146
+ {%- endfor %}
147
+ {%- if add_generation_prompt %}
148
+ {{- '<|im_start|>assistant\n' }}
149
+ {%- if enable_thinking is defined and enable_thinking is true %}
150
+ {{- '<think>\n' }}
151
+ {%- else %}
152
+ {{- '<think>\n\n</think>\n\n' }}
153
+ {%- endif %}
154
+ {%- endif %}
config.json ADDED
@@ -0,0 +1,80 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "Qwen3_5MoeForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "attn_output_gate": true,
8
+ "bos_token_id": null,
9
+ "dtype": "bfloat16",
10
+ "eos_token_id": 248044,
11
+ "full_attention_interval": 4,
12
+ "head_dim": 256,
13
+ "hidden_act": "silu",
14
+ "hidden_size": 2048,
15
+ "initializer_range": 0.02,
16
+ "layer_types": [
17
+ "linear_attention",
18
+ "linear_attention",
19
+ "linear_attention",
20
+ "full_attention",
21
+ "linear_attention",
22
+ "linear_attention",
23
+ "linear_attention",
24
+ "full_attention",
25
+ "linear_attention",
26
+ "linear_attention",
27
+ "linear_attention",
28
+ "full_attention",
29
+ "linear_attention",
30
+ "linear_attention",
31
+ "linear_attention",
32
+ "full_attention",
33
+ "linear_attention",
34
+ "linear_attention",
35
+ "linear_attention",
36
+ "full_attention",
37
+ "linear_attention",
38
+ "linear_attention",
39
+ "linear_attention",
40
+ "full_attention"
41
+ ],
42
+ "linear_conv_kernel_dim": 4,
43
+ "linear_key_head_dim": 128,
44
+ "linear_num_key_heads": 16,
45
+ "linear_num_value_heads": 16,
46
+ "linear_value_head_dim": 128,
47
+ "mamba_ssm_dtype": "float32",
48
+ "max_position_embeddings": 262144,
49
+ "mlp_only_layers": [],
50
+ "model_type": "qwen3_5_moe_text",
51
+ "moe_intermediate_size": 6784,
52
+ "mtp_num_hidden_layers": 1,
53
+ "mtp_use_dedicated_embeddings": false,
54
+ "num_attention_heads": 8,
55
+ "num_experts": 2,
56
+ "num_experts_per_tok": 1,
57
+ "num_hidden_layers": 24,
58
+ "num_key_value_heads": 2,
59
+ "output_router_logits": false,
60
+ "pad_token_id": null,
61
+ "partial_rotary_factor": 0.25,
62
+ "rms_norm_eps": 1e-06,
63
+ "rope_parameters": {
64
+ "mrope_interleaved": true,
65
+ "mrope_section": [
66
+ 11,
67
+ 11,
68
+ 10
69
+ ],
70
+ "partial_rotary_factor": 0.25,
71
+ "rope_theta": 10000000,
72
+ "rope_type": "default"
73
+ },
74
+ "router_aux_loss_coef": 0.001,
75
+ "shared_expert_intermediate_size": 6912,
76
+ "tie_word_embeddings": true,
77
+ "transformers_version": "5.13.0",
78
+ "use_cache": true,
79
+ "vocab_size": 248320
80
+ }
conversion_manifest.json ADDED
@@ -0,0 +1,12 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "source": "models/Qwen/Qwen3.5-2B",
3
+ "initial_function": "Qwen3.5-2B text model (dense MLP split 50/50)",
4
+ "total_parameters": 3995901760,
5
+ "active_parameters": 2995560256,
6
+ "num_experts": 2,
7
+ "experts_per_token": 1,
8
+ "shared_intermediate_size": 6912,
9
+ "routed_intermediate_size": 6784,
10
+ "vision_included": false,
11
+ "seed": 35
12
+ }
evaluation/chat_gate_60.json ADDED
@@ -0,0 +1,762 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "model": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
3
+ "total": 60,
4
+ "passed": 60,
5
+ "pass_rate": 1.0,
6
+ "gate_threshold": 0.9,
7
+ "gate_passed": true,
8
+ "by_language": {
9
+ "de": {
10
+ "passed": 10,
11
+ "total": 10,
12
+ "pass_rate": 1.0
13
+ },
14
+ "en": {
15
+ "passed": 10,
16
+ "total": 10,
17
+ "pass_rate": 1.0
18
+ },
19
+ "es": {
20
+ "passed": 10,
21
+ "total": 10,
22
+ "pass_rate": 1.0
23
+ },
24
+ "ja": {
25
+ "passed": 10,
26
+ "total": 10,
27
+ "pass_rate": 1.0
28
+ },
29
+ "ko": {
30
+ "passed": 10,
31
+ "total": 10,
32
+ "pass_rate": 1.0
33
+ },
34
+ "zh": {
35
+ "passed": 10,
36
+ "total": 10,
37
+ "pass_rate": 1.0
38
+ }
39
+ },
40
+ "results": [
41
+ {
42
+ "id": "ko_01",
43
+ "language": "ko",
44
+ "prompt": "프랑스의 수도는 어디인가요? 짧게 답하세요.",
45
+ "expected_any": [
46
+ "파리"
47
+ ],
48
+ "completion": "파리\n",
49
+ "passed": true,
50
+ "failures": [],
51
+ "new_tokens": 5,
52
+ "elapsed_seconds": 2.93180704198312
53
+ },
54
+ {
55
+ "id": "ko_02",
56
+ "language": "ko",
57
+ "prompt": "7 곱하기 8은 얼마인가요? 숫자로 답하세요.",
58
+ "expected_any": [
59
+ "56"
60
+ ],
61
+ "completion": "7 곱하기 8은 **56**입니다.\n",
62
+ "passed": true,
63
+ "failures": [],
64
+ "new_tokens": 16,
65
+ "elapsed_seconds": 3.4328290830017067
66
+ },
67
+ {
68
+ "id": "ko_03",
69
+ "language": "ko",
70
+ "prompt": "물의 화학식은 무엇인가요?",
71
+ "expected_any": [
72
+ "H2O",
73
+ "h2o"
74
+ ],
75
+ "completion": "물의 화학식은 **H₂O**입니다.\n\n이 화학식은 물의 구성 원자 수를 나타냅니다:\n* **H**: 수소 원자 2 개\n* **O**: 산소 원자 1",
76
+ "passed": true,
77
+ "failures": [],
78
+ "new_tokens": 48,
79
+ "elapsed_seconds": 10.234737457998563
80
+ },
81
+ {
82
+ "id": "ko_04",
83
+ "language": "ko",
84
+ "prompt": "태양계에서 가장 큰 행성은 무엇인가요?",
85
+ "expected_any": [
86
+ "목성"
87
+ ],
88
+ "completion": "태양계에서 가장 큰 행성은 **목성**입니다.\n\n목성은 태양계에서 가장 큰 행성으로, 지구의 약 318 배의 크기와 약 1,321 배의 질량을",
89
+ "passed": true,
90
+ "failures": [],
91
+ "new_tokens": 48,
92
+ "elapsed_seconds": 9.949900375038851
93
+ },
94
+ {
95
+ "id": "ko_05",
96
+ "language": "ko",
97
+ "prompt": "조용한 겨울 아침을 묘사하는 자연스러운 문장 두 개를 써 주세요.",
98
+ "completion": "1. 창문 너머로 비치는 차가운 햇살이 눈부시며, 거친 바람이 나무의 가지 사이로 스치며 겨울 아침의 고요함을 더 깊게 만듭니다.\n2. 눈",
99
+ "passed": true,
100
+ "failures": [],
101
+ "new_tokens": 48,
102
+ "elapsed_seconds": 10.018696333048865
103
+ },
104
+ {
105
+ "id": "ko_06",
106
+ "language": "ko",
107
+ "prompt": "친구에게 약속 시간을 10분 늦겠다고 정중히 알리는 한 문장을 써 주세요.",
108
+ "completion": "\"안녕하세요! 오늘 약속 시간이 10 분 늦어지니 미리 알려드리고 싶네요. 혹시 다른 시간으로 조정할 수 있을까요?\"\n",
109
+ "passed": true,
110
+ "failures": [],
111
+ "new_tokens": 36,
112
+ "elapsed_seconds": 7.496548916969914
113
+ },
114
+ {
115
+ "id": "ko_07",
116
+ "language": "ko",
117
+ "prompt": "인공지능 연구에서 재현성이 중요한 이유를 두 문장으로 설명하세요.",
118
+ "completion": "인공지능 연구에서 재현성은 모델의 신뢰성을 확보하고, 다른 연구자들이 동일한 결과를 검증할 수 있는 기준을 마련하기 위함입니다. 또한, 복잡한 알고리즘의 동작 원리를 투명하게 이해하고, 윤리적 문제",
119
+ "passed": true,
120
+ "failures": [],
121
+ "new_tokens": 48,
122
+ "elapsed_seconds": 9.953075792000163
123
+ },
124
+ {
125
+ "id": "ko_08",
126
+ "language": "ko",
127
+ "prompt": "다른 말 없이 정확히 '확인'이라고만 답하세요.",
128
+ "expected_any": [
129
+ "확인"
130
+ ],
131
+ "completion": "확인\n",
132
+ "passed": true,
133
+ "failures": [],
134
+ "new_tokens": 4,
135
+ "elapsed_seconds": 0.8486927500343882
136
+ },
137
+ {
138
+ "id": "ko_09",
139
+ "language": "ko",
140
+ "prompt": "사과, 바나나, 포도를 번호가 있는 세 항목으로 나열하세요.",
141
+ "completion": "네, 요청하신 대로 세 가지 항목을 번호로 나열했습니다.\n\n1. 사과\n2. 바나나\n3. 포도\n",
142
+ "passed": true,
143
+ "failures": [],
144
+ "new_tokens": 33,
145
+ "elapsed_seconds": 6.8605534169473685
146
+ },
147
+ {
148
+ "id": "ko_10",
149
+ "language": "ko",
150
+ "prompt": "'Good morning'을 자연스러운 한국어로 번역하세요.",
151
+ "expected_any": [
152
+ "좋은 아침",
153
+ "안녕하세요"
154
+ ],
155
+ "completion": "안녕하세요.\n",
156
+ "passed": true,
157
+ "failures": [],
158
+ "new_tokens": 7,
159
+ "elapsed_seconds": 1.5930803330265917
160
+ },
161
+ {
162
+ "id": "en_01",
163
+ "language": "en",
164
+ "prompt": "What is the capital of France? Answer briefly.",
165
+ "expected_any": [
166
+ "Paris"
167
+ ],
168
+ "completion": "The capital of France is **Paris**.\n",
169
+ "passed": true,
170
+ "failures": [],
171
+ "new_tokens": 11,
172
+ "elapsed_seconds": 2.369376624992583
173
+ },
174
+ {
175
+ "id": "en_02",
176
+ "language": "en",
177
+ "prompt": "What is 7 multiplied by 8? Answer with a number.",
178
+ "expected_any": [
179
+ "56"
180
+ ],
181
+ "completion": "56\n",
182
+ "passed": true,
183
+ "failures": [],
184
+ "new_tokens": 5,
185
+ "elapsed_seconds": 1.1381639169994742
186
+ },
187
+ {
188
+ "id": "en_03",
189
+ "language": "en",
190
+ "prompt": "What is the chemical formula for water?",
191
+ "expected_any": [
192
+ "H2O",
193
+ "h2o"
194
+ ],
195
+ "completion": "The chemical formula for water is **H₂O**.\n\nThis indicates that a single water molecule consists of two hydrogen atoms covalently bonded to one oxygen atom.\n",
196
+ "passed": true,
197
+ "failures": [],
198
+ "new_tokens": 36,
199
+ "elapsed_seconds": 7.6181977079831995
200
+ },
201
+ {
202
+ "id": "en_04",
203
+ "language": "en",
204
+ "prompt": "What is the largest planet in the Solar System?",
205
+ "expected_any": [
206
+ "Jupiter"
207
+ ],
208
+ "completion": "The largest planet in the Solar System is **Jupiter**.\n\nDiscovered by Galileo Galilei in 1610, Jupiter is a gas giant with a mass that is more than twice that of all the other planets combined",
209
+ "passed": true,
210
+ "failures": [],
211
+ "new_tokens": 48,
212
+ "elapsed_seconds": 9.993648000003304
213
+ },
214
+ {
215
+ "id": "en_05",
216
+ "language": "en",
217
+ "prompt": "Write two natural sentences describing a quiet winter morning.",
218
+ "completion": "The sun filters through the heavy, frost-covered branches of the oak trees, casting a soft, golden glow across the snow-dusted lawn.\nThe air is crisp and still, filled only with the gentle rustle of birch leaves and",
219
+ "passed": true,
220
+ "failures": [],
221
+ "new_tokens": 48,
222
+ "elapsed_seconds": 9.83665666700108
223
+ },
224
+ {
225
+ "id": "en_06",
226
+ "language": "en",
227
+ "prompt": "Write one polite sentence telling a friend you will be ten minutes late.",
228
+ "completion": "\"Hey, I'm heading out a bit early, so I'll be ten minutes late for our meeting.\"\n",
229
+ "passed": true,
230
+ "failures": [],
231
+ "new_tokens": 25,
232
+ "elapsed_seconds": 5.151561209000647
233
+ },
234
+ {
235
+ "id": "en_07",
236
+ "language": "en",
237
+ "prompt": "Explain in two sentences why reproducibility matters in AI research.",
238
+ "completion": "Reproducibility is essential because it allows the scientific community to verify that results are not artifacts of chance or specific, undocumented choices made by the original researchers. Without the ability to reproduce findings, the validity of AI models and their applications cannot be",
239
+ "passed": true,
240
+ "failures": [],
241
+ "new_tokens": 48,
242
+ "elapsed_seconds": 9.94930508401012
243
+ },
244
+ {
245
+ "id": "en_08",
246
+ "language": "en",
247
+ "prompt": "Reply with exactly the word 'confirmed' and nothing else.",
248
+ "expected_any": [
249
+ "confirmed"
250
+ ],
251
+ "completion": "confirmed\n",
252
+ "passed": true,
253
+ "failures": [],
254
+ "new_tokens": 4,
255
+ "elapsed_seconds": 0.9270532079972327
256
+ },
257
+ {
258
+ "id": "en_09",
259
+ "language": "en",
260
+ "prompt": "List apple, banana, and grape as three numbered items.",
261
+ "completion": "1. Apple\n2. Banana\n3. Grape\n",
262
+ "passed": true,
263
+ "failures": [],
264
+ "new_tokens": 14,
265
+ "elapsed_seconds": 2.939590915979352
266
+ },
267
+ {
268
+ "id": "en_10",
269
+ "language": "en",
270
+ "prompt": "Translate '좋은 아침입니다' into natural English.",
271
+ "expected_any": [
272
+ "good morning"
273
+ ],
274
+ "completion": "Good morning.\n",
275
+ "passed": true,
276
+ "failures": [],
277
+ "new_tokens": 6,
278
+ "elapsed_seconds": 1.294799041992519
279
+ },
280
+ {
281
+ "id": "zh_01",
282
+ "language": "zh",
283
+ "prompt": "法国的首都是哪里?请简短回答。",
284
+ "expected_any": [
285
+ "巴黎"
286
+ ],
287
+ "completion": "巴黎\n",
288
+ "passed": true,
289
+ "failures": [],
290
+ "new_tokens": 4,
291
+ "elapsed_seconds": 0.8428977089934051
292
+ },
293
+ {
294
+ "id": "zh_02",
295
+ "language": "zh",
296
+ "prompt": "7乘以8等于多少?请用数字回答。",
297
+ "expected_any": [
298
+ "56"
299
+ ],
300
+ "completion": "56\n",
301
+ "passed": true,
302
+ "failures": [],
303
+ "new_tokens": 5,
304
+ "elapsed_seconds": 1.042713167029433
305
+ },
306
+ {
307
+ "id": "zh_03",
308
+ "language": "zh",
309
+ "prompt": "水的化学式是什么?",
310
+ "expected_any": [
311
+ "H2O",
312
+ "h2o"
313
+ ],
314
+ "completion": "水的化学式是 **H₂O**。\n\n这个符号表示:\n* **H** 代表氢(Hydrogen)\n* **O** 代表氧(Oxygen)\n* 下标的 **",
315
+ "passed": true,
316
+ "failures": [],
317
+ "new_tokens": 48,
318
+ "elapsed_seconds": 9.921300665999297
319
+ },
320
+ {
321
+ "id": "zh_04",
322
+ "language": "zh",
323
+ "prompt": "太阳系中最大的行星是什么?",
324
+ "expected_any": [
325
+ "木星"
326
+ ],
327
+ "completion": "太阳系中最大的行星是**木星**(Jupiter)。\n\n它的质量约为其他所有行星总和的 2.5 倍,是太阳系中体积和质量最大的行星。木星是一颗气态巨行星,其直径约为地球直径",
328
+ "passed": true,
329
+ "failures": [],
330
+ "new_tokens": 48,
331
+ "elapsed_seconds": 9.965762625040952
332
+ },
333
+ {
334
+ "id": "zh_05",
335
+ "language": "zh",
336
+ "prompt": "用两个自然的句子描写安静的冬日清晨。",
337
+ "completion": "冬日的清晨,薄雾如轻纱般笼罩着沉睡的街道,将城市的喧嚣温柔地隔绝在外。\n此时,唯有远处偶尔传来的几声鸟鸣划破寂静,伴随着第一缕阳光穿透云层,洒在静谧",
338
+ "passed": true,
339
+ "failures": [],
340
+ "new_tokens": 48,
341
+ "elapsed_seconds": 10.13702458399348
342
+ },
343
+ {
344
+ "id": "zh_06",
345
+ "language": "zh",
346
+ "prompt": "写一句礼貌的话,告诉朋友你会迟到十分钟。",
347
+ "completion": "“抱歉啦,路上有点堵车,我大概会迟到十分钟,你等我一会儿就好!”\n",
348
+ "passed": true,
349
+ "failures": [],
350
+ "new_tokens": 22,
351
+ "elapsed_seconds": 4.585268583032303
352
+ },
353
+ {
354
+ "id": "zh_07",
355
+ "language": "zh",
356
+ "prompt": "用两句话解释可复现性为何对人工智能研究重要。",
357
+ "completion": "可复现性是人工智能研究的核心基石,它确保了模型在相同数据、超参数和计算环境下的行为一致性,从而允许研究者验证假设、发现规律并构建可靠的知识体系。\n\n此外,这一标准推动了社区协作与",
358
+ "passed": true,
359
+ "failures": [],
360
+ "new_tokens": 48,
361
+ "elapsed_seconds": 10.05659958301112
362
+ },
363
+ {
364
+ "id": "zh_08",
365
+ "language": "zh",
366
+ "prompt": "不要说别的,只回答“收到”。",
367
+ "expected_any": [
368
+ "收到"
369
+ ],
370
+ "completion": "收到\n",
371
+ "passed": true,
372
+ "failures": [],
373
+ "new_tokens": 4,
374
+ "elapsed_seconds": 0.8769729159539565
375
+ },
376
+ {
377
+ "id": "zh_09",
378
+ "language": "zh",
379
+ "prompt": "把苹果、香蕉和葡萄列成三个编号项目。",
380
+ "completion": "好的,以下是将苹果、香蕉和葡萄列成的三个编号项目:\n\n1. 苹果\n2. 香蕉\n3. 葡萄\n",
381
+ "passed": true,
382
+ "failures": [],
383
+ "new_tokens": 33,
384
+ "elapsed_seconds": 7.456039790995419
385
+ },
386
+ {
387
+ "id": "zh_10",
388
+ "language": "zh",
389
+ "prompt": "把“Good morning”翻译成自然的中文。",
390
+ "expected_any": [
391
+ "早上好",
392
+ "早安"
393
+ ],
394
+ "completion": "早上好。\n",
395
+ "passed": true,
396
+ "failures": [],
397
+ "new_tokens": 5,
398
+ "elapsed_seconds": 1.1610058340011165
399
+ },
400
+ {
401
+ "id": "ja_01",
402
+ "language": "ja",
403
+ "prompt": "フランスの首都はどこですか。短く答えてください。",
404
+ "expected_any": [
405
+ "パリ"
406
+ ],
407
+ "completion": "パリ\n",
408
+ "passed": true,
409
+ "failures": [],
410
+ "new_tokens": 4,
411
+ "elapsed_seconds": 0.9211049160221592
412
+ },
413
+ {
414
+ "id": "ja_02",
415
+ "language": "ja",
416
+ "prompt": "7かける8はいくつですか。数字で答えてください。",
417
+ "expected_any": [
418
+ "56"
419
+ ],
420
+ "completion": "7 かける 8 は、$7 \\times 8 = 56$ です。\n\n答え:56\n",
421
+ "passed": true,
422
+ "failures": [],
423
+ "new_tokens": 28,
424
+ "elapsed_seconds": 6.337640833982732
425
+ },
426
+ {
427
+ "id": "ja_03",
428
+ "language": "ja",
429
+ "prompt": "水の化学式は何ですか。",
430
+ "expected_any": [
431
+ "H2O",
432
+ "h2o"
433
+ ],
434
+ "completion": "水の化学式は**H₂O**です。\n\nこれは、水分子が「水素原子(H)」が 2 個、「酸素原子(O)」が 1 個から構成されていることを示しています。この",
435
+ "passed": true,
436
+ "failures": [],
437
+ "new_tokens": 48,
438
+ "elapsed_seconds": 10.600324957980774
439
+ },
440
+ {
441
+ "id": "ja_04",
442
+ "language": "ja",
443
+ "prompt": "太陽系で最も大きい惑星は何ですか。",
444
+ "expected_any": [
445
+ "木星"
446
+ ],
447
+ "completion": "太陽系で最も大きい惑星は**木星**です。\n\n木星は太陽系の中心から約 778 万 km 離れた位置にあり、直径は約 140,000 km にも",
448
+ "passed": true,
449
+ "failures": [],
450
+ "new_tokens": 48,
451
+ "elapsed_seconds": 10.53782487497665
452
+ },
453
+ {
454
+ "id": "ja_05",
455
+ "language": "ja",
456
+ "prompt": "静かな冬の朝を自然な二文で描写してください。",
457
+ "completion": "静かな冬の朝は、窓ガラスにまだ霜が結ばれたままの空に、遠くから雪の音が響き渡ります。\n窓際で揺れる枯れ草の影が、薄明かりに",
458
+ "passed": true,
459
+ "failures": [],
460
+ "new_tokens": 48,
461
+ "elapsed_seconds": 10.687851083057467
462
+ },
463
+ {
464
+ "id": "ja_06",
465
+ "language": "ja",
466
+ "prompt": "友人に10分遅れることを丁寧に伝える一文を書いてください。",
467
+ "completion": "ご友人に10 分遅れることを丁寧に伝えるための例文をいくつか作成しました。状況や関係性に合わせて使い分けてください。\n\n### 1. 標準的・丁寧な表現\n最も汎用性が高く",
468
+ "passed": true,
469
+ "failures": [],
470
+ "new_tokens": 48,
471
+ "elapsed_seconds": 10.560357165988535
472
+ },
473
+ {
474
+ "id": "ja_07",
475
+ "language": "ja",
476
+ "prompt": "AI研究で再現性が重要な理由を二文で説明してください。",
477
+ "completion": "AI 研究における再現性の重要性は、まず研究結果の客観性と信頼性を担保するためです。\nこれにより、他の研究者が同じ条件下で再現し、新たな発見や検証が可能になるからです。\n",
478
+ "passed": true,
479
+ "failures": [],
480
+ "new_tokens": 46,
481
+ "elapsed_seconds": 10.073462790984195
482
+ },
483
+ {
484
+ "id": "ja_08",
485
+ "language": "ja",
486
+ "prompt": "ほかの言葉を加えず「了解」とだけ答えてください。",
487
+ "expected_any": [
488
+ "了解"
489
+ ],
490
+ "completion": "了解\n",
491
+ "passed": true,
492
+ "failures": [],
493
+ "new_tokens": 4,
494
+ "elapsed_seconds": 0.9211295830318704
495
+ },
496
+ {
497
+ "id": "ja_09",
498
+ "language": "ja",
499
+ "prompt": "りんご、バナナ、ぶどうを番号付きの三項目で並べてください。",
500
+ "completion": "りんご、バナナ、ぶどうを番号付きの順に並べると、以下のようになります。\n\n1. りんご\n2. バナナ\n3. ぶどう\n",
501
+ "passed": true,
502
+ "failures": [],
503
+ "new_tokens": 43,
504
+ "elapsed_seconds": 9.689871000009589
505
+ },
506
+ {
507
+ "id": "ja_10",
508
+ "language": "ja",
509
+ "prompt": "「Good morning」を自然な日本語に訳してください。",
510
+ "expected_any": [
511
+ "おはよう"
512
+ ],
513
+ "completion": "おはようございます。\n",
514
+ "passed": true,
515
+ "failures": [],
516
+ "new_tokens": 8,
517
+ "elapsed_seconds": 1.8597023330512457
518
+ },
519
+ {
520
+ "id": "es_01",
521
+ "language": "es",
522
+ "prompt": "¿Cuál es la capital de Francia? Responde brevemente.",
523
+ "expected_any": [
524
+ "París",
525
+ "Paris"
526
+ ],
527
+ "completion": "La capital de Francia es **París**.\n",
528
+ "passed": true,
529
+ "failures": [],
530
+ "new_tokens": 12,
531
+ "elapsed_seconds": 2.6324190419982187
532
+ },
533
+ {
534
+ "id": "es_02",
535
+ "language": "es",
536
+ "prompt": "¿Cuánto es 7 por 8? Responde con un número.",
537
+ "expected_any": [
538
+ "56"
539
+ ],
540
+ "completion": "56\n",
541
+ "passed": true,
542
+ "failures": [],
543
+ "new_tokens": 5,
544
+ "elapsed_seconds": 1.209535208006855
545
+ },
546
+ {
547
+ "id": "es_03",
548
+ "language": "es",
549
+ "prompt": "¿Cuál es la fórmula química del agua?",
550
+ "expected_any": [
551
+ "H2O",
552
+ "h2o"
553
+ ],
554
+ "completion": "La fórmula química del agua es **H₂O**.\n\nEsto significa que una molécula de agua está compuesta por dos átomos de hidrógeno (H) unidos a un solo átomo de oxígeno (O",
555
+ "passed": true,
556
+ "failures": [],
557
+ "new_tokens": 48,
558
+ "elapsed_seconds": 10.877840041997842
559
+ },
560
+ {
561
+ "id": "es_04",
562
+ "language": "es",
563
+ "prompt": "¿Cuál es el planeta más grande del sistema solar?",
564
+ "expected_any": [
565
+ "Júpiter",
566
+ "Jupiter"
567
+ ],
568
+ "completion": "El planeta más grande del sistema solar es **Júpiter**.\n\nEs un gigante gaseoso que tiene un diámetro de aproximadamente 139,820 kilómetros, lo que lo hace más grande que todos los demás",
569
+ "passed": true,
570
+ "failures": [],
571
+ "new_tokens": 48,
572
+ "elapsed_seconds": 10.883560375019442
573
+ },
574
+ {
575
+ "id": "es_05",
576
+ "language": "es",
577
+ "prompt": "Escribe dos frases naturales que describan una tranquila mañana de invierno.",
578
+ "completion": "La nieve cae suavemente sobre el tejado de mi casa, creando un silencio absoluto que se extiende por toda la ciudad.\nEl aire frío y seco se siente en la piel, pero el sol de invierno brilla con una",
579
+ "passed": true,
580
+ "failures": [],
581
+ "new_tokens": 48,
582
+ "elapsed_seconds": 10.86848220799584
583
+ },
584
+ {
585
+ "id": "es_06",
586
+ "language": "es",
587
+ "prompt": "Escribe una frase cortés para decirle a un amigo que llegarás diez minutos tarde.",
588
+ "completion": "¡Hola! Me alegra mucho que estés aquí, pero por suerte llegaré diez minutos tarde.\n",
589
+ "passed": true,
590
+ "failures": [],
591
+ "new_tokens": 24,
592
+ "elapsed_seconds": 5.387403959000949
593
+ },
594
+ {
595
+ "id": "es_07",
596
+ "language": "es",
597
+ "prompt": "Explica en dos frases por qué la reproducibilidad importa en la investigación de IA.",
598
+ "completion": "La reproducibilidad es fundamental porque garantiza que los resultados obtenidos mediante modelos de IA sean verificados y replicados por otros investigadores, lo que valida la solidez de las conclusiones. Además, fomenta la transparencia y la confianza en",
599
+ "passed": true,
600
+ "failures": [],
601
+ "new_tokens": 48,
602
+ "elapsed_seconds": 10.632735249993857
603
+ },
604
+ {
605
+ "id": "es_08",
606
+ "language": "es",
607
+ "prompt": "Responde únicamente con la palabra 'entendido'.",
608
+ "expected_any": [
609
+ "entendido"
610
+ ],
611
+ "completion": "entendido\n",
612
+ "passed": true,
613
+ "failures": [],
614
+ "new_tokens": 5,
615
+ "elapsed_seconds": 1.210517042025458
616
+ },
617
+ {
618
+ "id": "es_09",
619
+ "language": "es",
620
+ "prompt": "Enumera manzana, plátano y uva como tres elementos numerados.",
621
+ "completion": "Aquí tienes la lista de los tres elementos numerados:\n\n1. Manzana\n2. Plátano\n3. Uva\n",
622
+ "passed": true,
623
+ "failures": [],
624
+ "new_tokens": 31,
625
+ "elapsed_seconds": 7.097092125040945
626
+ },
627
+ {
628
+ "id": "es_10",
629
+ "language": "es",
630
+ "prompt": "Traduce 'Good morning' a un español natural.",
631
+ "expected_any": [
632
+ "buenos días",
633
+ "buen día"
634
+ ],
635
+ "completion": "Buenos días.\n",
636
+ "passed": true,
637
+ "failures": [],
638
+ "new_tokens": 8,
639
+ "elapsed_seconds": 1.8161898750113323
640
+ },
641
+ {
642
+ "id": "de_01",
643
+ "language": "de",
644
+ "prompt": "Was ist die Hauptstadt von Frankreich? Antworte kurz.",
645
+ "expected_any": [
646
+ "Paris"
647
+ ],
648
+ "completion": "Die Hauptstadt von Frankreich ist Paris.\n",
649
+ "passed": true,
650
+ "failures": [],
651
+ "new_tokens": 10,
652
+ "elapsed_seconds": 2.182326541049406
653
+ },
654
+ {
655
+ "id": "de_02",
656
+ "language": "de",
657
+ "prompt": "Was ist 7 mal 8? Antworte mit einer Zahl.",
658
+ "expected_any": [
659
+ "56"
660
+ ],
661
+ "completion": "56\n",
662
+ "passed": true,
663
+ "failures": [],
664
+ "new_tokens": 5,
665
+ "elapsed_seconds": 1.1318954160087742
666
+ },
667
+ {
668
+ "id": "de_03",
669
+ "language": "de",
670
+ "prompt": "Wie lautet die chemische Formel für Wasser?",
671
+ "expected_any": [
672
+ "H2O",
673
+ "h2o"
674
+ ],
675
+ "completion": "Die chemische Formel für Wasser lautet **H₂O**.\n\nDies bedeutet, dass ein Wassermolekül aus zwei Atomen des Elements Wasserstoff (H) und einem Atom des Elements Sauerstoff (O) besteht. Die",
676
+ "passed": true,
677
+ "failures": [],
678
+ "new_tokens": 48,
679
+ "elapsed_seconds": 10.83673787501175
680
+ },
681
+ {
682
+ "id": "de_04",
683
+ "language": "de",
684
+ "prompt": "Welcher Planet ist der größte im Sonnensystem?",
685
+ "expected_any": [
686
+ "Jupiter"
687
+ ],
688
+ "completion": "Der Planet, der den größten Durchmesser und die größte Masse im Sonnensystem hat, ist **Jupiter**.\n\nEr ist ein Gasriesen, der etwa 11,2 Mal so groß wie die Erde ist und etwa 3",
689
+ "passed": true,
690
+ "failures": [],
691
+ "new_tokens": 48,
692
+ "elapsed_seconds": 10.766112500044983
693
+ },
694
+ {
695
+ "id": "de_05",
696
+ "language": "de",
697
+ "prompt": "Schreibe zwei natürliche Sätze über einen ruhigen Wintermorgen.",
698
+ "completion": "Der Wintermorgen ist so still, dass man kaum noch die eigenen Atemzüge hört, während der Schnee sanft über die Fensterläden fällt.\nDie Sonne scheint durch die dichten Wolken nur als ein warmes, goldenes",
699
+ "passed": true,
700
+ "failures": [],
701
+ "new_tokens": 48,
702
+ "elapsed_seconds": 10.828977917029988
703
+ },
704
+ {
705
+ "id": "de_06",
706
+ "language": "de",
707
+ "prompt": "Schreibe einen höflichen Satz, der einem Freund sagt, dass du zehn Minuten zu spät kommst.",
708
+ "completion": "\"Hey, ich bin ein bisschen zu spät, aber ich hoffe, du kannst mir trotzdem helfen, wenn du Zeit hast!\"\n",
709
+ "passed": true,
710
+ "failures": [],
711
+ "new_tokens": 28,
712
+ "elapsed_seconds": 6.2108728750026785
713
+ },
714
+ {
715
+ "id": "de_07",
716
+ "language": "de",
717
+ "prompt": "Erkläre in zwei Sätzen, warum Reproduzierbarkeit in der KI-Forschung wichtig ist.",
718
+ "completion": "Reproduzierbarkeit ist in der KI-Forschung entscheidend, um die Ergebnisse von Algorithmen nicht nur wissenschaftlich validieren zu können, sondern auch als vertrauenswürdige Vorhersagemodelle für die reale Welt einzuset",
719
+ "passed": true,
720
+ "failures": [],
721
+ "new_tokens": 48,
722
+ "elapsed_seconds": 10.962672458030283
723
+ },
724
+ {
725
+ "id": "de_08",
726
+ "language": "de",
727
+ "prompt": "Antworte ausschließlich mit dem Wort 'Verstanden'.",
728
+ "expected_any": [
729
+ "verstanden"
730
+ ],
731
+ "completion": "Verstanden\n",
732
+ "passed": true,
733
+ "failures": [],
734
+ "new_tokens": 5,
735
+ "elapsed_seconds": 1.1044922079890966
736
+ },
737
+ {
738
+ "id": "de_09",
739
+ "language": "de",
740
+ "prompt": "Liste Apfel, Banane und Traube als drei nummerierte Punkte auf.",
741
+ "completion": "Hier ist die Liste der drei Früchte als nummerierte Punkte:\n\n1. Apfel\n2. Banane\n3. Traube\n",
742
+ "passed": true,
743
+ "failures": [],
744
+ "new_tokens": 30,
745
+ "elapsed_seconds": 6.85629908303963
746
+ },
747
+ {
748
+ "id": "de_10",
749
+ "language": "de",
750
+ "prompt": "Übersetze 'Good morning' in natürliches Deutsch.",
751
+ "expected_any": [
752
+ "guten morgen"
753
+ ],
754
+ "completion": "Guten Morgen!\n",
755
+ "passed": true,
756
+ "failures": [],
757
+ "new_tokens": 7,
758
+ "elapsed_seconds": 1.618866834032815
759
+ }
760
+ ],
761
+ "rescored_from": "research/qwen35-moe-a3b/runs/chat-gate/v2-60.json"
762
+ }
evaluation/lm_eval_teacher4b_dev100.json ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "hellaswag": {
4
+ "name": "hellaswag",
5
+ "alias": "hellaswag",
6
+ "sample_len": 100,
7
+ "acc,none": 0.51,
8
+ "acc_stderr,none": 0.05024183937956913,
9
+ "acc_norm,none": 0.68,
10
+ "acc_norm_stderr,none": 0.046882617226215076
11
+ },
12
+ "arc_easy": {
13
+ "name": "arc_easy",
14
+ "alias": "arc_easy",
15
+ "sample_len": 100,
16
+ "acc,none": 0.8,
17
+ "acc_stderr,none": 0.04020151261036849,
18
+ "acc_norm,none": 0.81,
19
+ "acc_norm_stderr,none": 0.039427724440366255
20
+ }
21
+ },
22
+ "group_subtasks": {},
23
+ "configs": {
24
+ "arc_easy": {
25
+ "task": "arc_easy",
26
+ "dataset_path": "allenai/ai2_arc",
27
+ "dataset_name": "ARC-Easy",
28
+ "training_split": "train",
29
+ "validation_split": "validation",
30
+ "test_split": "test",
31
+ "doc_to_text": "Question: {{question}}\nAnswer:",
32
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
33
+ "unsafe_code": false,
34
+ "doc_to_choice": "{{choices.text}}",
35
+ "description": "",
36
+ "target_delimiter": " ",
37
+ "fewshot_delimiter": "\n\n",
38
+ "fewshot_config": {
39
+ "sampler": "default",
40
+ "split": null,
41
+ "process_docs": null,
42
+ "fewshot_indices": null,
43
+ "samples": null,
44
+ "doc_to_text": "Question: {{question}}\nAnswer:",
45
+ "doc_to_choice": "{{choices.text}}",
46
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
47
+ "gen_prefix": null,
48
+ "fewshot_delimiter": "\n\n",
49
+ "target_delimiter": " "
50
+ },
51
+ "num_fewshot": 0,
52
+ "metric_list": [
53
+ {
54
+ "metric": "acc",
55
+ "aggregation": "mean",
56
+ "higher_is_better": true
57
+ },
58
+ {
59
+ "metric": "acc_norm",
60
+ "aggregation": "mean",
61
+ "higher_is_better": true
62
+ }
63
+ ],
64
+ "output_type": "multiple_choice",
65
+ "repeats": 1,
66
+ "should_decontaminate": true,
67
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
68
+ "metadata": {
69
+ "version": 1.0,
70
+ "pretrained": "models/Qwen/Qwen3.5-4B",
71
+ "dtype": "bfloat16",
72
+ "config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/arc/arc_easy.yaml"
73
+ }
74
+ },
75
+ "hellaswag": {
76
+ "task": "hellaswag",
77
+ "dataset_path": "Rowan/hellaswag",
78
+ "training_split": "train",
79
+ "validation_split": "validation",
80
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
81
+ "doc_to_text": "{{query}}",
82
+ "doc_to_target": "{{label}}",
83
+ "unsafe_code": false,
84
+ "doc_to_choice": "choices",
85
+ "description": "",
86
+ "target_delimiter": " ",
87
+ "fewshot_delimiter": "\n\n",
88
+ "fewshot_config": {
89
+ "sampler": "default",
90
+ "split": null,
91
+ "process_docs": "<function process_docs at 0x13ebcfd70>",
92
+ "fewshot_indices": null,
93
+ "samples": null,
94
+ "doc_to_text": "{{query}}",
95
+ "doc_to_choice": "choices",
96
+ "doc_to_target": "{{label}}",
97
+ "gen_prefix": null,
98
+ "fewshot_delimiter": "\n\n",
99
+ "target_delimiter": " "
100
+ },
101
+ "num_fewshot": 0,
102
+ "metric_list": [
103
+ {
104
+ "metric": "acc",
105
+ "aggregation": "mean",
106
+ "higher_is_better": true
107
+ },
108
+ {
109
+ "metric": "acc_norm",
110
+ "aggregation": "mean",
111
+ "higher_is_better": true
112
+ }
113
+ ],
114
+ "output_type": "multiple_choice",
115
+ "repeats": 1,
116
+ "should_decontaminate": false,
117
+ "metadata": {
118
+ "version": 1.0,
119
+ "pretrained": "models/Qwen/Qwen3.5-4B",
120
+ "dtype": "bfloat16",
121
+ "config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml"
122
+ }
123
+ }
124
+ },
125
+ "versions": {
126
+ "arc_easy": 1.0,
127
+ "hellaswag": 1.0
128
+ },
129
+ "n-shot": {
130
+ "arc_easy": 0,
131
+ "hellaswag": 0
132
+ },
133
+ "higher_is_better": {
134
+ "arc_easy": {
135
+ "acc": true,
136
+ "acc_norm": true
137
+ },
138
+ "hellaswag": {
139
+ "acc": true,
140
+ "acc_norm": true
141
+ }
142
+ },
143
+ "n-samples": {
144
+ "hellaswag": {
145
+ "original": 10042,
146
+ "effective": 100
147
+ },
148
+ "arc_easy": {
149
+ "original": 2376,
150
+ "effective": 100
151
+ }
152
+ },
153
+ "config": {
154
+ "model": "hf",
155
+ "model_args": {
156
+ "pretrained": "models/Qwen/Qwen3.5-4B",
157
+ "dtype": "bfloat16"
158
+ },
159
+ "model_num_parameters": 4205751296,
160
+ "model_dtype": "torch.bfloat16",
161
+ "model_revision": "main",
162
+ "model_sha": "",
163
+ "batch_size": "1",
164
+ "batch_sizes": [],
165
+ "device": "mps",
166
+ "use_cache": null,
167
+ "limit": 100.0,
168
+ "bootstrap_iters": 100000,
169
+ "gen_kwargs": {},
170
+ "random_seed": 0,
171
+ "numpy_seed": 1234,
172
+ "torch_seed": 1234,
173
+ "fewshot_seed": 1234
174
+ },
175
+ "git_hash": null,
176
+ "date": 1787550063.559031,
177
+ "pretty_env_info": "PyTorch version: 2.11.0\nIs debug build: False\nCUDA used to build PyTorch: None\nROCM used to build PyTorch: N/A\n\nOS: macOS 26.4.1 (arm64)\nGCC version: Could not collect\nClang version: 21.0.0 (clang-2100.0.123.102)\nCMake version: version 4.4.0\nLibc version: N/A\n\nPython version: 3.14.6 (main, Jun 10 2026, 10:03:53) [Clang 21.0.0 (clang-2100.0.123.102)] (64-bit runtime)\nPython platform: macOS-26.4.1-arm64-arm-64bit-Mach-O\nIs CUDA available: False\nCUDA runtime version: No CUDA\nCUDA_MODULE_LOADING set to: N/A\nGPU models and configuration: No CUDA\nNvidia driver version: No CUDA\ncuDNN version: No CUDA\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nApple M4\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.4\n[pip3] torch==2.11.0\n[conda] Could not collect",
178
+ "transformers_version": "5.13.0",
179
+ "lm_eval_version": "0.4.12",
180
+ "upper_git_hash": null,
181
+ "tokenizer_pad_token": [
182
+ "<|endoftext|>",
183
+ "248044"
184
+ ],
185
+ "tokenizer_eos_token": [
186
+ "<|im_end|>",
187
+ "248046"
188
+ ],
189
+ "tokenizer_bos_token": [
190
+ null,
191
+ "None"
192
+ ],
193
+ "eot_token_id": 248046,
194
+ "max_length": 262144,
195
+ "task_hashes": {
196
+ "hellaswag": "4f7d86a1e256013e93651a0d8510973180c5f31142ca184d9677955a29676af1",
197
+ "arc_easy": "fd3a493579cfdccf229a32d7c26710006fc4edf7c024add74f70c416c6dab3ce"
198
+ },
199
+ "model_source": "hf",
200
+ "model_name": "models/Qwen/Qwen3.5-4B",
201
+ "model_name_sanitized": "models__Qwen__Qwen3.5-4B",
202
+ "system_instruction": null,
203
+ "system_instruction_sha": null,
204
+ "fewshot_as_multiturn": null,
205
+ "chat_template": null,
206
+ "chat_template_sha": null,
207
+ "total_evaluation_time_seconds": "257.4533243330079"
208
+ }
evaluation/lm_eval_v2_dev100.json ADDED
@@ -0,0 +1,208 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "results": {
3
+ "hellaswag": {
4
+ "name": "hellaswag",
5
+ "alias": "hellaswag",
6
+ "sample_len": 100,
7
+ "acc,none": 0.44,
8
+ "acc_stderr,none": 0.049888765156985884,
9
+ "acc_norm,none": 0.62,
10
+ "acc_norm_stderr,none": 0.04878317312145634
11
+ },
12
+ "arc_easy": {
13
+ "name": "arc_easy",
14
+ "alias": "arc_easy",
15
+ "sample_len": 100,
16
+ "acc,none": 0.71,
17
+ "acc_stderr,none": 0.045604802157206865,
18
+ "acc_norm,none": 0.73,
19
+ "acc_norm_stderr,none": 0.04461960433384737
20
+ }
21
+ },
22
+ "group_subtasks": {},
23
+ "configs": {
24
+ "arc_easy": {
25
+ "task": "arc_easy",
26
+ "dataset_path": "allenai/ai2_arc",
27
+ "dataset_name": "ARC-Easy",
28
+ "training_split": "train",
29
+ "validation_split": "validation",
30
+ "test_split": "test",
31
+ "doc_to_text": "Question: {{question}}\nAnswer:",
32
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
33
+ "unsafe_code": false,
34
+ "doc_to_choice": "{{choices.text}}",
35
+ "description": "",
36
+ "target_delimiter": " ",
37
+ "fewshot_delimiter": "\n\n",
38
+ "fewshot_config": {
39
+ "sampler": "default",
40
+ "split": null,
41
+ "process_docs": null,
42
+ "fewshot_indices": null,
43
+ "samples": null,
44
+ "doc_to_text": "Question: {{question}}\nAnswer:",
45
+ "doc_to_choice": "{{choices.text}}",
46
+ "doc_to_target": "{{choices.label.index(answerKey)}}",
47
+ "gen_prefix": null,
48
+ "fewshot_delimiter": "\n\n",
49
+ "target_delimiter": " "
50
+ },
51
+ "num_fewshot": 0,
52
+ "metric_list": [
53
+ {
54
+ "metric": "acc",
55
+ "aggregation": "mean",
56
+ "higher_is_better": true
57
+ },
58
+ {
59
+ "metric": "acc_norm",
60
+ "aggregation": "mean",
61
+ "higher_is_better": true
62
+ }
63
+ ],
64
+ "output_type": "multiple_choice",
65
+ "repeats": 1,
66
+ "should_decontaminate": true,
67
+ "doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
68
+ "metadata": {
69
+ "version": 1.0,
70
+ "pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
71
+ "dtype": "bfloat16",
72
+ "config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/arc/arc_easy.yaml"
73
+ }
74
+ },
75
+ "hellaswag": {
76
+ "task": "hellaswag",
77
+ "dataset_path": "Rowan/hellaswag",
78
+ "training_split": "train",
79
+ "validation_split": "validation",
80
+ "process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
81
+ "doc_to_text": "{{query}}",
82
+ "doc_to_target": "{{label}}",
83
+ "unsafe_code": false,
84
+ "doc_to_choice": "choices",
85
+ "description": "",
86
+ "target_delimiter": " ",
87
+ "fewshot_delimiter": "\n\n",
88
+ "fewshot_config": {
89
+ "sampler": "default",
90
+ "split": null,
91
+ "process_docs": "<function process_docs at 0x174a5fed0>",
92
+ "fewshot_indices": null,
93
+ "samples": null,
94
+ "doc_to_text": "{{query}}",
95
+ "doc_to_choice": "choices",
96
+ "doc_to_target": "{{label}}",
97
+ "gen_prefix": null,
98
+ "fewshot_delimiter": "\n\n",
99
+ "target_delimiter": " "
100
+ },
101
+ "num_fewshot": 0,
102
+ "metric_list": [
103
+ {
104
+ "metric": "acc",
105
+ "aggregation": "mean",
106
+ "higher_is_better": true
107
+ },
108
+ {
109
+ "metric": "acc_norm",
110
+ "aggregation": "mean",
111
+ "higher_is_better": true
112
+ }
113
+ ],
114
+ "output_type": "multiple_choice",
115
+ "repeats": 1,
116
+ "should_decontaminate": false,
117
+ "metadata": {
118
+ "version": 1.0,
119
+ "pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
120
+ "dtype": "bfloat16",
121
+ "config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml"
122
+ }
123
+ }
124
+ },
125
+ "versions": {
126
+ "arc_easy": 1.0,
127
+ "hellaswag": 1.0
128
+ },
129
+ "n-shot": {
130
+ "arc_easy": 0,
131
+ "hellaswag": 0
132
+ },
133
+ "higher_is_better": {
134
+ "arc_easy": {
135
+ "acc": true,
136
+ "acc_norm": true
137
+ },
138
+ "hellaswag": {
139
+ "acc": true,
140
+ "acc_norm": true
141
+ }
142
+ },
143
+ "n-samples": {
144
+ "hellaswag": {
145
+ "original": 10042,
146
+ "effective": 100
147
+ },
148
+ "arc_easy": {
149
+ "original": 2376,
150
+ "effective": 100
151
+ }
152
+ },
153
+ "config": {
154
+ "model": "hf",
155
+ "model_args": {
156
+ "pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
157
+ "dtype": "bfloat16"
158
+ },
159
+ "model_num_parameters": 3995901760,
160
+ "model_dtype": "torch.bfloat16",
161
+ "model_revision": "main",
162
+ "model_sha": "",
163
+ "batch_size": "1",
164
+ "batch_sizes": [],
165
+ "device": "mps",
166
+ "use_cache": null,
167
+ "limit": 100.0,
168
+ "bootstrap_iters": 100000,
169
+ "gen_kwargs": {},
170
+ "random_seed": 0,
171
+ "numpy_seed": 1234,
172
+ "torch_seed": 1234,
173
+ "fewshot_seed": 1234
174
+ },
175
+ "git_hash": null,
176
+ "date": 1787549803.098937,
177
+ "pretty_env_info": "PyTorch version: 2.11.0\nIs debug build: False\nCUDA used to build PyTorch: None\nROCM used to build PyTorch: N/A\n\nOS: macOS 26.4.1 (arm64)\nGCC version: Could not collect\nClang version: 21.0.0 (clang-2100.0.123.102)\nCMake version: version 4.4.0\nLibc version: N/A\n\nPython version: 3.14.6 (main, Jun 10 2026, 10:03:53) [Clang 21.0.0 (clang-2100.0.123.102)] (64-bit runtime)\nPython platform: macOS-26.4.1-arm64-arm-64bit-Mach-O\nIs CUDA available: False\nCUDA runtime version: No CUDA\nCUDA_MODULE_LOADING set to: N/A\nGPU models and configuration: No CUDA\nNvidia driver version: No CUDA\ncuDNN version: No CUDA\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nApple M4\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.4\n[pip3] torch==2.11.0\n[conda] Could not collect",
178
+ "transformers_version": "5.13.0",
179
+ "lm_eval_version": "0.4.12",
180
+ "upper_git_hash": null,
181
+ "tokenizer_pad_token": [
182
+ "<|endoftext|>",
183
+ "248044"
184
+ ],
185
+ "tokenizer_eos_token": [
186
+ "<|im_end|>",
187
+ "248046"
188
+ ],
189
+ "tokenizer_bos_token": [
190
+ null,
191
+ "None"
192
+ ],
193
+ "eot_token_id": 248046,
194
+ "max_length": 262144,
195
+ "task_hashes": {
196
+ "hellaswag": "4f7d86a1e256013e93651a0d8510973180c5f31142ca184d9677955a29676af1",
197
+ "arc_easy": "fd3a493579cfdccf229a32d7c26710006fc4edf7c024add74f70c416c6dab3ce"
198
+ },
199
+ "model_source": "hf",
200
+ "model_name": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
201
+ "model_name_sanitized": "models__Qwen__Qwen3.5-4B-A3B-Student-v2",
202
+ "system_instruction": null,
203
+ "system_instruction_sha": null,
204
+ "fewshot_as_multiturn": null,
205
+ "chat_template": null,
206
+ "chat_template_sha": null,
207
+ "total_evaluation_time_seconds": "244.92804725002497"
208
+ }
evaluation/multilingual_lm_loss.json ADDED
@@ -0,0 +1,115 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "sequence_length": 128,
3
+ "documents_per_language": 4,
4
+ "languages": [
5
+ "de",
6
+ "en",
7
+ "es",
8
+ "ja",
9
+ "ko",
10
+ "zh"
11
+ ],
12
+ "models": {
13
+ "qwen35_2b": {
14
+ "ko": {
15
+ "loss": 3.2937907576560974,
16
+ "documents": 4,
17
+ "tokens": 512
18
+ },
19
+ "en": {
20
+ "loss": 2.9049559831619263,
21
+ "documents": 4,
22
+ "tokens": 512
23
+ },
24
+ "zh": {
25
+ "loss": 3.7243993878364563,
26
+ "documents": 4,
27
+ "tokens": 512
28
+ },
29
+ "ja": {
30
+ "loss": 3.160854697227478,
31
+ "documents": 4,
32
+ "tokens": 512
33
+ },
34
+ "es": {
35
+ "loss": 2.9567737579345703,
36
+ "documents": 4,
37
+ "tokens": 512
38
+ },
39
+ "de": {
40
+ "loss": 3.21018385887146,
41
+ "documents": 4,
42
+ "tokens": 512
43
+ }
44
+ },
45
+ "a3b_v2": {
46
+ "ko": {
47
+ "loss": 3.2937907576560974,
48
+ "documents": 4,
49
+ "tokens": 512
50
+ },
51
+ "en": {
52
+ "loss": 2.9049559831619263,
53
+ "documents": 4,
54
+ "tokens": 512
55
+ },
56
+ "zh": {
57
+ "loss": 3.7243993878364563,
58
+ "documents": 4,
59
+ "tokens": 512
60
+ },
61
+ "ja": {
62
+ "loss": 3.160854697227478,
63
+ "documents": 4,
64
+ "tokens": 512
65
+ },
66
+ "es": {
67
+ "loss": 2.9567737579345703,
68
+ "documents": 4,
69
+ "tokens": 512
70
+ },
71
+ "de": {
72
+ "loss": 3.21018385887146,
73
+ "documents": 4,
74
+ "tokens": 512
75
+ }
76
+ },
77
+ "qwen35_4b": {
78
+ "ko": {
79
+ "loss": 2.9032246470451355,
80
+ "documents": 4,
81
+ "tokens": 512
82
+ },
83
+ "en": {
84
+ "loss": 2.770678699016571,
85
+ "documents": 4,
86
+ "tokens": 512
87
+ },
88
+ "zh": {
89
+ "loss": 3.378099739551544,
90
+ "documents": 4,
91
+ "tokens": 512
92
+ },
93
+ "ja": {
94
+ "loss": 2.9211148023605347,
95
+ "documents": 4,
96
+ "tokens": 512
97
+ },
98
+ "es": {
99
+ "loss": 2.6984021067619324,
100
+ "documents": 4,
101
+ "tokens": 512
102
+ },
103
+ "de": {
104
+ "loss": 2.9680771231651306,
105
+ "documents": 4,
106
+ "tokens": 512
107
+ }
108
+ }
109
+ },
110
+ "mean_losses": {
111
+ "qwen35_2b": 3.2084930737813315,
112
+ "a3b_v2": 3.2084930737813315,
113
+ "qwen35_4b": 2.9399328529834747
114
+ }
115
+ }
evaluation/openai_service_20.json ADDED
@@ -0,0 +1,165 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "requests": 20,
3
+ "passed": 20,
4
+ "all_passed": true,
5
+ "latency_mean_seconds": 2.5972848790552234,
6
+ "latency_p50_seconds": 2.9117911249923054,
7
+ "latency_p95_seconds": 3.568211792036891,
8
+ "health_before": {
9
+ "status": "ok",
10
+ "model": "qwen3.5-4b-a3b-student-v2",
11
+ "model_path": "../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2",
12
+ "device": "mps",
13
+ "parameters": 3995901760,
14
+ "requests": 0,
15
+ "latency_mean_seconds": null,
16
+ "latency_p95_seconds": null,
17
+ "rss_mb": null,
18
+ "mps_allocated_mb": 7621.5859375
19
+ },
20
+ "health_after": {
21
+ "status": "ok",
22
+ "model": "qwen3.5-4b-a3b-student-v2",
23
+ "model_path": "../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2",
24
+ "device": "mps",
25
+ "parameters": 3995901760,
26
+ "requests": 20,
27
+ "latency_mean_seconds": 2.5897246561566134,
28
+ "latency_p95_seconds": 3.5636009160079993,
29
+ "rss_mb": null,
30
+ "mps_allocated_mb": 7621.5859375
31
+ },
32
+ "models": {
33
+ "object": "list",
34
+ "data": [
35
+ {
36
+ "id": "qwen3.5-4b-a3b-student-v2",
37
+ "object": "model",
38
+ "created": 1787551049,
39
+ "owned_by": "local"
40
+ }
41
+ ]
42
+ },
43
+ "results": [
44
+ {
45
+ "index": 1,
46
+ "passed": true,
47
+ "elapsed_seconds": 1.8874667910276912,
48
+ "content": "서울"
49
+ },
50
+ {
51
+ "index": 2,
52
+ "passed": true,
53
+ "elapsed_seconds": 2.6587018750142306,
54
+ "content": "The capital of France is **Paris**."
55
+ },
56
+ {
57
+ "index": 3,
58
+ "passed": true,
59
+ "elapsed_seconds": 3.4837433329666965,
60
+ "content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
61
+ },
62
+ {
63
+ "index": 4,
64
+ "passed": true,
65
+ "elapsed_seconds": 3.3025628330069594,
66
+ "content": "Hello there! How's your day going so far?"
67
+ },
68
+ {
69
+ "index": 5,
70
+ "passed": true,
71
+ "elapsed_seconds": 0.9436147080268711,
72
+ "content": "서울"
73
+ },
74
+ {
75
+ "index": 6,
76
+ "passed": true,
77
+ "elapsed_seconds": 2.515450624981895,
78
+ "content": "The capital of France is **Paris**."
79
+ },
80
+ {
81
+ "index": 7,
82
+ "passed": true,
83
+ "elapsed_seconds": 3.568211792036891,
84
+ "content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
85
+ },
86
+ {
87
+ "index": 8,
88
+ "passed": true,
89
+ "elapsed_seconds": 3.2445592500152998,
90
+ "content": "Hello there! How's your day going so far?"
91
+ },
92
+ {
93
+ "index": 9,
94
+ "passed": true,
95
+ "elapsed_seconds": 0.9237514159758575,
96
+ "content": "서울"
97
+ },
98
+ {
99
+ "index": 10,
100
+ "passed": true,
101
+ "elapsed_seconds": 2.4350813750061207,
102
+ "content": "The capital of France is **Paris**."
103
+ },
104
+ {
105
+ "index": 11,
106
+ "passed": true,
107
+ "elapsed_seconds": 3.567038333043456,
108
+ "content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
109
+ },
110
+ {
111
+ "index": 12,
112
+ "passed": true,
113
+ "elapsed_seconds": 3.16488037497038,
114
+ "content": "Hello there! How's your day going so far?"
115
+ },
116
+ {
117
+ "index": 13,
118
+ "passed": true,
119
+ "elapsed_seconds": 0.9253840000019409,
120
+ "content": "서울"
121
+ },
122
+ {
123
+ "index": 14,
124
+ "passed": true,
125
+ "elapsed_seconds": 2.5322331670322455,
126
+ "content": "The capital of France is **Paris**."
127
+ },
128
+ {
129
+ "index": 15,
130
+ "passed": true,
131
+ "elapsed_seconds": 3.6034239170257933,
132
+ "content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
133
+ },
134
+ {
135
+ "index": 16,
136
+ "passed": true,
137
+ "elapsed_seconds": 3.2237549159908667,
138
+ "content": "Hello there! How's your day going so far?"
139
+ },
140
+ {
141
+ "index": 17,
142
+ "passed": true,
143
+ "elapsed_seconds": 0.9134719159919769,
144
+ "content": "서울"
145
+ },
146
+ {
147
+ "index": 18,
148
+ "passed": true,
149
+ "elapsed_seconds": 2.403110374987591,
150
+ "content": "The capital of France is **Paris**."
151
+ },
152
+ {
153
+ "index": 19,
154
+ "passed": true,
155
+ "elapsed_seconds": 3.4773494169930927,
156
+ "content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
157
+ },
158
+ {
159
+ "index": 20,
160
+ "passed": true,
161
+ "elapsed_seconds": 3.1719071670086123,
162
+ "content": "Hello there! How's your day going so far?"
163
+ }
164
+ ]
165
+ }
merges.txt ADDED
The diff for this file is too large to render. See raw diff
 
model-common.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:40a4b73b7901fd219c6814fa5dd05b0ff522389de8d87f9995e505a7331a958b
3
+ size 1951744432
model-layer-00.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:050bc55fb97a2759ac2cfc3cc224c75e4b3ea6deef0ea2d3116360c03c176703
3
+ size 251671392
model-layer-01.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:584ac63f1b1322eb754152b6f1cc8f58d474aa74590eec6dd9a352d967e7d06a
3
+ size 251671392
model-layer-02.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6b8317a7c0167103676cdde7422620670be06a5a68c0c7fa4e8bc925197ac432
3
+ size 251671392
model-layer-03.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:05dfd5ad3bf73b4b17c634b69db3dec4fc14c57dbaab642e3b312308a1ded5fa
3
+ size 251671392
model-layer-04.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2a63f3e6d42b3ed951be93367a030bd1d7d015724b659eeaa592358c0cc02d07
3
+ size 251671392
model-layer-05.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2f83c8222cdec99b41fca8532dca86c065fedbcd3f5c9be0bc53786cabbba303
3
+ size 251671392
model-layer-06.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac849368a9a338b977fcbdc3e23d8363da3f40e13349d8801fb4674dee5e7739
3
+ size 251671392
model-layer-07.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:427086a8339f5295e956024355beec362c73dae9ad7bec79d88a8860dddfb883
3
+ size 251671392
model-layer-08.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fd30e7089c9510120fa19be4a19ff251613523cb489e491ee8aa3d84985bb86e
3
+ size 251671392
model-layer-09.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:fb834374a42eaf0681fb90bd18a1dc066f7d17bf5b68333ec3e3020d9e097aef
3
+ size 251671392
model-layer-10.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3eaddc0892948def344e80877b98448f574f8a86794e5c36bd54fea9366fb29a
3
+ size 251671400
model-layer-11.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f4551261310cefbcb24e1434e424a6807c78805797a1936f92987ac61d5f87ec
3
+ size 251671400
model-layer-12.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:19d0e2f3c546506b5f173b768124427579e8f9523b31292f7bc10b3a9a8cdf17
3
+ size 251671400
model-layer-13.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:18d8275c1334ea12f3b4390ef9850143904f2536782ffd96016ccbdbbef043d2
3
+ size 251671400
model-layer-14.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:110a065ec5d6a927ac5eb22156f7ce90c94f7dd6caea067bec74eee303619b41
3
+ size 251671400
model-layer-15.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7f36395bdc43afcf968ab2e766c840e6c71c32053a5d54596978bd0518820661
3
+ size 251671400
model-layer-16.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f33286ee6aede3f08e4c4ce152813b7d9d9ae61636fb0ec5528352832f06c971
3
+ size 251671400
model-layer-17.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:86775994e4e45afefb7a6e791cd83a378620bd500e874bfcfc8ec3deb6bc2acc
3
+ size 251671400
model-layer-18.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:50246c386a0f863d04ce770675b9100726eb9d59ef2a73c9a4ed42e9031901b3
3
+ size 251671400
model-layer-19.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c47bb7ec84481206c1931adf2c1691b6a051ef29e9480158838fc9e7050ab3b9
3
+ size 251671400
model-layer-20.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:96c783fdc495547c37c9835cbc1214342befaf29daaa376d0a197c3539729417
3
+ size 251671400
model-layer-21.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2615d98f592d082e06595c94e3c77d1a1188ac26a1b009f8ab0b9daac0f72458
3
+ size 251671400
model-layer-22.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e44a9882ba2c1ab0f461ee1026de10c92aa96e6d5710463a32c151554bbf3a5c
3
+ size 251671400
model-layer-23.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e764ddf68a2162aa3df83a3ff7f5bbb40a9e71fc438b95f63b51a88f0e032402
3
+ size 251671400
model.safetensors.index.json ADDED
@@ -0,0 +1,423 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "metadata": {
3
+ "total_size": 7991808704
4
+ },
5
+ "weight_map": {
6
+ "model.embed_tokens.weight": "model-common.safetensors",
7
+ "model.layers.0.input_layernorm.weight": "model-common.safetensors",
8
+ "model.layers.0.linear_attn.A_log": "model-common.safetensors",
9
+ "model.layers.0.linear_attn.conv1d.weight": "model-common.safetensors",
10
+ "model.layers.0.linear_attn.dt_bias": "model-common.safetensors",
11
+ "model.layers.0.linear_attn.in_proj_a.weight": "model-common.safetensors",
12
+ "model.layers.0.linear_attn.in_proj_b.weight": "model-common.safetensors",
13
+ "model.layers.0.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
14
+ "model.layers.0.linear_attn.in_proj_z.weight": "model-common.safetensors",
15
+ "model.layers.0.linear_attn.norm.weight": "model-common.safetensors",
16
+ "model.layers.0.linear_attn.out_proj.weight": "model-common.safetensors",
17
+ "model.layers.0.mlp.experts.down_proj": "model-layer-00.safetensors",
18
+ "model.layers.0.mlp.experts.gate_up_proj": "model-layer-00.safetensors",
19
+ "model.layers.0.mlp.gate.weight": "model-layer-00.safetensors",
20
+ "model.layers.0.mlp.shared_expert.down_proj.weight": "model-layer-00.safetensors",
21
+ "model.layers.0.mlp.shared_expert.gate_proj.weight": "model-layer-00.safetensors",
22
+ "model.layers.0.mlp.shared_expert.up_proj.weight": "model-layer-00.safetensors",
23
+ "model.layers.0.mlp.shared_expert_gate.weight": "model-layer-00.safetensors",
24
+ "model.layers.0.post_attention_layernorm.weight": "model-common.safetensors",
25
+ "model.layers.1.input_layernorm.weight": "model-common.safetensors",
26
+ "model.layers.1.linear_attn.A_log": "model-common.safetensors",
27
+ "model.layers.1.linear_attn.conv1d.weight": "model-common.safetensors",
28
+ "model.layers.1.linear_attn.dt_bias": "model-common.safetensors",
29
+ "model.layers.1.linear_attn.in_proj_a.weight": "model-common.safetensors",
30
+ "model.layers.1.linear_attn.in_proj_b.weight": "model-common.safetensors",
31
+ "model.layers.1.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
32
+ "model.layers.1.linear_attn.in_proj_z.weight": "model-common.safetensors",
33
+ "model.layers.1.linear_attn.norm.weight": "model-common.safetensors",
34
+ "model.layers.1.linear_attn.out_proj.weight": "model-common.safetensors",
35
+ "model.layers.1.mlp.experts.down_proj": "model-layer-01.safetensors",
36
+ "model.layers.1.mlp.experts.gate_up_proj": "model-layer-01.safetensors",
37
+ "model.layers.1.mlp.gate.weight": "model-layer-01.safetensors",
38
+ "model.layers.1.mlp.shared_expert.down_proj.weight": "model-layer-01.safetensors",
39
+ "model.layers.1.mlp.shared_expert.gate_proj.weight": "model-layer-01.safetensors",
40
+ "model.layers.1.mlp.shared_expert.up_proj.weight": "model-layer-01.safetensors",
41
+ "model.layers.1.mlp.shared_expert_gate.weight": "model-layer-01.safetensors",
42
+ "model.layers.1.post_attention_layernorm.weight": "model-common.safetensors",
43
+ "model.layers.10.input_layernorm.weight": "model-common.safetensors",
44
+ "model.layers.10.linear_attn.A_log": "model-common.safetensors",
45
+ "model.layers.10.linear_attn.conv1d.weight": "model-common.safetensors",
46
+ "model.layers.10.linear_attn.dt_bias": "model-common.safetensors",
47
+ "model.layers.10.linear_attn.in_proj_a.weight": "model-common.safetensors",
48
+ "model.layers.10.linear_attn.in_proj_b.weight": "model-common.safetensors",
49
+ "model.layers.10.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
50
+ "model.layers.10.linear_attn.in_proj_z.weight": "model-common.safetensors",
51
+ "model.layers.10.linear_attn.norm.weight": "model-common.safetensors",
52
+ "model.layers.10.linear_attn.out_proj.weight": "model-common.safetensors",
53
+ "model.layers.10.mlp.experts.down_proj": "model-layer-10.safetensors",
54
+ "model.layers.10.mlp.experts.gate_up_proj": "model-layer-10.safetensors",
55
+ "model.layers.10.mlp.gate.weight": "model-layer-10.safetensors",
56
+ "model.layers.10.mlp.shared_expert.down_proj.weight": "model-layer-10.safetensors",
57
+ "model.layers.10.mlp.shared_expert.gate_proj.weight": "model-layer-10.safetensors",
58
+ "model.layers.10.mlp.shared_expert.up_proj.weight": "model-layer-10.safetensors",
59
+ "model.layers.10.mlp.shared_expert_gate.weight": "model-layer-10.safetensors",
60
+ "model.layers.10.post_attention_layernorm.weight": "model-common.safetensors",
61
+ "model.layers.11.input_layernorm.weight": "model-common.safetensors",
62
+ "model.layers.11.mlp.experts.down_proj": "model-layer-11.safetensors",
63
+ "model.layers.11.mlp.experts.gate_up_proj": "model-layer-11.safetensors",
64
+ "model.layers.11.mlp.gate.weight": "model-layer-11.safetensors",
65
+ "model.layers.11.mlp.shared_expert.down_proj.weight": "model-layer-11.safetensors",
66
+ "model.layers.11.mlp.shared_expert.gate_proj.weight": "model-layer-11.safetensors",
67
+ "model.layers.11.mlp.shared_expert.up_proj.weight": "model-layer-11.safetensors",
68
+ "model.layers.11.mlp.shared_expert_gate.weight": "model-layer-11.safetensors",
69
+ "model.layers.11.post_attention_layernorm.weight": "model-common.safetensors",
70
+ "model.layers.11.self_attn.k_norm.weight": "model-common.safetensors",
71
+ "model.layers.11.self_attn.k_proj.weight": "model-common.safetensors",
72
+ "model.layers.11.self_attn.o_proj.weight": "model-common.safetensors",
73
+ "model.layers.11.self_attn.q_norm.weight": "model-common.safetensors",
74
+ "model.layers.11.self_attn.q_proj.weight": "model-common.safetensors",
75
+ "model.layers.11.self_attn.v_proj.weight": "model-common.safetensors",
76
+ "model.layers.12.input_layernorm.weight": "model-common.safetensors",
77
+ "model.layers.12.linear_attn.A_log": "model-common.safetensors",
78
+ "model.layers.12.linear_attn.conv1d.weight": "model-common.safetensors",
79
+ "model.layers.12.linear_attn.dt_bias": "model-common.safetensors",
80
+ "model.layers.12.linear_attn.in_proj_a.weight": "model-common.safetensors",
81
+ "model.layers.12.linear_attn.in_proj_b.weight": "model-common.safetensors",
82
+ "model.layers.12.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
83
+ "model.layers.12.linear_attn.in_proj_z.weight": "model-common.safetensors",
84
+ "model.layers.12.linear_attn.norm.weight": "model-common.safetensors",
85
+ "model.layers.12.linear_attn.out_proj.weight": "model-common.safetensors",
86
+ "model.layers.12.mlp.experts.down_proj": "model-layer-12.safetensors",
87
+ "model.layers.12.mlp.experts.gate_up_proj": "model-layer-12.safetensors",
88
+ "model.layers.12.mlp.gate.weight": "model-layer-12.safetensors",
89
+ "model.layers.12.mlp.shared_expert.down_proj.weight": "model-layer-12.safetensors",
90
+ "model.layers.12.mlp.shared_expert.gate_proj.weight": "model-layer-12.safetensors",
91
+ "model.layers.12.mlp.shared_expert.up_proj.weight": "model-layer-12.safetensors",
92
+ "model.layers.12.mlp.shared_expert_gate.weight": "model-layer-12.safetensors",
93
+ "model.layers.12.post_attention_layernorm.weight": "model-common.safetensors",
94
+ "model.layers.13.input_layernorm.weight": "model-common.safetensors",
95
+ "model.layers.13.linear_attn.A_log": "model-common.safetensors",
96
+ "model.layers.13.linear_attn.conv1d.weight": "model-common.safetensors",
97
+ "model.layers.13.linear_attn.dt_bias": "model-common.safetensors",
98
+ "model.layers.13.linear_attn.in_proj_a.weight": "model-common.safetensors",
99
+ "model.layers.13.linear_attn.in_proj_b.weight": "model-common.safetensors",
100
+ "model.layers.13.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
101
+ "model.layers.13.linear_attn.in_proj_z.weight": "model-common.safetensors",
102
+ "model.layers.13.linear_attn.norm.weight": "model-common.safetensors",
103
+ "model.layers.13.linear_attn.out_proj.weight": "model-common.safetensors",
104
+ "model.layers.13.mlp.experts.down_proj": "model-layer-13.safetensors",
105
+ "model.layers.13.mlp.experts.gate_up_proj": "model-layer-13.safetensors",
106
+ "model.layers.13.mlp.gate.weight": "model-layer-13.safetensors",
107
+ "model.layers.13.mlp.shared_expert.down_proj.weight": "model-layer-13.safetensors",
108
+ "model.layers.13.mlp.shared_expert.gate_proj.weight": "model-layer-13.safetensors",
109
+ "model.layers.13.mlp.shared_expert.up_proj.weight": "model-layer-13.safetensors",
110
+ "model.layers.13.mlp.shared_expert_gate.weight": "model-layer-13.safetensors",
111
+ "model.layers.13.post_attention_layernorm.weight": "model-common.safetensors",
112
+ "model.layers.14.input_layernorm.weight": "model-common.safetensors",
113
+ "model.layers.14.linear_attn.A_log": "model-common.safetensors",
114
+ "model.layers.14.linear_attn.conv1d.weight": "model-common.safetensors",
115
+ "model.layers.14.linear_attn.dt_bias": "model-common.safetensors",
116
+ "model.layers.14.linear_attn.in_proj_a.weight": "model-common.safetensors",
117
+ "model.layers.14.linear_attn.in_proj_b.weight": "model-common.safetensors",
118
+ "model.layers.14.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
119
+ "model.layers.14.linear_attn.in_proj_z.weight": "model-common.safetensors",
120
+ "model.layers.14.linear_attn.norm.weight": "model-common.safetensors",
121
+ "model.layers.14.linear_attn.out_proj.weight": "model-common.safetensors",
122
+ "model.layers.14.mlp.experts.down_proj": "model-layer-14.safetensors",
123
+ "model.layers.14.mlp.experts.gate_up_proj": "model-layer-14.safetensors",
124
+ "model.layers.14.mlp.gate.weight": "model-layer-14.safetensors",
125
+ "model.layers.14.mlp.shared_expert.down_proj.weight": "model-layer-14.safetensors",
126
+ "model.layers.14.mlp.shared_expert.gate_proj.weight": "model-layer-14.safetensors",
127
+ "model.layers.14.mlp.shared_expert.up_proj.weight": "model-layer-14.safetensors",
128
+ "model.layers.14.mlp.shared_expert_gate.weight": "model-layer-14.safetensors",
129
+ "model.layers.14.post_attention_layernorm.weight": "model-common.safetensors",
130
+ "model.layers.15.input_layernorm.weight": "model-common.safetensors",
131
+ "model.layers.15.mlp.experts.down_proj": "model-layer-15.safetensors",
132
+ "model.layers.15.mlp.experts.gate_up_proj": "model-layer-15.safetensors",
133
+ "model.layers.15.mlp.gate.weight": "model-layer-15.safetensors",
134
+ "model.layers.15.mlp.shared_expert.down_proj.weight": "model-layer-15.safetensors",
135
+ "model.layers.15.mlp.shared_expert.gate_proj.weight": "model-layer-15.safetensors",
136
+ "model.layers.15.mlp.shared_expert.up_proj.weight": "model-layer-15.safetensors",
137
+ "model.layers.15.mlp.shared_expert_gate.weight": "model-layer-15.safetensors",
138
+ "model.layers.15.post_attention_layernorm.weight": "model-common.safetensors",
139
+ "model.layers.15.self_attn.k_norm.weight": "model-common.safetensors",
140
+ "model.layers.15.self_attn.k_proj.weight": "model-common.safetensors",
141
+ "model.layers.15.self_attn.o_proj.weight": "model-common.safetensors",
142
+ "model.layers.15.self_attn.q_norm.weight": "model-common.safetensors",
143
+ "model.layers.15.self_attn.q_proj.weight": "model-common.safetensors",
144
+ "model.layers.15.self_attn.v_proj.weight": "model-common.safetensors",
145
+ "model.layers.16.input_layernorm.weight": "model-common.safetensors",
146
+ "model.layers.16.linear_attn.A_log": "model-common.safetensors",
147
+ "model.layers.16.linear_attn.conv1d.weight": "model-common.safetensors",
148
+ "model.layers.16.linear_attn.dt_bias": "model-common.safetensors",
149
+ "model.layers.16.linear_attn.in_proj_a.weight": "model-common.safetensors",
150
+ "model.layers.16.linear_attn.in_proj_b.weight": "model-common.safetensors",
151
+ "model.layers.16.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
152
+ "model.layers.16.linear_attn.in_proj_z.weight": "model-common.safetensors",
153
+ "model.layers.16.linear_attn.norm.weight": "model-common.safetensors",
154
+ "model.layers.16.linear_attn.out_proj.weight": "model-common.safetensors",
155
+ "model.layers.16.mlp.experts.down_proj": "model-layer-16.safetensors",
156
+ "model.layers.16.mlp.experts.gate_up_proj": "model-layer-16.safetensors",
157
+ "model.layers.16.mlp.gate.weight": "model-layer-16.safetensors",
158
+ "model.layers.16.mlp.shared_expert.down_proj.weight": "model-layer-16.safetensors",
159
+ "model.layers.16.mlp.shared_expert.gate_proj.weight": "model-layer-16.safetensors",
160
+ "model.layers.16.mlp.shared_expert.up_proj.weight": "model-layer-16.safetensors",
161
+ "model.layers.16.mlp.shared_expert_gate.weight": "model-layer-16.safetensors",
162
+ "model.layers.16.post_attention_layernorm.weight": "model-common.safetensors",
163
+ "model.layers.17.input_layernorm.weight": "model-common.safetensors",
164
+ "model.layers.17.linear_attn.A_log": "model-common.safetensors",
165
+ "model.layers.17.linear_attn.conv1d.weight": "model-common.safetensors",
166
+ "model.layers.17.linear_attn.dt_bias": "model-common.safetensors",
167
+ "model.layers.17.linear_attn.in_proj_a.weight": "model-common.safetensors",
168
+ "model.layers.17.linear_attn.in_proj_b.weight": "model-common.safetensors",
169
+ "model.layers.17.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
170
+ "model.layers.17.linear_attn.in_proj_z.weight": "model-common.safetensors",
171
+ "model.layers.17.linear_attn.norm.weight": "model-common.safetensors",
172
+ "model.layers.17.linear_attn.out_proj.weight": "model-common.safetensors",
173
+ "model.layers.17.mlp.experts.down_proj": "model-layer-17.safetensors",
174
+ "model.layers.17.mlp.experts.gate_up_proj": "model-layer-17.safetensors",
175
+ "model.layers.17.mlp.gate.weight": "model-layer-17.safetensors",
176
+ "model.layers.17.mlp.shared_expert.down_proj.weight": "model-layer-17.safetensors",
177
+ "model.layers.17.mlp.shared_expert.gate_proj.weight": "model-layer-17.safetensors",
178
+ "model.layers.17.mlp.shared_expert.up_proj.weight": "model-layer-17.safetensors",
179
+ "model.layers.17.mlp.shared_expert_gate.weight": "model-layer-17.safetensors",
180
+ "model.layers.17.post_attention_layernorm.weight": "model-common.safetensors",
181
+ "model.layers.18.input_layernorm.weight": "model-common.safetensors",
182
+ "model.layers.18.linear_attn.A_log": "model-common.safetensors",
183
+ "model.layers.18.linear_attn.conv1d.weight": "model-common.safetensors",
184
+ "model.layers.18.linear_attn.dt_bias": "model-common.safetensors",
185
+ "model.layers.18.linear_attn.in_proj_a.weight": "model-common.safetensors",
186
+ "model.layers.18.linear_attn.in_proj_b.weight": "model-common.safetensors",
187
+ "model.layers.18.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
188
+ "model.layers.18.linear_attn.in_proj_z.weight": "model-common.safetensors",
189
+ "model.layers.18.linear_attn.norm.weight": "model-common.safetensors",
190
+ "model.layers.18.linear_attn.out_proj.weight": "model-common.safetensors",
191
+ "model.layers.18.mlp.experts.down_proj": "model-layer-18.safetensors",
192
+ "model.layers.18.mlp.experts.gate_up_proj": "model-layer-18.safetensors",
193
+ "model.layers.18.mlp.gate.weight": "model-layer-18.safetensors",
194
+ "model.layers.18.mlp.shared_expert.down_proj.weight": "model-layer-18.safetensors",
195
+ "model.layers.18.mlp.shared_expert.gate_proj.weight": "model-layer-18.safetensors",
196
+ "model.layers.18.mlp.shared_expert.up_proj.weight": "model-layer-18.safetensors",
197
+ "model.layers.18.mlp.shared_expert_gate.weight": "model-layer-18.safetensors",
198
+ "model.layers.18.post_attention_layernorm.weight": "model-common.safetensors",
199
+ "model.layers.19.input_layernorm.weight": "model-common.safetensors",
200
+ "model.layers.19.mlp.experts.down_proj": "model-layer-19.safetensors",
201
+ "model.layers.19.mlp.experts.gate_up_proj": "model-layer-19.safetensors",
202
+ "model.layers.19.mlp.gate.weight": "model-layer-19.safetensors",
203
+ "model.layers.19.mlp.shared_expert.down_proj.weight": "model-layer-19.safetensors",
204
+ "model.layers.19.mlp.shared_expert.gate_proj.weight": "model-layer-19.safetensors",
205
+ "model.layers.19.mlp.shared_expert.up_proj.weight": "model-layer-19.safetensors",
206
+ "model.layers.19.mlp.shared_expert_gate.weight": "model-layer-19.safetensors",
207
+ "model.layers.19.post_attention_layernorm.weight": "model-common.safetensors",
208
+ "model.layers.19.self_attn.k_norm.weight": "model-common.safetensors",
209
+ "model.layers.19.self_attn.k_proj.weight": "model-common.safetensors",
210
+ "model.layers.19.self_attn.o_proj.weight": "model-common.safetensors",
211
+ "model.layers.19.self_attn.q_norm.weight": "model-common.safetensors",
212
+ "model.layers.19.self_attn.q_proj.weight": "model-common.safetensors",
213
+ "model.layers.19.self_attn.v_proj.weight": "model-common.safetensors",
214
+ "model.layers.2.input_layernorm.weight": "model-common.safetensors",
215
+ "model.layers.2.linear_attn.A_log": "model-common.safetensors",
216
+ "model.layers.2.linear_attn.conv1d.weight": "model-common.safetensors",
217
+ "model.layers.2.linear_attn.dt_bias": "model-common.safetensors",
218
+ "model.layers.2.linear_attn.in_proj_a.weight": "model-common.safetensors",
219
+ "model.layers.2.linear_attn.in_proj_b.weight": "model-common.safetensors",
220
+ "model.layers.2.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
221
+ "model.layers.2.linear_attn.in_proj_z.weight": "model-common.safetensors",
222
+ "model.layers.2.linear_attn.norm.weight": "model-common.safetensors",
223
+ "model.layers.2.linear_attn.out_proj.weight": "model-common.safetensors",
224
+ "model.layers.2.mlp.experts.down_proj": "model-layer-02.safetensors",
225
+ "model.layers.2.mlp.experts.gate_up_proj": "model-layer-02.safetensors",
226
+ "model.layers.2.mlp.gate.weight": "model-layer-02.safetensors",
227
+ "model.layers.2.mlp.shared_expert.down_proj.weight": "model-layer-02.safetensors",
228
+ "model.layers.2.mlp.shared_expert.gate_proj.weight": "model-layer-02.safetensors",
229
+ "model.layers.2.mlp.shared_expert.up_proj.weight": "model-layer-02.safetensors",
230
+ "model.layers.2.mlp.shared_expert_gate.weight": "model-layer-02.safetensors",
231
+ "model.layers.2.post_attention_layernorm.weight": "model-common.safetensors",
232
+ "model.layers.20.input_layernorm.weight": "model-common.safetensors",
233
+ "model.layers.20.linear_attn.A_log": "model-common.safetensors",
234
+ "model.layers.20.linear_attn.conv1d.weight": "model-common.safetensors",
235
+ "model.layers.20.linear_attn.dt_bias": "model-common.safetensors",
236
+ "model.layers.20.linear_attn.in_proj_a.weight": "model-common.safetensors",
237
+ "model.layers.20.linear_attn.in_proj_b.weight": "model-common.safetensors",
238
+ "model.layers.20.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
239
+ "model.layers.20.linear_attn.in_proj_z.weight": "model-common.safetensors",
240
+ "model.layers.20.linear_attn.norm.weight": "model-common.safetensors",
241
+ "model.layers.20.linear_attn.out_proj.weight": "model-common.safetensors",
242
+ "model.layers.20.mlp.experts.down_proj": "model-layer-20.safetensors",
243
+ "model.layers.20.mlp.experts.gate_up_proj": "model-layer-20.safetensors",
244
+ "model.layers.20.mlp.gate.weight": "model-layer-20.safetensors",
245
+ "model.layers.20.mlp.shared_expert.down_proj.weight": "model-layer-20.safetensors",
246
+ "model.layers.20.mlp.shared_expert.gate_proj.weight": "model-layer-20.safetensors",
247
+ "model.layers.20.mlp.shared_expert.up_proj.weight": "model-layer-20.safetensors",
248
+ "model.layers.20.mlp.shared_expert_gate.weight": "model-layer-20.safetensors",
249
+ "model.layers.20.post_attention_layernorm.weight": "model-common.safetensors",
250
+ "model.layers.21.input_layernorm.weight": "model-common.safetensors",
251
+ "model.layers.21.linear_attn.A_log": "model-common.safetensors",
252
+ "model.layers.21.linear_attn.conv1d.weight": "model-common.safetensors",
253
+ "model.layers.21.linear_attn.dt_bias": "model-common.safetensors",
254
+ "model.layers.21.linear_attn.in_proj_a.weight": "model-common.safetensors",
255
+ "model.layers.21.linear_attn.in_proj_b.weight": "model-common.safetensors",
256
+ "model.layers.21.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
257
+ "model.layers.21.linear_attn.in_proj_z.weight": "model-common.safetensors",
258
+ "model.layers.21.linear_attn.norm.weight": "model-common.safetensors",
259
+ "model.layers.21.linear_attn.out_proj.weight": "model-common.safetensors",
260
+ "model.layers.21.mlp.experts.down_proj": "model-layer-21.safetensors",
261
+ "model.layers.21.mlp.experts.gate_up_proj": "model-layer-21.safetensors",
262
+ "model.layers.21.mlp.gate.weight": "model-layer-21.safetensors",
263
+ "model.layers.21.mlp.shared_expert.down_proj.weight": "model-layer-21.safetensors",
264
+ "model.layers.21.mlp.shared_expert.gate_proj.weight": "model-layer-21.safetensors",
265
+ "model.layers.21.mlp.shared_expert.up_proj.weight": "model-layer-21.safetensors",
266
+ "model.layers.21.mlp.shared_expert_gate.weight": "model-layer-21.safetensors",
267
+ "model.layers.21.post_attention_layernorm.weight": "model-common.safetensors",
268
+ "model.layers.22.input_layernorm.weight": "model-common.safetensors",
269
+ "model.layers.22.linear_attn.A_log": "model-common.safetensors",
270
+ "model.layers.22.linear_attn.conv1d.weight": "model-common.safetensors",
271
+ "model.layers.22.linear_attn.dt_bias": "model-common.safetensors",
272
+ "model.layers.22.linear_attn.in_proj_a.weight": "model-common.safetensors",
273
+ "model.layers.22.linear_attn.in_proj_b.weight": "model-common.safetensors",
274
+ "model.layers.22.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
275
+ "model.layers.22.linear_attn.in_proj_z.weight": "model-common.safetensors",
276
+ "model.layers.22.linear_attn.norm.weight": "model-common.safetensors",
277
+ "model.layers.22.linear_attn.out_proj.weight": "model-common.safetensors",
278
+ "model.layers.22.mlp.experts.down_proj": "model-layer-22.safetensors",
279
+ "model.layers.22.mlp.experts.gate_up_proj": "model-layer-22.safetensors",
280
+ "model.layers.22.mlp.gate.weight": "model-layer-22.safetensors",
281
+ "model.layers.22.mlp.shared_expert.down_proj.weight": "model-layer-22.safetensors",
282
+ "model.layers.22.mlp.shared_expert.gate_proj.weight": "model-layer-22.safetensors",
283
+ "model.layers.22.mlp.shared_expert.up_proj.weight": "model-layer-22.safetensors",
284
+ "model.layers.22.mlp.shared_expert_gate.weight": "model-layer-22.safetensors",
285
+ "model.layers.22.post_attention_layernorm.weight": "model-common.safetensors",
286
+ "model.layers.23.input_layernorm.weight": "model-common.safetensors",
287
+ "model.layers.23.mlp.experts.down_proj": "model-layer-23.safetensors",
288
+ "model.layers.23.mlp.experts.gate_up_proj": "model-layer-23.safetensors",
289
+ "model.layers.23.mlp.gate.weight": "model-layer-23.safetensors",
290
+ "model.layers.23.mlp.shared_expert.down_proj.weight": "model-layer-23.safetensors",
291
+ "model.layers.23.mlp.shared_expert.gate_proj.weight": "model-layer-23.safetensors",
292
+ "model.layers.23.mlp.shared_expert.up_proj.weight": "model-layer-23.safetensors",
293
+ "model.layers.23.mlp.shared_expert_gate.weight": "model-layer-23.safetensors",
294
+ "model.layers.23.post_attention_layernorm.weight": "model-common.safetensors",
295
+ "model.layers.23.self_attn.k_norm.weight": "model-common.safetensors",
296
+ "model.layers.23.self_attn.k_proj.weight": "model-common.safetensors",
297
+ "model.layers.23.self_attn.o_proj.weight": "model-common.safetensors",
298
+ "model.layers.23.self_attn.q_norm.weight": "model-common.safetensors",
299
+ "model.layers.23.self_attn.q_proj.weight": "model-common.safetensors",
300
+ "model.layers.23.self_attn.v_proj.weight": "model-common.safetensors",
301
+ "model.layers.3.input_layernorm.weight": "model-common.safetensors",
302
+ "model.layers.3.mlp.experts.down_proj": "model-layer-03.safetensors",
303
+ "model.layers.3.mlp.experts.gate_up_proj": "model-layer-03.safetensors",
304
+ "model.layers.3.mlp.gate.weight": "model-layer-03.safetensors",
305
+ "model.layers.3.mlp.shared_expert.down_proj.weight": "model-layer-03.safetensors",
306
+ "model.layers.3.mlp.shared_expert.gate_proj.weight": "model-layer-03.safetensors",
307
+ "model.layers.3.mlp.shared_expert.up_proj.weight": "model-layer-03.safetensors",
308
+ "model.layers.3.mlp.shared_expert_gate.weight": "model-layer-03.safetensors",
309
+ "model.layers.3.post_attention_layernorm.weight": "model-common.safetensors",
310
+ "model.layers.3.self_attn.k_norm.weight": "model-common.safetensors",
311
+ "model.layers.3.self_attn.k_proj.weight": "model-common.safetensors",
312
+ "model.layers.3.self_attn.o_proj.weight": "model-common.safetensors",
313
+ "model.layers.3.self_attn.q_norm.weight": "model-common.safetensors",
314
+ "model.layers.3.self_attn.q_proj.weight": "model-common.safetensors",
315
+ "model.layers.3.self_attn.v_proj.weight": "model-common.safetensors",
316
+ "model.layers.4.input_layernorm.weight": "model-common.safetensors",
317
+ "model.layers.4.linear_attn.A_log": "model-common.safetensors",
318
+ "model.layers.4.linear_attn.conv1d.weight": "model-common.safetensors",
319
+ "model.layers.4.linear_attn.dt_bias": "model-common.safetensors",
320
+ "model.layers.4.linear_attn.in_proj_a.weight": "model-common.safetensors",
321
+ "model.layers.4.linear_attn.in_proj_b.weight": "model-common.safetensors",
322
+ "model.layers.4.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
323
+ "model.layers.4.linear_attn.in_proj_z.weight": "model-common.safetensors",
324
+ "model.layers.4.linear_attn.norm.weight": "model-common.safetensors",
325
+ "model.layers.4.linear_attn.out_proj.weight": "model-common.safetensors",
326
+ "model.layers.4.mlp.experts.down_proj": "model-layer-04.safetensors",
327
+ "model.layers.4.mlp.experts.gate_up_proj": "model-layer-04.safetensors",
328
+ "model.layers.4.mlp.gate.weight": "model-layer-04.safetensors",
329
+ "model.layers.4.mlp.shared_expert.down_proj.weight": "model-layer-04.safetensors",
330
+ "model.layers.4.mlp.shared_expert.gate_proj.weight": "model-layer-04.safetensors",
331
+ "model.layers.4.mlp.shared_expert.up_proj.weight": "model-layer-04.safetensors",
332
+ "model.layers.4.mlp.shared_expert_gate.weight": "model-layer-04.safetensors",
333
+ "model.layers.4.post_attention_layernorm.weight": "model-common.safetensors",
334
+ "model.layers.5.input_layernorm.weight": "model-common.safetensors",
335
+ "model.layers.5.linear_attn.A_log": "model-common.safetensors",
336
+ "model.layers.5.linear_attn.conv1d.weight": "model-common.safetensors",
337
+ "model.layers.5.linear_attn.dt_bias": "model-common.safetensors",
338
+ "model.layers.5.linear_attn.in_proj_a.weight": "model-common.safetensors",
339
+ "model.layers.5.linear_attn.in_proj_b.weight": "model-common.safetensors",
340
+ "model.layers.5.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
341
+ "model.layers.5.linear_attn.in_proj_z.weight": "model-common.safetensors",
342
+ "model.layers.5.linear_attn.norm.weight": "model-common.safetensors",
343
+ "model.layers.5.linear_attn.out_proj.weight": "model-common.safetensors",
344
+ "model.layers.5.mlp.experts.down_proj": "model-layer-05.safetensors",
345
+ "model.layers.5.mlp.experts.gate_up_proj": "model-layer-05.safetensors",
346
+ "model.layers.5.mlp.gate.weight": "model-layer-05.safetensors",
347
+ "model.layers.5.mlp.shared_expert.down_proj.weight": "model-layer-05.safetensors",
348
+ "model.layers.5.mlp.shared_expert.gate_proj.weight": "model-layer-05.safetensors",
349
+ "model.layers.5.mlp.shared_expert.up_proj.weight": "model-layer-05.safetensors",
350
+ "model.layers.5.mlp.shared_expert_gate.weight": "model-layer-05.safetensors",
351
+ "model.layers.5.post_attention_layernorm.weight": "model-common.safetensors",
352
+ "model.layers.6.input_layernorm.weight": "model-common.safetensors",
353
+ "model.layers.6.linear_attn.A_log": "model-common.safetensors",
354
+ "model.layers.6.linear_attn.conv1d.weight": "model-common.safetensors",
355
+ "model.layers.6.linear_attn.dt_bias": "model-common.safetensors",
356
+ "model.layers.6.linear_attn.in_proj_a.weight": "model-common.safetensors",
357
+ "model.layers.6.linear_attn.in_proj_b.weight": "model-common.safetensors",
358
+ "model.layers.6.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
359
+ "model.layers.6.linear_attn.in_proj_z.weight": "model-common.safetensors",
360
+ "model.layers.6.linear_attn.norm.weight": "model-common.safetensors",
361
+ "model.layers.6.linear_attn.out_proj.weight": "model-common.safetensors",
362
+ "model.layers.6.mlp.experts.down_proj": "model-layer-06.safetensors",
363
+ "model.layers.6.mlp.experts.gate_up_proj": "model-layer-06.safetensors",
364
+ "model.layers.6.mlp.gate.weight": "model-layer-06.safetensors",
365
+ "model.layers.6.mlp.shared_expert.down_proj.weight": "model-layer-06.safetensors",
366
+ "model.layers.6.mlp.shared_expert.gate_proj.weight": "model-layer-06.safetensors",
367
+ "model.layers.6.mlp.shared_expert.up_proj.weight": "model-layer-06.safetensors",
368
+ "model.layers.6.mlp.shared_expert_gate.weight": "model-layer-06.safetensors",
369
+ "model.layers.6.post_attention_layernorm.weight": "model-common.safetensors",
370
+ "model.layers.7.input_layernorm.weight": "model-common.safetensors",
371
+ "model.layers.7.mlp.experts.down_proj": "model-layer-07.safetensors",
372
+ "model.layers.7.mlp.experts.gate_up_proj": "model-layer-07.safetensors",
373
+ "model.layers.7.mlp.gate.weight": "model-layer-07.safetensors",
374
+ "model.layers.7.mlp.shared_expert.down_proj.weight": "model-layer-07.safetensors",
375
+ "model.layers.7.mlp.shared_expert.gate_proj.weight": "model-layer-07.safetensors",
376
+ "model.layers.7.mlp.shared_expert.up_proj.weight": "model-layer-07.safetensors",
377
+ "model.layers.7.mlp.shared_expert_gate.weight": "model-layer-07.safetensors",
378
+ "model.layers.7.post_attention_layernorm.weight": "model-common.safetensors",
379
+ "model.layers.7.self_attn.k_norm.weight": "model-common.safetensors",
380
+ "model.layers.7.self_attn.k_proj.weight": "model-common.safetensors",
381
+ "model.layers.7.self_attn.o_proj.weight": "model-common.safetensors",
382
+ "model.layers.7.self_attn.q_norm.weight": "model-common.safetensors",
383
+ "model.layers.7.self_attn.q_proj.weight": "model-common.safetensors",
384
+ "model.layers.7.self_attn.v_proj.weight": "model-common.safetensors",
385
+ "model.layers.8.input_layernorm.weight": "model-common.safetensors",
386
+ "model.layers.8.linear_attn.A_log": "model-common.safetensors",
387
+ "model.layers.8.linear_attn.conv1d.weight": "model-common.safetensors",
388
+ "model.layers.8.linear_attn.dt_bias": "model-common.safetensors",
389
+ "model.layers.8.linear_attn.in_proj_a.weight": "model-common.safetensors",
390
+ "model.layers.8.linear_attn.in_proj_b.weight": "model-common.safetensors",
391
+ "model.layers.8.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
392
+ "model.layers.8.linear_attn.in_proj_z.weight": "model-common.safetensors",
393
+ "model.layers.8.linear_attn.norm.weight": "model-common.safetensors",
394
+ "model.layers.8.linear_attn.out_proj.weight": "model-common.safetensors",
395
+ "model.layers.8.mlp.experts.down_proj": "model-layer-08.safetensors",
396
+ "model.layers.8.mlp.experts.gate_up_proj": "model-layer-08.safetensors",
397
+ "model.layers.8.mlp.gate.weight": "model-layer-08.safetensors",
398
+ "model.layers.8.mlp.shared_expert.down_proj.weight": "model-layer-08.safetensors",
399
+ "model.layers.8.mlp.shared_expert.gate_proj.weight": "model-layer-08.safetensors",
400
+ "model.layers.8.mlp.shared_expert.up_proj.weight": "model-layer-08.safetensors",
401
+ "model.layers.8.mlp.shared_expert_gate.weight": "model-layer-08.safetensors",
402
+ "model.layers.8.post_attention_layernorm.weight": "model-common.safetensors",
403
+ "model.layers.9.input_layernorm.weight": "model-common.safetensors",
404
+ "model.layers.9.linear_attn.A_log": "model-common.safetensors",
405
+ "model.layers.9.linear_attn.conv1d.weight": "model-common.safetensors",
406
+ "model.layers.9.linear_attn.dt_bias": "model-common.safetensors",
407
+ "model.layers.9.linear_attn.in_proj_a.weight": "model-common.safetensors",
408
+ "model.layers.9.linear_attn.in_proj_b.weight": "model-common.safetensors",
409
+ "model.layers.9.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
410
+ "model.layers.9.linear_attn.in_proj_z.weight": "model-common.safetensors",
411
+ "model.layers.9.linear_attn.norm.weight": "model-common.safetensors",
412
+ "model.layers.9.linear_attn.out_proj.weight": "model-common.safetensors",
413
+ "model.layers.9.mlp.experts.down_proj": "model-layer-09.safetensors",
414
+ "model.layers.9.mlp.experts.gate_up_proj": "model-layer-09.safetensors",
415
+ "model.layers.9.mlp.gate.weight": "model-layer-09.safetensors",
416
+ "model.layers.9.mlp.shared_expert.down_proj.weight": "model-layer-09.safetensors",
417
+ "model.layers.9.mlp.shared_expert.gate_proj.weight": "model-layer-09.safetensors",
418
+ "model.layers.9.mlp.shared_expert.up_proj.weight": "model-layer-09.safetensors",
419
+ "model.layers.9.mlp.shared_expert_gate.weight": "model-layer-09.safetensors",
420
+ "model.layers.9.post_attention_layernorm.weight": "model-common.safetensors",
421
+ "model.norm.weight": "model-common.safetensors"
422
+ }
423
+ }
research_code/chat-gate-60.jsonl ADDED
@@ -0,0 +1,60 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {"id":"ko_01","language":"ko","prompt":"프랑스의 수도는 어디인가요? 짧게 답하세요.","expected_any":["파리"]}
2
+ {"id":"ko_02","language":"ko","prompt":"7 곱하기 8은 얼마인가요? 숫자로 답하세요.","expected_any":["56"]}
3
+ {"id":"ko_03","language":"ko","prompt":"물의 화학식은 무엇인가요?","expected_any":["H2O","h2o"]}
4
+ {"id":"ko_04","language":"ko","prompt":"태양계에서 가장 큰 행성은 무엇인가요?","expected_any":["목성"]}
5
+ {"id":"ko_05","language":"ko","prompt":"조용한 겨울 아침을 묘사하는 자연스러운 문장 두 개를 써 주세요."}
6
+ {"id":"ko_06","language":"ko","prompt":"친구에게 약속 시간을 10분 늦겠다고 정중히 알리는 한 문장을 써 주세요."}
7
+ {"id":"ko_07","language":"ko","prompt":"인공지능 연구에서 재현성이 중요한 이유를 두 문장으로 설명하세요."}
8
+ {"id":"ko_08","language":"ko","prompt":"다른 말 없이 정확히 '확인'이라고만 답하세요.","expected_any":["확인"]}
9
+ {"id":"ko_09","language":"ko","prompt":"사과, 바나나, 포도를 번호가 있는 세 항목으로 나열하세요."}
10
+ {"id":"ko_10","language":"ko","prompt":"'Good morning'을 자연스러운 한국어로 번역하세요.","expected_any":["좋은 아침","안녕하세요"]}
11
+ {"id":"en_01","language":"en","prompt":"What is the capital of France? Answer briefly.","expected_any":["Paris"]}
12
+ {"id":"en_02","language":"en","prompt":"What is 7 multiplied by 8? Answer with a number.","expected_any":["56"]}
13
+ {"id":"en_03","language":"en","prompt":"What is the chemical formula for water?","expected_any":["H2O","h2o"]}
14
+ {"id":"en_04","language":"en","prompt":"What is the largest planet in the Solar System?","expected_any":["Jupiter"]}
15
+ {"id":"en_05","language":"en","prompt":"Write two natural sentences describing a quiet winter morning."}
16
+ {"id":"en_06","language":"en","prompt":"Write one polite sentence telling a friend you will be ten minutes late."}
17
+ {"id":"en_07","language":"en","prompt":"Explain in two sentences why reproducibility matters in AI research."}
18
+ {"id":"en_08","language":"en","prompt":"Reply with exactly the word 'confirmed' and nothing else.","expected_any":["confirmed"]}
19
+ {"id":"en_09","language":"en","prompt":"List apple, banana, and grape as three numbered items."}
20
+ {"id":"en_10","language":"en","prompt":"Translate '좋은 아침입니다' into natural English.","expected_any":["good morning"]}
21
+ {"id":"zh_01","language":"zh","prompt":"法国的首都是哪里?请简短回答。","expected_any":["巴黎"]}
22
+ {"id":"zh_02","language":"zh","prompt":"7乘以8等于多少?请用数字回答。","expected_any":["56"]}
23
+ {"id":"zh_03","language":"zh","prompt":"水的化学式是什么?","expected_any":["H2O","h2o"]}
24
+ {"id":"zh_04","language":"zh","prompt":"太阳系中最大的行星是什么?","expected_any":["木星"]}
25
+ {"id":"zh_05","language":"zh","prompt":"用两个自然的句子描写安静的冬日清晨。"}
26
+ {"id":"zh_06","language":"zh","prompt":"写一句礼貌的话,告诉朋友你会迟到十分钟。"}
27
+ {"id":"zh_07","language":"zh","prompt":"用两句话解释可复现性为何对人工智能研究重要。"}
28
+ {"id":"zh_08","language":"zh","prompt":"不要说别的,只回答“收到”。","expected_any":["收到"]}
29
+ {"id":"zh_09","language":"zh","prompt":"把苹果、香蕉和葡萄列成三个编号项目。"}
30
+ {"id":"zh_10","language":"zh","prompt":"把“Good morning”翻译成自然的中文。","expected_any":["早上好","早安"]}
31
+ {"id":"ja_01","language":"ja","prompt":"フランスの首都はどこですか。短く答えてください。","expected_any":["パリ"]}
32
+ {"id":"ja_02","language":"ja","prompt":"7かける8はいくつですか。数字で答えてください。","expected_any":["56"]}
33
+ {"id":"ja_03","language":"ja","prompt":"水の化学式は何ですか。","expected_any":["H2O","h2o"]}
34
+ {"id":"ja_04","language":"ja","prompt":"太陽系で最も大きい惑星は何ですか。","expected_any":["木星"]}
35
+ {"id":"ja_05","language":"ja","prompt":"静かな冬の朝を自然な二文で描写してください。"}
36
+ {"id":"ja_06","language":"ja","prompt":"友人に10分遅れることを丁寧に伝える一文を書いてください。"}
37
+ {"id":"ja_07","language":"ja","prompt":"AI研究で再現性が重要な理由を二文で説明してください。"}
38
+ {"id":"ja_08","language":"ja","prompt":"ほかの言葉を加えず「了解」とだけ答えてください。","expected_any":["了解"]}
39
+ {"id":"ja_09","language":"ja","prompt":"りんご、バナナ、ぶどうを番号付きの三項目で並べてください。"}
40
+ {"id":"ja_10","language":"ja","prompt":"「Good morning」を自然な日本語に訳してください。","expected_any":["おはよう"]}
41
+ {"id":"es_01","language":"es","prompt":"¿Cuál es la capital de Francia? Responde brevemente.","expected_any":["París","Paris"]}
42
+ {"id":"es_02","language":"es","prompt":"¿Cuánto es 7 por 8? Responde con un número.","expected_any":["56"]}
43
+ {"id":"es_03","language":"es","prompt":"¿Cuál es la fórmula química del agua?","expected_any":["H2O","h2o"]}
44
+ {"id":"es_04","language":"es","prompt":"¿Cuál es el planeta más grande del sistema solar?","expected_any":["Júpiter","Jupiter"]}
45
+ {"id":"es_05","language":"es","prompt":"Escribe dos frases naturales que describan una tranquila mañana de invierno."}
46
+ {"id":"es_06","language":"es","prompt":"Escribe una frase cortés para decirle a un amigo que llegarás diez minutos tarde."}
47
+ {"id":"es_07","language":"es","prompt":"Explica en dos frases por qué la reproducibilidad importa en la investigación de IA."}
48
+ {"id":"es_08","language":"es","prompt":"Responde únicamente con la palabra 'entendido'.","expected_any":["entendido"]}
49
+ {"id":"es_09","language":"es","prompt":"Enumera manzana, plátano y uva como tres elementos numerados."}
50
+ {"id":"es_10","language":"es","prompt":"Traduce 'Good morning' a un español natural.","expected_any":["buenos días","buen día"]}
51
+ {"id":"de_01","language":"de","prompt":"Was ist die Hauptstadt von Frankreich? Antworte kurz.","expected_any":["Paris"]}
52
+ {"id":"de_02","language":"de","prompt":"Was ist 7 mal 8? Antworte mit einer Zahl.","expected_any":["56"]}
53
+ {"id":"de_03","language":"de","prompt":"Wie lautet die chemische Formel für Wasser?","expected_any":["H2O","h2o"]}
54
+ {"id":"de_04","language":"de","prompt":"Welcher Planet ist der größte im Sonnensystem?","expected_any":["Jupiter"]}
55
+ {"id":"de_05","language":"de","prompt":"Schreibe zwei natürliche Sätze über einen ruhigen Wintermorgen."}
56
+ {"id":"de_06","language":"de","prompt":"Schreibe einen höflichen Satz, der einem Freund sagt, dass du zehn Minuten zu spät kommst."}
57
+ {"id":"de_07","language":"de","prompt":"Erkläre in zwei Sätzen, warum Reproduzierbarkeit in der KI-Forschung wichtig ist."}
58
+ {"id":"de_08","language":"de","prompt":"Antworte ausschließlich mit dem Wort 'Verstanden'.","expected_any":["verstanden"]}
59
+ {"id":"de_09","language":"de","prompt":"Liste Apfel, Banane und Traube als drei nummerierte Punkte auf."}
60
+ {"id":"de_10","language":"de","prompt":"Übersetze 'Good morning' in natürliches Deutsch.","expected_any":["guten morgen"]}
research_code/compare_lm_loss.py ADDED
@@ -0,0 +1,116 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import gc
6
+ import json
7
+ from pathlib import Path
8
+
9
+ import torch
10
+ from transformers import AutoConfig, AutoTokenizer
11
+ from transformers import Qwen3_5ForConditionalGeneration
12
+ from transformers import Qwen3_5MoeForCausalLM
13
+ from transformers import Qwen3_5MoeForConditionalGeneration
14
+
15
+
16
+ def parse_model(value: str) -> tuple[str, Path]:
17
+ if "=" not in value:
18
+ raise argparse.ArgumentTypeError("model must use NAME=PATH")
19
+ name, path = value.split("=", 1)
20
+ return name, Path(path)
21
+
22
+
23
+ def load_rows(path: Path, per_language: int) -> dict[str, list[str]]:
24
+ rows: dict[str, list[str]] = {}
25
+ with path.open(encoding="utf-8") as handle:
26
+ for line in handle:
27
+ row = json.loads(line)
28
+ if row.get("split") != "eval":
29
+ continue
30
+ bucket = rows.setdefault(row["language"], [])
31
+ if len(bucket) < per_language:
32
+ bucket.append(row["text"])
33
+ return rows
34
+
35
+
36
+ def model_class(path: Path):
37
+ config = AutoConfig.from_pretrained(path)
38
+ if config.model_type == "qwen3_5_moe_text":
39
+ return Qwen3_5MoeForCausalLM
40
+ if config.model_type == "qwen3_5_moe":
41
+ return Qwen3_5MoeForConditionalGeneration
42
+ return Qwen3_5ForConditionalGeneration
43
+
44
+
45
+ def evaluate(path: Path, tokenizer, rows, sequence_length: int, device: str):
46
+ model = model_class(path).from_pretrained(path, dtype=torch.bfloat16).to(device).eval()
47
+ model.config.use_cache = False
48
+ results = {}
49
+ with torch.no_grad():
50
+ for language, texts in rows.items():
51
+ losses = []
52
+ token_count = 0
53
+ for text in texts:
54
+ batch = tokenizer(
55
+ text,
56
+ return_tensors="pt",
57
+ truncation=True,
58
+ max_length=sequence_length,
59
+ )
60
+ batch = {key: value.to(device) for key, value in batch.items()}
61
+ output = model(
62
+ **batch,
63
+ labels=batch["input_ids"],
64
+ use_cache=False,
65
+ output_router_logits=False,
66
+ )
67
+ losses.append(float(output.loss.cpu()))
68
+ token_count += int(batch["attention_mask"].sum())
69
+ results[language] = {
70
+ "loss": sum(losses) / len(losses),
71
+ "documents": len(losses),
72
+ "tokens": token_count,
73
+ }
74
+ del model
75
+ gc.collect()
76
+ if device == "mps":
77
+ torch.mps.empty_cache()
78
+ return results
79
+
80
+
81
+ def main() -> None:
82
+ parser = argparse.ArgumentParser()
83
+ parser.add_argument("--model", action="append", type=parse_model, required=True)
84
+ parser.add_argument("--tokenizer", type=Path, required=True)
85
+ parser.add_argument("--corpus", type=Path, required=True)
86
+ parser.add_argument("--output", type=Path, required=True)
87
+ parser.add_argument("--documents-per-language", type=int, default=4)
88
+ parser.add_argument("--sequence-length", type=int, default=128)
89
+ parser.add_argument("--device", default="mps")
90
+ args = parser.parse_args()
91
+
92
+ rows = load_rows(args.corpus, args.documents_per_language)
93
+ tokenizer = AutoTokenizer.from_pretrained(args.tokenizer)
94
+ models = {}
95
+ for name, path in args.model:
96
+ print(f"evaluating {name}", flush=True)
97
+ models[name] = evaluate(path, tokenizer, rows, args.sequence_length, args.device)
98
+
99
+ means = {
100
+ name: sum(row["loss"] for row in result.values()) / len(result)
101
+ for name, result in models.items()
102
+ }
103
+ report = {
104
+ "sequence_length": args.sequence_length,
105
+ "documents_per_language": args.documents_per_language,
106
+ "languages": sorted(rows),
107
+ "models": models,
108
+ "mean_losses": means,
109
+ }
110
+ args.output.parent.mkdir(parents=True, exist_ok=True)
111
+ args.output.write_text(json.dumps(report, indent=2))
112
+ print(json.dumps(report, indent=2))
113
+
114
+
115
+ if __name__ == "__main__":
116
+ main()
research_code/convert_2b_to_4b_a3b.py ADDED
@@ -0,0 +1,228 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ """Expand Qwen3.5-2B into a text-only 4B-total/3B-active sparse MoE.
3
+
4
+ The initial checkpoint preserves the dense 2B MLP exactly: the shared path and
5
+ the selected routed expert each contribute one half of the original output.
6
+ Extra neurons start output-neutral but trainable.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import json
12
+ import shutil
13
+ from pathlib import Path
14
+
15
+ import torch
16
+ from safetensors import safe_open
17
+ from safetensors.torch import save_file
18
+ from transformers import Qwen3_5MoeForCausalLM, Qwen3_5MoeTextConfig
19
+
20
+
21
+ NUM_EXPERTS = 2
22
+ EXPERTS_PER_TOKEN = 1
23
+ SHARED_WIDTH = 6912
24
+ ROUTED_WIDTH = 6784
25
+ SEED = 35
26
+
27
+
28
+ def source_tensor(source: Path, weight_map: dict[str, str], name: str) -> torch.Tensor:
29
+ with safe_open(source / weight_map[name], framework="pt", device="cpu") as handle:
30
+ return handle.get_tensor(name)
31
+
32
+
33
+ def save_shard(
34
+ output: Path,
35
+ filename: str,
36
+ tensors: dict[str, torch.Tensor],
37
+ weight_map: dict[str, str],
38
+ ) -> int:
39
+ tensors = {name: tensor.contiguous() for name, tensor in tensors.items()}
40
+ save_file(tensors, output / filename, metadata={"format": "pt"})
41
+ weight_map.update({name: filename for name in tensors})
42
+ return sum(tensor.numel() * tensor.element_size() for tensor in tensors.values())
43
+
44
+
45
+ def build_config(source_config: dict) -> dict:
46
+ text = dict(source_config["text_config"])
47
+ dense_width = int(text.pop("intermediate_size"))
48
+ if text["hidden_size"] != 2048 or text["num_hidden_layers"] != 24 or dense_width != 6144:
49
+ raise ValueError("converter is intentionally pinned to Qwen3.5-2B")
50
+ text.update({
51
+ "architectures": ["Qwen3_5MoeForCausalLM"],
52
+ "model_type": "qwen3_5_moe_text",
53
+ "num_experts": NUM_EXPERTS,
54
+ "num_experts_per_tok": EXPERTS_PER_TOKEN,
55
+ "moe_intermediate_size": ROUTED_WIDTH,
56
+ "shared_expert_intermediate_size": SHARED_WIDTH,
57
+ "router_aux_loss_coef": 1e-3,
58
+ "output_router_logits": False,
59
+ })
60
+ return text
61
+
62
+
63
+ def main() -> None:
64
+ parser = argparse.ArgumentParser()
65
+ parser.add_argument("--source", type=Path, required=True)
66
+ parser.add_argument("--output", type=Path, required=True)
67
+ args = parser.parse_args()
68
+
69
+ if args.output.exists() and any(args.output.iterdir()):
70
+ raise SystemExit(f"Refusing non-empty output directory: {args.output}")
71
+ args.output.mkdir(parents=True, exist_ok=True)
72
+
73
+ source_config = json.loads((args.source / "config.json").read_text())
74
+ config_dict = build_config(source_config)
75
+ config = Qwen3_5MoeTextConfig(**config_dict)
76
+ with torch.device("meta"):
77
+ target = Qwen3_5MoeForCausalLM(config)
78
+ target_keys = set(target.state_dict())
79
+ total_parameters = sum(parameter.numel() for parameter in target.parameters())
80
+ inactive = (
81
+ config.num_hidden_layers
82
+ * (NUM_EXPERTS - EXPERTS_PER_TOKEN)
83
+ * 3
84
+ * config.hidden_size
85
+ * ROUTED_WIDTH
86
+ )
87
+ active_parameters = total_parameters - inactive
88
+
89
+ source_index = json.loads((args.source / "model.safetensors.index.json").read_text())
90
+ source_map: dict[str, str] = source_index["weight_map"]
91
+ output_map: dict[str, str] = {}
92
+ total_bytes = 0
93
+
94
+ # Copy the text backbone while dropping the vision tower and dense MLPs.
95
+ common: dict[str, torch.Tensor] = {}
96
+ prefix = "model.language_model."
97
+ for shard in sorted(set(source_map.values())):
98
+ with safe_open(args.source / shard, framework="pt", device="cpu") as handle:
99
+ for source_name in handle.keys():
100
+ if not source_name.startswith(prefix) or ".mlp." in source_name:
101
+ continue
102
+ target_name = "model." + source_name[len(prefix):]
103
+ if target_name in target_keys:
104
+ common[target_name] = handle.get_tensor(source_name)
105
+ total_bytes += save_shard(args.output, "model-common.safetensors", common, output_map)
106
+
107
+ generator = torch.Generator(device="cpu").manual_seed(SEED)
108
+ dense_width = 6144
109
+ hidden = config.hidden_size
110
+ init_std = float(config.initializer_range)
111
+ for layer in range(config.num_hidden_layers):
112
+ src = f"model.language_model.layers.{layer}.mlp"
113
+ dst = f"model.layers.{layer}.mlp"
114
+ dense_gate = source_tensor(args.source, source_map, f"{src}.gate_proj.weight")
115
+ dense_up = source_tensor(args.source, source_map, f"{src}.up_proj.weight")
116
+ dense_down = source_tensor(args.source, source_map, f"{src}.down_proj.weight")
117
+ dtype = dense_gate.dtype
118
+
119
+ shared_gate = torch.randn(SHARED_WIDTH, hidden, generator=generator, dtype=torch.float32)
120
+ shared_up = torch.randn(SHARED_WIDTH, hidden, generator=generator, dtype=torch.float32)
121
+ shared_down = torch.zeros(hidden, SHARED_WIDTH, dtype=dtype)
122
+ shared_gate.mul_(init_std).to(dtype=dtype)
123
+ shared_up.mul_(init_std).to(dtype=dtype)
124
+ shared_gate = shared_gate.to(dtype)
125
+ shared_up = shared_up.to(dtype)
126
+ shared_gate[:dense_width] = dense_gate
127
+ shared_up[:dense_width] = dense_up
128
+ # sigmoid(shared_expert_gate=0) gives the shared path a 0.5 multiplier.
129
+ shared_down[:, :dense_width] = dense_down
130
+
131
+ routed_gate_up = torch.empty(
132
+ NUM_EXPERTS, 2 * ROUTED_WIDTH, hidden, dtype=dtype
133
+ )
134
+ routed_down = torch.zeros(NUM_EXPERTS, hidden, ROUTED_WIDTH, dtype=dtype)
135
+ for expert in range(NUM_EXPERTS):
136
+ extra_gate = torch.randn(
137
+ ROUTED_WIDTH, hidden, generator=generator, dtype=torch.float32
138
+ ).mul_(init_std).to(dtype)
139
+ extra_up = torch.randn(
140
+ ROUTED_WIDTH, hidden, generator=generator, dtype=torch.float32
141
+ ).mul_(init_std).to(dtype)
142
+ extra_gate[:dense_width] = dense_gate
143
+ extra_up[:dense_width] = dense_up
144
+ routed_gate_up[expert, :ROUTED_WIDTH] = extra_gate
145
+ routed_gate_up[expert, ROUTED_WIDTH:] = extra_up
146
+ # Routed path supplies the other half of the original dense output.
147
+ routed_down[expert, :, :dense_width] = dense_down * 0.5
148
+
149
+ tensors = {
150
+ f"{dst}.gate.weight": torch.zeros(NUM_EXPERTS, hidden, dtype=dtype),
151
+ f"{dst}.experts.gate_up_proj": routed_gate_up,
152
+ f"{dst}.experts.down_proj": routed_down,
153
+ f"{dst}.shared_expert.gate_proj.weight": shared_gate,
154
+ f"{dst}.shared_expert.up_proj.weight": shared_up,
155
+ f"{dst}.shared_expert.down_proj.weight": shared_down,
156
+ f"{dst}.shared_expert_gate.weight": torch.zeros(1, hidden, dtype=dtype),
157
+ }
158
+ total_bytes += save_shard(
159
+ args.output, f"model-layer-{layer:02d}.safetensors", tensors, output_map
160
+ )
161
+ print(f"converted layer {layer + 1}/{config.num_hidden_layers}", flush=True)
162
+
163
+ missing = sorted(target_keys - set(output_map) - {"lm_head.weight"})
164
+ unexpected = sorted(set(output_map) - target_keys)
165
+ if missing or unexpected:
166
+ raise RuntimeError(f"key audit failed: missing={missing[:20]} unexpected={unexpected[:20]}")
167
+
168
+ (args.output / "model.safetensors.index.json").write_text(json.dumps({
169
+ "metadata": {"total_size": total_bytes},
170
+ "weight_map": dict(sorted(output_map.items())),
171
+ }, indent=2))
172
+ (args.output / "config.json").write_text(config.to_json_string())
173
+ (args.output / "conversion_manifest.json").write_text(json.dumps({
174
+ "source": str(args.source),
175
+ "initial_function": "Qwen3.5-2B text model (dense MLP split 50/50)",
176
+ "total_parameters": total_parameters,
177
+ "active_parameters": active_parameters,
178
+ "num_experts": NUM_EXPERTS,
179
+ "experts_per_token": EXPERTS_PER_TOKEN,
180
+ "shared_intermediate_size": SHARED_WIDTH,
181
+ "routed_intermediate_size": ROUTED_WIDTH,
182
+ "vision_included": False,
183
+ "seed": SEED,
184
+ }, indent=2))
185
+
186
+ for filename in ("chat_template.jinja", "merges.txt", "tokenizer.json",
187
+ "tokenizer_config.json", "vocab.json", "LICENSE"):
188
+ source_file = args.source / filename
189
+ if source_file.exists():
190
+ shutil.copy2(source_file, args.output / filename)
191
+
192
+ (args.output / "README.md").write_text(f"""---
193
+ license: apache-2.0
194
+ base_model:
195
+ - Qwen/Qwen3.5-2B
196
+ - Qwen/Qwen3.5-4B
197
+ library_name: transformers
198
+ pipeline_tag: text-generation
199
+ tags: [qwen3_5_moe, moe, upcycled, research]
200
+ ---
201
+
202
+ # Qwen3.5-4B-A3B-Student-v2
203
+
204
+ Text-only sparse-MoE research checkpoint initialized to preserve the text
205
+ generation function of Qwen3.5-2B. It is intended for distillation from
206
+ Qwen3.5-4B and is not yet claimed to match the 4B teacher.
207
+
208
+ | Property | Value |
209
+ |---|---:|
210
+ | Total parameters | {total_parameters:,} |
211
+ | Active parameters/token | {active_parameters:,} |
212
+ | Experts / selected | {NUM_EXPERTS} / {EXPERTS_PER_TOKEN} |
213
+ | Shared / routed width | {SHARED_WIDTH} / {ROUTED_WIDTH} |
214
+ | Vision | No |
215
+
216
+ The shared path and selected routed path initially contribute half of the dense
217
+ Qwen3.5-2B MLP each. Extra neurons are output-neutral at initialization but can
218
+ learn during distillation. See `conversion_manifest.json` for exact metadata.
219
+ """)
220
+ print(json.dumps({
221
+ "total_parameters": total_parameters,
222
+ "active_parameters": active_parameters,
223
+ "total_bytes": total_bytes,
224
+ }, indent=2))
225
+
226
+
227
+ if __name__ == "__main__":
228
+ main()
research_code/evaluate_chat_gate.py ADDED
@@ -0,0 +1,121 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import re
7
+ import time
8
+ from pathlib import Path
9
+
10
+ import torch
11
+ from transformers import AutoConfig, AutoTokenizer
12
+ from transformers import Qwen3_5ForConditionalGeneration
13
+ from transformers import Qwen3_5MoeForCausalLM
14
+ from transformers import Qwen3_5MoeForConditionalGeneration
15
+
16
+
17
+ def load_model(path: Path, device: str):
18
+ config = AutoConfig.from_pretrained(path)
19
+ if config.model_type == "qwen3_5_moe_text":
20
+ cls = Qwen3_5MoeForCausalLM
21
+ elif config.model_type == "qwen3_5_moe":
22
+ cls = Qwen3_5MoeForConditionalGeneration
23
+ else:
24
+ cls = Qwen3_5ForConditionalGeneration
25
+ return cls.from_pretrained(path, dtype=torch.bfloat16).to(device).eval()
26
+
27
+
28
+ def normalize(value: str) -> str:
29
+ subscripts = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
30
+ return " ".join(value.casefold().translate(subscripts).split())
31
+
32
+
33
+ def score(item: dict, completion: str) -> tuple[bool, list[str]]:
34
+ text = completion.strip()
35
+ failures = []
36
+ if len(text) < 2:
37
+ failures.append("empty_or_too_short")
38
+ if text and sum(character.isspace() for character in text) / len(text) > 0.5:
39
+ failures.append("whitespace_dominated")
40
+ if re.search(r"(.)\1{7,}", text, flags=re.DOTALL):
41
+ failures.append("character_repetition")
42
+ words = re.findall(r"\w+", text.casefold())
43
+ if len(words) >= 12 and len(set(words)) < 4:
44
+ failures.append("word_repetition")
45
+ expected = item.get("expected_any")
46
+ if expected and not any(normalize(value) in normalize(text) for value in expected):
47
+ failures.append("expected_answer_missing")
48
+ return not failures, failures
49
+
50
+
51
+ def main() -> None:
52
+ parser = argparse.ArgumentParser()
53
+ parser.add_argument("--model", type=Path, required=True)
54
+ parser.add_argument("--prompts", type=Path, required=True)
55
+ parser.add_argument("--output", type=Path, required=True)
56
+ parser.add_argument("--device", default="mps")
57
+ parser.add_argument("--max-new-tokens", type=int, default=48)
58
+ args = parser.parse_args()
59
+
60
+ items = [json.loads(line) for line in args.prompts.read_text().splitlines() if line]
61
+ tokenizer = AutoTokenizer.from_pretrained(args.model)
62
+ model = load_model(args.model, args.device)
63
+ results = []
64
+ for index, item in enumerate(items, 1):
65
+ batch = tokenizer.apply_chat_template(
66
+ [{"role": "user", "content": item["prompt"]}],
67
+ tokenize=True,
68
+ add_generation_prompt=True,
69
+ enable_thinking=False,
70
+ return_tensors="pt",
71
+ return_dict=True,
72
+ ).to(args.device)
73
+ started = time.perf_counter()
74
+ with torch.no_grad():
75
+ output = model.generate(
76
+ **batch,
77
+ max_new_tokens=args.max_new_tokens,
78
+ do_sample=False,
79
+ use_cache=True,
80
+ )
81
+ elapsed = time.perf_counter() - started
82
+ ids = output[0, batch["input_ids"].shape[1]:]
83
+ completion = tokenizer.decode(ids, skip_special_tokens=True)
84
+ passed, failures = score(item, completion)
85
+ result = {
86
+ **item,
87
+ "completion": completion,
88
+ "passed": passed,
89
+ "failures": failures,
90
+ "new_tokens": int(ids.numel()),
91
+ "elapsed_seconds": elapsed,
92
+ }
93
+ results.append(result)
94
+ print(f"[{index:02d}/{len(items)}] {item['id']} {'PASS' if passed else 'FAIL'}", flush=True)
95
+
96
+ by_language = {}
97
+ for language in sorted({item["language"] for item in items}):
98
+ subset = [row for row in results if row["language"] == language]
99
+ by_language[language] = {
100
+ "passed": sum(row["passed"] for row in subset),
101
+ "total": len(subset),
102
+ "pass_rate": sum(row["passed"] for row in subset) / len(subset),
103
+ }
104
+ passed = sum(row["passed"] for row in results)
105
+ report = {
106
+ "model": str(args.model),
107
+ "total": len(results),
108
+ "passed": passed,
109
+ "pass_rate": passed / len(results),
110
+ "gate_threshold": 0.90,
111
+ "gate_passed": passed / len(results) >= 0.90,
112
+ "by_language": by_language,
113
+ "results": results,
114
+ }
115
+ args.output.parent.mkdir(parents=True, exist_ok=True)
116
+ args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
117
+ print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
118
+
119
+
120
+ if __name__ == "__main__":
121
+ main()
research_code/rescore_chat_gate.py ADDED
@@ -0,0 +1,35 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import sys
7
+ from pathlib import Path
8
+
9
+ sys.path.insert(0, str(Path(__file__).resolve().parent))
10
+ from evaluate_chat_gate import score
11
+
12
+
13
+ def main() -> None:
14
+ parser = argparse.ArgumentParser()
15
+ parser.add_argument("--input", type=Path, required=True)
16
+ parser.add_argument("--output", type=Path, required=True)
17
+ args = parser.parse_args()
18
+
19
+ report = json.loads(args.input.read_text())
20
+ for row in report["results"]:
21
+ row["passed"], row["failures"] = score(row, row["completion"])
22
+ for language, summary in report["by_language"].items():
23
+ rows = [row for row in report["results"] if row["language"] == language]
24
+ summary["passed"] = sum(row["passed"] for row in rows)
25
+ summary["pass_rate"] = summary["passed"] / summary["total"]
26
+ report["passed"] = sum(row["passed"] for row in report["results"])
27
+ report["pass_rate"] = report["passed"] / report["total"]
28
+ report["gate_passed"] = report["pass_rate"] >= report["gate_threshold"]
29
+ report["rescored_from"] = str(args.input)
30
+ args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
31
+ print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
32
+
33
+
34
+ if __name__ == "__main__":
35
+ main()
research_code/service/README.md ADDED
@@ -0,0 +1,20 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ # Local OpenAI-compatible service
2
+
3
+ Run with the service directory as the working directory:
4
+
5
+ ```bash
6
+ cd research/qwen35-moe-a3b/service
7
+ PYTORCH_ENABLE_MPS_FALLBACK=1 \
8
+ MODEL_PATH=../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2 \
9
+ python3 -m uvicorn app:app --host 127.0.0.1 --port 8088
10
+ ```
11
+
12
+ Implemented endpoints:
13
+
14
+ - `GET /health`
15
+ - `GET /v1/models`
16
+ - `POST /v1/chat/completions` (non-streaming text requests)
17
+
18
+ This is a single-process research service. It serializes generation calls to
19
+ protect the shared MPS model. Authentication, TLS, streaming, tool calls, and
20
+ multi-worker deployment are intentionally out of scope for this local gate.
research_code/service/app.py ADDED
@@ -0,0 +1,167 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ import threading
5
+ import time
6
+ import uuid
7
+ from contextlib import asynccontextmanager
8
+ from pathlib import Path
9
+ from typing import Literal
10
+
11
+ import torch
12
+ from fastapi import FastAPI, HTTPException
13
+ from pydantic import BaseModel, Field
14
+ from transformers import AutoTokenizer, Qwen3_5MoeForCausalLM
15
+
16
+
17
+ MODEL_ID = "qwen3.5-4b-a3b-student-v2"
18
+ DEFAULT_MODEL_PATH = Path(__file__).resolve().parents[3] / "models/Qwen/Qwen3.5-4B-A3B-Student-v2"
19
+ state: dict = {"latencies": [], "requests": 0}
20
+ generation_lock = threading.Lock()
21
+
22
+
23
+ class Message(BaseModel):
24
+ role: Literal["developer", "system", "user", "assistant"]
25
+ content: str
26
+
27
+
28
+ class ChatCompletionRequest(BaseModel):
29
+ model: str
30
+ messages: list[Message] = Field(min_length=1)
31
+ max_tokens: int | None = Field(default=None, ge=1, le=2048)
32
+ max_completion_tokens: int | None = Field(default=None, ge=1, le=2048)
33
+ temperature: float = Field(default=0.0, ge=0.0, le=2.0)
34
+ top_p: float = Field(default=1.0, gt=0.0, le=1.0)
35
+ stream: bool = False
36
+ stop: str | list[str] | None = None
37
+
38
+
39
+ def memory_metrics() -> dict[str, float | None]:
40
+ try:
41
+ import psutil
42
+ rss_mb = psutil.Process().memory_info().rss / 1024**2
43
+ except Exception:
44
+ rss_mb = None
45
+ mps_mb = torch.mps.current_allocated_memory() / 1024**2 if torch.backends.mps.is_available() else None
46
+ return {"rss_mb": rss_mb, "mps_allocated_mb": mps_mb}
47
+
48
+
49
+ @asynccontextmanager
50
+ async def lifespan(app: FastAPI):
51
+ model_path = Path(os.environ.get("MODEL_PATH", DEFAULT_MODEL_PATH))
52
+ device = os.environ.get("MODEL_DEVICE", "mps" if torch.backends.mps.is_available() else "cpu")
53
+ tokenizer = AutoTokenizer.from_pretrained(model_path)
54
+ model = Qwen3_5MoeForCausalLM.from_pretrained(model_path, dtype=torch.bfloat16).to(device).eval()
55
+ state.update({
56
+ "model": model,
57
+ "tokenizer": tokenizer,
58
+ "model_path": str(model_path),
59
+ "device": device,
60
+ "parameters": sum(parameter.numel() for parameter in model.parameters()),
61
+ "started_at": int(time.time()),
62
+ })
63
+ yield
64
+ state.pop("model", None)
65
+ state.pop("tokenizer", None)
66
+ if device == "mps":
67
+ torch.mps.empty_cache()
68
+
69
+
70
+ app = FastAPI(title="Qwen3.5 A3B Local API", version="0.1.0", lifespan=lifespan)
71
+
72
+
73
+ @app.get("/health")
74
+ def health():
75
+ latencies = state["latencies"]
76
+ ordered = sorted(latencies)
77
+ p95 = ordered[max(0, int(len(ordered) * 0.95) - 1)] if ordered else None
78
+ return {
79
+ "status": "ok" if "model" in state else "starting",
80
+ "model": MODEL_ID,
81
+ "model_path": state.get("model_path"),
82
+ "device": state.get("device"),
83
+ "parameters": state.get("parameters"),
84
+ "requests": state["requests"],
85
+ "latency_mean_seconds": sum(latencies) / len(latencies) if latencies else None,
86
+ "latency_p95_seconds": p95,
87
+ **memory_metrics(),
88
+ }
89
+
90
+
91
+ @app.get("/v1/models")
92
+ def list_models():
93
+ return {
94
+ "object": "list",
95
+ "data": [{
96
+ "id": MODEL_ID,
97
+ "object": "model",
98
+ "created": state.get("started_at", int(time.time())),
99
+ "owned_by": "local",
100
+ }],
101
+ }
102
+
103
+
104
+ @app.post("/v1/chat/completions")
105
+ def chat_completions(request: ChatCompletionRequest):
106
+ if request.stream:
107
+ raise HTTPException(status_code=400, detail="stream=true is not implemented in this research server")
108
+ if request.model not in {MODEL_ID, "local", state.get("model_path")}:
109
+ raise HTTPException(status_code=404, detail=f"unknown model: {request.model}")
110
+
111
+ messages = [message.model_dump() for message in request.messages]
112
+ for message in messages:
113
+ if message["role"] == "developer":
114
+ message["role"] = "system"
115
+ tokenizer = state["tokenizer"]
116
+ model = state["model"]
117
+ device = state["device"]
118
+ batch = tokenizer.apply_chat_template(
119
+ messages,
120
+ tokenize=True,
121
+ add_generation_prompt=True,
122
+ enable_thinking=False,
123
+ return_tensors="pt",
124
+ return_dict=True,
125
+ ).to(device)
126
+ max_new_tokens = request.max_completion_tokens or request.max_tokens or 128
127
+ generation_kwargs = {
128
+ "max_new_tokens": max_new_tokens,
129
+ "do_sample": request.temperature > 0,
130
+ "use_cache": True,
131
+ }
132
+ if request.temperature > 0:
133
+ generation_kwargs.update(temperature=request.temperature, top_p=request.top_p)
134
+ started = time.perf_counter()
135
+ with generation_lock, torch.no_grad():
136
+ output = model.generate(
137
+ **batch,
138
+ **generation_kwargs,
139
+ )
140
+ elapsed = time.perf_counter() - started
141
+ completion_ids = output[0, batch["input_ids"].shape[1]:]
142
+ content = tokenizer.decode(completion_ids, skip_special_tokens=True)
143
+ stops = [request.stop] if isinstance(request.stop, str) else request.stop or []
144
+ for stop in stops:
145
+ if stop in content:
146
+ content = content.split(stop, 1)[0]
147
+ state["latencies"].append(elapsed)
148
+ state["requests"] += 1
149
+ prompt_tokens = int(batch["attention_mask"].sum())
150
+ completion_tokens = int(completion_ids.numel())
151
+ return {
152
+ "id": f"chatcmpl-{uuid.uuid4().hex}",
153
+ "object": "chat.completion",
154
+ "created": int(time.time()),
155
+ "model": MODEL_ID,
156
+ "choices": [{
157
+ "index": 0,
158
+ "message": {"role": "assistant", "content": content, "refusal": None},
159
+ "finish_reason": "stop" if completion_tokens < max_new_tokens else "length",
160
+ "logprobs": None,
161
+ }],
162
+ "usage": {
163
+ "prompt_tokens": prompt_tokens,
164
+ "completion_tokens": completion_tokens,
165
+ "total_tokens": prompt_tokens + completion_tokens,
166
+ },
167
+ }
research_code/service/test_openai_service.py ADDED
@@ -0,0 +1,69 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import statistics
7
+ import time
8
+ from pathlib import Path
9
+
10
+ import httpx
11
+
12
+
13
+ PROMPTS = [
14
+ "대한민국의 수도는 어디인가요? 짧게 답하세요.",
15
+ "What is the capital of France? Answer briefly.",
16
+ "7 곱하기 8은 얼마인가요?",
17
+ "Write one friendly greeting.",
18
+ ]
19
+
20
+
21
+ def main() -> None:
22
+ parser = argparse.ArgumentParser()
23
+ parser.add_argument("--base-url", default="http://127.0.0.1:8088")
24
+ parser.add_argument("--requests", type=int, default=20)
25
+ parser.add_argument("--output", type=Path, required=True)
26
+ args = parser.parse_args()
27
+
28
+ rows = []
29
+ with httpx.Client(timeout=120) as client:
30
+ health_before = client.get(f"{args.base_url}/health").raise_for_status().json()
31
+ models = client.get(f"{args.base_url}/v1/models").raise_for_status().json()
32
+ for index in range(args.requests):
33
+ started = time.perf_counter()
34
+ response = client.post(f"{args.base_url}/v1/chat/completions", json={
35
+ "model": "qwen3.5-4b-a3b-student-v2",
36
+ "messages": [{"role": "user", "content": PROMPTS[index % len(PROMPTS)]}],
37
+ "max_completion_tokens": 16,
38
+ "temperature": 0,
39
+ })
40
+ elapsed = time.perf_counter() - started
41
+ response.raise_for_status()
42
+ body = response.json()
43
+ content = body["choices"][0]["message"]["content"].strip()
44
+ passed = body["object"] == "chat.completion" and bool(content)
45
+ rows.append({"index": index + 1, "passed": passed, "elapsed_seconds": elapsed, "content": content})
46
+ print(f"[{index + 1:02d}/{args.requests}] {'PASS' if passed else 'FAIL'} {elapsed:.3f}s", flush=True)
47
+ health_after = client.get(f"{args.base_url}/health").raise_for_status().json()
48
+
49
+ latencies = [row["elapsed_seconds"] for row in rows]
50
+ ordered = sorted(latencies)
51
+ report = {
52
+ "requests": args.requests,
53
+ "passed": sum(row["passed"] for row in rows),
54
+ "all_passed": all(row["passed"] for row in rows),
55
+ "latency_mean_seconds": statistics.mean(latencies),
56
+ "latency_p50_seconds": statistics.median(latencies),
57
+ "latency_p95_seconds": ordered[max(0, int(len(ordered) * 0.95) - 1)],
58
+ "health_before": health_before,
59
+ "health_after": health_after,
60
+ "models": models,
61
+ "results": rows,
62
+ }
63
+ args.output.parent.mkdir(parents=True, exist_ok=True)
64
+ args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
65
+ print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
66
+
67
+
68
+ if __name__ == "__main__":
69
+ main()
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42
3
+ size 12807982
tokenizer_config.json ADDED
@@ -0,0 +1,305 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "added_tokens_decoder": {
4
+ "248044": {
5
+ "content": "<|endoftext|>",
6
+ "lstrip": false,
7
+ "normalized": false,
8
+ "rstrip": false,
9
+ "single_word": false,
10
+ "special": true
11
+ },
12
+ "248045": {
13
+ "content": "<|im_start|>",
14
+ "lstrip": false,
15
+ "normalized": false,
16
+ "rstrip": false,
17
+ "single_word": false,
18
+ "special": true
19
+ },
20
+ "248046": {
21
+ "content": "<|im_end|>",
22
+ "lstrip": false,
23
+ "normalized": false,
24
+ "rstrip": false,
25
+ "single_word": false,
26
+ "special": true
27
+ },
28
+ "248047": {
29
+ "content": "<|object_ref_start|>",
30
+ "lstrip": false,
31
+ "normalized": false,
32
+ "rstrip": false,
33
+ "single_word": false,
34
+ "special": true
35
+ },
36
+ "248048": {
37
+ "content": "<|object_ref_end|>",
38
+ "lstrip": false,
39
+ "normalized": false,
40
+ "rstrip": false,
41
+ "single_word": false,
42
+ "special": true
43
+ },
44
+ "248049": {
45
+ "content": "<|box_start|>",
46
+ "lstrip": false,
47
+ "normalized": false,
48
+ "rstrip": false,
49
+ "single_word": false,
50
+ "special": true
51
+ },
52
+ "248050": {
53
+ "content": "<|box_end|>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": false,
57
+ "single_word": false,
58
+ "special": true
59
+ },
60
+ "248051": {
61
+ "content": "<|quad_start|>",
62
+ "lstrip": false,
63
+ "normalized": false,
64
+ "rstrip": false,
65
+ "single_word": false,
66
+ "special": true
67
+ },
68
+ "248052": {
69
+ "content": "<|quad_end|>",
70
+ "lstrip": false,
71
+ "normalized": false,
72
+ "rstrip": false,
73
+ "single_word": false,
74
+ "special": true
75
+ },
76
+ "248053": {
77
+ "content": "<|vision_start|>",
78
+ "lstrip": false,
79
+ "normalized": false,
80
+ "rstrip": false,
81
+ "single_word": false,
82
+ "special": true
83
+ },
84
+ "248054": {
85
+ "content": "<|vision_end|>",
86
+ "lstrip": false,
87
+ "normalized": false,
88
+ "rstrip": false,
89
+ "single_word": false,
90
+ "special": true
91
+ },
92
+ "248055": {
93
+ "content": "<|vision_pad|>",
94
+ "lstrip": false,
95
+ "normalized": false,
96
+ "rstrip": false,
97
+ "single_word": false,
98
+ "special": true
99
+ },
100
+ "248056": {
101
+ "content": "<|image_pad|>",
102
+ "lstrip": false,
103
+ "normalized": false,
104
+ "rstrip": false,
105
+ "single_word": false,
106
+ "special": true
107
+ },
108
+ "248057": {
109
+ "content": "<|video_pad|>",
110
+ "lstrip": false,
111
+ "normalized": false,
112
+ "rstrip": false,
113
+ "single_word": false,
114
+ "special": true
115
+ },
116
+ "248058": {
117
+ "content": "<tool_call>",
118
+ "lstrip": false,
119
+ "normalized": false,
120
+ "rstrip": false,
121
+ "single_word": false,
122
+ "special": false
123
+ },
124
+ "248059": {
125
+ "content": "</tool_call>",
126
+ "lstrip": false,
127
+ "normalized": false,
128
+ "rstrip": false,
129
+ "single_word": false,
130
+ "special": false
131
+ },
132
+ "248060": {
133
+ "content": "<|fim_prefix|>",
134
+ "lstrip": false,
135
+ "normalized": false,
136
+ "rstrip": false,
137
+ "single_word": false,
138
+ "special": false
139
+ },
140
+ "248061": {
141
+ "content": "<|fim_middle|>",
142
+ "lstrip": false,
143
+ "normalized": false,
144
+ "rstrip": false,
145
+ "single_word": false,
146
+ "special": false
147
+ },
148
+ "248062": {
149
+ "content": "<|fim_suffix|>",
150
+ "lstrip": false,
151
+ "normalized": false,
152
+ "rstrip": false,
153
+ "single_word": false,
154
+ "special": false
155
+ },
156
+ "248063": {
157
+ "content": "<|fim_pad|>",
158
+ "lstrip": false,
159
+ "normalized": false,
160
+ "rstrip": false,
161
+ "single_word": false,
162
+ "special": false
163
+ },
164
+ "248064": {
165
+ "content": "<|repo_name|>",
166
+ "lstrip": false,
167
+ "normalized": false,
168
+ "rstrip": false,
169
+ "single_word": false,
170
+ "special": false
171
+ },
172
+ "248065": {
173
+ "content": "<|file_sep|>",
174
+ "lstrip": false,
175
+ "normalized": false,
176
+ "rstrip": false,
177
+ "single_word": false,
178
+ "special": false
179
+ },
180
+ "248066": {
181
+ "content": "<tool_response>",
182
+ "lstrip": false,
183
+ "normalized": false,
184
+ "rstrip": false,
185
+ "single_word": false,
186
+ "special": false
187
+ },
188
+ "248067": {
189
+ "content": "</tool_response>",
190
+ "lstrip": false,
191
+ "normalized": false,
192
+ "rstrip": false,
193
+ "single_word": false,
194
+ "special": false
195
+ },
196
+ "248068": {
197
+ "content": "<think>",
198
+ "lstrip": false,
199
+ "normalized": false,
200
+ "rstrip": false,
201
+ "single_word": false,
202
+ "special": false
203
+ },
204
+ "248069": {
205
+ "content": "</think>",
206
+ "lstrip": false,
207
+ "normalized": false,
208
+ "rstrip": false,
209
+ "single_word": false,
210
+ "special": false
211
+ },
212
+ "248070": {
213
+ "content": "<|audio_start|>",
214
+ "lstrip": false,
215
+ "normalized": false,
216
+ "rstrip": false,
217
+ "single_word": false,
218
+ "special": true
219
+ },
220
+ "248071": {
221
+ "content": "<|audio_end|>",
222
+ "lstrip": false,
223
+ "normalized": false,
224
+ "rstrip": false,
225
+ "single_word": false,
226
+ "special": true
227
+ },
228
+ "248072": {
229
+ "content": "<tts_pad>",
230
+ "lstrip": false,
231
+ "normalized": false,
232
+ "rstrip": false,
233
+ "single_word": false,
234
+ "special": true
235
+ },
236
+ "248073": {
237
+ "content": "<tts_text_bos>",
238
+ "lstrip": false,
239
+ "normalized": false,
240
+ "rstrip": false,
241
+ "single_word": false,
242
+ "special": true
243
+ },
244
+ "248074": {
245
+ "content": "<tts_text_eod>",
246
+ "lstrip": false,
247
+ "normalized": false,
248
+ "rstrip": false,
249
+ "single_word": false,
250
+ "special": true
251
+ },
252
+ "248075": {
253
+ "content": "<tts_text_bos_single>",
254
+ "lstrip": false,
255
+ "normalized": false,
256
+ "rstrip": false,
257
+ "single_word": false,
258
+ "special": true
259
+ },
260
+ "248076": {
261
+ "content": "<|audio_pad|>",
262
+ "lstrip": false,
263
+ "normalized": false,
264
+ "rstrip": false,
265
+ "single_word": false,
266
+ "special": true
267
+ }
268
+ },
269
+ "additional_special_tokens": [
270
+ "<|im_start|>",
271
+ "<|im_end|>",
272
+ "<|object_ref_start|>",
273
+ "<|object_ref_end|>",
274
+ "<|box_start|>",
275
+ "<|box_end|>",
276
+ "<|quad_start|>",
277
+ "<|quad_end|>",
278
+ "<|vision_start|>",
279
+ "<|vision_end|>",
280
+ "<|vision_pad|>",
281
+ "<|image_pad|>",
282
+ "<|video_pad|>"
283
+ ],
284
+ "bos_token": null,
285
+ "chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {{- '<|im_start|>system\\n' + content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is true %}\n {{- '<think>\\n' }}\n {%- else %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- endif %}\n{%- endif %}",
286
+ "clean_up_tokenization_spaces": false,
287
+ "eos_token": "<|im_end|>",
288
+ "errors": "replace",
289
+ "model_max_length": 262144,
290
+ "pad_token": "<|endoftext|>",
291
+ "split_special_tokens": false,
292
+ "tokenizer_class": "Qwen2Tokenizer",
293
+ "unk_token": null,
294
+ "add_bos_token": false,
295
+ "pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
296
+ "extra_special_tokens": {
297
+ "audio_bos_token": "<|audio_start|>",
298
+ "audio_eos_token": "<|audio_end|>",
299
+ "audio_token": "<|audio_pad|>",
300
+ "image_token": "<|image_pad|>",
301
+ "video_token": "<|video_pad|>",
302
+ "vision_bos_token": "<|vision_start|>",
303
+ "vision_eos_token": "<|vision_end|>"
304
+ }
305
+ }
vocab.json ADDED
The diff for this file is too large to render. See raw diff