Text Generation
Transformers
Safetensors
qwen3_5_moe_text
qwen3_5_moe
Mixture of Experts
upcycled
research
conversational
Instructions to use sepsy070716/Qwen3.5-4B-A3B-Student-v2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use sepsy070716/Qwen3.5-4B-A3B-Student-v2 with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("text-generation", model="sepsy070716/Qwen3.5-4B-A3B-Student-v2") messages = [ {"role": "user", "content": "Who are you?"}, ] pipe(messages)# Load model directly from transformers import AutoTokenizer, AutoModelForCausalLM tokenizer = AutoTokenizer.from_pretrained("sepsy070716/Qwen3.5-4B-A3B-Student-v2") model = AutoModelForCausalLM.from_pretrained("sepsy070716/Qwen3.5-4B-A3B-Student-v2", device_map="auto") messages = [ {"role": "user", "content": "Who are you?"}, ] inputs = tokenizer.apply_chat_template( messages, add_generation_prompt=True, tokenize=True, return_dict=True, return_tensors="pt", ).to(model.device) outputs = model.generate(**inputs, max_new_tokens=40) print(tokenizer.decode(outputs[0][inputs["input_ids"].shape[-1]:])) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use sepsy070716/Qwen3.5-4B-A3B-Student-v2 with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "sepsy070716/Qwen3.5-4B-A3B-Student-v2" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sepsy070716/Qwen3.5-4B-A3B-Student-v2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker
docker model run hf.co/sepsy070716/Qwen3.5-4B-A3B-Student-v2
- SGLang
How to use sepsy070716/Qwen3.5-4B-A3B-Student-v2 with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "sepsy070716/Qwen3.5-4B-A3B-Student-v2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sepsy070716/Qwen3.5-4B-A3B-Student-v2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "sepsy070716/Qwen3.5-4B-A3B-Student-v2" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "sepsy070716/Qwen3.5-4B-A3B-Student-v2", "messages": [ { "role": "user", "content": "What is the capital of France?" } ] }' - Docker Model Runner
How to use sepsy070716/Qwen3.5-4B-A3B-Student-v2 with Docker Model Runner:
docker model run hf.co/sepsy070716/Qwen3.5-4B-A3B-Student-v2
Release Qwen3.5-4B-A3B Student v2 with evaluation evidence
Browse files- .gitattributes +1 -0
- LICENSE +202 -0
- README.md +155 -0
- chat_template.jinja +154 -0
- config.json +80 -0
- conversion_manifest.json +12 -0
- evaluation/chat_gate_60.json +762 -0
- evaluation/lm_eval_teacher4b_dev100.json +208 -0
- evaluation/lm_eval_v2_dev100.json +208 -0
- evaluation/multilingual_lm_loss.json +115 -0
- evaluation/openai_service_20.json +165 -0
- merges.txt +0 -0
- model-common.safetensors +3 -0
- model-layer-00.safetensors +3 -0
- model-layer-01.safetensors +3 -0
- model-layer-02.safetensors +3 -0
- model-layer-03.safetensors +3 -0
- model-layer-04.safetensors +3 -0
- model-layer-05.safetensors +3 -0
- model-layer-06.safetensors +3 -0
- model-layer-07.safetensors +3 -0
- model-layer-08.safetensors +3 -0
- model-layer-09.safetensors +3 -0
- model-layer-10.safetensors +3 -0
- model-layer-11.safetensors +3 -0
- model-layer-12.safetensors +3 -0
- model-layer-13.safetensors +3 -0
- model-layer-14.safetensors +3 -0
- model-layer-15.safetensors +3 -0
- model-layer-16.safetensors +3 -0
- model-layer-17.safetensors +3 -0
- model-layer-18.safetensors +3 -0
- model-layer-19.safetensors +3 -0
- model-layer-20.safetensors +3 -0
- model-layer-21.safetensors +3 -0
- model-layer-22.safetensors +3 -0
- model-layer-23.safetensors +3 -0
- model.safetensors.index.json +423 -0
- research_code/chat-gate-60.jsonl +60 -0
- research_code/compare_lm_loss.py +116 -0
- research_code/convert_2b_to_4b_a3b.py +228 -0
- research_code/evaluate_chat_gate.py +121 -0
- research_code/rescore_chat_gate.py +35 -0
- research_code/service/README.md +20 -0
- research_code/service/app.py +167 -0
- research_code/service/test_openai_service.py +69 -0
- tokenizer.json +3 -0
- tokenizer_config.json +305 -0
- vocab.json +0 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,202 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
|
| 2 |
+
Apache License
|
| 3 |
+
Version 2.0, January 2004
|
| 4 |
+
http://www.apache.org/licenses/
|
| 5 |
+
|
| 6 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 7 |
+
|
| 8 |
+
1. Definitions.
|
| 9 |
+
|
| 10 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 11 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 12 |
+
|
| 13 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 14 |
+
the copyright owner that is granting the License.
|
| 15 |
+
|
| 16 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 17 |
+
other entities that control, are controlled by, or are under common
|
| 18 |
+
control with that entity. For the purposes of this definition,
|
| 19 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 20 |
+
direction or management of such entity, whether by contract or
|
| 21 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 22 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 23 |
+
|
| 24 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 25 |
+
exercising permissions granted by this License.
|
| 26 |
+
|
| 27 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 28 |
+
including but not limited to software source code, documentation
|
| 29 |
+
source, and configuration files.
|
| 30 |
+
|
| 31 |
+
"Object" form shall mean any form resulting from mechanical
|
| 32 |
+
transformation or translation of a Source form, including but
|
| 33 |
+
not limited to compiled object code, generated documentation,
|
| 34 |
+
and conversions to other media types.
|
| 35 |
+
|
| 36 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 37 |
+
Object form, made available under the License, as indicated by a
|
| 38 |
+
copyright notice that is included in or attached to the work
|
| 39 |
+
(an example is provided in the Appendix below).
|
| 40 |
+
|
| 41 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 42 |
+
form, that is based on (or derived from) the Work and for which the
|
| 43 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 44 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 45 |
+
of this License, Derivative Works shall not include works that remain
|
| 46 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 47 |
+
the Work and Derivative Works thereof.
|
| 48 |
+
|
| 49 |
+
"Contribution" shall mean any work of authorship, including
|
| 50 |
+
the original version of the Work and any modifications or additions
|
| 51 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 52 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 53 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 54 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 55 |
+
means any form of electronic, verbal, or written communication sent
|
| 56 |
+
to the Licensor or its representatives, including but not limited to
|
| 57 |
+
communication on electronic mailing lists, source code control systems,
|
| 58 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 59 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 60 |
+
excluding communication that is conspicuously marked or otherwise
|
| 61 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 62 |
+
|
| 63 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 64 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 65 |
+
subsequently incorporated within the Work.
|
| 66 |
+
|
| 67 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 68 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 69 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 70 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 71 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 72 |
+
Work and such Derivative Works in Source or Object form.
|
| 73 |
+
|
| 74 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 75 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 76 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 77 |
+
(except as stated in this section) patent license to make, have made,
|
| 78 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 79 |
+
where such license applies only to those patent claims licensable
|
| 80 |
+
by such Contributor that are necessarily infringed by their
|
| 81 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 82 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 83 |
+
institute patent litigation against any entity (including a
|
| 84 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 85 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 86 |
+
or contributory patent infringement, then any patent licenses
|
| 87 |
+
granted to You under this License for that Work shall terminate
|
| 88 |
+
as of the date such litigation is filed.
|
| 89 |
+
|
| 90 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 91 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 92 |
+
modifications, and in Source or Object form, provided that You
|
| 93 |
+
meet the following conditions:
|
| 94 |
+
|
| 95 |
+
(a) You must give any other recipients of the Work or
|
| 96 |
+
Derivative Works a copy of this License; and
|
| 97 |
+
|
| 98 |
+
(b) You must cause any modified files to carry prominent notices
|
| 99 |
+
stating that You changed the files; and
|
| 100 |
+
|
| 101 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 102 |
+
that You distribute, all copyright, patent, trademark, and
|
| 103 |
+
attribution notices from the Source form of the Work,
|
| 104 |
+
excluding those notices that do not pertain to any part of
|
| 105 |
+
the Derivative Works; and
|
| 106 |
+
|
| 107 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 108 |
+
distribution, then any Derivative Works that You distribute must
|
| 109 |
+
include a readable copy of the attribution notices contained
|
| 110 |
+
within such NOTICE file, excluding those notices that do not
|
| 111 |
+
pertain to any part of the Derivative Works, in at least one
|
| 112 |
+
of the following places: within a NOTICE text file distributed
|
| 113 |
+
as part of the Derivative Works; within the Source form or
|
| 114 |
+
documentation, if provided along with the Derivative Works; or,
|
| 115 |
+
within a display generated by the Derivative Works, if and
|
| 116 |
+
wherever such third-party notices normally appear. The contents
|
| 117 |
+
of the NOTICE file are for informational purposes only and
|
| 118 |
+
do not modify the License. You may add Your own attribution
|
| 119 |
+
notices within Derivative Works that You distribute, alongside
|
| 120 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 121 |
+
that such additional attribution notices cannot be construed
|
| 122 |
+
as modifying the License.
|
| 123 |
+
|
| 124 |
+
You may add Your own copyright statement to Your modifications and
|
| 125 |
+
may provide additional or different license terms and conditions
|
| 126 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 127 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 128 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 129 |
+
the conditions stated in this License.
|
| 130 |
+
|
| 131 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 132 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 133 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 134 |
+
this License, without any additional terms or conditions.
|
| 135 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 136 |
+
the terms of any separate license agreement you may have executed
|
| 137 |
+
with Licensor regarding such Contributions.
|
| 138 |
+
|
| 139 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 140 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 141 |
+
except as required for reasonable and customary use in describing the
|
| 142 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 143 |
+
|
| 144 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 145 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 146 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 147 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 148 |
+
implied, including, without limitation, any warranties or conditions
|
| 149 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 150 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 151 |
+
appropriateness of using or redistributing the Work and assume any
|
| 152 |
+
risks associated with Your exercise of permissions under this License.
|
| 153 |
+
|
| 154 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 155 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 156 |
+
unless required by applicable law (such as deliberate and grossly
|
| 157 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 158 |
+
liable to You for damages, including any direct, indirect, special,
|
| 159 |
+
incidental, or consequential damages of any character arising as a
|
| 160 |
+
result of this License or out of the use or inability to use the
|
| 161 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 162 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 163 |
+
other commercial damages or losses), even if such Contributor
|
| 164 |
+
has been advised of the possibility of such damages.
|
| 165 |
+
|
| 166 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 167 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 168 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 169 |
+
or other liability obligations and/or rights consistent with this
|
| 170 |
+
License. However, in accepting such obligations, You may act only
|
| 171 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 172 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 173 |
+
defend, and hold each Contributor harmless for any liability
|
| 174 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 175 |
+
of your accepting any such warranty or additional liability.
|
| 176 |
+
|
| 177 |
+
END OF TERMS AND CONDITIONS
|
| 178 |
+
|
| 179 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 180 |
+
|
| 181 |
+
To apply the Apache License to your work, attach the following
|
| 182 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 183 |
+
replaced with your own identifying information. (Don't include
|
| 184 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 185 |
+
comment syntax for the file format. We also recommend that a
|
| 186 |
+
file or class name and description of purpose be included on the
|
| 187 |
+
same "printed page" as the copyright notice for easier
|
| 188 |
+
identification within third-party archives.
|
| 189 |
+
|
| 190 |
+
Copyright 2026 Alibaba Cloud
|
| 191 |
+
|
| 192 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 193 |
+
you may not use this file except in compliance with the License.
|
| 194 |
+
You may obtain a copy of the License at
|
| 195 |
+
|
| 196 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 197 |
+
|
| 198 |
+
Unless required by applicable law or agreed to in writing, software
|
| 199 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 200 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 201 |
+
See the License for the specific language governing permissions and
|
| 202 |
+
limitations under the License.
|
README.md
ADDED
|
@@ -0,0 +1,155 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
base_model:
|
| 4 |
+
- Qwen/Qwen3.5-2B
|
| 5 |
+
- Qwen/Qwen3.5-4B
|
| 6 |
+
library_name: transformers
|
| 7 |
+
pipeline_tag: text-generation
|
| 8 |
+
tags:
|
| 9 |
+
- qwen3_5_moe
|
| 10 |
+
- moe
|
| 11 |
+
- upcycled
|
| 12 |
+
- text-generation
|
| 13 |
+
- research
|
| 14 |
+
language: [ko, en, zh, ja, es, de]
|
| 15 |
+
---
|
| 16 |
+
|
| 17 |
+
# Qwen3.5-4B-A3B-Student-v2
|
| 18 |
+
|
| 19 |
+
A text-only sparse-MoE release candidate built as a practical local alternative
|
| 20 |
+
to Qwen3.5-4B. It has 4.0B total parameters and 3.0B active parameters per
|
| 21 |
+
token. The initialization preserves Qwen3.5-2B behavior, while adding
|
| 22 |
+
output-neutral trainable capacity for later Qwen3.5-4B distillation.
|
| 23 |
+
|
| 24 |
+
This is an independently measured research release. It is not an official Qwen
|
| 25 |
+
model and does not include vision.
|
| 26 |
+
|
| 27 |
+
## Architecture
|
| 28 |
+
|
| 29 |
+
| Property | Value |
|
| 30 |
+
|---|---:|
|
| 31 |
+
| Total parameters | 3,995,901,760 |
|
| 32 |
+
| Active parameters/token | 2,995,560,256 |
|
| 33 |
+
| Transformer layers / hidden size | 24 / 2,048 |
|
| 34 |
+
| Experts / selected per token | 2 / 1 |
|
| 35 |
+
| Shared / routed intermediate width | 6,912 / 6,784 |
|
| 36 |
+
| Vision tower | No |
|
| 37 |
+
| Weight dtype | BF16 |
|
| 38 |
+
|
| 39 |
+
Each MoE layer initially computes one half of the original Qwen3.5-2B dense MLP
|
| 40 |
+
through the shared path and one half through the selected routed expert. The
|
| 41 |
+
two routed experts begin functionally identical. Additional neurons have random
|
| 42 |
+
gate/up projections and zero down projections, making them output-neutral but
|
| 43 |
+
trainable. Exact conversion metadata is in `conversion_manifest.json`.
|
| 44 |
+
|
| 45 |
+
## Evaluation
|
| 46 |
+
|
| 47 |
+
All reported results were produced locally on an Apple M4 with 32GB unified
|
| 48 |
+
memory. Raw JSON reports are included in `evaluation/`.
|
| 49 |
+
|
| 50 |
+
### Chat and sentence generation
|
| 51 |
+
|
| 52 |
+
The fixed gate contains 60 prompts: 10 each in Korean, English, Chinese,
|
| 53 |
+
Japanese, Spanish, and German. It covers facts, arithmetic, translation,
|
| 54 |
+
instruction following, and free-form sentence generation.
|
| 55 |
+
|
| 56 |
+
| Result | Score |
|
| 57 |
+
|---|---:|
|
| 58 |
+
| Non-degenerate/correct automatic checks | 60 / 60 |
|
| 59 |
+
| Languages meeting the gate | 6 / 6 |
|
| 60 |
+
|
| 61 |
+
Six chemical-formula answers used the correct Unicode spelling `H₂O`; the
|
| 62 |
+
scorer normalizes Unicode subscripts before comparison.
|
| 63 |
+
|
| 64 |
+
### Multilingual held-out LM loss
|
| 65 |
+
|
| 66 |
+
Four held-out FineWeb/FineWeb2 documents per language, 128 tokens per document:
|
| 67 |
+
|
| 68 |
+
| Model | Mean loss | Relative to Qwen3.5-4B |
|
| 69 |
+
|---|---:|---:|
|
| 70 |
+
| Qwen3.5-4B | 2.9399 | 1.000x |
|
| 71 |
+
| This model | 3.2085 | 1.091x |
|
| 72 |
+
| Qwen3.5-2B | 3.2085 | 1.091x |
|
| 73 |
+
|
| 74 |
+
### Standard benchmark development subset
|
| 75 |
+
|
| 76 |
+
EleutherAI `lm-evaluation-harness==0.4.12`, zero-shot, BF16, first 100 examples
|
| 77 |
+
per task. These limited results are development indicators, **not full-task
|
| 78 |
+
benchmark claims**.
|
| 79 |
+
|
| 80 |
+
| Model | ARC-Easy acc_norm | HellaSwag acc_norm | Mean |
|
| 81 |
+
|---|---:|---:|---:|
|
| 82 |
+
| Qwen3.5-4B | 0.81 | 0.68 | 0.745 |
|
| 83 |
+
| This model | 0.73 | 0.62 | 0.675 |
|
| 84 |
+
|
| 85 |
+
The subset mean is 90.6% of the Qwen3.5-4B teacher mean.
|
| 86 |
+
|
| 87 |
+
### Local service gate
|
| 88 |
+
|
| 89 |
+
The included FastAPI service completed 20/20 consecutive non-streaming
|
| 90 |
+
`POST /v1/chat/completions` requests:
|
| 91 |
+
|
| 92 |
+
| Metric | Value |
|
| 93 |
+
|---|---:|
|
| 94 |
+
| Successful requests | 20 / 20 |
|
| 95 |
+
| Mean latency | 2.60 s |
|
| 96 |
+
| p95 latency | 3.57 s |
|
| 97 |
+
| MPS allocated memory | 7.62 GB |
|
| 98 |
+
|
| 99 |
+
Requests generated up to 16 new tokens. See `evaluation/openai_service_20.json`.
|
| 100 |
+
|
| 101 |
+
## Transformers usage
|
| 102 |
+
|
| 103 |
+
Use Transformers 5.13.0 or another version that provides
|
| 104 |
+
`Qwen3_5MoeForCausalLM`:
|
| 105 |
+
|
| 106 |
+
```python
|
| 107 |
+
import torch
|
| 108 |
+
from transformers import AutoTokenizer, Qwen3_5MoeForCausalLM
|
| 109 |
+
|
| 110 |
+
model_id = "sepsy070716/Qwen3.5-4B-A3B-Student-v2"
|
| 111 |
+
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
| 112 |
+
model = Qwen3_5MoeForCausalLM.from_pretrained(
|
| 113 |
+
model_id,
|
| 114 |
+
dtype=torch.bfloat16,
|
| 115 |
+
device_map="auto",
|
| 116 |
+
)
|
| 117 |
+
|
| 118 |
+
messages = [{"role": "user", "content": "대한민국의 수도는 어디인가요?"}]
|
| 119 |
+
inputs = tokenizer.apply_chat_template(
|
| 120 |
+
messages,
|
| 121 |
+
tokenize=True,
|
| 122 |
+
add_generation_prompt=True,
|
| 123 |
+
enable_thinking=False,
|
| 124 |
+
return_tensors="pt",
|
| 125 |
+
return_dict=True,
|
| 126 |
+
).to(model.device)
|
| 127 |
+
output = model.generate(**inputs, max_new_tokens=64, do_sample=False)
|
| 128 |
+
print(tokenizer.decode(output[0, inputs["input_ids"].shape[1]:], skip_special_tokens=True))
|
| 129 |
+
```
|
| 130 |
+
|
| 131 |
+
## Reproduction and service
|
| 132 |
+
|
| 133 |
+
`research_code/` contains the converter, multilingual loss comparison, 60-prompt
|
| 134 |
+
gate and scorer, plus the local OpenAI-compatible service and its 20-request
|
| 135 |
+
test. The service implements `GET /health`, `GET /v1/models`, and non-streaming
|
| 136 |
+
`POST /v1/chat/completions`.
|
| 137 |
+
|
| 138 |
+
## Limitations
|
| 139 |
+
|
| 140 |
+
- Current quality is inherited primarily from Qwen3.5-2B; the extra capacity has
|
| 141 |
+
not yet received large-scale continued pretraining or teacher distillation.
|
| 142 |
+
- The 100-example ARC-Easy/HellaSwag figures are small development subsets and
|
| 143 |
+
have substantial sampling uncertainty. Run the full tasks before making
|
| 144 |
+
publication or production claims.
|
| 145 |
+
- This model is text-only and cannot replace the original model's vision path.
|
| 146 |
+
- The included server is a single-process local research server. It has no
|
| 147 |
+
authentication, TLS, streaming, tool calling, or multi-worker support.
|
| 148 |
+
- Apply the same safety, bias, privacy, and factuality evaluation required for
|
| 149 |
+
any deployment of the upstream Qwen models.
|
| 150 |
+
|
| 151 |
+
## License and attribution
|
| 152 |
+
|
| 153 |
+
Released under Apache-2.0, following the included upstream license. Derived from
|
| 154 |
+
Qwen3.5-2B weights and evaluated against Qwen3.5-4B. Qwen model names and
|
| 155 |
+
trademarks belong to their respective owners.
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,154 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- set image_count = namespace(value=0) %}
|
| 2 |
+
{%- set video_count = namespace(value=0) %}
|
| 3 |
+
{%- macro render_content(content, do_vision_count, is_system_content=false) %}
|
| 4 |
+
{%- if content is string %}
|
| 5 |
+
{{- content }}
|
| 6 |
+
{%- elif content is iterable and content is not mapping %}
|
| 7 |
+
{%- for item in content %}
|
| 8 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
|
| 9 |
+
{%- if is_system_content %}
|
| 10 |
+
{{- raise_exception('System message cannot contain images.') }}
|
| 11 |
+
{%- endif %}
|
| 12 |
+
{%- if do_vision_count %}
|
| 13 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{%- if add_vision_id %}
|
| 16 |
+
{{- 'Picture ' ~ image_count.value ~ ': ' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{{- '<|vision_start|><|image_pad|><|vision_end|>' }}
|
| 19 |
+
{%- elif 'video' in item or item.type == 'video' %}
|
| 20 |
+
{%- if is_system_content %}
|
| 21 |
+
{{- raise_exception('System message cannot contain videos.') }}
|
| 22 |
+
{%- endif %}
|
| 23 |
+
{%- if do_vision_count %}
|
| 24 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 25 |
+
{%- endif %}
|
| 26 |
+
{%- if add_vision_id %}
|
| 27 |
+
{{- 'Video ' ~ video_count.value ~ ': ' }}
|
| 28 |
+
{%- endif %}
|
| 29 |
+
{{- '<|vision_start|><|video_pad|><|vision_end|>' }}
|
| 30 |
+
{%- elif 'text' in item %}
|
| 31 |
+
{{- item.text }}
|
| 32 |
+
{%- else %}
|
| 33 |
+
{{- raise_exception('Unexpected item type in content.') }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endfor %}
|
| 36 |
+
{%- elif content is none or content is undefined %}
|
| 37 |
+
{{- '' }}
|
| 38 |
+
{%- else %}
|
| 39 |
+
{{- raise_exception('Unexpected content type.') }}
|
| 40 |
+
{%- endif %}
|
| 41 |
+
{%- endmacro %}
|
| 42 |
+
{%- if not messages %}
|
| 43 |
+
{{- raise_exception('No messages provided.') }}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- if tools and tools is iterable and tools is not mapping %}
|
| 46 |
+
{{- '<|im_start|>system\n' }}
|
| 47 |
+
{{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
|
| 48 |
+
{%- for tool in tools %}
|
| 49 |
+
{{- "\n" }}
|
| 50 |
+
{{- tool | tojson }}
|
| 51 |
+
{%- endfor %}
|
| 52 |
+
{{- "\n</tools>" }}
|
| 53 |
+
{{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
|
| 54 |
+
{%- if messages[0].role == 'system' %}
|
| 55 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 56 |
+
{%- if content %}
|
| 57 |
+
{{- '\n\n' + content }}
|
| 58 |
+
{%- endif %}
|
| 59 |
+
{%- endif %}
|
| 60 |
+
{{- '<|im_end|>\n' }}
|
| 61 |
+
{%- else %}
|
| 62 |
+
{%- if messages[0].role == 'system' %}
|
| 63 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 64 |
+
{{- '<|im_start|>system\n' + content + '<|im_end|>\n' }}
|
| 65 |
+
{%- endif %}
|
| 66 |
+
{%- endif %}
|
| 67 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 68 |
+
{%- for message in messages[::-1] %}
|
| 69 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 70 |
+
{%- if ns.multi_step_tool and message.role == "user" %}
|
| 71 |
+
{%- set content = render_content(message.content, false)|trim %}
|
| 72 |
+
{%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
|
| 73 |
+
{%- set ns.multi_step_tool = false %}
|
| 74 |
+
{%- set ns.last_query_index = index %}
|
| 75 |
+
{%- endif %}
|
| 76 |
+
{%- endif %}
|
| 77 |
+
{%- endfor %}
|
| 78 |
+
{%- if ns.multi_step_tool %}
|
| 79 |
+
{{- raise_exception('No user query found in messages.') }}
|
| 80 |
+
{%- endif %}
|
| 81 |
+
{%- for message in messages %}
|
| 82 |
+
{%- set content = render_content(message.content, true)|trim %}
|
| 83 |
+
{%- if message.role == "system" %}
|
| 84 |
+
{%- if not loop.first %}
|
| 85 |
+
{{- raise_exception('System message must be at the beginning.') }}
|
| 86 |
+
{%- endif %}
|
| 87 |
+
{%- elif message.role == "user" %}
|
| 88 |
+
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 89 |
+
{%- elif message.role == "assistant" %}
|
| 90 |
+
{%- set reasoning_content = '' %}
|
| 91 |
+
{%- if message.reasoning_content is string %}
|
| 92 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 93 |
+
{%- else %}
|
| 94 |
+
{%- if '</think>' in content %}
|
| 95 |
+
{%- set reasoning_content = content.split('</think>')[0].rstrip('\n').split('<think>')[-1].lstrip('\n') %}
|
| 96 |
+
{%- set content = content.split('</think>')[-1].lstrip('\n') %}
|
| 97 |
+
{%- endif %}
|
| 98 |
+
{%- endif %}
|
| 99 |
+
{%- set reasoning_content = reasoning_content|trim %}
|
| 100 |
+
{%- if loop.index0 > ns.last_query_index %}
|
| 101 |
+
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
|
| 102 |
+
{%- else %}
|
| 103 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 104 |
+
{%- endif %}
|
| 105 |
+
{%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
|
| 106 |
+
{%- for tool_call in message.tool_calls %}
|
| 107 |
+
{%- if tool_call.function is defined %}
|
| 108 |
+
{%- set tool_call = tool_call.function %}
|
| 109 |
+
{%- endif %}
|
| 110 |
+
{%- if loop.first %}
|
| 111 |
+
{%- if content|trim %}
|
| 112 |
+
{{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 113 |
+
{%- else %}
|
| 114 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 115 |
+
{%- endif %}
|
| 116 |
+
{%- else %}
|
| 117 |
+
{{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 118 |
+
{%- endif %}
|
| 119 |
+
{%- if tool_call.arguments is defined %}
|
| 120 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 121 |
+
{{- '<parameter=' + args_name + '>\n' }}
|
| 122 |
+
{%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}
|
| 123 |
+
{{- args_value }}
|
| 124 |
+
{{- '\n</parameter>\n' }}
|
| 125 |
+
{%- endfor %}
|
| 126 |
+
{%- endif %}
|
| 127 |
+
{{- '</function>\n</tool_call>' }}
|
| 128 |
+
{%- endfor %}
|
| 129 |
+
{%- endif %}
|
| 130 |
+
{{- '<|im_end|>\n' }}
|
| 131 |
+
{%- elif message.role == "tool" %}
|
| 132 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 133 |
+
{{- '<|im_start|>user' }}
|
| 134 |
+
{%- endif %}
|
| 135 |
+
{{- '\n<tool_response>\n' }}
|
| 136 |
+
{{- content }}
|
| 137 |
+
{{- '\n</tool_response>' }}
|
| 138 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 139 |
+
{{- '<|im_end|>\n' }}
|
| 140 |
+
{%- elif loop.last %}
|
| 141 |
+
{{- '<|im_end|>\n' }}
|
| 142 |
+
{%- endif %}
|
| 143 |
+
{%- else %}
|
| 144 |
+
{{- raise_exception('Unexpected message role.') }}
|
| 145 |
+
{%- endif %}
|
| 146 |
+
{%- endfor %}
|
| 147 |
+
{%- if add_generation_prompt %}
|
| 148 |
+
{{- '<|im_start|>assistant\n' }}
|
| 149 |
+
{%- if enable_thinking is defined and enable_thinking is true %}
|
| 150 |
+
{{- '<think>\n' }}
|
| 151 |
+
{%- else %}
|
| 152 |
+
{{- '<think>\n\n</think>\n\n' }}
|
| 153 |
+
{%- endif %}
|
| 154 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,80 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"Qwen3_5MoeForCausalLM"
|
| 4 |
+
],
|
| 5 |
+
"attention_bias": false,
|
| 6 |
+
"attention_dropout": 0.0,
|
| 7 |
+
"attn_output_gate": true,
|
| 8 |
+
"bos_token_id": null,
|
| 9 |
+
"dtype": "bfloat16",
|
| 10 |
+
"eos_token_id": 248044,
|
| 11 |
+
"full_attention_interval": 4,
|
| 12 |
+
"head_dim": 256,
|
| 13 |
+
"hidden_act": "silu",
|
| 14 |
+
"hidden_size": 2048,
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"layer_types": [
|
| 17 |
+
"linear_attention",
|
| 18 |
+
"linear_attention",
|
| 19 |
+
"linear_attention",
|
| 20 |
+
"full_attention",
|
| 21 |
+
"linear_attention",
|
| 22 |
+
"linear_attention",
|
| 23 |
+
"linear_attention",
|
| 24 |
+
"full_attention",
|
| 25 |
+
"linear_attention",
|
| 26 |
+
"linear_attention",
|
| 27 |
+
"linear_attention",
|
| 28 |
+
"full_attention",
|
| 29 |
+
"linear_attention",
|
| 30 |
+
"linear_attention",
|
| 31 |
+
"linear_attention",
|
| 32 |
+
"full_attention",
|
| 33 |
+
"linear_attention",
|
| 34 |
+
"linear_attention",
|
| 35 |
+
"linear_attention",
|
| 36 |
+
"full_attention",
|
| 37 |
+
"linear_attention",
|
| 38 |
+
"linear_attention",
|
| 39 |
+
"linear_attention",
|
| 40 |
+
"full_attention"
|
| 41 |
+
],
|
| 42 |
+
"linear_conv_kernel_dim": 4,
|
| 43 |
+
"linear_key_head_dim": 128,
|
| 44 |
+
"linear_num_key_heads": 16,
|
| 45 |
+
"linear_num_value_heads": 16,
|
| 46 |
+
"linear_value_head_dim": 128,
|
| 47 |
+
"mamba_ssm_dtype": "float32",
|
| 48 |
+
"max_position_embeddings": 262144,
|
| 49 |
+
"mlp_only_layers": [],
|
| 50 |
+
"model_type": "qwen3_5_moe_text",
|
| 51 |
+
"moe_intermediate_size": 6784,
|
| 52 |
+
"mtp_num_hidden_layers": 1,
|
| 53 |
+
"mtp_use_dedicated_embeddings": false,
|
| 54 |
+
"num_attention_heads": 8,
|
| 55 |
+
"num_experts": 2,
|
| 56 |
+
"num_experts_per_tok": 1,
|
| 57 |
+
"num_hidden_layers": 24,
|
| 58 |
+
"num_key_value_heads": 2,
|
| 59 |
+
"output_router_logits": false,
|
| 60 |
+
"pad_token_id": null,
|
| 61 |
+
"partial_rotary_factor": 0.25,
|
| 62 |
+
"rms_norm_eps": 1e-06,
|
| 63 |
+
"rope_parameters": {
|
| 64 |
+
"mrope_interleaved": true,
|
| 65 |
+
"mrope_section": [
|
| 66 |
+
11,
|
| 67 |
+
11,
|
| 68 |
+
10
|
| 69 |
+
],
|
| 70 |
+
"partial_rotary_factor": 0.25,
|
| 71 |
+
"rope_theta": 10000000,
|
| 72 |
+
"rope_type": "default"
|
| 73 |
+
},
|
| 74 |
+
"router_aux_loss_coef": 0.001,
|
| 75 |
+
"shared_expert_intermediate_size": 6912,
|
| 76 |
+
"tie_word_embeddings": true,
|
| 77 |
+
"transformers_version": "5.13.0",
|
| 78 |
+
"use_cache": true,
|
| 79 |
+
"vocab_size": 248320
|
| 80 |
+
}
|
conversion_manifest.json
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"source": "models/Qwen/Qwen3.5-2B",
|
| 3 |
+
"initial_function": "Qwen3.5-2B text model (dense MLP split 50/50)",
|
| 4 |
+
"total_parameters": 3995901760,
|
| 5 |
+
"active_parameters": 2995560256,
|
| 6 |
+
"num_experts": 2,
|
| 7 |
+
"experts_per_token": 1,
|
| 8 |
+
"shared_intermediate_size": 6912,
|
| 9 |
+
"routed_intermediate_size": 6784,
|
| 10 |
+
"vision_included": false,
|
| 11 |
+
"seed": 35
|
| 12 |
+
}
|
evaluation/chat_gate_60.json
ADDED
|
@@ -0,0 +1,762 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"model": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 3 |
+
"total": 60,
|
| 4 |
+
"passed": 60,
|
| 5 |
+
"pass_rate": 1.0,
|
| 6 |
+
"gate_threshold": 0.9,
|
| 7 |
+
"gate_passed": true,
|
| 8 |
+
"by_language": {
|
| 9 |
+
"de": {
|
| 10 |
+
"passed": 10,
|
| 11 |
+
"total": 10,
|
| 12 |
+
"pass_rate": 1.0
|
| 13 |
+
},
|
| 14 |
+
"en": {
|
| 15 |
+
"passed": 10,
|
| 16 |
+
"total": 10,
|
| 17 |
+
"pass_rate": 1.0
|
| 18 |
+
},
|
| 19 |
+
"es": {
|
| 20 |
+
"passed": 10,
|
| 21 |
+
"total": 10,
|
| 22 |
+
"pass_rate": 1.0
|
| 23 |
+
},
|
| 24 |
+
"ja": {
|
| 25 |
+
"passed": 10,
|
| 26 |
+
"total": 10,
|
| 27 |
+
"pass_rate": 1.0
|
| 28 |
+
},
|
| 29 |
+
"ko": {
|
| 30 |
+
"passed": 10,
|
| 31 |
+
"total": 10,
|
| 32 |
+
"pass_rate": 1.0
|
| 33 |
+
},
|
| 34 |
+
"zh": {
|
| 35 |
+
"passed": 10,
|
| 36 |
+
"total": 10,
|
| 37 |
+
"pass_rate": 1.0
|
| 38 |
+
}
|
| 39 |
+
},
|
| 40 |
+
"results": [
|
| 41 |
+
{
|
| 42 |
+
"id": "ko_01",
|
| 43 |
+
"language": "ko",
|
| 44 |
+
"prompt": "프랑스의 수도는 어디인가요? 짧게 답하세요.",
|
| 45 |
+
"expected_any": [
|
| 46 |
+
"파리"
|
| 47 |
+
],
|
| 48 |
+
"completion": "파리\n",
|
| 49 |
+
"passed": true,
|
| 50 |
+
"failures": [],
|
| 51 |
+
"new_tokens": 5,
|
| 52 |
+
"elapsed_seconds": 2.93180704198312
|
| 53 |
+
},
|
| 54 |
+
{
|
| 55 |
+
"id": "ko_02",
|
| 56 |
+
"language": "ko",
|
| 57 |
+
"prompt": "7 곱하기 8은 얼마인가요? 숫자로 답하세요.",
|
| 58 |
+
"expected_any": [
|
| 59 |
+
"56"
|
| 60 |
+
],
|
| 61 |
+
"completion": "7 곱하기 8은 **56**입니다.\n",
|
| 62 |
+
"passed": true,
|
| 63 |
+
"failures": [],
|
| 64 |
+
"new_tokens": 16,
|
| 65 |
+
"elapsed_seconds": 3.4328290830017067
|
| 66 |
+
},
|
| 67 |
+
{
|
| 68 |
+
"id": "ko_03",
|
| 69 |
+
"language": "ko",
|
| 70 |
+
"prompt": "물의 화학식은 무엇인가요?",
|
| 71 |
+
"expected_any": [
|
| 72 |
+
"H2O",
|
| 73 |
+
"h2o"
|
| 74 |
+
],
|
| 75 |
+
"completion": "물의 화학식은 **H₂O**입니다.\n\n이 화학식은 물의 구성 원자 수를 나타냅니다:\n* **H**: 수소 원자 2 개\n* **O**: 산소 원자 1",
|
| 76 |
+
"passed": true,
|
| 77 |
+
"failures": [],
|
| 78 |
+
"new_tokens": 48,
|
| 79 |
+
"elapsed_seconds": 10.234737457998563
|
| 80 |
+
},
|
| 81 |
+
{
|
| 82 |
+
"id": "ko_04",
|
| 83 |
+
"language": "ko",
|
| 84 |
+
"prompt": "태양계에서 가장 큰 행성은 무엇인가요?",
|
| 85 |
+
"expected_any": [
|
| 86 |
+
"목성"
|
| 87 |
+
],
|
| 88 |
+
"completion": "태양계에서 가장 큰 행성은 **목성**입니다.\n\n목성은 태양계에서 가장 큰 행성으로, 지구의 약 318 배의 크기와 약 1,321 배의 질량을",
|
| 89 |
+
"passed": true,
|
| 90 |
+
"failures": [],
|
| 91 |
+
"new_tokens": 48,
|
| 92 |
+
"elapsed_seconds": 9.949900375038851
|
| 93 |
+
},
|
| 94 |
+
{
|
| 95 |
+
"id": "ko_05",
|
| 96 |
+
"language": "ko",
|
| 97 |
+
"prompt": "조용한 겨울 아침을 묘사하는 자연스러운 문장 두 개를 써 주세요.",
|
| 98 |
+
"completion": "1. 창문 너머로 비치는 차가운 햇살이 눈부시며, 거친 바람이 나무의 가지 사이로 스치며 겨울 아침의 고요함을 더 깊게 만듭니다.\n2. 눈",
|
| 99 |
+
"passed": true,
|
| 100 |
+
"failures": [],
|
| 101 |
+
"new_tokens": 48,
|
| 102 |
+
"elapsed_seconds": 10.018696333048865
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"id": "ko_06",
|
| 106 |
+
"language": "ko",
|
| 107 |
+
"prompt": "친구에게 약속 시간을 10분 늦겠다고 정중히 알리는 한 문장을 써 주세요.",
|
| 108 |
+
"completion": "\"안녕하세요! 오늘 약속 시간이 10 분 늦어지니 미리 알려드리고 싶네요. 혹시 다른 시간으로 조정할 수 있을까요?\"\n",
|
| 109 |
+
"passed": true,
|
| 110 |
+
"failures": [],
|
| 111 |
+
"new_tokens": 36,
|
| 112 |
+
"elapsed_seconds": 7.496548916969914
|
| 113 |
+
},
|
| 114 |
+
{
|
| 115 |
+
"id": "ko_07",
|
| 116 |
+
"language": "ko",
|
| 117 |
+
"prompt": "인공지능 연구에서 재현성이 중요한 이유를 두 문장으로 설명하세요.",
|
| 118 |
+
"completion": "인공지능 연구에서 재현성은 모델의 신뢰성을 확보하고, 다른 연구자들이 동일한 결과를 검증할 수 있는 기준을 마련하기 위함입니다. 또한, 복잡한 알고리즘의 동작 원리를 투명하게 이해하고, 윤리적 문제",
|
| 119 |
+
"passed": true,
|
| 120 |
+
"failures": [],
|
| 121 |
+
"new_tokens": 48,
|
| 122 |
+
"elapsed_seconds": 9.953075792000163
|
| 123 |
+
},
|
| 124 |
+
{
|
| 125 |
+
"id": "ko_08",
|
| 126 |
+
"language": "ko",
|
| 127 |
+
"prompt": "다른 말 없이 정확히 '확인'이라고만 답하세요.",
|
| 128 |
+
"expected_any": [
|
| 129 |
+
"확인"
|
| 130 |
+
],
|
| 131 |
+
"completion": "확인\n",
|
| 132 |
+
"passed": true,
|
| 133 |
+
"failures": [],
|
| 134 |
+
"new_tokens": 4,
|
| 135 |
+
"elapsed_seconds": 0.8486927500343882
|
| 136 |
+
},
|
| 137 |
+
{
|
| 138 |
+
"id": "ko_09",
|
| 139 |
+
"language": "ko",
|
| 140 |
+
"prompt": "사과, 바나나, 포도를 번호가 있는 세 항목으로 나열하세요.",
|
| 141 |
+
"completion": "네, 요청하신 대로 세 가지 항목을 번호로 나열했습니다.\n\n1. 사과\n2. 바나나\n3. 포도\n",
|
| 142 |
+
"passed": true,
|
| 143 |
+
"failures": [],
|
| 144 |
+
"new_tokens": 33,
|
| 145 |
+
"elapsed_seconds": 6.8605534169473685
|
| 146 |
+
},
|
| 147 |
+
{
|
| 148 |
+
"id": "ko_10",
|
| 149 |
+
"language": "ko",
|
| 150 |
+
"prompt": "'Good morning'을 자연스러운 한국어로 번역하세요.",
|
| 151 |
+
"expected_any": [
|
| 152 |
+
"좋은 아침",
|
| 153 |
+
"안녕하세요"
|
| 154 |
+
],
|
| 155 |
+
"completion": "안녕하세요.\n",
|
| 156 |
+
"passed": true,
|
| 157 |
+
"failures": [],
|
| 158 |
+
"new_tokens": 7,
|
| 159 |
+
"elapsed_seconds": 1.5930803330265917
|
| 160 |
+
},
|
| 161 |
+
{
|
| 162 |
+
"id": "en_01",
|
| 163 |
+
"language": "en",
|
| 164 |
+
"prompt": "What is the capital of France? Answer briefly.",
|
| 165 |
+
"expected_any": [
|
| 166 |
+
"Paris"
|
| 167 |
+
],
|
| 168 |
+
"completion": "The capital of France is **Paris**.\n",
|
| 169 |
+
"passed": true,
|
| 170 |
+
"failures": [],
|
| 171 |
+
"new_tokens": 11,
|
| 172 |
+
"elapsed_seconds": 2.369376624992583
|
| 173 |
+
},
|
| 174 |
+
{
|
| 175 |
+
"id": "en_02",
|
| 176 |
+
"language": "en",
|
| 177 |
+
"prompt": "What is 7 multiplied by 8? Answer with a number.",
|
| 178 |
+
"expected_any": [
|
| 179 |
+
"56"
|
| 180 |
+
],
|
| 181 |
+
"completion": "56\n",
|
| 182 |
+
"passed": true,
|
| 183 |
+
"failures": [],
|
| 184 |
+
"new_tokens": 5,
|
| 185 |
+
"elapsed_seconds": 1.1381639169994742
|
| 186 |
+
},
|
| 187 |
+
{
|
| 188 |
+
"id": "en_03",
|
| 189 |
+
"language": "en",
|
| 190 |
+
"prompt": "What is the chemical formula for water?",
|
| 191 |
+
"expected_any": [
|
| 192 |
+
"H2O",
|
| 193 |
+
"h2o"
|
| 194 |
+
],
|
| 195 |
+
"completion": "The chemical formula for water is **H₂O**.\n\nThis indicates that a single water molecule consists of two hydrogen atoms covalently bonded to one oxygen atom.\n",
|
| 196 |
+
"passed": true,
|
| 197 |
+
"failures": [],
|
| 198 |
+
"new_tokens": 36,
|
| 199 |
+
"elapsed_seconds": 7.6181977079831995
|
| 200 |
+
},
|
| 201 |
+
{
|
| 202 |
+
"id": "en_04",
|
| 203 |
+
"language": "en",
|
| 204 |
+
"prompt": "What is the largest planet in the Solar System?",
|
| 205 |
+
"expected_any": [
|
| 206 |
+
"Jupiter"
|
| 207 |
+
],
|
| 208 |
+
"completion": "The largest planet in the Solar System is **Jupiter**.\n\nDiscovered by Galileo Galilei in 1610, Jupiter is a gas giant with a mass that is more than twice that of all the other planets combined",
|
| 209 |
+
"passed": true,
|
| 210 |
+
"failures": [],
|
| 211 |
+
"new_tokens": 48,
|
| 212 |
+
"elapsed_seconds": 9.993648000003304
|
| 213 |
+
},
|
| 214 |
+
{
|
| 215 |
+
"id": "en_05",
|
| 216 |
+
"language": "en",
|
| 217 |
+
"prompt": "Write two natural sentences describing a quiet winter morning.",
|
| 218 |
+
"completion": "The sun filters through the heavy, frost-covered branches of the oak trees, casting a soft, golden glow across the snow-dusted lawn.\nThe air is crisp and still, filled only with the gentle rustle of birch leaves and",
|
| 219 |
+
"passed": true,
|
| 220 |
+
"failures": [],
|
| 221 |
+
"new_tokens": 48,
|
| 222 |
+
"elapsed_seconds": 9.83665666700108
|
| 223 |
+
},
|
| 224 |
+
{
|
| 225 |
+
"id": "en_06",
|
| 226 |
+
"language": "en",
|
| 227 |
+
"prompt": "Write one polite sentence telling a friend you will be ten minutes late.",
|
| 228 |
+
"completion": "\"Hey, I'm heading out a bit early, so I'll be ten minutes late for our meeting.\"\n",
|
| 229 |
+
"passed": true,
|
| 230 |
+
"failures": [],
|
| 231 |
+
"new_tokens": 25,
|
| 232 |
+
"elapsed_seconds": 5.151561209000647
|
| 233 |
+
},
|
| 234 |
+
{
|
| 235 |
+
"id": "en_07",
|
| 236 |
+
"language": "en",
|
| 237 |
+
"prompt": "Explain in two sentences why reproducibility matters in AI research.",
|
| 238 |
+
"completion": "Reproducibility is essential because it allows the scientific community to verify that results are not artifacts of chance or specific, undocumented choices made by the original researchers. Without the ability to reproduce findings, the validity of AI models and their applications cannot be",
|
| 239 |
+
"passed": true,
|
| 240 |
+
"failures": [],
|
| 241 |
+
"new_tokens": 48,
|
| 242 |
+
"elapsed_seconds": 9.94930508401012
|
| 243 |
+
},
|
| 244 |
+
{
|
| 245 |
+
"id": "en_08",
|
| 246 |
+
"language": "en",
|
| 247 |
+
"prompt": "Reply with exactly the word 'confirmed' and nothing else.",
|
| 248 |
+
"expected_any": [
|
| 249 |
+
"confirmed"
|
| 250 |
+
],
|
| 251 |
+
"completion": "confirmed\n",
|
| 252 |
+
"passed": true,
|
| 253 |
+
"failures": [],
|
| 254 |
+
"new_tokens": 4,
|
| 255 |
+
"elapsed_seconds": 0.9270532079972327
|
| 256 |
+
},
|
| 257 |
+
{
|
| 258 |
+
"id": "en_09",
|
| 259 |
+
"language": "en",
|
| 260 |
+
"prompt": "List apple, banana, and grape as three numbered items.",
|
| 261 |
+
"completion": "1. Apple\n2. Banana\n3. Grape\n",
|
| 262 |
+
"passed": true,
|
| 263 |
+
"failures": [],
|
| 264 |
+
"new_tokens": 14,
|
| 265 |
+
"elapsed_seconds": 2.939590915979352
|
| 266 |
+
},
|
| 267 |
+
{
|
| 268 |
+
"id": "en_10",
|
| 269 |
+
"language": "en",
|
| 270 |
+
"prompt": "Translate '좋은 아침입니다' into natural English.",
|
| 271 |
+
"expected_any": [
|
| 272 |
+
"good morning"
|
| 273 |
+
],
|
| 274 |
+
"completion": "Good morning.\n",
|
| 275 |
+
"passed": true,
|
| 276 |
+
"failures": [],
|
| 277 |
+
"new_tokens": 6,
|
| 278 |
+
"elapsed_seconds": 1.294799041992519
|
| 279 |
+
},
|
| 280 |
+
{
|
| 281 |
+
"id": "zh_01",
|
| 282 |
+
"language": "zh",
|
| 283 |
+
"prompt": "法国的首都是哪里?请简短回答。",
|
| 284 |
+
"expected_any": [
|
| 285 |
+
"巴黎"
|
| 286 |
+
],
|
| 287 |
+
"completion": "巴黎\n",
|
| 288 |
+
"passed": true,
|
| 289 |
+
"failures": [],
|
| 290 |
+
"new_tokens": 4,
|
| 291 |
+
"elapsed_seconds": 0.8428977089934051
|
| 292 |
+
},
|
| 293 |
+
{
|
| 294 |
+
"id": "zh_02",
|
| 295 |
+
"language": "zh",
|
| 296 |
+
"prompt": "7乘以8等于多少?请用数字回答。",
|
| 297 |
+
"expected_any": [
|
| 298 |
+
"56"
|
| 299 |
+
],
|
| 300 |
+
"completion": "56\n",
|
| 301 |
+
"passed": true,
|
| 302 |
+
"failures": [],
|
| 303 |
+
"new_tokens": 5,
|
| 304 |
+
"elapsed_seconds": 1.042713167029433
|
| 305 |
+
},
|
| 306 |
+
{
|
| 307 |
+
"id": "zh_03",
|
| 308 |
+
"language": "zh",
|
| 309 |
+
"prompt": "水的化学式是什么?",
|
| 310 |
+
"expected_any": [
|
| 311 |
+
"H2O",
|
| 312 |
+
"h2o"
|
| 313 |
+
],
|
| 314 |
+
"completion": "水的化学式是 **H₂O**。\n\n这个符号表示:\n* **H** 代表氢(Hydrogen)\n* **O** 代表氧(Oxygen)\n* 下标的 **",
|
| 315 |
+
"passed": true,
|
| 316 |
+
"failures": [],
|
| 317 |
+
"new_tokens": 48,
|
| 318 |
+
"elapsed_seconds": 9.921300665999297
|
| 319 |
+
},
|
| 320 |
+
{
|
| 321 |
+
"id": "zh_04",
|
| 322 |
+
"language": "zh",
|
| 323 |
+
"prompt": "太阳系中最大的行星是什么?",
|
| 324 |
+
"expected_any": [
|
| 325 |
+
"木星"
|
| 326 |
+
],
|
| 327 |
+
"completion": "太阳系中最大的行星是**木星**(Jupiter)。\n\n它的质量约为其他所有行星总和的 2.5 倍,是太阳系中体积和质量最大的行星。木星是一颗气态巨行星,其直径约为地球直径",
|
| 328 |
+
"passed": true,
|
| 329 |
+
"failures": [],
|
| 330 |
+
"new_tokens": 48,
|
| 331 |
+
"elapsed_seconds": 9.965762625040952
|
| 332 |
+
},
|
| 333 |
+
{
|
| 334 |
+
"id": "zh_05",
|
| 335 |
+
"language": "zh",
|
| 336 |
+
"prompt": "用两个自然的句子描写安静的冬日清晨。",
|
| 337 |
+
"completion": "冬日的清晨,薄雾如轻纱般笼罩着沉睡的街道,将城市的喧嚣温柔地隔绝在外。\n此时,唯有远处偶尔传来的几声鸟鸣划破寂静,伴随着第一缕阳光穿透云层,洒在静谧",
|
| 338 |
+
"passed": true,
|
| 339 |
+
"failures": [],
|
| 340 |
+
"new_tokens": 48,
|
| 341 |
+
"elapsed_seconds": 10.13702458399348
|
| 342 |
+
},
|
| 343 |
+
{
|
| 344 |
+
"id": "zh_06",
|
| 345 |
+
"language": "zh",
|
| 346 |
+
"prompt": "写一句礼貌的话,告诉朋友你会迟到十分钟。",
|
| 347 |
+
"completion": "“抱歉啦,路上有点堵车,我大概会迟到十分钟,你等我一会儿就好!”\n",
|
| 348 |
+
"passed": true,
|
| 349 |
+
"failures": [],
|
| 350 |
+
"new_tokens": 22,
|
| 351 |
+
"elapsed_seconds": 4.585268583032303
|
| 352 |
+
},
|
| 353 |
+
{
|
| 354 |
+
"id": "zh_07",
|
| 355 |
+
"language": "zh",
|
| 356 |
+
"prompt": "用两句话解释可复现性为何对人工智能研究重要。",
|
| 357 |
+
"completion": "可复现性是人工智能研究的核心基石,它确保了模型在相同数据、超参数和计算环境下的行为一致性,从而允许研究者验证假设、发现规律并构建可靠的知识体系。\n\n此外,这一标准推动了社区协作与",
|
| 358 |
+
"passed": true,
|
| 359 |
+
"failures": [],
|
| 360 |
+
"new_tokens": 48,
|
| 361 |
+
"elapsed_seconds": 10.05659958301112
|
| 362 |
+
},
|
| 363 |
+
{
|
| 364 |
+
"id": "zh_08",
|
| 365 |
+
"language": "zh",
|
| 366 |
+
"prompt": "不要说别的,只回答“收到”。",
|
| 367 |
+
"expected_any": [
|
| 368 |
+
"收到"
|
| 369 |
+
],
|
| 370 |
+
"completion": "收到\n",
|
| 371 |
+
"passed": true,
|
| 372 |
+
"failures": [],
|
| 373 |
+
"new_tokens": 4,
|
| 374 |
+
"elapsed_seconds": 0.8769729159539565
|
| 375 |
+
},
|
| 376 |
+
{
|
| 377 |
+
"id": "zh_09",
|
| 378 |
+
"language": "zh",
|
| 379 |
+
"prompt": "把苹果、香蕉和葡萄列成三个编号项目。",
|
| 380 |
+
"completion": "好的,以下是将苹果、香蕉和葡萄列成的三个编号项目:\n\n1. 苹果\n2. 香蕉\n3. 葡萄\n",
|
| 381 |
+
"passed": true,
|
| 382 |
+
"failures": [],
|
| 383 |
+
"new_tokens": 33,
|
| 384 |
+
"elapsed_seconds": 7.456039790995419
|
| 385 |
+
},
|
| 386 |
+
{
|
| 387 |
+
"id": "zh_10",
|
| 388 |
+
"language": "zh",
|
| 389 |
+
"prompt": "把“Good morning”翻译成自然的中文。",
|
| 390 |
+
"expected_any": [
|
| 391 |
+
"早上好",
|
| 392 |
+
"早安"
|
| 393 |
+
],
|
| 394 |
+
"completion": "早上好。\n",
|
| 395 |
+
"passed": true,
|
| 396 |
+
"failures": [],
|
| 397 |
+
"new_tokens": 5,
|
| 398 |
+
"elapsed_seconds": 1.1610058340011165
|
| 399 |
+
},
|
| 400 |
+
{
|
| 401 |
+
"id": "ja_01",
|
| 402 |
+
"language": "ja",
|
| 403 |
+
"prompt": "フランスの首都はどこですか。短く答えてください。",
|
| 404 |
+
"expected_any": [
|
| 405 |
+
"パリ"
|
| 406 |
+
],
|
| 407 |
+
"completion": "パリ\n",
|
| 408 |
+
"passed": true,
|
| 409 |
+
"failures": [],
|
| 410 |
+
"new_tokens": 4,
|
| 411 |
+
"elapsed_seconds": 0.9211049160221592
|
| 412 |
+
},
|
| 413 |
+
{
|
| 414 |
+
"id": "ja_02",
|
| 415 |
+
"language": "ja",
|
| 416 |
+
"prompt": "7かける8はいくつですか。数字で答えてください。",
|
| 417 |
+
"expected_any": [
|
| 418 |
+
"56"
|
| 419 |
+
],
|
| 420 |
+
"completion": "7 かける 8 は、$7 \\times 8 = 56$ です。\n\n答え:56\n",
|
| 421 |
+
"passed": true,
|
| 422 |
+
"failures": [],
|
| 423 |
+
"new_tokens": 28,
|
| 424 |
+
"elapsed_seconds": 6.337640833982732
|
| 425 |
+
},
|
| 426 |
+
{
|
| 427 |
+
"id": "ja_03",
|
| 428 |
+
"language": "ja",
|
| 429 |
+
"prompt": "水の化学式は何ですか。",
|
| 430 |
+
"expected_any": [
|
| 431 |
+
"H2O",
|
| 432 |
+
"h2o"
|
| 433 |
+
],
|
| 434 |
+
"completion": "水の化学式は**H₂O**です。\n\nこれは、水分子が「水素原子(H)」が 2 個、「酸素原子(O)」が 1 個から構成されていることを示しています。この",
|
| 435 |
+
"passed": true,
|
| 436 |
+
"failures": [],
|
| 437 |
+
"new_tokens": 48,
|
| 438 |
+
"elapsed_seconds": 10.600324957980774
|
| 439 |
+
},
|
| 440 |
+
{
|
| 441 |
+
"id": "ja_04",
|
| 442 |
+
"language": "ja",
|
| 443 |
+
"prompt": "太陽系で最も大きい惑星は何ですか。",
|
| 444 |
+
"expected_any": [
|
| 445 |
+
"木星"
|
| 446 |
+
],
|
| 447 |
+
"completion": "太陽系で最も大きい惑星は**木星**です。\n\n木星は太陽系の中心から約 778 万 km 離れた位置にあり、直径は約 140,000 km にも",
|
| 448 |
+
"passed": true,
|
| 449 |
+
"failures": [],
|
| 450 |
+
"new_tokens": 48,
|
| 451 |
+
"elapsed_seconds": 10.53782487497665
|
| 452 |
+
},
|
| 453 |
+
{
|
| 454 |
+
"id": "ja_05",
|
| 455 |
+
"language": "ja",
|
| 456 |
+
"prompt": "静かな冬の朝を自然な二文で描写してください。",
|
| 457 |
+
"completion": "静かな冬の朝は、窓ガラスにまだ霜が結ばれたままの空に、遠くから雪の音が響き渡ります。\n窓際で揺れる枯れ草の影が、薄明かりに",
|
| 458 |
+
"passed": true,
|
| 459 |
+
"failures": [],
|
| 460 |
+
"new_tokens": 48,
|
| 461 |
+
"elapsed_seconds": 10.687851083057467
|
| 462 |
+
},
|
| 463 |
+
{
|
| 464 |
+
"id": "ja_06",
|
| 465 |
+
"language": "ja",
|
| 466 |
+
"prompt": "友人に10分遅れることを丁寧に伝える一文を書いてください。",
|
| 467 |
+
"completion": "ご友人に10 分遅れることを丁寧に伝えるための例文をいくつか作成しました。状況や関係性に合わせて使い分けてください。\n\n### 1. 標準的・丁寧な表現\n最も汎用性が高く",
|
| 468 |
+
"passed": true,
|
| 469 |
+
"failures": [],
|
| 470 |
+
"new_tokens": 48,
|
| 471 |
+
"elapsed_seconds": 10.560357165988535
|
| 472 |
+
},
|
| 473 |
+
{
|
| 474 |
+
"id": "ja_07",
|
| 475 |
+
"language": "ja",
|
| 476 |
+
"prompt": "AI研究で再現性が重要な理由を二文で説明してください。",
|
| 477 |
+
"completion": "AI 研究における再現性の重要性は、まず研究結果の客観性と信頼性を担保するためです。\nこれにより、他の研究者が同じ条件下で再現し、新たな発見や検証が可能になるからです。\n",
|
| 478 |
+
"passed": true,
|
| 479 |
+
"failures": [],
|
| 480 |
+
"new_tokens": 46,
|
| 481 |
+
"elapsed_seconds": 10.073462790984195
|
| 482 |
+
},
|
| 483 |
+
{
|
| 484 |
+
"id": "ja_08",
|
| 485 |
+
"language": "ja",
|
| 486 |
+
"prompt": "ほかの言葉を加えず「了解」とだけ答えてください。",
|
| 487 |
+
"expected_any": [
|
| 488 |
+
"了解"
|
| 489 |
+
],
|
| 490 |
+
"completion": "了解\n",
|
| 491 |
+
"passed": true,
|
| 492 |
+
"failures": [],
|
| 493 |
+
"new_tokens": 4,
|
| 494 |
+
"elapsed_seconds": 0.9211295830318704
|
| 495 |
+
},
|
| 496 |
+
{
|
| 497 |
+
"id": "ja_09",
|
| 498 |
+
"language": "ja",
|
| 499 |
+
"prompt": "りんご、バナナ、ぶどうを番号付きの三項目で並べてください。",
|
| 500 |
+
"completion": "りんご、バナナ、ぶどうを番号付きの順に並べると、以下のようになります。\n\n1. りんご\n2. バナナ\n3. ぶどう\n",
|
| 501 |
+
"passed": true,
|
| 502 |
+
"failures": [],
|
| 503 |
+
"new_tokens": 43,
|
| 504 |
+
"elapsed_seconds": 9.689871000009589
|
| 505 |
+
},
|
| 506 |
+
{
|
| 507 |
+
"id": "ja_10",
|
| 508 |
+
"language": "ja",
|
| 509 |
+
"prompt": "「Good morning」を自然な日本語に訳してください。",
|
| 510 |
+
"expected_any": [
|
| 511 |
+
"おはよう"
|
| 512 |
+
],
|
| 513 |
+
"completion": "おはようございます。\n",
|
| 514 |
+
"passed": true,
|
| 515 |
+
"failures": [],
|
| 516 |
+
"new_tokens": 8,
|
| 517 |
+
"elapsed_seconds": 1.8597023330512457
|
| 518 |
+
},
|
| 519 |
+
{
|
| 520 |
+
"id": "es_01",
|
| 521 |
+
"language": "es",
|
| 522 |
+
"prompt": "¿Cuál es la capital de Francia? Responde brevemente.",
|
| 523 |
+
"expected_any": [
|
| 524 |
+
"París",
|
| 525 |
+
"Paris"
|
| 526 |
+
],
|
| 527 |
+
"completion": "La capital de Francia es **París**.\n",
|
| 528 |
+
"passed": true,
|
| 529 |
+
"failures": [],
|
| 530 |
+
"new_tokens": 12,
|
| 531 |
+
"elapsed_seconds": 2.6324190419982187
|
| 532 |
+
},
|
| 533 |
+
{
|
| 534 |
+
"id": "es_02",
|
| 535 |
+
"language": "es",
|
| 536 |
+
"prompt": "¿Cuánto es 7 por 8? Responde con un número.",
|
| 537 |
+
"expected_any": [
|
| 538 |
+
"56"
|
| 539 |
+
],
|
| 540 |
+
"completion": "56\n",
|
| 541 |
+
"passed": true,
|
| 542 |
+
"failures": [],
|
| 543 |
+
"new_tokens": 5,
|
| 544 |
+
"elapsed_seconds": 1.209535208006855
|
| 545 |
+
},
|
| 546 |
+
{
|
| 547 |
+
"id": "es_03",
|
| 548 |
+
"language": "es",
|
| 549 |
+
"prompt": "¿Cuál es la fórmula química del agua?",
|
| 550 |
+
"expected_any": [
|
| 551 |
+
"H2O",
|
| 552 |
+
"h2o"
|
| 553 |
+
],
|
| 554 |
+
"completion": "La fórmula química del agua es **H₂O**.\n\nEsto significa que una molécula de agua está compuesta por dos átomos de hidrógeno (H) unidos a un solo átomo de oxígeno (O",
|
| 555 |
+
"passed": true,
|
| 556 |
+
"failures": [],
|
| 557 |
+
"new_tokens": 48,
|
| 558 |
+
"elapsed_seconds": 10.877840041997842
|
| 559 |
+
},
|
| 560 |
+
{
|
| 561 |
+
"id": "es_04",
|
| 562 |
+
"language": "es",
|
| 563 |
+
"prompt": "¿Cuál es el planeta más grande del sistema solar?",
|
| 564 |
+
"expected_any": [
|
| 565 |
+
"Júpiter",
|
| 566 |
+
"Jupiter"
|
| 567 |
+
],
|
| 568 |
+
"completion": "El planeta más grande del sistema solar es **Júpiter**.\n\nEs un gigante gaseoso que tiene un diámetro de aproximadamente 139,820 kilómetros, lo que lo hace más grande que todos los demás",
|
| 569 |
+
"passed": true,
|
| 570 |
+
"failures": [],
|
| 571 |
+
"new_tokens": 48,
|
| 572 |
+
"elapsed_seconds": 10.883560375019442
|
| 573 |
+
},
|
| 574 |
+
{
|
| 575 |
+
"id": "es_05",
|
| 576 |
+
"language": "es",
|
| 577 |
+
"prompt": "Escribe dos frases naturales que describan una tranquila mañana de invierno.",
|
| 578 |
+
"completion": "La nieve cae suavemente sobre el tejado de mi casa, creando un silencio absoluto que se extiende por toda la ciudad.\nEl aire frío y seco se siente en la piel, pero el sol de invierno brilla con una",
|
| 579 |
+
"passed": true,
|
| 580 |
+
"failures": [],
|
| 581 |
+
"new_tokens": 48,
|
| 582 |
+
"elapsed_seconds": 10.86848220799584
|
| 583 |
+
},
|
| 584 |
+
{
|
| 585 |
+
"id": "es_06",
|
| 586 |
+
"language": "es",
|
| 587 |
+
"prompt": "Escribe una frase cortés para decirle a un amigo que llegarás diez minutos tarde.",
|
| 588 |
+
"completion": "¡Hola! Me alegra mucho que estés aquí, pero por suerte llegaré diez minutos tarde.\n",
|
| 589 |
+
"passed": true,
|
| 590 |
+
"failures": [],
|
| 591 |
+
"new_tokens": 24,
|
| 592 |
+
"elapsed_seconds": 5.387403959000949
|
| 593 |
+
},
|
| 594 |
+
{
|
| 595 |
+
"id": "es_07",
|
| 596 |
+
"language": "es",
|
| 597 |
+
"prompt": "Explica en dos frases por qué la reproducibilidad importa en la investigación de IA.",
|
| 598 |
+
"completion": "La reproducibilidad es fundamental porque garantiza que los resultados obtenidos mediante modelos de IA sean verificados y replicados por otros investigadores, lo que valida la solidez de las conclusiones. Además, fomenta la transparencia y la confianza en",
|
| 599 |
+
"passed": true,
|
| 600 |
+
"failures": [],
|
| 601 |
+
"new_tokens": 48,
|
| 602 |
+
"elapsed_seconds": 10.632735249993857
|
| 603 |
+
},
|
| 604 |
+
{
|
| 605 |
+
"id": "es_08",
|
| 606 |
+
"language": "es",
|
| 607 |
+
"prompt": "Responde únicamente con la palabra 'entendido'.",
|
| 608 |
+
"expected_any": [
|
| 609 |
+
"entendido"
|
| 610 |
+
],
|
| 611 |
+
"completion": "entendido\n",
|
| 612 |
+
"passed": true,
|
| 613 |
+
"failures": [],
|
| 614 |
+
"new_tokens": 5,
|
| 615 |
+
"elapsed_seconds": 1.210517042025458
|
| 616 |
+
},
|
| 617 |
+
{
|
| 618 |
+
"id": "es_09",
|
| 619 |
+
"language": "es",
|
| 620 |
+
"prompt": "Enumera manzana, plátano y uva como tres elementos numerados.",
|
| 621 |
+
"completion": "Aquí tienes la lista de los tres elementos numerados:\n\n1. Manzana\n2. Plátano\n3. Uva\n",
|
| 622 |
+
"passed": true,
|
| 623 |
+
"failures": [],
|
| 624 |
+
"new_tokens": 31,
|
| 625 |
+
"elapsed_seconds": 7.097092125040945
|
| 626 |
+
},
|
| 627 |
+
{
|
| 628 |
+
"id": "es_10",
|
| 629 |
+
"language": "es",
|
| 630 |
+
"prompt": "Traduce 'Good morning' a un español natural.",
|
| 631 |
+
"expected_any": [
|
| 632 |
+
"buenos días",
|
| 633 |
+
"buen día"
|
| 634 |
+
],
|
| 635 |
+
"completion": "Buenos días.\n",
|
| 636 |
+
"passed": true,
|
| 637 |
+
"failures": [],
|
| 638 |
+
"new_tokens": 8,
|
| 639 |
+
"elapsed_seconds": 1.8161898750113323
|
| 640 |
+
},
|
| 641 |
+
{
|
| 642 |
+
"id": "de_01",
|
| 643 |
+
"language": "de",
|
| 644 |
+
"prompt": "Was ist die Hauptstadt von Frankreich? Antworte kurz.",
|
| 645 |
+
"expected_any": [
|
| 646 |
+
"Paris"
|
| 647 |
+
],
|
| 648 |
+
"completion": "Die Hauptstadt von Frankreich ist Paris.\n",
|
| 649 |
+
"passed": true,
|
| 650 |
+
"failures": [],
|
| 651 |
+
"new_tokens": 10,
|
| 652 |
+
"elapsed_seconds": 2.182326541049406
|
| 653 |
+
},
|
| 654 |
+
{
|
| 655 |
+
"id": "de_02",
|
| 656 |
+
"language": "de",
|
| 657 |
+
"prompt": "Was ist 7 mal 8? Antworte mit einer Zahl.",
|
| 658 |
+
"expected_any": [
|
| 659 |
+
"56"
|
| 660 |
+
],
|
| 661 |
+
"completion": "56\n",
|
| 662 |
+
"passed": true,
|
| 663 |
+
"failures": [],
|
| 664 |
+
"new_tokens": 5,
|
| 665 |
+
"elapsed_seconds": 1.1318954160087742
|
| 666 |
+
},
|
| 667 |
+
{
|
| 668 |
+
"id": "de_03",
|
| 669 |
+
"language": "de",
|
| 670 |
+
"prompt": "Wie lautet die chemische Formel für Wasser?",
|
| 671 |
+
"expected_any": [
|
| 672 |
+
"H2O",
|
| 673 |
+
"h2o"
|
| 674 |
+
],
|
| 675 |
+
"completion": "Die chemische Formel für Wasser lautet **H₂O**.\n\nDies bedeutet, dass ein Wassermolekül aus zwei Atomen des Elements Wasserstoff (H) und einem Atom des Elements Sauerstoff (O) besteht. Die",
|
| 676 |
+
"passed": true,
|
| 677 |
+
"failures": [],
|
| 678 |
+
"new_tokens": 48,
|
| 679 |
+
"elapsed_seconds": 10.83673787501175
|
| 680 |
+
},
|
| 681 |
+
{
|
| 682 |
+
"id": "de_04",
|
| 683 |
+
"language": "de",
|
| 684 |
+
"prompt": "Welcher Planet ist der größte im Sonnensystem?",
|
| 685 |
+
"expected_any": [
|
| 686 |
+
"Jupiter"
|
| 687 |
+
],
|
| 688 |
+
"completion": "Der Planet, der den größten Durchmesser und die größte Masse im Sonnensystem hat, ist **Jupiter**.\n\nEr ist ein Gasriesen, der etwa 11,2 Mal so groß wie die Erde ist und etwa 3",
|
| 689 |
+
"passed": true,
|
| 690 |
+
"failures": [],
|
| 691 |
+
"new_tokens": 48,
|
| 692 |
+
"elapsed_seconds": 10.766112500044983
|
| 693 |
+
},
|
| 694 |
+
{
|
| 695 |
+
"id": "de_05",
|
| 696 |
+
"language": "de",
|
| 697 |
+
"prompt": "Schreibe zwei natürliche Sätze über einen ruhigen Wintermorgen.",
|
| 698 |
+
"completion": "Der Wintermorgen ist so still, dass man kaum noch die eigenen Atemzüge hört, während der Schnee sanft über die Fensterläden fällt.\nDie Sonne scheint durch die dichten Wolken nur als ein warmes, goldenes",
|
| 699 |
+
"passed": true,
|
| 700 |
+
"failures": [],
|
| 701 |
+
"new_tokens": 48,
|
| 702 |
+
"elapsed_seconds": 10.828977917029988
|
| 703 |
+
},
|
| 704 |
+
{
|
| 705 |
+
"id": "de_06",
|
| 706 |
+
"language": "de",
|
| 707 |
+
"prompt": "Schreibe einen höflichen Satz, der einem Freund sagt, dass du zehn Minuten zu spät kommst.",
|
| 708 |
+
"completion": "\"Hey, ich bin ein bisschen zu spät, aber ich hoffe, du kannst mir trotzdem helfen, wenn du Zeit hast!\"\n",
|
| 709 |
+
"passed": true,
|
| 710 |
+
"failures": [],
|
| 711 |
+
"new_tokens": 28,
|
| 712 |
+
"elapsed_seconds": 6.2108728750026785
|
| 713 |
+
},
|
| 714 |
+
{
|
| 715 |
+
"id": "de_07",
|
| 716 |
+
"language": "de",
|
| 717 |
+
"prompt": "Erkläre in zwei Sätzen, warum Reproduzierbarkeit in der KI-Forschung wichtig ist.",
|
| 718 |
+
"completion": "Reproduzierbarkeit ist in der KI-Forschung entscheidend, um die Ergebnisse von Algorithmen nicht nur wissenschaftlich validieren zu können, sondern auch als vertrauenswürdige Vorhersagemodelle für die reale Welt einzuset",
|
| 719 |
+
"passed": true,
|
| 720 |
+
"failures": [],
|
| 721 |
+
"new_tokens": 48,
|
| 722 |
+
"elapsed_seconds": 10.962672458030283
|
| 723 |
+
},
|
| 724 |
+
{
|
| 725 |
+
"id": "de_08",
|
| 726 |
+
"language": "de",
|
| 727 |
+
"prompt": "Antworte ausschließlich mit dem Wort 'Verstanden'.",
|
| 728 |
+
"expected_any": [
|
| 729 |
+
"verstanden"
|
| 730 |
+
],
|
| 731 |
+
"completion": "Verstanden\n",
|
| 732 |
+
"passed": true,
|
| 733 |
+
"failures": [],
|
| 734 |
+
"new_tokens": 5,
|
| 735 |
+
"elapsed_seconds": 1.1044922079890966
|
| 736 |
+
},
|
| 737 |
+
{
|
| 738 |
+
"id": "de_09",
|
| 739 |
+
"language": "de",
|
| 740 |
+
"prompt": "Liste Apfel, Banane und Traube als drei nummerierte Punkte auf.",
|
| 741 |
+
"completion": "Hier ist die Liste der drei Früchte als nummerierte Punkte:\n\n1. Apfel\n2. Banane\n3. Traube\n",
|
| 742 |
+
"passed": true,
|
| 743 |
+
"failures": [],
|
| 744 |
+
"new_tokens": 30,
|
| 745 |
+
"elapsed_seconds": 6.85629908303963
|
| 746 |
+
},
|
| 747 |
+
{
|
| 748 |
+
"id": "de_10",
|
| 749 |
+
"language": "de",
|
| 750 |
+
"prompt": "Übersetze 'Good morning' in natürliches Deutsch.",
|
| 751 |
+
"expected_any": [
|
| 752 |
+
"guten morgen"
|
| 753 |
+
],
|
| 754 |
+
"completion": "Guten Morgen!\n",
|
| 755 |
+
"passed": true,
|
| 756 |
+
"failures": [],
|
| 757 |
+
"new_tokens": 7,
|
| 758 |
+
"elapsed_seconds": 1.618866834032815
|
| 759 |
+
}
|
| 760 |
+
],
|
| 761 |
+
"rescored_from": "research/qwen35-moe-a3b/runs/chat-gate/v2-60.json"
|
| 762 |
+
}
|
evaluation/lm_eval_teacher4b_dev100.json
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"hellaswag": {
|
| 4 |
+
"name": "hellaswag",
|
| 5 |
+
"alias": "hellaswag",
|
| 6 |
+
"sample_len": 100,
|
| 7 |
+
"acc,none": 0.51,
|
| 8 |
+
"acc_stderr,none": 0.05024183937956913,
|
| 9 |
+
"acc_norm,none": 0.68,
|
| 10 |
+
"acc_norm_stderr,none": 0.046882617226215076
|
| 11 |
+
},
|
| 12 |
+
"arc_easy": {
|
| 13 |
+
"name": "arc_easy",
|
| 14 |
+
"alias": "arc_easy",
|
| 15 |
+
"sample_len": 100,
|
| 16 |
+
"acc,none": 0.8,
|
| 17 |
+
"acc_stderr,none": 0.04020151261036849,
|
| 18 |
+
"acc_norm,none": 0.81,
|
| 19 |
+
"acc_norm_stderr,none": 0.039427724440366255
|
| 20 |
+
}
|
| 21 |
+
},
|
| 22 |
+
"group_subtasks": {},
|
| 23 |
+
"configs": {
|
| 24 |
+
"arc_easy": {
|
| 25 |
+
"task": "arc_easy",
|
| 26 |
+
"dataset_path": "allenai/ai2_arc",
|
| 27 |
+
"dataset_name": "ARC-Easy",
|
| 28 |
+
"training_split": "train",
|
| 29 |
+
"validation_split": "validation",
|
| 30 |
+
"test_split": "test",
|
| 31 |
+
"doc_to_text": "Question: {{question}}\nAnswer:",
|
| 32 |
+
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
| 33 |
+
"unsafe_code": false,
|
| 34 |
+
"doc_to_choice": "{{choices.text}}",
|
| 35 |
+
"description": "",
|
| 36 |
+
"target_delimiter": " ",
|
| 37 |
+
"fewshot_delimiter": "\n\n",
|
| 38 |
+
"fewshot_config": {
|
| 39 |
+
"sampler": "default",
|
| 40 |
+
"split": null,
|
| 41 |
+
"process_docs": null,
|
| 42 |
+
"fewshot_indices": null,
|
| 43 |
+
"samples": null,
|
| 44 |
+
"doc_to_text": "Question: {{question}}\nAnswer:",
|
| 45 |
+
"doc_to_choice": "{{choices.text}}",
|
| 46 |
+
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
| 47 |
+
"gen_prefix": null,
|
| 48 |
+
"fewshot_delimiter": "\n\n",
|
| 49 |
+
"target_delimiter": " "
|
| 50 |
+
},
|
| 51 |
+
"num_fewshot": 0,
|
| 52 |
+
"metric_list": [
|
| 53 |
+
{
|
| 54 |
+
"metric": "acc",
|
| 55 |
+
"aggregation": "mean",
|
| 56 |
+
"higher_is_better": true
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"metric": "acc_norm",
|
| 60 |
+
"aggregation": "mean",
|
| 61 |
+
"higher_is_better": true
|
| 62 |
+
}
|
| 63 |
+
],
|
| 64 |
+
"output_type": "multiple_choice",
|
| 65 |
+
"repeats": 1,
|
| 66 |
+
"should_decontaminate": true,
|
| 67 |
+
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
|
| 68 |
+
"metadata": {
|
| 69 |
+
"version": 1.0,
|
| 70 |
+
"pretrained": "models/Qwen/Qwen3.5-4B",
|
| 71 |
+
"dtype": "bfloat16",
|
| 72 |
+
"config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/arc/arc_easy.yaml"
|
| 73 |
+
}
|
| 74 |
+
},
|
| 75 |
+
"hellaswag": {
|
| 76 |
+
"task": "hellaswag",
|
| 77 |
+
"dataset_path": "Rowan/hellaswag",
|
| 78 |
+
"training_split": "train",
|
| 79 |
+
"validation_split": "validation",
|
| 80 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 81 |
+
"doc_to_text": "{{query}}",
|
| 82 |
+
"doc_to_target": "{{label}}",
|
| 83 |
+
"unsafe_code": false,
|
| 84 |
+
"doc_to_choice": "choices",
|
| 85 |
+
"description": "",
|
| 86 |
+
"target_delimiter": " ",
|
| 87 |
+
"fewshot_delimiter": "\n\n",
|
| 88 |
+
"fewshot_config": {
|
| 89 |
+
"sampler": "default",
|
| 90 |
+
"split": null,
|
| 91 |
+
"process_docs": "<function process_docs at 0x13ebcfd70>",
|
| 92 |
+
"fewshot_indices": null,
|
| 93 |
+
"samples": null,
|
| 94 |
+
"doc_to_text": "{{query}}",
|
| 95 |
+
"doc_to_choice": "choices",
|
| 96 |
+
"doc_to_target": "{{label}}",
|
| 97 |
+
"gen_prefix": null,
|
| 98 |
+
"fewshot_delimiter": "\n\n",
|
| 99 |
+
"target_delimiter": " "
|
| 100 |
+
},
|
| 101 |
+
"num_fewshot": 0,
|
| 102 |
+
"metric_list": [
|
| 103 |
+
{
|
| 104 |
+
"metric": "acc",
|
| 105 |
+
"aggregation": "mean",
|
| 106 |
+
"higher_is_better": true
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"metric": "acc_norm",
|
| 110 |
+
"aggregation": "mean",
|
| 111 |
+
"higher_is_better": true
|
| 112 |
+
}
|
| 113 |
+
],
|
| 114 |
+
"output_type": "multiple_choice",
|
| 115 |
+
"repeats": 1,
|
| 116 |
+
"should_decontaminate": false,
|
| 117 |
+
"metadata": {
|
| 118 |
+
"version": 1.0,
|
| 119 |
+
"pretrained": "models/Qwen/Qwen3.5-4B",
|
| 120 |
+
"dtype": "bfloat16",
|
| 121 |
+
"config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml"
|
| 122 |
+
}
|
| 123 |
+
}
|
| 124 |
+
},
|
| 125 |
+
"versions": {
|
| 126 |
+
"arc_easy": 1.0,
|
| 127 |
+
"hellaswag": 1.0
|
| 128 |
+
},
|
| 129 |
+
"n-shot": {
|
| 130 |
+
"arc_easy": 0,
|
| 131 |
+
"hellaswag": 0
|
| 132 |
+
},
|
| 133 |
+
"higher_is_better": {
|
| 134 |
+
"arc_easy": {
|
| 135 |
+
"acc": true,
|
| 136 |
+
"acc_norm": true
|
| 137 |
+
},
|
| 138 |
+
"hellaswag": {
|
| 139 |
+
"acc": true,
|
| 140 |
+
"acc_norm": true
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"n-samples": {
|
| 144 |
+
"hellaswag": {
|
| 145 |
+
"original": 10042,
|
| 146 |
+
"effective": 100
|
| 147 |
+
},
|
| 148 |
+
"arc_easy": {
|
| 149 |
+
"original": 2376,
|
| 150 |
+
"effective": 100
|
| 151 |
+
}
|
| 152 |
+
},
|
| 153 |
+
"config": {
|
| 154 |
+
"model": "hf",
|
| 155 |
+
"model_args": {
|
| 156 |
+
"pretrained": "models/Qwen/Qwen3.5-4B",
|
| 157 |
+
"dtype": "bfloat16"
|
| 158 |
+
},
|
| 159 |
+
"model_num_parameters": 4205751296,
|
| 160 |
+
"model_dtype": "torch.bfloat16",
|
| 161 |
+
"model_revision": "main",
|
| 162 |
+
"model_sha": "",
|
| 163 |
+
"batch_size": "1",
|
| 164 |
+
"batch_sizes": [],
|
| 165 |
+
"device": "mps",
|
| 166 |
+
"use_cache": null,
|
| 167 |
+
"limit": 100.0,
|
| 168 |
+
"bootstrap_iters": 100000,
|
| 169 |
+
"gen_kwargs": {},
|
| 170 |
+
"random_seed": 0,
|
| 171 |
+
"numpy_seed": 1234,
|
| 172 |
+
"torch_seed": 1234,
|
| 173 |
+
"fewshot_seed": 1234
|
| 174 |
+
},
|
| 175 |
+
"git_hash": null,
|
| 176 |
+
"date": 1787550063.559031,
|
| 177 |
+
"pretty_env_info": "PyTorch version: 2.11.0\nIs debug build: False\nCUDA used to build PyTorch: None\nROCM used to build PyTorch: N/A\n\nOS: macOS 26.4.1 (arm64)\nGCC version: Could not collect\nClang version: 21.0.0 (clang-2100.0.123.102)\nCMake version: version 4.4.0\nLibc version: N/A\n\nPython version: 3.14.6 (main, Jun 10 2026, 10:03:53) [Clang 21.0.0 (clang-2100.0.123.102)] (64-bit runtime)\nPython platform: macOS-26.4.1-arm64-arm-64bit-Mach-O\nIs CUDA available: False\nCUDA runtime version: No CUDA\nCUDA_MODULE_LOADING set to: N/A\nGPU models and configuration: No CUDA\nNvidia driver version: No CUDA\ncuDNN version: No CUDA\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nApple M4\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.4\n[pip3] torch==2.11.0\n[conda] Could not collect",
|
| 178 |
+
"transformers_version": "5.13.0",
|
| 179 |
+
"lm_eval_version": "0.4.12",
|
| 180 |
+
"upper_git_hash": null,
|
| 181 |
+
"tokenizer_pad_token": [
|
| 182 |
+
"<|endoftext|>",
|
| 183 |
+
"248044"
|
| 184 |
+
],
|
| 185 |
+
"tokenizer_eos_token": [
|
| 186 |
+
"<|im_end|>",
|
| 187 |
+
"248046"
|
| 188 |
+
],
|
| 189 |
+
"tokenizer_bos_token": [
|
| 190 |
+
null,
|
| 191 |
+
"None"
|
| 192 |
+
],
|
| 193 |
+
"eot_token_id": 248046,
|
| 194 |
+
"max_length": 262144,
|
| 195 |
+
"task_hashes": {
|
| 196 |
+
"hellaswag": "4f7d86a1e256013e93651a0d8510973180c5f31142ca184d9677955a29676af1",
|
| 197 |
+
"arc_easy": "fd3a493579cfdccf229a32d7c26710006fc4edf7c024add74f70c416c6dab3ce"
|
| 198 |
+
},
|
| 199 |
+
"model_source": "hf",
|
| 200 |
+
"model_name": "models/Qwen/Qwen3.5-4B",
|
| 201 |
+
"model_name_sanitized": "models__Qwen__Qwen3.5-4B",
|
| 202 |
+
"system_instruction": null,
|
| 203 |
+
"system_instruction_sha": null,
|
| 204 |
+
"fewshot_as_multiturn": null,
|
| 205 |
+
"chat_template": null,
|
| 206 |
+
"chat_template_sha": null,
|
| 207 |
+
"total_evaluation_time_seconds": "257.4533243330079"
|
| 208 |
+
}
|
evaluation/lm_eval_v2_dev100.json
ADDED
|
@@ -0,0 +1,208 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"results": {
|
| 3 |
+
"hellaswag": {
|
| 4 |
+
"name": "hellaswag",
|
| 5 |
+
"alias": "hellaswag",
|
| 6 |
+
"sample_len": 100,
|
| 7 |
+
"acc,none": 0.44,
|
| 8 |
+
"acc_stderr,none": 0.049888765156985884,
|
| 9 |
+
"acc_norm,none": 0.62,
|
| 10 |
+
"acc_norm_stderr,none": 0.04878317312145634
|
| 11 |
+
},
|
| 12 |
+
"arc_easy": {
|
| 13 |
+
"name": "arc_easy",
|
| 14 |
+
"alias": "arc_easy",
|
| 15 |
+
"sample_len": 100,
|
| 16 |
+
"acc,none": 0.71,
|
| 17 |
+
"acc_stderr,none": 0.045604802157206865,
|
| 18 |
+
"acc_norm,none": 0.73,
|
| 19 |
+
"acc_norm_stderr,none": 0.04461960433384737
|
| 20 |
+
}
|
| 21 |
+
},
|
| 22 |
+
"group_subtasks": {},
|
| 23 |
+
"configs": {
|
| 24 |
+
"arc_easy": {
|
| 25 |
+
"task": "arc_easy",
|
| 26 |
+
"dataset_path": "allenai/ai2_arc",
|
| 27 |
+
"dataset_name": "ARC-Easy",
|
| 28 |
+
"training_split": "train",
|
| 29 |
+
"validation_split": "validation",
|
| 30 |
+
"test_split": "test",
|
| 31 |
+
"doc_to_text": "Question: {{question}}\nAnswer:",
|
| 32 |
+
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
| 33 |
+
"unsafe_code": false,
|
| 34 |
+
"doc_to_choice": "{{choices.text}}",
|
| 35 |
+
"description": "",
|
| 36 |
+
"target_delimiter": " ",
|
| 37 |
+
"fewshot_delimiter": "\n\n",
|
| 38 |
+
"fewshot_config": {
|
| 39 |
+
"sampler": "default",
|
| 40 |
+
"split": null,
|
| 41 |
+
"process_docs": null,
|
| 42 |
+
"fewshot_indices": null,
|
| 43 |
+
"samples": null,
|
| 44 |
+
"doc_to_text": "Question: {{question}}\nAnswer:",
|
| 45 |
+
"doc_to_choice": "{{choices.text}}",
|
| 46 |
+
"doc_to_target": "{{choices.label.index(answerKey)}}",
|
| 47 |
+
"gen_prefix": null,
|
| 48 |
+
"fewshot_delimiter": "\n\n",
|
| 49 |
+
"target_delimiter": " "
|
| 50 |
+
},
|
| 51 |
+
"num_fewshot": 0,
|
| 52 |
+
"metric_list": [
|
| 53 |
+
{
|
| 54 |
+
"metric": "acc",
|
| 55 |
+
"aggregation": "mean",
|
| 56 |
+
"higher_is_better": true
|
| 57 |
+
},
|
| 58 |
+
{
|
| 59 |
+
"metric": "acc_norm",
|
| 60 |
+
"aggregation": "mean",
|
| 61 |
+
"higher_is_better": true
|
| 62 |
+
}
|
| 63 |
+
],
|
| 64 |
+
"output_type": "multiple_choice",
|
| 65 |
+
"repeats": 1,
|
| 66 |
+
"should_decontaminate": true,
|
| 67 |
+
"doc_to_decontamination_query": "Question: {{question}}\nAnswer:",
|
| 68 |
+
"metadata": {
|
| 69 |
+
"version": 1.0,
|
| 70 |
+
"pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 71 |
+
"dtype": "bfloat16",
|
| 72 |
+
"config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/arc/arc_easy.yaml"
|
| 73 |
+
}
|
| 74 |
+
},
|
| 75 |
+
"hellaswag": {
|
| 76 |
+
"task": "hellaswag",
|
| 77 |
+
"dataset_path": "Rowan/hellaswag",
|
| 78 |
+
"training_split": "train",
|
| 79 |
+
"validation_split": "validation",
|
| 80 |
+
"process_docs": "def process_docs(dataset: datasets.Dataset) -> datasets.Dataset:\n def _process_doc(doc):\n ctx = doc[\"ctx_a\"] + \" \" + doc[\"ctx_b\"].capitalize()\n out_doc = {\n \"query\": preprocess(doc[\"activity_label\"] + \": \" + ctx),\n \"choices\": [preprocess(ending) for ending in doc[\"endings\"]],\n \"gold\": int(doc[\"label\"]),\n }\n return out_doc\n\n return dataset.map(_process_doc)\n",
|
| 81 |
+
"doc_to_text": "{{query}}",
|
| 82 |
+
"doc_to_target": "{{label}}",
|
| 83 |
+
"unsafe_code": false,
|
| 84 |
+
"doc_to_choice": "choices",
|
| 85 |
+
"description": "",
|
| 86 |
+
"target_delimiter": " ",
|
| 87 |
+
"fewshot_delimiter": "\n\n",
|
| 88 |
+
"fewshot_config": {
|
| 89 |
+
"sampler": "default",
|
| 90 |
+
"split": null,
|
| 91 |
+
"process_docs": "<function process_docs at 0x174a5fed0>",
|
| 92 |
+
"fewshot_indices": null,
|
| 93 |
+
"samples": null,
|
| 94 |
+
"doc_to_text": "{{query}}",
|
| 95 |
+
"doc_to_choice": "choices",
|
| 96 |
+
"doc_to_target": "{{label}}",
|
| 97 |
+
"gen_prefix": null,
|
| 98 |
+
"fewshot_delimiter": "\n\n",
|
| 99 |
+
"target_delimiter": " "
|
| 100 |
+
},
|
| 101 |
+
"num_fewshot": 0,
|
| 102 |
+
"metric_list": [
|
| 103 |
+
{
|
| 104 |
+
"metric": "acc",
|
| 105 |
+
"aggregation": "mean",
|
| 106 |
+
"higher_is_better": true
|
| 107 |
+
},
|
| 108 |
+
{
|
| 109 |
+
"metric": "acc_norm",
|
| 110 |
+
"aggregation": "mean",
|
| 111 |
+
"higher_is_better": true
|
| 112 |
+
}
|
| 113 |
+
],
|
| 114 |
+
"output_type": "multiple_choice",
|
| 115 |
+
"repeats": 1,
|
| 116 |
+
"should_decontaminate": false,
|
| 117 |
+
"metadata": {
|
| 118 |
+
"version": 1.0,
|
| 119 |
+
"pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 120 |
+
"dtype": "bfloat16",
|
| 121 |
+
"config_source": "/Volumes/외장디스크1/오픈소스 모델/research/qwen35-moe-a3b/.venv-eval/lib/python3.14/site-packages/lm_eval/tasks/hellaswag/hellaswag.yaml"
|
| 122 |
+
}
|
| 123 |
+
}
|
| 124 |
+
},
|
| 125 |
+
"versions": {
|
| 126 |
+
"arc_easy": 1.0,
|
| 127 |
+
"hellaswag": 1.0
|
| 128 |
+
},
|
| 129 |
+
"n-shot": {
|
| 130 |
+
"arc_easy": 0,
|
| 131 |
+
"hellaswag": 0
|
| 132 |
+
},
|
| 133 |
+
"higher_is_better": {
|
| 134 |
+
"arc_easy": {
|
| 135 |
+
"acc": true,
|
| 136 |
+
"acc_norm": true
|
| 137 |
+
},
|
| 138 |
+
"hellaswag": {
|
| 139 |
+
"acc": true,
|
| 140 |
+
"acc_norm": true
|
| 141 |
+
}
|
| 142 |
+
},
|
| 143 |
+
"n-samples": {
|
| 144 |
+
"hellaswag": {
|
| 145 |
+
"original": 10042,
|
| 146 |
+
"effective": 100
|
| 147 |
+
},
|
| 148 |
+
"arc_easy": {
|
| 149 |
+
"original": 2376,
|
| 150 |
+
"effective": 100
|
| 151 |
+
}
|
| 152 |
+
},
|
| 153 |
+
"config": {
|
| 154 |
+
"model": "hf",
|
| 155 |
+
"model_args": {
|
| 156 |
+
"pretrained": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 157 |
+
"dtype": "bfloat16"
|
| 158 |
+
},
|
| 159 |
+
"model_num_parameters": 3995901760,
|
| 160 |
+
"model_dtype": "torch.bfloat16",
|
| 161 |
+
"model_revision": "main",
|
| 162 |
+
"model_sha": "",
|
| 163 |
+
"batch_size": "1",
|
| 164 |
+
"batch_sizes": [],
|
| 165 |
+
"device": "mps",
|
| 166 |
+
"use_cache": null,
|
| 167 |
+
"limit": 100.0,
|
| 168 |
+
"bootstrap_iters": 100000,
|
| 169 |
+
"gen_kwargs": {},
|
| 170 |
+
"random_seed": 0,
|
| 171 |
+
"numpy_seed": 1234,
|
| 172 |
+
"torch_seed": 1234,
|
| 173 |
+
"fewshot_seed": 1234
|
| 174 |
+
},
|
| 175 |
+
"git_hash": null,
|
| 176 |
+
"date": 1787549803.098937,
|
| 177 |
+
"pretty_env_info": "PyTorch version: 2.11.0\nIs debug build: False\nCUDA used to build PyTorch: None\nROCM used to build PyTorch: N/A\n\nOS: macOS 26.4.1 (arm64)\nGCC version: Could not collect\nClang version: 21.0.0 (clang-2100.0.123.102)\nCMake version: version 4.4.0\nLibc version: N/A\n\nPython version: 3.14.6 (main, Jun 10 2026, 10:03:53) [Clang 21.0.0 (clang-2100.0.123.102)] (64-bit runtime)\nPython platform: macOS-26.4.1-arm64-arm-64bit-Mach-O\nIs CUDA available: False\nCUDA runtime version: No CUDA\nCUDA_MODULE_LOADING set to: N/A\nGPU models and configuration: No CUDA\nNvidia driver version: No CUDA\ncuDNN version: No CUDA\nIs XPU available: False\nHIP runtime version: N/A\nMIOpen runtime version: N/A\nIs XNNPACK available: True\nCaching allocator config: N/A\n\nCPU:\nApple M4\n\nVersions of relevant libraries:\n[pip3] numpy==2.4.4\n[pip3] torch==2.11.0\n[conda] Could not collect",
|
| 178 |
+
"transformers_version": "5.13.0",
|
| 179 |
+
"lm_eval_version": "0.4.12",
|
| 180 |
+
"upper_git_hash": null,
|
| 181 |
+
"tokenizer_pad_token": [
|
| 182 |
+
"<|endoftext|>",
|
| 183 |
+
"248044"
|
| 184 |
+
],
|
| 185 |
+
"tokenizer_eos_token": [
|
| 186 |
+
"<|im_end|>",
|
| 187 |
+
"248046"
|
| 188 |
+
],
|
| 189 |
+
"tokenizer_bos_token": [
|
| 190 |
+
null,
|
| 191 |
+
"None"
|
| 192 |
+
],
|
| 193 |
+
"eot_token_id": 248046,
|
| 194 |
+
"max_length": 262144,
|
| 195 |
+
"task_hashes": {
|
| 196 |
+
"hellaswag": "4f7d86a1e256013e93651a0d8510973180c5f31142ca184d9677955a29676af1",
|
| 197 |
+
"arc_easy": "fd3a493579cfdccf229a32d7c26710006fc4edf7c024add74f70c416c6dab3ce"
|
| 198 |
+
},
|
| 199 |
+
"model_source": "hf",
|
| 200 |
+
"model_name": "models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 201 |
+
"model_name_sanitized": "models__Qwen__Qwen3.5-4B-A3B-Student-v2",
|
| 202 |
+
"system_instruction": null,
|
| 203 |
+
"system_instruction_sha": null,
|
| 204 |
+
"fewshot_as_multiturn": null,
|
| 205 |
+
"chat_template": null,
|
| 206 |
+
"chat_template_sha": null,
|
| 207 |
+
"total_evaluation_time_seconds": "244.92804725002497"
|
| 208 |
+
}
|
evaluation/multilingual_lm_loss.json
ADDED
|
@@ -0,0 +1,115 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"sequence_length": 128,
|
| 3 |
+
"documents_per_language": 4,
|
| 4 |
+
"languages": [
|
| 5 |
+
"de",
|
| 6 |
+
"en",
|
| 7 |
+
"es",
|
| 8 |
+
"ja",
|
| 9 |
+
"ko",
|
| 10 |
+
"zh"
|
| 11 |
+
],
|
| 12 |
+
"models": {
|
| 13 |
+
"qwen35_2b": {
|
| 14 |
+
"ko": {
|
| 15 |
+
"loss": 3.2937907576560974,
|
| 16 |
+
"documents": 4,
|
| 17 |
+
"tokens": 512
|
| 18 |
+
},
|
| 19 |
+
"en": {
|
| 20 |
+
"loss": 2.9049559831619263,
|
| 21 |
+
"documents": 4,
|
| 22 |
+
"tokens": 512
|
| 23 |
+
},
|
| 24 |
+
"zh": {
|
| 25 |
+
"loss": 3.7243993878364563,
|
| 26 |
+
"documents": 4,
|
| 27 |
+
"tokens": 512
|
| 28 |
+
},
|
| 29 |
+
"ja": {
|
| 30 |
+
"loss": 3.160854697227478,
|
| 31 |
+
"documents": 4,
|
| 32 |
+
"tokens": 512
|
| 33 |
+
},
|
| 34 |
+
"es": {
|
| 35 |
+
"loss": 2.9567737579345703,
|
| 36 |
+
"documents": 4,
|
| 37 |
+
"tokens": 512
|
| 38 |
+
},
|
| 39 |
+
"de": {
|
| 40 |
+
"loss": 3.21018385887146,
|
| 41 |
+
"documents": 4,
|
| 42 |
+
"tokens": 512
|
| 43 |
+
}
|
| 44 |
+
},
|
| 45 |
+
"a3b_v2": {
|
| 46 |
+
"ko": {
|
| 47 |
+
"loss": 3.2937907576560974,
|
| 48 |
+
"documents": 4,
|
| 49 |
+
"tokens": 512
|
| 50 |
+
},
|
| 51 |
+
"en": {
|
| 52 |
+
"loss": 2.9049559831619263,
|
| 53 |
+
"documents": 4,
|
| 54 |
+
"tokens": 512
|
| 55 |
+
},
|
| 56 |
+
"zh": {
|
| 57 |
+
"loss": 3.7243993878364563,
|
| 58 |
+
"documents": 4,
|
| 59 |
+
"tokens": 512
|
| 60 |
+
},
|
| 61 |
+
"ja": {
|
| 62 |
+
"loss": 3.160854697227478,
|
| 63 |
+
"documents": 4,
|
| 64 |
+
"tokens": 512
|
| 65 |
+
},
|
| 66 |
+
"es": {
|
| 67 |
+
"loss": 2.9567737579345703,
|
| 68 |
+
"documents": 4,
|
| 69 |
+
"tokens": 512
|
| 70 |
+
},
|
| 71 |
+
"de": {
|
| 72 |
+
"loss": 3.21018385887146,
|
| 73 |
+
"documents": 4,
|
| 74 |
+
"tokens": 512
|
| 75 |
+
}
|
| 76 |
+
},
|
| 77 |
+
"qwen35_4b": {
|
| 78 |
+
"ko": {
|
| 79 |
+
"loss": 2.9032246470451355,
|
| 80 |
+
"documents": 4,
|
| 81 |
+
"tokens": 512
|
| 82 |
+
},
|
| 83 |
+
"en": {
|
| 84 |
+
"loss": 2.770678699016571,
|
| 85 |
+
"documents": 4,
|
| 86 |
+
"tokens": 512
|
| 87 |
+
},
|
| 88 |
+
"zh": {
|
| 89 |
+
"loss": 3.378099739551544,
|
| 90 |
+
"documents": 4,
|
| 91 |
+
"tokens": 512
|
| 92 |
+
},
|
| 93 |
+
"ja": {
|
| 94 |
+
"loss": 2.9211148023605347,
|
| 95 |
+
"documents": 4,
|
| 96 |
+
"tokens": 512
|
| 97 |
+
},
|
| 98 |
+
"es": {
|
| 99 |
+
"loss": 2.6984021067619324,
|
| 100 |
+
"documents": 4,
|
| 101 |
+
"tokens": 512
|
| 102 |
+
},
|
| 103 |
+
"de": {
|
| 104 |
+
"loss": 2.9680771231651306,
|
| 105 |
+
"documents": 4,
|
| 106 |
+
"tokens": 512
|
| 107 |
+
}
|
| 108 |
+
}
|
| 109 |
+
},
|
| 110 |
+
"mean_losses": {
|
| 111 |
+
"qwen35_2b": 3.2084930737813315,
|
| 112 |
+
"a3b_v2": 3.2084930737813315,
|
| 113 |
+
"qwen35_4b": 2.9399328529834747
|
| 114 |
+
}
|
| 115 |
+
}
|
evaluation/openai_service_20.json
ADDED
|
@@ -0,0 +1,165 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"requests": 20,
|
| 3 |
+
"passed": 20,
|
| 4 |
+
"all_passed": true,
|
| 5 |
+
"latency_mean_seconds": 2.5972848790552234,
|
| 6 |
+
"latency_p50_seconds": 2.9117911249923054,
|
| 7 |
+
"latency_p95_seconds": 3.568211792036891,
|
| 8 |
+
"health_before": {
|
| 9 |
+
"status": "ok",
|
| 10 |
+
"model": "qwen3.5-4b-a3b-student-v2",
|
| 11 |
+
"model_path": "../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 12 |
+
"device": "mps",
|
| 13 |
+
"parameters": 3995901760,
|
| 14 |
+
"requests": 0,
|
| 15 |
+
"latency_mean_seconds": null,
|
| 16 |
+
"latency_p95_seconds": null,
|
| 17 |
+
"rss_mb": null,
|
| 18 |
+
"mps_allocated_mb": 7621.5859375
|
| 19 |
+
},
|
| 20 |
+
"health_after": {
|
| 21 |
+
"status": "ok",
|
| 22 |
+
"model": "qwen3.5-4b-a3b-student-v2",
|
| 23 |
+
"model_path": "../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2",
|
| 24 |
+
"device": "mps",
|
| 25 |
+
"parameters": 3995901760,
|
| 26 |
+
"requests": 20,
|
| 27 |
+
"latency_mean_seconds": 2.5897246561566134,
|
| 28 |
+
"latency_p95_seconds": 3.5636009160079993,
|
| 29 |
+
"rss_mb": null,
|
| 30 |
+
"mps_allocated_mb": 7621.5859375
|
| 31 |
+
},
|
| 32 |
+
"models": {
|
| 33 |
+
"object": "list",
|
| 34 |
+
"data": [
|
| 35 |
+
{
|
| 36 |
+
"id": "qwen3.5-4b-a3b-student-v2",
|
| 37 |
+
"object": "model",
|
| 38 |
+
"created": 1787551049,
|
| 39 |
+
"owned_by": "local"
|
| 40 |
+
}
|
| 41 |
+
]
|
| 42 |
+
},
|
| 43 |
+
"results": [
|
| 44 |
+
{
|
| 45 |
+
"index": 1,
|
| 46 |
+
"passed": true,
|
| 47 |
+
"elapsed_seconds": 1.8874667910276912,
|
| 48 |
+
"content": "서울"
|
| 49 |
+
},
|
| 50 |
+
{
|
| 51 |
+
"index": 2,
|
| 52 |
+
"passed": true,
|
| 53 |
+
"elapsed_seconds": 2.6587018750142306,
|
| 54 |
+
"content": "The capital of France is **Paris**."
|
| 55 |
+
},
|
| 56 |
+
{
|
| 57 |
+
"index": 3,
|
| 58 |
+
"passed": true,
|
| 59 |
+
"elapsed_seconds": 3.4837433329666965,
|
| 60 |
+
"content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
|
| 61 |
+
},
|
| 62 |
+
{
|
| 63 |
+
"index": 4,
|
| 64 |
+
"passed": true,
|
| 65 |
+
"elapsed_seconds": 3.3025628330069594,
|
| 66 |
+
"content": "Hello there! How's your day going so far?"
|
| 67 |
+
},
|
| 68 |
+
{
|
| 69 |
+
"index": 5,
|
| 70 |
+
"passed": true,
|
| 71 |
+
"elapsed_seconds": 0.9436147080268711,
|
| 72 |
+
"content": "서울"
|
| 73 |
+
},
|
| 74 |
+
{
|
| 75 |
+
"index": 6,
|
| 76 |
+
"passed": true,
|
| 77 |
+
"elapsed_seconds": 2.515450624981895,
|
| 78 |
+
"content": "The capital of France is **Paris**."
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"index": 7,
|
| 82 |
+
"passed": true,
|
| 83 |
+
"elapsed_seconds": 3.568211792036891,
|
| 84 |
+
"content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
|
| 85 |
+
},
|
| 86 |
+
{
|
| 87 |
+
"index": 8,
|
| 88 |
+
"passed": true,
|
| 89 |
+
"elapsed_seconds": 3.2445592500152998,
|
| 90 |
+
"content": "Hello there! How's your day going so far?"
|
| 91 |
+
},
|
| 92 |
+
{
|
| 93 |
+
"index": 9,
|
| 94 |
+
"passed": true,
|
| 95 |
+
"elapsed_seconds": 0.9237514159758575,
|
| 96 |
+
"content": "서울"
|
| 97 |
+
},
|
| 98 |
+
{
|
| 99 |
+
"index": 10,
|
| 100 |
+
"passed": true,
|
| 101 |
+
"elapsed_seconds": 2.4350813750061207,
|
| 102 |
+
"content": "The capital of France is **Paris**."
|
| 103 |
+
},
|
| 104 |
+
{
|
| 105 |
+
"index": 11,
|
| 106 |
+
"passed": true,
|
| 107 |
+
"elapsed_seconds": 3.567038333043456,
|
| 108 |
+
"content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
|
| 109 |
+
},
|
| 110 |
+
{
|
| 111 |
+
"index": 12,
|
| 112 |
+
"passed": true,
|
| 113 |
+
"elapsed_seconds": 3.16488037497038,
|
| 114 |
+
"content": "Hello there! How's your day going so far?"
|
| 115 |
+
},
|
| 116 |
+
{
|
| 117 |
+
"index": 13,
|
| 118 |
+
"passed": true,
|
| 119 |
+
"elapsed_seconds": 0.9253840000019409,
|
| 120 |
+
"content": "서울"
|
| 121 |
+
},
|
| 122 |
+
{
|
| 123 |
+
"index": 14,
|
| 124 |
+
"passed": true,
|
| 125 |
+
"elapsed_seconds": 2.5322331670322455,
|
| 126 |
+
"content": "The capital of France is **Paris**."
|
| 127 |
+
},
|
| 128 |
+
{
|
| 129 |
+
"index": 15,
|
| 130 |
+
"passed": true,
|
| 131 |
+
"elapsed_seconds": 3.6034239170257933,
|
| 132 |
+
"content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
|
| 133 |
+
},
|
| 134 |
+
{
|
| 135 |
+
"index": 16,
|
| 136 |
+
"passed": true,
|
| 137 |
+
"elapsed_seconds": 3.2237549159908667,
|
| 138 |
+
"content": "Hello there! How's your day going so far?"
|
| 139 |
+
},
|
| 140 |
+
{
|
| 141 |
+
"index": 17,
|
| 142 |
+
"passed": true,
|
| 143 |
+
"elapsed_seconds": 0.9134719159919769,
|
| 144 |
+
"content": "서울"
|
| 145 |
+
},
|
| 146 |
+
{
|
| 147 |
+
"index": 18,
|
| 148 |
+
"passed": true,
|
| 149 |
+
"elapsed_seconds": 2.403110374987591,
|
| 150 |
+
"content": "The capital of France is **Paris**."
|
| 151 |
+
},
|
| 152 |
+
{
|
| 153 |
+
"index": 19,
|
| 154 |
+
"passed": true,
|
| 155 |
+
"elapsed_seconds": 3.4773494169930927,
|
| 156 |
+
"content": "7 곱하기 8을 계산하면 다음과 같습니다.\n\n$7 \\"
|
| 157 |
+
},
|
| 158 |
+
{
|
| 159 |
+
"index": 20,
|
| 160 |
+
"passed": true,
|
| 161 |
+
"elapsed_seconds": 3.1719071670086123,
|
| 162 |
+
"content": "Hello there! How's your day going so far?"
|
| 163 |
+
}
|
| 164 |
+
]
|
| 165 |
+
}
|
merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
model-common.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:40a4b73b7901fd219c6814fa5dd05b0ff522389de8d87f9995e505a7331a958b
|
| 3 |
+
size 1951744432
|
model-layer-00.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:050bc55fb97a2759ac2cfc3cc224c75e4b3ea6deef0ea2d3116360c03c176703
|
| 3 |
+
size 251671392
|
model-layer-01.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:584ac63f1b1322eb754152b6f1cc8f58d474aa74590eec6dd9a352d967e7d06a
|
| 3 |
+
size 251671392
|
model-layer-02.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:6b8317a7c0167103676cdde7422620670be06a5a68c0c7fa4e8bc925197ac432
|
| 3 |
+
size 251671392
|
model-layer-03.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:05dfd5ad3bf73b4b17c634b69db3dec4fc14c57dbaab642e3b312308a1ded5fa
|
| 3 |
+
size 251671392
|
model-layer-04.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2a63f3e6d42b3ed951be93367a030bd1d7d015724b659eeaa592358c0cc02d07
|
| 3 |
+
size 251671392
|
model-layer-05.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2f83c8222cdec99b41fca8532dca86c065fedbcd3f5c9be0bc53786cabbba303
|
| 3 |
+
size 251671392
|
model-layer-06.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ac849368a9a338b977fcbdc3e23d8363da3f40e13349d8801fb4674dee5e7739
|
| 3 |
+
size 251671392
|
model-layer-07.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:427086a8339f5295e956024355beec362c73dae9ad7bec79d88a8860dddfb883
|
| 3 |
+
size 251671392
|
model-layer-08.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fd30e7089c9510120fa19be4a19ff251613523cb489e491ee8aa3d84985bb86e
|
| 3 |
+
size 251671392
|
model-layer-09.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fb834374a42eaf0681fb90bd18a1dc066f7d17bf5b68333ec3e3020d9e097aef
|
| 3 |
+
size 251671392
|
model-layer-10.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:3eaddc0892948def344e80877b98448f574f8a86794e5c36bd54fea9366fb29a
|
| 3 |
+
size 251671400
|
model-layer-11.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f4551261310cefbcb24e1434e424a6807c78805797a1936f92987ac61d5f87ec
|
| 3 |
+
size 251671400
|
model-layer-12.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:19d0e2f3c546506b5f173b768124427579e8f9523b31292f7bc10b3a9a8cdf17
|
| 3 |
+
size 251671400
|
model-layer-13.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:18d8275c1334ea12f3b4390ef9850143904f2536782ffd96016ccbdbbef043d2
|
| 3 |
+
size 251671400
|
model-layer-14.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:110a065ec5d6a927ac5eb22156f7ce90c94f7dd6caea067bec74eee303619b41
|
| 3 |
+
size 251671400
|
model-layer-15.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7f36395bdc43afcf968ab2e766c840e6c71c32053a5d54596978bd0518820661
|
| 3 |
+
size 251671400
|
model-layer-16.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f33286ee6aede3f08e4c4ce152813b7d9d9ae61636fb0ec5528352832f06c971
|
| 3 |
+
size 251671400
|
model-layer-17.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:86775994e4e45afefb7a6e791cd83a378620bd500e874bfcfc8ec3deb6bc2acc
|
| 3 |
+
size 251671400
|
model-layer-18.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:50246c386a0f863d04ce770675b9100726eb9d59ef2a73c9a4ed42e9031901b3
|
| 3 |
+
size 251671400
|
model-layer-19.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c47bb7ec84481206c1931adf2c1691b6a051ef29e9480158838fc9e7050ab3b9
|
| 3 |
+
size 251671400
|
model-layer-20.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:96c783fdc495547c37c9835cbc1214342befaf29daaa376d0a197c3539729417
|
| 3 |
+
size 251671400
|
model-layer-21.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2615d98f592d082e06595c94e3c77d1a1188ac26a1b009f8ab0b9daac0f72458
|
| 3 |
+
size 251671400
|
model-layer-22.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e44a9882ba2c1ab0f461ee1026de10c92aa96e6d5710463a32c151554bbf3a5c
|
| 3 |
+
size 251671400
|
model-layer-23.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:e764ddf68a2162aa3df83a3ff7f5bbb40a9e71fc438b95f63b51a88f0e032402
|
| 3 |
+
size 251671400
|
model.safetensors.index.json
ADDED
|
@@ -0,0 +1,423 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"metadata": {
|
| 3 |
+
"total_size": 7991808704
|
| 4 |
+
},
|
| 5 |
+
"weight_map": {
|
| 6 |
+
"model.embed_tokens.weight": "model-common.safetensors",
|
| 7 |
+
"model.layers.0.input_layernorm.weight": "model-common.safetensors",
|
| 8 |
+
"model.layers.0.linear_attn.A_log": "model-common.safetensors",
|
| 9 |
+
"model.layers.0.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 10 |
+
"model.layers.0.linear_attn.dt_bias": "model-common.safetensors",
|
| 11 |
+
"model.layers.0.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 12 |
+
"model.layers.0.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 13 |
+
"model.layers.0.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 14 |
+
"model.layers.0.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 15 |
+
"model.layers.0.linear_attn.norm.weight": "model-common.safetensors",
|
| 16 |
+
"model.layers.0.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 17 |
+
"model.layers.0.mlp.experts.down_proj": "model-layer-00.safetensors",
|
| 18 |
+
"model.layers.0.mlp.experts.gate_up_proj": "model-layer-00.safetensors",
|
| 19 |
+
"model.layers.0.mlp.gate.weight": "model-layer-00.safetensors",
|
| 20 |
+
"model.layers.0.mlp.shared_expert.down_proj.weight": "model-layer-00.safetensors",
|
| 21 |
+
"model.layers.0.mlp.shared_expert.gate_proj.weight": "model-layer-00.safetensors",
|
| 22 |
+
"model.layers.0.mlp.shared_expert.up_proj.weight": "model-layer-00.safetensors",
|
| 23 |
+
"model.layers.0.mlp.shared_expert_gate.weight": "model-layer-00.safetensors",
|
| 24 |
+
"model.layers.0.post_attention_layernorm.weight": "model-common.safetensors",
|
| 25 |
+
"model.layers.1.input_layernorm.weight": "model-common.safetensors",
|
| 26 |
+
"model.layers.1.linear_attn.A_log": "model-common.safetensors",
|
| 27 |
+
"model.layers.1.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 28 |
+
"model.layers.1.linear_attn.dt_bias": "model-common.safetensors",
|
| 29 |
+
"model.layers.1.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 30 |
+
"model.layers.1.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 31 |
+
"model.layers.1.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 32 |
+
"model.layers.1.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 33 |
+
"model.layers.1.linear_attn.norm.weight": "model-common.safetensors",
|
| 34 |
+
"model.layers.1.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 35 |
+
"model.layers.1.mlp.experts.down_proj": "model-layer-01.safetensors",
|
| 36 |
+
"model.layers.1.mlp.experts.gate_up_proj": "model-layer-01.safetensors",
|
| 37 |
+
"model.layers.1.mlp.gate.weight": "model-layer-01.safetensors",
|
| 38 |
+
"model.layers.1.mlp.shared_expert.down_proj.weight": "model-layer-01.safetensors",
|
| 39 |
+
"model.layers.1.mlp.shared_expert.gate_proj.weight": "model-layer-01.safetensors",
|
| 40 |
+
"model.layers.1.mlp.shared_expert.up_proj.weight": "model-layer-01.safetensors",
|
| 41 |
+
"model.layers.1.mlp.shared_expert_gate.weight": "model-layer-01.safetensors",
|
| 42 |
+
"model.layers.1.post_attention_layernorm.weight": "model-common.safetensors",
|
| 43 |
+
"model.layers.10.input_layernorm.weight": "model-common.safetensors",
|
| 44 |
+
"model.layers.10.linear_attn.A_log": "model-common.safetensors",
|
| 45 |
+
"model.layers.10.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 46 |
+
"model.layers.10.linear_attn.dt_bias": "model-common.safetensors",
|
| 47 |
+
"model.layers.10.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 48 |
+
"model.layers.10.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 49 |
+
"model.layers.10.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 50 |
+
"model.layers.10.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 51 |
+
"model.layers.10.linear_attn.norm.weight": "model-common.safetensors",
|
| 52 |
+
"model.layers.10.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 53 |
+
"model.layers.10.mlp.experts.down_proj": "model-layer-10.safetensors",
|
| 54 |
+
"model.layers.10.mlp.experts.gate_up_proj": "model-layer-10.safetensors",
|
| 55 |
+
"model.layers.10.mlp.gate.weight": "model-layer-10.safetensors",
|
| 56 |
+
"model.layers.10.mlp.shared_expert.down_proj.weight": "model-layer-10.safetensors",
|
| 57 |
+
"model.layers.10.mlp.shared_expert.gate_proj.weight": "model-layer-10.safetensors",
|
| 58 |
+
"model.layers.10.mlp.shared_expert.up_proj.weight": "model-layer-10.safetensors",
|
| 59 |
+
"model.layers.10.mlp.shared_expert_gate.weight": "model-layer-10.safetensors",
|
| 60 |
+
"model.layers.10.post_attention_layernorm.weight": "model-common.safetensors",
|
| 61 |
+
"model.layers.11.input_layernorm.weight": "model-common.safetensors",
|
| 62 |
+
"model.layers.11.mlp.experts.down_proj": "model-layer-11.safetensors",
|
| 63 |
+
"model.layers.11.mlp.experts.gate_up_proj": "model-layer-11.safetensors",
|
| 64 |
+
"model.layers.11.mlp.gate.weight": "model-layer-11.safetensors",
|
| 65 |
+
"model.layers.11.mlp.shared_expert.down_proj.weight": "model-layer-11.safetensors",
|
| 66 |
+
"model.layers.11.mlp.shared_expert.gate_proj.weight": "model-layer-11.safetensors",
|
| 67 |
+
"model.layers.11.mlp.shared_expert.up_proj.weight": "model-layer-11.safetensors",
|
| 68 |
+
"model.layers.11.mlp.shared_expert_gate.weight": "model-layer-11.safetensors",
|
| 69 |
+
"model.layers.11.post_attention_layernorm.weight": "model-common.safetensors",
|
| 70 |
+
"model.layers.11.self_attn.k_norm.weight": "model-common.safetensors",
|
| 71 |
+
"model.layers.11.self_attn.k_proj.weight": "model-common.safetensors",
|
| 72 |
+
"model.layers.11.self_attn.o_proj.weight": "model-common.safetensors",
|
| 73 |
+
"model.layers.11.self_attn.q_norm.weight": "model-common.safetensors",
|
| 74 |
+
"model.layers.11.self_attn.q_proj.weight": "model-common.safetensors",
|
| 75 |
+
"model.layers.11.self_attn.v_proj.weight": "model-common.safetensors",
|
| 76 |
+
"model.layers.12.input_layernorm.weight": "model-common.safetensors",
|
| 77 |
+
"model.layers.12.linear_attn.A_log": "model-common.safetensors",
|
| 78 |
+
"model.layers.12.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 79 |
+
"model.layers.12.linear_attn.dt_bias": "model-common.safetensors",
|
| 80 |
+
"model.layers.12.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 81 |
+
"model.layers.12.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 82 |
+
"model.layers.12.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 83 |
+
"model.layers.12.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 84 |
+
"model.layers.12.linear_attn.norm.weight": "model-common.safetensors",
|
| 85 |
+
"model.layers.12.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 86 |
+
"model.layers.12.mlp.experts.down_proj": "model-layer-12.safetensors",
|
| 87 |
+
"model.layers.12.mlp.experts.gate_up_proj": "model-layer-12.safetensors",
|
| 88 |
+
"model.layers.12.mlp.gate.weight": "model-layer-12.safetensors",
|
| 89 |
+
"model.layers.12.mlp.shared_expert.down_proj.weight": "model-layer-12.safetensors",
|
| 90 |
+
"model.layers.12.mlp.shared_expert.gate_proj.weight": "model-layer-12.safetensors",
|
| 91 |
+
"model.layers.12.mlp.shared_expert.up_proj.weight": "model-layer-12.safetensors",
|
| 92 |
+
"model.layers.12.mlp.shared_expert_gate.weight": "model-layer-12.safetensors",
|
| 93 |
+
"model.layers.12.post_attention_layernorm.weight": "model-common.safetensors",
|
| 94 |
+
"model.layers.13.input_layernorm.weight": "model-common.safetensors",
|
| 95 |
+
"model.layers.13.linear_attn.A_log": "model-common.safetensors",
|
| 96 |
+
"model.layers.13.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 97 |
+
"model.layers.13.linear_attn.dt_bias": "model-common.safetensors",
|
| 98 |
+
"model.layers.13.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 99 |
+
"model.layers.13.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 100 |
+
"model.layers.13.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 101 |
+
"model.layers.13.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 102 |
+
"model.layers.13.linear_attn.norm.weight": "model-common.safetensors",
|
| 103 |
+
"model.layers.13.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 104 |
+
"model.layers.13.mlp.experts.down_proj": "model-layer-13.safetensors",
|
| 105 |
+
"model.layers.13.mlp.experts.gate_up_proj": "model-layer-13.safetensors",
|
| 106 |
+
"model.layers.13.mlp.gate.weight": "model-layer-13.safetensors",
|
| 107 |
+
"model.layers.13.mlp.shared_expert.down_proj.weight": "model-layer-13.safetensors",
|
| 108 |
+
"model.layers.13.mlp.shared_expert.gate_proj.weight": "model-layer-13.safetensors",
|
| 109 |
+
"model.layers.13.mlp.shared_expert.up_proj.weight": "model-layer-13.safetensors",
|
| 110 |
+
"model.layers.13.mlp.shared_expert_gate.weight": "model-layer-13.safetensors",
|
| 111 |
+
"model.layers.13.post_attention_layernorm.weight": "model-common.safetensors",
|
| 112 |
+
"model.layers.14.input_layernorm.weight": "model-common.safetensors",
|
| 113 |
+
"model.layers.14.linear_attn.A_log": "model-common.safetensors",
|
| 114 |
+
"model.layers.14.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 115 |
+
"model.layers.14.linear_attn.dt_bias": "model-common.safetensors",
|
| 116 |
+
"model.layers.14.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 117 |
+
"model.layers.14.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 118 |
+
"model.layers.14.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 119 |
+
"model.layers.14.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 120 |
+
"model.layers.14.linear_attn.norm.weight": "model-common.safetensors",
|
| 121 |
+
"model.layers.14.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 122 |
+
"model.layers.14.mlp.experts.down_proj": "model-layer-14.safetensors",
|
| 123 |
+
"model.layers.14.mlp.experts.gate_up_proj": "model-layer-14.safetensors",
|
| 124 |
+
"model.layers.14.mlp.gate.weight": "model-layer-14.safetensors",
|
| 125 |
+
"model.layers.14.mlp.shared_expert.down_proj.weight": "model-layer-14.safetensors",
|
| 126 |
+
"model.layers.14.mlp.shared_expert.gate_proj.weight": "model-layer-14.safetensors",
|
| 127 |
+
"model.layers.14.mlp.shared_expert.up_proj.weight": "model-layer-14.safetensors",
|
| 128 |
+
"model.layers.14.mlp.shared_expert_gate.weight": "model-layer-14.safetensors",
|
| 129 |
+
"model.layers.14.post_attention_layernorm.weight": "model-common.safetensors",
|
| 130 |
+
"model.layers.15.input_layernorm.weight": "model-common.safetensors",
|
| 131 |
+
"model.layers.15.mlp.experts.down_proj": "model-layer-15.safetensors",
|
| 132 |
+
"model.layers.15.mlp.experts.gate_up_proj": "model-layer-15.safetensors",
|
| 133 |
+
"model.layers.15.mlp.gate.weight": "model-layer-15.safetensors",
|
| 134 |
+
"model.layers.15.mlp.shared_expert.down_proj.weight": "model-layer-15.safetensors",
|
| 135 |
+
"model.layers.15.mlp.shared_expert.gate_proj.weight": "model-layer-15.safetensors",
|
| 136 |
+
"model.layers.15.mlp.shared_expert.up_proj.weight": "model-layer-15.safetensors",
|
| 137 |
+
"model.layers.15.mlp.shared_expert_gate.weight": "model-layer-15.safetensors",
|
| 138 |
+
"model.layers.15.post_attention_layernorm.weight": "model-common.safetensors",
|
| 139 |
+
"model.layers.15.self_attn.k_norm.weight": "model-common.safetensors",
|
| 140 |
+
"model.layers.15.self_attn.k_proj.weight": "model-common.safetensors",
|
| 141 |
+
"model.layers.15.self_attn.o_proj.weight": "model-common.safetensors",
|
| 142 |
+
"model.layers.15.self_attn.q_norm.weight": "model-common.safetensors",
|
| 143 |
+
"model.layers.15.self_attn.q_proj.weight": "model-common.safetensors",
|
| 144 |
+
"model.layers.15.self_attn.v_proj.weight": "model-common.safetensors",
|
| 145 |
+
"model.layers.16.input_layernorm.weight": "model-common.safetensors",
|
| 146 |
+
"model.layers.16.linear_attn.A_log": "model-common.safetensors",
|
| 147 |
+
"model.layers.16.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 148 |
+
"model.layers.16.linear_attn.dt_bias": "model-common.safetensors",
|
| 149 |
+
"model.layers.16.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 150 |
+
"model.layers.16.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 151 |
+
"model.layers.16.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 152 |
+
"model.layers.16.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 153 |
+
"model.layers.16.linear_attn.norm.weight": "model-common.safetensors",
|
| 154 |
+
"model.layers.16.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 155 |
+
"model.layers.16.mlp.experts.down_proj": "model-layer-16.safetensors",
|
| 156 |
+
"model.layers.16.mlp.experts.gate_up_proj": "model-layer-16.safetensors",
|
| 157 |
+
"model.layers.16.mlp.gate.weight": "model-layer-16.safetensors",
|
| 158 |
+
"model.layers.16.mlp.shared_expert.down_proj.weight": "model-layer-16.safetensors",
|
| 159 |
+
"model.layers.16.mlp.shared_expert.gate_proj.weight": "model-layer-16.safetensors",
|
| 160 |
+
"model.layers.16.mlp.shared_expert.up_proj.weight": "model-layer-16.safetensors",
|
| 161 |
+
"model.layers.16.mlp.shared_expert_gate.weight": "model-layer-16.safetensors",
|
| 162 |
+
"model.layers.16.post_attention_layernorm.weight": "model-common.safetensors",
|
| 163 |
+
"model.layers.17.input_layernorm.weight": "model-common.safetensors",
|
| 164 |
+
"model.layers.17.linear_attn.A_log": "model-common.safetensors",
|
| 165 |
+
"model.layers.17.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 166 |
+
"model.layers.17.linear_attn.dt_bias": "model-common.safetensors",
|
| 167 |
+
"model.layers.17.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 168 |
+
"model.layers.17.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 169 |
+
"model.layers.17.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 170 |
+
"model.layers.17.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 171 |
+
"model.layers.17.linear_attn.norm.weight": "model-common.safetensors",
|
| 172 |
+
"model.layers.17.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 173 |
+
"model.layers.17.mlp.experts.down_proj": "model-layer-17.safetensors",
|
| 174 |
+
"model.layers.17.mlp.experts.gate_up_proj": "model-layer-17.safetensors",
|
| 175 |
+
"model.layers.17.mlp.gate.weight": "model-layer-17.safetensors",
|
| 176 |
+
"model.layers.17.mlp.shared_expert.down_proj.weight": "model-layer-17.safetensors",
|
| 177 |
+
"model.layers.17.mlp.shared_expert.gate_proj.weight": "model-layer-17.safetensors",
|
| 178 |
+
"model.layers.17.mlp.shared_expert.up_proj.weight": "model-layer-17.safetensors",
|
| 179 |
+
"model.layers.17.mlp.shared_expert_gate.weight": "model-layer-17.safetensors",
|
| 180 |
+
"model.layers.17.post_attention_layernorm.weight": "model-common.safetensors",
|
| 181 |
+
"model.layers.18.input_layernorm.weight": "model-common.safetensors",
|
| 182 |
+
"model.layers.18.linear_attn.A_log": "model-common.safetensors",
|
| 183 |
+
"model.layers.18.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 184 |
+
"model.layers.18.linear_attn.dt_bias": "model-common.safetensors",
|
| 185 |
+
"model.layers.18.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 186 |
+
"model.layers.18.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 187 |
+
"model.layers.18.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 188 |
+
"model.layers.18.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 189 |
+
"model.layers.18.linear_attn.norm.weight": "model-common.safetensors",
|
| 190 |
+
"model.layers.18.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 191 |
+
"model.layers.18.mlp.experts.down_proj": "model-layer-18.safetensors",
|
| 192 |
+
"model.layers.18.mlp.experts.gate_up_proj": "model-layer-18.safetensors",
|
| 193 |
+
"model.layers.18.mlp.gate.weight": "model-layer-18.safetensors",
|
| 194 |
+
"model.layers.18.mlp.shared_expert.down_proj.weight": "model-layer-18.safetensors",
|
| 195 |
+
"model.layers.18.mlp.shared_expert.gate_proj.weight": "model-layer-18.safetensors",
|
| 196 |
+
"model.layers.18.mlp.shared_expert.up_proj.weight": "model-layer-18.safetensors",
|
| 197 |
+
"model.layers.18.mlp.shared_expert_gate.weight": "model-layer-18.safetensors",
|
| 198 |
+
"model.layers.18.post_attention_layernorm.weight": "model-common.safetensors",
|
| 199 |
+
"model.layers.19.input_layernorm.weight": "model-common.safetensors",
|
| 200 |
+
"model.layers.19.mlp.experts.down_proj": "model-layer-19.safetensors",
|
| 201 |
+
"model.layers.19.mlp.experts.gate_up_proj": "model-layer-19.safetensors",
|
| 202 |
+
"model.layers.19.mlp.gate.weight": "model-layer-19.safetensors",
|
| 203 |
+
"model.layers.19.mlp.shared_expert.down_proj.weight": "model-layer-19.safetensors",
|
| 204 |
+
"model.layers.19.mlp.shared_expert.gate_proj.weight": "model-layer-19.safetensors",
|
| 205 |
+
"model.layers.19.mlp.shared_expert.up_proj.weight": "model-layer-19.safetensors",
|
| 206 |
+
"model.layers.19.mlp.shared_expert_gate.weight": "model-layer-19.safetensors",
|
| 207 |
+
"model.layers.19.post_attention_layernorm.weight": "model-common.safetensors",
|
| 208 |
+
"model.layers.19.self_attn.k_norm.weight": "model-common.safetensors",
|
| 209 |
+
"model.layers.19.self_attn.k_proj.weight": "model-common.safetensors",
|
| 210 |
+
"model.layers.19.self_attn.o_proj.weight": "model-common.safetensors",
|
| 211 |
+
"model.layers.19.self_attn.q_norm.weight": "model-common.safetensors",
|
| 212 |
+
"model.layers.19.self_attn.q_proj.weight": "model-common.safetensors",
|
| 213 |
+
"model.layers.19.self_attn.v_proj.weight": "model-common.safetensors",
|
| 214 |
+
"model.layers.2.input_layernorm.weight": "model-common.safetensors",
|
| 215 |
+
"model.layers.2.linear_attn.A_log": "model-common.safetensors",
|
| 216 |
+
"model.layers.2.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 217 |
+
"model.layers.2.linear_attn.dt_bias": "model-common.safetensors",
|
| 218 |
+
"model.layers.2.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 219 |
+
"model.layers.2.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 220 |
+
"model.layers.2.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 221 |
+
"model.layers.2.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 222 |
+
"model.layers.2.linear_attn.norm.weight": "model-common.safetensors",
|
| 223 |
+
"model.layers.2.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 224 |
+
"model.layers.2.mlp.experts.down_proj": "model-layer-02.safetensors",
|
| 225 |
+
"model.layers.2.mlp.experts.gate_up_proj": "model-layer-02.safetensors",
|
| 226 |
+
"model.layers.2.mlp.gate.weight": "model-layer-02.safetensors",
|
| 227 |
+
"model.layers.2.mlp.shared_expert.down_proj.weight": "model-layer-02.safetensors",
|
| 228 |
+
"model.layers.2.mlp.shared_expert.gate_proj.weight": "model-layer-02.safetensors",
|
| 229 |
+
"model.layers.2.mlp.shared_expert.up_proj.weight": "model-layer-02.safetensors",
|
| 230 |
+
"model.layers.2.mlp.shared_expert_gate.weight": "model-layer-02.safetensors",
|
| 231 |
+
"model.layers.2.post_attention_layernorm.weight": "model-common.safetensors",
|
| 232 |
+
"model.layers.20.input_layernorm.weight": "model-common.safetensors",
|
| 233 |
+
"model.layers.20.linear_attn.A_log": "model-common.safetensors",
|
| 234 |
+
"model.layers.20.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 235 |
+
"model.layers.20.linear_attn.dt_bias": "model-common.safetensors",
|
| 236 |
+
"model.layers.20.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 237 |
+
"model.layers.20.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 238 |
+
"model.layers.20.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 239 |
+
"model.layers.20.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 240 |
+
"model.layers.20.linear_attn.norm.weight": "model-common.safetensors",
|
| 241 |
+
"model.layers.20.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 242 |
+
"model.layers.20.mlp.experts.down_proj": "model-layer-20.safetensors",
|
| 243 |
+
"model.layers.20.mlp.experts.gate_up_proj": "model-layer-20.safetensors",
|
| 244 |
+
"model.layers.20.mlp.gate.weight": "model-layer-20.safetensors",
|
| 245 |
+
"model.layers.20.mlp.shared_expert.down_proj.weight": "model-layer-20.safetensors",
|
| 246 |
+
"model.layers.20.mlp.shared_expert.gate_proj.weight": "model-layer-20.safetensors",
|
| 247 |
+
"model.layers.20.mlp.shared_expert.up_proj.weight": "model-layer-20.safetensors",
|
| 248 |
+
"model.layers.20.mlp.shared_expert_gate.weight": "model-layer-20.safetensors",
|
| 249 |
+
"model.layers.20.post_attention_layernorm.weight": "model-common.safetensors",
|
| 250 |
+
"model.layers.21.input_layernorm.weight": "model-common.safetensors",
|
| 251 |
+
"model.layers.21.linear_attn.A_log": "model-common.safetensors",
|
| 252 |
+
"model.layers.21.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 253 |
+
"model.layers.21.linear_attn.dt_bias": "model-common.safetensors",
|
| 254 |
+
"model.layers.21.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 255 |
+
"model.layers.21.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 256 |
+
"model.layers.21.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 257 |
+
"model.layers.21.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 258 |
+
"model.layers.21.linear_attn.norm.weight": "model-common.safetensors",
|
| 259 |
+
"model.layers.21.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 260 |
+
"model.layers.21.mlp.experts.down_proj": "model-layer-21.safetensors",
|
| 261 |
+
"model.layers.21.mlp.experts.gate_up_proj": "model-layer-21.safetensors",
|
| 262 |
+
"model.layers.21.mlp.gate.weight": "model-layer-21.safetensors",
|
| 263 |
+
"model.layers.21.mlp.shared_expert.down_proj.weight": "model-layer-21.safetensors",
|
| 264 |
+
"model.layers.21.mlp.shared_expert.gate_proj.weight": "model-layer-21.safetensors",
|
| 265 |
+
"model.layers.21.mlp.shared_expert.up_proj.weight": "model-layer-21.safetensors",
|
| 266 |
+
"model.layers.21.mlp.shared_expert_gate.weight": "model-layer-21.safetensors",
|
| 267 |
+
"model.layers.21.post_attention_layernorm.weight": "model-common.safetensors",
|
| 268 |
+
"model.layers.22.input_layernorm.weight": "model-common.safetensors",
|
| 269 |
+
"model.layers.22.linear_attn.A_log": "model-common.safetensors",
|
| 270 |
+
"model.layers.22.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 271 |
+
"model.layers.22.linear_attn.dt_bias": "model-common.safetensors",
|
| 272 |
+
"model.layers.22.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 273 |
+
"model.layers.22.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 274 |
+
"model.layers.22.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 275 |
+
"model.layers.22.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 276 |
+
"model.layers.22.linear_attn.norm.weight": "model-common.safetensors",
|
| 277 |
+
"model.layers.22.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 278 |
+
"model.layers.22.mlp.experts.down_proj": "model-layer-22.safetensors",
|
| 279 |
+
"model.layers.22.mlp.experts.gate_up_proj": "model-layer-22.safetensors",
|
| 280 |
+
"model.layers.22.mlp.gate.weight": "model-layer-22.safetensors",
|
| 281 |
+
"model.layers.22.mlp.shared_expert.down_proj.weight": "model-layer-22.safetensors",
|
| 282 |
+
"model.layers.22.mlp.shared_expert.gate_proj.weight": "model-layer-22.safetensors",
|
| 283 |
+
"model.layers.22.mlp.shared_expert.up_proj.weight": "model-layer-22.safetensors",
|
| 284 |
+
"model.layers.22.mlp.shared_expert_gate.weight": "model-layer-22.safetensors",
|
| 285 |
+
"model.layers.22.post_attention_layernorm.weight": "model-common.safetensors",
|
| 286 |
+
"model.layers.23.input_layernorm.weight": "model-common.safetensors",
|
| 287 |
+
"model.layers.23.mlp.experts.down_proj": "model-layer-23.safetensors",
|
| 288 |
+
"model.layers.23.mlp.experts.gate_up_proj": "model-layer-23.safetensors",
|
| 289 |
+
"model.layers.23.mlp.gate.weight": "model-layer-23.safetensors",
|
| 290 |
+
"model.layers.23.mlp.shared_expert.down_proj.weight": "model-layer-23.safetensors",
|
| 291 |
+
"model.layers.23.mlp.shared_expert.gate_proj.weight": "model-layer-23.safetensors",
|
| 292 |
+
"model.layers.23.mlp.shared_expert.up_proj.weight": "model-layer-23.safetensors",
|
| 293 |
+
"model.layers.23.mlp.shared_expert_gate.weight": "model-layer-23.safetensors",
|
| 294 |
+
"model.layers.23.post_attention_layernorm.weight": "model-common.safetensors",
|
| 295 |
+
"model.layers.23.self_attn.k_norm.weight": "model-common.safetensors",
|
| 296 |
+
"model.layers.23.self_attn.k_proj.weight": "model-common.safetensors",
|
| 297 |
+
"model.layers.23.self_attn.o_proj.weight": "model-common.safetensors",
|
| 298 |
+
"model.layers.23.self_attn.q_norm.weight": "model-common.safetensors",
|
| 299 |
+
"model.layers.23.self_attn.q_proj.weight": "model-common.safetensors",
|
| 300 |
+
"model.layers.23.self_attn.v_proj.weight": "model-common.safetensors",
|
| 301 |
+
"model.layers.3.input_layernorm.weight": "model-common.safetensors",
|
| 302 |
+
"model.layers.3.mlp.experts.down_proj": "model-layer-03.safetensors",
|
| 303 |
+
"model.layers.3.mlp.experts.gate_up_proj": "model-layer-03.safetensors",
|
| 304 |
+
"model.layers.3.mlp.gate.weight": "model-layer-03.safetensors",
|
| 305 |
+
"model.layers.3.mlp.shared_expert.down_proj.weight": "model-layer-03.safetensors",
|
| 306 |
+
"model.layers.3.mlp.shared_expert.gate_proj.weight": "model-layer-03.safetensors",
|
| 307 |
+
"model.layers.3.mlp.shared_expert.up_proj.weight": "model-layer-03.safetensors",
|
| 308 |
+
"model.layers.3.mlp.shared_expert_gate.weight": "model-layer-03.safetensors",
|
| 309 |
+
"model.layers.3.post_attention_layernorm.weight": "model-common.safetensors",
|
| 310 |
+
"model.layers.3.self_attn.k_norm.weight": "model-common.safetensors",
|
| 311 |
+
"model.layers.3.self_attn.k_proj.weight": "model-common.safetensors",
|
| 312 |
+
"model.layers.3.self_attn.o_proj.weight": "model-common.safetensors",
|
| 313 |
+
"model.layers.3.self_attn.q_norm.weight": "model-common.safetensors",
|
| 314 |
+
"model.layers.3.self_attn.q_proj.weight": "model-common.safetensors",
|
| 315 |
+
"model.layers.3.self_attn.v_proj.weight": "model-common.safetensors",
|
| 316 |
+
"model.layers.4.input_layernorm.weight": "model-common.safetensors",
|
| 317 |
+
"model.layers.4.linear_attn.A_log": "model-common.safetensors",
|
| 318 |
+
"model.layers.4.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 319 |
+
"model.layers.4.linear_attn.dt_bias": "model-common.safetensors",
|
| 320 |
+
"model.layers.4.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 321 |
+
"model.layers.4.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 322 |
+
"model.layers.4.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 323 |
+
"model.layers.4.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 324 |
+
"model.layers.4.linear_attn.norm.weight": "model-common.safetensors",
|
| 325 |
+
"model.layers.4.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 326 |
+
"model.layers.4.mlp.experts.down_proj": "model-layer-04.safetensors",
|
| 327 |
+
"model.layers.4.mlp.experts.gate_up_proj": "model-layer-04.safetensors",
|
| 328 |
+
"model.layers.4.mlp.gate.weight": "model-layer-04.safetensors",
|
| 329 |
+
"model.layers.4.mlp.shared_expert.down_proj.weight": "model-layer-04.safetensors",
|
| 330 |
+
"model.layers.4.mlp.shared_expert.gate_proj.weight": "model-layer-04.safetensors",
|
| 331 |
+
"model.layers.4.mlp.shared_expert.up_proj.weight": "model-layer-04.safetensors",
|
| 332 |
+
"model.layers.4.mlp.shared_expert_gate.weight": "model-layer-04.safetensors",
|
| 333 |
+
"model.layers.4.post_attention_layernorm.weight": "model-common.safetensors",
|
| 334 |
+
"model.layers.5.input_layernorm.weight": "model-common.safetensors",
|
| 335 |
+
"model.layers.5.linear_attn.A_log": "model-common.safetensors",
|
| 336 |
+
"model.layers.5.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 337 |
+
"model.layers.5.linear_attn.dt_bias": "model-common.safetensors",
|
| 338 |
+
"model.layers.5.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 339 |
+
"model.layers.5.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 340 |
+
"model.layers.5.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 341 |
+
"model.layers.5.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 342 |
+
"model.layers.5.linear_attn.norm.weight": "model-common.safetensors",
|
| 343 |
+
"model.layers.5.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 344 |
+
"model.layers.5.mlp.experts.down_proj": "model-layer-05.safetensors",
|
| 345 |
+
"model.layers.5.mlp.experts.gate_up_proj": "model-layer-05.safetensors",
|
| 346 |
+
"model.layers.5.mlp.gate.weight": "model-layer-05.safetensors",
|
| 347 |
+
"model.layers.5.mlp.shared_expert.down_proj.weight": "model-layer-05.safetensors",
|
| 348 |
+
"model.layers.5.mlp.shared_expert.gate_proj.weight": "model-layer-05.safetensors",
|
| 349 |
+
"model.layers.5.mlp.shared_expert.up_proj.weight": "model-layer-05.safetensors",
|
| 350 |
+
"model.layers.5.mlp.shared_expert_gate.weight": "model-layer-05.safetensors",
|
| 351 |
+
"model.layers.5.post_attention_layernorm.weight": "model-common.safetensors",
|
| 352 |
+
"model.layers.6.input_layernorm.weight": "model-common.safetensors",
|
| 353 |
+
"model.layers.6.linear_attn.A_log": "model-common.safetensors",
|
| 354 |
+
"model.layers.6.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 355 |
+
"model.layers.6.linear_attn.dt_bias": "model-common.safetensors",
|
| 356 |
+
"model.layers.6.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 357 |
+
"model.layers.6.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 358 |
+
"model.layers.6.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 359 |
+
"model.layers.6.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 360 |
+
"model.layers.6.linear_attn.norm.weight": "model-common.safetensors",
|
| 361 |
+
"model.layers.6.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 362 |
+
"model.layers.6.mlp.experts.down_proj": "model-layer-06.safetensors",
|
| 363 |
+
"model.layers.6.mlp.experts.gate_up_proj": "model-layer-06.safetensors",
|
| 364 |
+
"model.layers.6.mlp.gate.weight": "model-layer-06.safetensors",
|
| 365 |
+
"model.layers.6.mlp.shared_expert.down_proj.weight": "model-layer-06.safetensors",
|
| 366 |
+
"model.layers.6.mlp.shared_expert.gate_proj.weight": "model-layer-06.safetensors",
|
| 367 |
+
"model.layers.6.mlp.shared_expert.up_proj.weight": "model-layer-06.safetensors",
|
| 368 |
+
"model.layers.6.mlp.shared_expert_gate.weight": "model-layer-06.safetensors",
|
| 369 |
+
"model.layers.6.post_attention_layernorm.weight": "model-common.safetensors",
|
| 370 |
+
"model.layers.7.input_layernorm.weight": "model-common.safetensors",
|
| 371 |
+
"model.layers.7.mlp.experts.down_proj": "model-layer-07.safetensors",
|
| 372 |
+
"model.layers.7.mlp.experts.gate_up_proj": "model-layer-07.safetensors",
|
| 373 |
+
"model.layers.7.mlp.gate.weight": "model-layer-07.safetensors",
|
| 374 |
+
"model.layers.7.mlp.shared_expert.down_proj.weight": "model-layer-07.safetensors",
|
| 375 |
+
"model.layers.7.mlp.shared_expert.gate_proj.weight": "model-layer-07.safetensors",
|
| 376 |
+
"model.layers.7.mlp.shared_expert.up_proj.weight": "model-layer-07.safetensors",
|
| 377 |
+
"model.layers.7.mlp.shared_expert_gate.weight": "model-layer-07.safetensors",
|
| 378 |
+
"model.layers.7.post_attention_layernorm.weight": "model-common.safetensors",
|
| 379 |
+
"model.layers.7.self_attn.k_norm.weight": "model-common.safetensors",
|
| 380 |
+
"model.layers.7.self_attn.k_proj.weight": "model-common.safetensors",
|
| 381 |
+
"model.layers.7.self_attn.o_proj.weight": "model-common.safetensors",
|
| 382 |
+
"model.layers.7.self_attn.q_norm.weight": "model-common.safetensors",
|
| 383 |
+
"model.layers.7.self_attn.q_proj.weight": "model-common.safetensors",
|
| 384 |
+
"model.layers.7.self_attn.v_proj.weight": "model-common.safetensors",
|
| 385 |
+
"model.layers.8.input_layernorm.weight": "model-common.safetensors",
|
| 386 |
+
"model.layers.8.linear_attn.A_log": "model-common.safetensors",
|
| 387 |
+
"model.layers.8.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 388 |
+
"model.layers.8.linear_attn.dt_bias": "model-common.safetensors",
|
| 389 |
+
"model.layers.8.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 390 |
+
"model.layers.8.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 391 |
+
"model.layers.8.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 392 |
+
"model.layers.8.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 393 |
+
"model.layers.8.linear_attn.norm.weight": "model-common.safetensors",
|
| 394 |
+
"model.layers.8.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 395 |
+
"model.layers.8.mlp.experts.down_proj": "model-layer-08.safetensors",
|
| 396 |
+
"model.layers.8.mlp.experts.gate_up_proj": "model-layer-08.safetensors",
|
| 397 |
+
"model.layers.8.mlp.gate.weight": "model-layer-08.safetensors",
|
| 398 |
+
"model.layers.8.mlp.shared_expert.down_proj.weight": "model-layer-08.safetensors",
|
| 399 |
+
"model.layers.8.mlp.shared_expert.gate_proj.weight": "model-layer-08.safetensors",
|
| 400 |
+
"model.layers.8.mlp.shared_expert.up_proj.weight": "model-layer-08.safetensors",
|
| 401 |
+
"model.layers.8.mlp.shared_expert_gate.weight": "model-layer-08.safetensors",
|
| 402 |
+
"model.layers.8.post_attention_layernorm.weight": "model-common.safetensors",
|
| 403 |
+
"model.layers.9.input_layernorm.weight": "model-common.safetensors",
|
| 404 |
+
"model.layers.9.linear_attn.A_log": "model-common.safetensors",
|
| 405 |
+
"model.layers.9.linear_attn.conv1d.weight": "model-common.safetensors",
|
| 406 |
+
"model.layers.9.linear_attn.dt_bias": "model-common.safetensors",
|
| 407 |
+
"model.layers.9.linear_attn.in_proj_a.weight": "model-common.safetensors",
|
| 408 |
+
"model.layers.9.linear_attn.in_proj_b.weight": "model-common.safetensors",
|
| 409 |
+
"model.layers.9.linear_attn.in_proj_qkv.weight": "model-common.safetensors",
|
| 410 |
+
"model.layers.9.linear_attn.in_proj_z.weight": "model-common.safetensors",
|
| 411 |
+
"model.layers.9.linear_attn.norm.weight": "model-common.safetensors",
|
| 412 |
+
"model.layers.9.linear_attn.out_proj.weight": "model-common.safetensors",
|
| 413 |
+
"model.layers.9.mlp.experts.down_proj": "model-layer-09.safetensors",
|
| 414 |
+
"model.layers.9.mlp.experts.gate_up_proj": "model-layer-09.safetensors",
|
| 415 |
+
"model.layers.9.mlp.gate.weight": "model-layer-09.safetensors",
|
| 416 |
+
"model.layers.9.mlp.shared_expert.down_proj.weight": "model-layer-09.safetensors",
|
| 417 |
+
"model.layers.9.mlp.shared_expert.gate_proj.weight": "model-layer-09.safetensors",
|
| 418 |
+
"model.layers.9.mlp.shared_expert.up_proj.weight": "model-layer-09.safetensors",
|
| 419 |
+
"model.layers.9.mlp.shared_expert_gate.weight": "model-layer-09.safetensors",
|
| 420 |
+
"model.layers.9.post_attention_layernorm.weight": "model-common.safetensors",
|
| 421 |
+
"model.norm.weight": "model-common.safetensors"
|
| 422 |
+
}
|
| 423 |
+
}
|
research_code/chat-gate-60.jsonl
ADDED
|
@@ -0,0 +1,60 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{"id":"ko_01","language":"ko","prompt":"프랑스의 수도는 어디인가요? 짧게 답하세요.","expected_any":["파리"]}
|
| 2 |
+
{"id":"ko_02","language":"ko","prompt":"7 곱하기 8은 얼마인가요? 숫자로 답하세요.","expected_any":["56"]}
|
| 3 |
+
{"id":"ko_03","language":"ko","prompt":"물의 화학식은 무엇인가요?","expected_any":["H2O","h2o"]}
|
| 4 |
+
{"id":"ko_04","language":"ko","prompt":"태양계에서 가장 큰 행성은 무엇인가요?","expected_any":["목성"]}
|
| 5 |
+
{"id":"ko_05","language":"ko","prompt":"조용한 겨울 아침을 묘사하는 자연스러운 문장 두 개를 써 주세요."}
|
| 6 |
+
{"id":"ko_06","language":"ko","prompt":"친구에게 약속 시간을 10분 늦겠다고 정중히 알리는 한 문장을 써 주세요."}
|
| 7 |
+
{"id":"ko_07","language":"ko","prompt":"인공지능 연구에서 재현성이 중요한 이유를 두 문장으로 설명하세요."}
|
| 8 |
+
{"id":"ko_08","language":"ko","prompt":"다른 말 없이 정확히 '확인'이라고만 답하세요.","expected_any":["확인"]}
|
| 9 |
+
{"id":"ko_09","language":"ko","prompt":"사과, 바나나, 포도를 번호가 있는 세 항목으로 나열하세요."}
|
| 10 |
+
{"id":"ko_10","language":"ko","prompt":"'Good morning'을 자연스러운 한국어로 번역하세요.","expected_any":["좋은 아침","안녕하세요"]}
|
| 11 |
+
{"id":"en_01","language":"en","prompt":"What is the capital of France? Answer briefly.","expected_any":["Paris"]}
|
| 12 |
+
{"id":"en_02","language":"en","prompt":"What is 7 multiplied by 8? Answer with a number.","expected_any":["56"]}
|
| 13 |
+
{"id":"en_03","language":"en","prompt":"What is the chemical formula for water?","expected_any":["H2O","h2o"]}
|
| 14 |
+
{"id":"en_04","language":"en","prompt":"What is the largest planet in the Solar System?","expected_any":["Jupiter"]}
|
| 15 |
+
{"id":"en_05","language":"en","prompt":"Write two natural sentences describing a quiet winter morning."}
|
| 16 |
+
{"id":"en_06","language":"en","prompt":"Write one polite sentence telling a friend you will be ten minutes late."}
|
| 17 |
+
{"id":"en_07","language":"en","prompt":"Explain in two sentences why reproducibility matters in AI research."}
|
| 18 |
+
{"id":"en_08","language":"en","prompt":"Reply with exactly the word 'confirmed' and nothing else.","expected_any":["confirmed"]}
|
| 19 |
+
{"id":"en_09","language":"en","prompt":"List apple, banana, and grape as three numbered items."}
|
| 20 |
+
{"id":"en_10","language":"en","prompt":"Translate '좋은 아침입니다' into natural English.","expected_any":["good morning"]}
|
| 21 |
+
{"id":"zh_01","language":"zh","prompt":"法国的首都是哪里?请简短回答。","expected_any":["巴黎"]}
|
| 22 |
+
{"id":"zh_02","language":"zh","prompt":"7乘以8等于多少?请用数字回答。","expected_any":["56"]}
|
| 23 |
+
{"id":"zh_03","language":"zh","prompt":"水的化学式是什么?","expected_any":["H2O","h2o"]}
|
| 24 |
+
{"id":"zh_04","language":"zh","prompt":"太阳系中最大的行星是什么?","expected_any":["木星"]}
|
| 25 |
+
{"id":"zh_05","language":"zh","prompt":"用两个自然的句子描写安静的冬日清晨。"}
|
| 26 |
+
{"id":"zh_06","language":"zh","prompt":"写一句礼貌的话,告诉朋友你会迟到十分钟。"}
|
| 27 |
+
{"id":"zh_07","language":"zh","prompt":"用两句话解释可复现性为何对人工智能研究重要。"}
|
| 28 |
+
{"id":"zh_08","language":"zh","prompt":"不要说别的,只回答“收到”。","expected_any":["收到"]}
|
| 29 |
+
{"id":"zh_09","language":"zh","prompt":"把苹果、香蕉和葡萄列成三个编号项目。"}
|
| 30 |
+
{"id":"zh_10","language":"zh","prompt":"把“Good morning”翻译成自然的中文。","expected_any":["早上好","早安"]}
|
| 31 |
+
{"id":"ja_01","language":"ja","prompt":"フランスの首都はどこですか。短く答えてください。","expected_any":["パリ"]}
|
| 32 |
+
{"id":"ja_02","language":"ja","prompt":"7かける8はいくつですか。数字で答えてください。","expected_any":["56"]}
|
| 33 |
+
{"id":"ja_03","language":"ja","prompt":"水の化学式は何ですか。","expected_any":["H2O","h2o"]}
|
| 34 |
+
{"id":"ja_04","language":"ja","prompt":"太陽系で最も大きい惑星は何ですか。","expected_any":["木星"]}
|
| 35 |
+
{"id":"ja_05","language":"ja","prompt":"静かな冬の朝を自然な二文で描写してください。"}
|
| 36 |
+
{"id":"ja_06","language":"ja","prompt":"友人に10分遅れることを丁寧に伝える一文を書いてください。"}
|
| 37 |
+
{"id":"ja_07","language":"ja","prompt":"AI研究で再現性が重要な理由を二文で説明してください。"}
|
| 38 |
+
{"id":"ja_08","language":"ja","prompt":"ほかの言葉を加えず「了解」とだけ答えてください。","expected_any":["了解"]}
|
| 39 |
+
{"id":"ja_09","language":"ja","prompt":"りんご、バナナ、ぶどうを番号付きの三項目で並べてください。"}
|
| 40 |
+
{"id":"ja_10","language":"ja","prompt":"「Good morning」を自然な日本語に訳してください。","expected_any":["おはよう"]}
|
| 41 |
+
{"id":"es_01","language":"es","prompt":"¿Cuál es la capital de Francia? Responde brevemente.","expected_any":["París","Paris"]}
|
| 42 |
+
{"id":"es_02","language":"es","prompt":"¿Cuánto es 7 por 8? Responde con un número.","expected_any":["56"]}
|
| 43 |
+
{"id":"es_03","language":"es","prompt":"¿Cuál es la fórmula química del agua?","expected_any":["H2O","h2o"]}
|
| 44 |
+
{"id":"es_04","language":"es","prompt":"¿Cuál es el planeta más grande del sistema solar?","expected_any":["Júpiter","Jupiter"]}
|
| 45 |
+
{"id":"es_05","language":"es","prompt":"Escribe dos frases naturales que describan una tranquila mañana de invierno."}
|
| 46 |
+
{"id":"es_06","language":"es","prompt":"Escribe una frase cortés para decirle a un amigo que llegarás diez minutos tarde."}
|
| 47 |
+
{"id":"es_07","language":"es","prompt":"Explica en dos frases por qué la reproducibilidad importa en la investigación de IA."}
|
| 48 |
+
{"id":"es_08","language":"es","prompt":"Responde únicamente con la palabra 'entendido'.","expected_any":["entendido"]}
|
| 49 |
+
{"id":"es_09","language":"es","prompt":"Enumera manzana, plátano y uva como tres elementos numerados."}
|
| 50 |
+
{"id":"es_10","language":"es","prompt":"Traduce 'Good morning' a un español natural.","expected_any":["buenos días","buen día"]}
|
| 51 |
+
{"id":"de_01","language":"de","prompt":"Was ist die Hauptstadt von Frankreich? Antworte kurz.","expected_any":["Paris"]}
|
| 52 |
+
{"id":"de_02","language":"de","prompt":"Was ist 7 mal 8? Antworte mit einer Zahl.","expected_any":["56"]}
|
| 53 |
+
{"id":"de_03","language":"de","prompt":"Wie lautet die chemische Formel für Wasser?","expected_any":["H2O","h2o"]}
|
| 54 |
+
{"id":"de_04","language":"de","prompt":"Welcher Planet ist der größte im Sonnensystem?","expected_any":["Jupiter"]}
|
| 55 |
+
{"id":"de_05","language":"de","prompt":"Schreibe zwei natürliche Sätze über einen ruhigen Wintermorgen."}
|
| 56 |
+
{"id":"de_06","language":"de","prompt":"Schreibe einen höflichen Satz, der einem Freund sagt, dass du zehn Minuten zu spät kommst."}
|
| 57 |
+
{"id":"de_07","language":"de","prompt":"Erkläre in zwei Sätzen, warum Reproduzierbarkeit in der KI-Forschung wichtig ist."}
|
| 58 |
+
{"id":"de_08","language":"de","prompt":"Antworte ausschließlich mit dem Wort 'Verstanden'.","expected_any":["verstanden"]}
|
| 59 |
+
{"id":"de_09","language":"de","prompt":"Liste Apfel, Banane und Traube als drei nummerierte Punkte auf."}
|
| 60 |
+
{"id":"de_10","language":"de","prompt":"Übersetze 'Good morning' in natürliches Deutsch.","expected_any":["guten morgen"]}
|
research_code/compare_lm_loss.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import argparse
|
| 5 |
+
import gc
|
| 6 |
+
import json
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
import torch
|
| 10 |
+
from transformers import AutoConfig, AutoTokenizer
|
| 11 |
+
from transformers import Qwen3_5ForConditionalGeneration
|
| 12 |
+
from transformers import Qwen3_5MoeForCausalLM
|
| 13 |
+
from transformers import Qwen3_5MoeForConditionalGeneration
|
| 14 |
+
|
| 15 |
+
|
| 16 |
+
def parse_model(value: str) -> tuple[str, Path]:
|
| 17 |
+
if "=" not in value:
|
| 18 |
+
raise argparse.ArgumentTypeError("model must use NAME=PATH")
|
| 19 |
+
name, path = value.split("=", 1)
|
| 20 |
+
return name, Path(path)
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
def load_rows(path: Path, per_language: int) -> dict[str, list[str]]:
|
| 24 |
+
rows: dict[str, list[str]] = {}
|
| 25 |
+
with path.open(encoding="utf-8") as handle:
|
| 26 |
+
for line in handle:
|
| 27 |
+
row = json.loads(line)
|
| 28 |
+
if row.get("split") != "eval":
|
| 29 |
+
continue
|
| 30 |
+
bucket = rows.setdefault(row["language"], [])
|
| 31 |
+
if len(bucket) < per_language:
|
| 32 |
+
bucket.append(row["text"])
|
| 33 |
+
return rows
|
| 34 |
+
|
| 35 |
+
|
| 36 |
+
def model_class(path: Path):
|
| 37 |
+
config = AutoConfig.from_pretrained(path)
|
| 38 |
+
if config.model_type == "qwen3_5_moe_text":
|
| 39 |
+
return Qwen3_5MoeForCausalLM
|
| 40 |
+
if config.model_type == "qwen3_5_moe":
|
| 41 |
+
return Qwen3_5MoeForConditionalGeneration
|
| 42 |
+
return Qwen3_5ForConditionalGeneration
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def evaluate(path: Path, tokenizer, rows, sequence_length: int, device: str):
|
| 46 |
+
model = model_class(path).from_pretrained(path, dtype=torch.bfloat16).to(device).eval()
|
| 47 |
+
model.config.use_cache = False
|
| 48 |
+
results = {}
|
| 49 |
+
with torch.no_grad():
|
| 50 |
+
for language, texts in rows.items():
|
| 51 |
+
losses = []
|
| 52 |
+
token_count = 0
|
| 53 |
+
for text in texts:
|
| 54 |
+
batch = tokenizer(
|
| 55 |
+
text,
|
| 56 |
+
return_tensors="pt",
|
| 57 |
+
truncation=True,
|
| 58 |
+
max_length=sequence_length,
|
| 59 |
+
)
|
| 60 |
+
batch = {key: value.to(device) for key, value in batch.items()}
|
| 61 |
+
output = model(
|
| 62 |
+
**batch,
|
| 63 |
+
labels=batch["input_ids"],
|
| 64 |
+
use_cache=False,
|
| 65 |
+
output_router_logits=False,
|
| 66 |
+
)
|
| 67 |
+
losses.append(float(output.loss.cpu()))
|
| 68 |
+
token_count += int(batch["attention_mask"].sum())
|
| 69 |
+
results[language] = {
|
| 70 |
+
"loss": sum(losses) / len(losses),
|
| 71 |
+
"documents": len(losses),
|
| 72 |
+
"tokens": token_count,
|
| 73 |
+
}
|
| 74 |
+
del model
|
| 75 |
+
gc.collect()
|
| 76 |
+
if device == "mps":
|
| 77 |
+
torch.mps.empty_cache()
|
| 78 |
+
return results
|
| 79 |
+
|
| 80 |
+
|
| 81 |
+
def main() -> None:
|
| 82 |
+
parser = argparse.ArgumentParser()
|
| 83 |
+
parser.add_argument("--model", action="append", type=parse_model, required=True)
|
| 84 |
+
parser.add_argument("--tokenizer", type=Path, required=True)
|
| 85 |
+
parser.add_argument("--corpus", type=Path, required=True)
|
| 86 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 87 |
+
parser.add_argument("--documents-per-language", type=int, default=4)
|
| 88 |
+
parser.add_argument("--sequence-length", type=int, default=128)
|
| 89 |
+
parser.add_argument("--device", default="mps")
|
| 90 |
+
args = parser.parse_args()
|
| 91 |
+
|
| 92 |
+
rows = load_rows(args.corpus, args.documents_per_language)
|
| 93 |
+
tokenizer = AutoTokenizer.from_pretrained(args.tokenizer)
|
| 94 |
+
models = {}
|
| 95 |
+
for name, path in args.model:
|
| 96 |
+
print(f"evaluating {name}", flush=True)
|
| 97 |
+
models[name] = evaluate(path, tokenizer, rows, args.sequence_length, args.device)
|
| 98 |
+
|
| 99 |
+
means = {
|
| 100 |
+
name: sum(row["loss"] for row in result.values()) / len(result)
|
| 101 |
+
for name, result in models.items()
|
| 102 |
+
}
|
| 103 |
+
report = {
|
| 104 |
+
"sequence_length": args.sequence_length,
|
| 105 |
+
"documents_per_language": args.documents_per_language,
|
| 106 |
+
"languages": sorted(rows),
|
| 107 |
+
"models": models,
|
| 108 |
+
"mean_losses": means,
|
| 109 |
+
}
|
| 110 |
+
args.output.parent.mkdir(parents=True, exist_ok=True)
|
| 111 |
+
args.output.write_text(json.dumps(report, indent=2))
|
| 112 |
+
print(json.dumps(report, indent=2))
|
| 113 |
+
|
| 114 |
+
|
| 115 |
+
if __name__ == "__main__":
|
| 116 |
+
main()
|
research_code/convert_2b_to_4b_a3b.py
ADDED
|
@@ -0,0 +1,228 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Expand Qwen3.5-2B into a text-only 4B-total/3B-active sparse MoE.
|
| 3 |
+
|
| 4 |
+
The initial checkpoint preserves the dense 2B MLP exactly: the shared path and
|
| 5 |
+
the selected routed expert each contribute one half of the original output.
|
| 6 |
+
Extra neurons start output-neutral but trainable.
|
| 7 |
+
"""
|
| 8 |
+
from __future__ import annotations
|
| 9 |
+
|
| 10 |
+
import argparse
|
| 11 |
+
import json
|
| 12 |
+
import shutil
|
| 13 |
+
from pathlib import Path
|
| 14 |
+
|
| 15 |
+
import torch
|
| 16 |
+
from safetensors import safe_open
|
| 17 |
+
from safetensors.torch import save_file
|
| 18 |
+
from transformers import Qwen3_5MoeForCausalLM, Qwen3_5MoeTextConfig
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
NUM_EXPERTS = 2
|
| 22 |
+
EXPERTS_PER_TOKEN = 1
|
| 23 |
+
SHARED_WIDTH = 6912
|
| 24 |
+
ROUTED_WIDTH = 6784
|
| 25 |
+
SEED = 35
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def source_tensor(source: Path, weight_map: dict[str, str], name: str) -> torch.Tensor:
|
| 29 |
+
with safe_open(source / weight_map[name], framework="pt", device="cpu") as handle:
|
| 30 |
+
return handle.get_tensor(name)
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def save_shard(
|
| 34 |
+
output: Path,
|
| 35 |
+
filename: str,
|
| 36 |
+
tensors: dict[str, torch.Tensor],
|
| 37 |
+
weight_map: dict[str, str],
|
| 38 |
+
) -> int:
|
| 39 |
+
tensors = {name: tensor.contiguous() for name, tensor in tensors.items()}
|
| 40 |
+
save_file(tensors, output / filename, metadata={"format": "pt"})
|
| 41 |
+
weight_map.update({name: filename for name in tensors})
|
| 42 |
+
return sum(tensor.numel() * tensor.element_size() for tensor in tensors.values())
|
| 43 |
+
|
| 44 |
+
|
| 45 |
+
def build_config(source_config: dict) -> dict:
|
| 46 |
+
text = dict(source_config["text_config"])
|
| 47 |
+
dense_width = int(text.pop("intermediate_size"))
|
| 48 |
+
if text["hidden_size"] != 2048 or text["num_hidden_layers"] != 24 or dense_width != 6144:
|
| 49 |
+
raise ValueError("converter is intentionally pinned to Qwen3.5-2B")
|
| 50 |
+
text.update({
|
| 51 |
+
"architectures": ["Qwen3_5MoeForCausalLM"],
|
| 52 |
+
"model_type": "qwen3_5_moe_text",
|
| 53 |
+
"num_experts": NUM_EXPERTS,
|
| 54 |
+
"num_experts_per_tok": EXPERTS_PER_TOKEN,
|
| 55 |
+
"moe_intermediate_size": ROUTED_WIDTH,
|
| 56 |
+
"shared_expert_intermediate_size": SHARED_WIDTH,
|
| 57 |
+
"router_aux_loss_coef": 1e-3,
|
| 58 |
+
"output_router_logits": False,
|
| 59 |
+
})
|
| 60 |
+
return text
|
| 61 |
+
|
| 62 |
+
|
| 63 |
+
def main() -> None:
|
| 64 |
+
parser = argparse.ArgumentParser()
|
| 65 |
+
parser.add_argument("--source", type=Path, required=True)
|
| 66 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 67 |
+
args = parser.parse_args()
|
| 68 |
+
|
| 69 |
+
if args.output.exists() and any(args.output.iterdir()):
|
| 70 |
+
raise SystemExit(f"Refusing non-empty output directory: {args.output}")
|
| 71 |
+
args.output.mkdir(parents=True, exist_ok=True)
|
| 72 |
+
|
| 73 |
+
source_config = json.loads((args.source / "config.json").read_text())
|
| 74 |
+
config_dict = build_config(source_config)
|
| 75 |
+
config = Qwen3_5MoeTextConfig(**config_dict)
|
| 76 |
+
with torch.device("meta"):
|
| 77 |
+
target = Qwen3_5MoeForCausalLM(config)
|
| 78 |
+
target_keys = set(target.state_dict())
|
| 79 |
+
total_parameters = sum(parameter.numel() for parameter in target.parameters())
|
| 80 |
+
inactive = (
|
| 81 |
+
config.num_hidden_layers
|
| 82 |
+
* (NUM_EXPERTS - EXPERTS_PER_TOKEN)
|
| 83 |
+
* 3
|
| 84 |
+
* config.hidden_size
|
| 85 |
+
* ROUTED_WIDTH
|
| 86 |
+
)
|
| 87 |
+
active_parameters = total_parameters - inactive
|
| 88 |
+
|
| 89 |
+
source_index = json.loads((args.source / "model.safetensors.index.json").read_text())
|
| 90 |
+
source_map: dict[str, str] = source_index["weight_map"]
|
| 91 |
+
output_map: dict[str, str] = {}
|
| 92 |
+
total_bytes = 0
|
| 93 |
+
|
| 94 |
+
# Copy the text backbone while dropping the vision tower and dense MLPs.
|
| 95 |
+
common: dict[str, torch.Tensor] = {}
|
| 96 |
+
prefix = "model.language_model."
|
| 97 |
+
for shard in sorted(set(source_map.values())):
|
| 98 |
+
with safe_open(args.source / shard, framework="pt", device="cpu") as handle:
|
| 99 |
+
for source_name in handle.keys():
|
| 100 |
+
if not source_name.startswith(prefix) or ".mlp." in source_name:
|
| 101 |
+
continue
|
| 102 |
+
target_name = "model." + source_name[len(prefix):]
|
| 103 |
+
if target_name in target_keys:
|
| 104 |
+
common[target_name] = handle.get_tensor(source_name)
|
| 105 |
+
total_bytes += save_shard(args.output, "model-common.safetensors", common, output_map)
|
| 106 |
+
|
| 107 |
+
generator = torch.Generator(device="cpu").manual_seed(SEED)
|
| 108 |
+
dense_width = 6144
|
| 109 |
+
hidden = config.hidden_size
|
| 110 |
+
init_std = float(config.initializer_range)
|
| 111 |
+
for layer in range(config.num_hidden_layers):
|
| 112 |
+
src = f"model.language_model.layers.{layer}.mlp"
|
| 113 |
+
dst = f"model.layers.{layer}.mlp"
|
| 114 |
+
dense_gate = source_tensor(args.source, source_map, f"{src}.gate_proj.weight")
|
| 115 |
+
dense_up = source_tensor(args.source, source_map, f"{src}.up_proj.weight")
|
| 116 |
+
dense_down = source_tensor(args.source, source_map, f"{src}.down_proj.weight")
|
| 117 |
+
dtype = dense_gate.dtype
|
| 118 |
+
|
| 119 |
+
shared_gate = torch.randn(SHARED_WIDTH, hidden, generator=generator, dtype=torch.float32)
|
| 120 |
+
shared_up = torch.randn(SHARED_WIDTH, hidden, generator=generator, dtype=torch.float32)
|
| 121 |
+
shared_down = torch.zeros(hidden, SHARED_WIDTH, dtype=dtype)
|
| 122 |
+
shared_gate.mul_(init_std).to(dtype=dtype)
|
| 123 |
+
shared_up.mul_(init_std).to(dtype=dtype)
|
| 124 |
+
shared_gate = shared_gate.to(dtype)
|
| 125 |
+
shared_up = shared_up.to(dtype)
|
| 126 |
+
shared_gate[:dense_width] = dense_gate
|
| 127 |
+
shared_up[:dense_width] = dense_up
|
| 128 |
+
# sigmoid(shared_expert_gate=0) gives the shared path a 0.5 multiplier.
|
| 129 |
+
shared_down[:, :dense_width] = dense_down
|
| 130 |
+
|
| 131 |
+
routed_gate_up = torch.empty(
|
| 132 |
+
NUM_EXPERTS, 2 * ROUTED_WIDTH, hidden, dtype=dtype
|
| 133 |
+
)
|
| 134 |
+
routed_down = torch.zeros(NUM_EXPERTS, hidden, ROUTED_WIDTH, dtype=dtype)
|
| 135 |
+
for expert in range(NUM_EXPERTS):
|
| 136 |
+
extra_gate = torch.randn(
|
| 137 |
+
ROUTED_WIDTH, hidden, generator=generator, dtype=torch.float32
|
| 138 |
+
).mul_(init_std).to(dtype)
|
| 139 |
+
extra_up = torch.randn(
|
| 140 |
+
ROUTED_WIDTH, hidden, generator=generator, dtype=torch.float32
|
| 141 |
+
).mul_(init_std).to(dtype)
|
| 142 |
+
extra_gate[:dense_width] = dense_gate
|
| 143 |
+
extra_up[:dense_width] = dense_up
|
| 144 |
+
routed_gate_up[expert, :ROUTED_WIDTH] = extra_gate
|
| 145 |
+
routed_gate_up[expert, ROUTED_WIDTH:] = extra_up
|
| 146 |
+
# Routed path supplies the other half of the original dense output.
|
| 147 |
+
routed_down[expert, :, :dense_width] = dense_down * 0.5
|
| 148 |
+
|
| 149 |
+
tensors = {
|
| 150 |
+
f"{dst}.gate.weight": torch.zeros(NUM_EXPERTS, hidden, dtype=dtype),
|
| 151 |
+
f"{dst}.experts.gate_up_proj": routed_gate_up,
|
| 152 |
+
f"{dst}.experts.down_proj": routed_down,
|
| 153 |
+
f"{dst}.shared_expert.gate_proj.weight": shared_gate,
|
| 154 |
+
f"{dst}.shared_expert.up_proj.weight": shared_up,
|
| 155 |
+
f"{dst}.shared_expert.down_proj.weight": shared_down,
|
| 156 |
+
f"{dst}.shared_expert_gate.weight": torch.zeros(1, hidden, dtype=dtype),
|
| 157 |
+
}
|
| 158 |
+
total_bytes += save_shard(
|
| 159 |
+
args.output, f"model-layer-{layer:02d}.safetensors", tensors, output_map
|
| 160 |
+
)
|
| 161 |
+
print(f"converted layer {layer + 1}/{config.num_hidden_layers}", flush=True)
|
| 162 |
+
|
| 163 |
+
missing = sorted(target_keys - set(output_map) - {"lm_head.weight"})
|
| 164 |
+
unexpected = sorted(set(output_map) - target_keys)
|
| 165 |
+
if missing or unexpected:
|
| 166 |
+
raise RuntimeError(f"key audit failed: missing={missing[:20]} unexpected={unexpected[:20]}")
|
| 167 |
+
|
| 168 |
+
(args.output / "model.safetensors.index.json").write_text(json.dumps({
|
| 169 |
+
"metadata": {"total_size": total_bytes},
|
| 170 |
+
"weight_map": dict(sorted(output_map.items())),
|
| 171 |
+
}, indent=2))
|
| 172 |
+
(args.output / "config.json").write_text(config.to_json_string())
|
| 173 |
+
(args.output / "conversion_manifest.json").write_text(json.dumps({
|
| 174 |
+
"source": str(args.source),
|
| 175 |
+
"initial_function": "Qwen3.5-2B text model (dense MLP split 50/50)",
|
| 176 |
+
"total_parameters": total_parameters,
|
| 177 |
+
"active_parameters": active_parameters,
|
| 178 |
+
"num_experts": NUM_EXPERTS,
|
| 179 |
+
"experts_per_token": EXPERTS_PER_TOKEN,
|
| 180 |
+
"shared_intermediate_size": SHARED_WIDTH,
|
| 181 |
+
"routed_intermediate_size": ROUTED_WIDTH,
|
| 182 |
+
"vision_included": False,
|
| 183 |
+
"seed": SEED,
|
| 184 |
+
}, indent=2))
|
| 185 |
+
|
| 186 |
+
for filename in ("chat_template.jinja", "merges.txt", "tokenizer.json",
|
| 187 |
+
"tokenizer_config.json", "vocab.json", "LICENSE"):
|
| 188 |
+
source_file = args.source / filename
|
| 189 |
+
if source_file.exists():
|
| 190 |
+
shutil.copy2(source_file, args.output / filename)
|
| 191 |
+
|
| 192 |
+
(args.output / "README.md").write_text(f"""---
|
| 193 |
+
license: apache-2.0
|
| 194 |
+
base_model:
|
| 195 |
+
- Qwen/Qwen3.5-2B
|
| 196 |
+
- Qwen/Qwen3.5-4B
|
| 197 |
+
library_name: transformers
|
| 198 |
+
pipeline_tag: text-generation
|
| 199 |
+
tags: [qwen3_5_moe, moe, upcycled, research]
|
| 200 |
+
---
|
| 201 |
+
|
| 202 |
+
# Qwen3.5-4B-A3B-Student-v2
|
| 203 |
+
|
| 204 |
+
Text-only sparse-MoE research checkpoint initialized to preserve the text
|
| 205 |
+
generation function of Qwen3.5-2B. It is intended for distillation from
|
| 206 |
+
Qwen3.5-4B and is not yet claimed to match the 4B teacher.
|
| 207 |
+
|
| 208 |
+
| Property | Value |
|
| 209 |
+
|---|---:|
|
| 210 |
+
| Total parameters | {total_parameters:,} |
|
| 211 |
+
| Active parameters/token | {active_parameters:,} |
|
| 212 |
+
| Experts / selected | {NUM_EXPERTS} / {EXPERTS_PER_TOKEN} |
|
| 213 |
+
| Shared / routed width | {SHARED_WIDTH} / {ROUTED_WIDTH} |
|
| 214 |
+
| Vision | No |
|
| 215 |
+
|
| 216 |
+
The shared path and selected routed path initially contribute half of the dense
|
| 217 |
+
Qwen3.5-2B MLP each. Extra neurons are output-neutral at initialization but can
|
| 218 |
+
learn during distillation. See `conversion_manifest.json` for exact metadata.
|
| 219 |
+
""")
|
| 220 |
+
print(json.dumps({
|
| 221 |
+
"total_parameters": total_parameters,
|
| 222 |
+
"active_parameters": active_parameters,
|
| 223 |
+
"total_bytes": total_bytes,
|
| 224 |
+
}, indent=2))
|
| 225 |
+
|
| 226 |
+
|
| 227 |
+
if __name__ == "__main__":
|
| 228 |
+
main()
|
research_code/evaluate_chat_gate.py
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import argparse
|
| 5 |
+
import json
|
| 6 |
+
import re
|
| 7 |
+
import time
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
import torch
|
| 11 |
+
from transformers import AutoConfig, AutoTokenizer
|
| 12 |
+
from transformers import Qwen3_5ForConditionalGeneration
|
| 13 |
+
from transformers import Qwen3_5MoeForCausalLM
|
| 14 |
+
from transformers import Qwen3_5MoeForConditionalGeneration
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
def load_model(path: Path, device: str):
|
| 18 |
+
config = AutoConfig.from_pretrained(path)
|
| 19 |
+
if config.model_type == "qwen3_5_moe_text":
|
| 20 |
+
cls = Qwen3_5MoeForCausalLM
|
| 21 |
+
elif config.model_type == "qwen3_5_moe":
|
| 22 |
+
cls = Qwen3_5MoeForConditionalGeneration
|
| 23 |
+
else:
|
| 24 |
+
cls = Qwen3_5ForConditionalGeneration
|
| 25 |
+
return cls.from_pretrained(path, dtype=torch.bfloat16).to(device).eval()
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
def normalize(value: str) -> str:
|
| 29 |
+
subscripts = str.maketrans("₀₁₂₃₄₅₆₇₈₉", "0123456789")
|
| 30 |
+
return " ".join(value.casefold().translate(subscripts).split())
|
| 31 |
+
|
| 32 |
+
|
| 33 |
+
def score(item: dict, completion: str) -> tuple[bool, list[str]]:
|
| 34 |
+
text = completion.strip()
|
| 35 |
+
failures = []
|
| 36 |
+
if len(text) < 2:
|
| 37 |
+
failures.append("empty_or_too_short")
|
| 38 |
+
if text and sum(character.isspace() for character in text) / len(text) > 0.5:
|
| 39 |
+
failures.append("whitespace_dominated")
|
| 40 |
+
if re.search(r"(.)\1{7,}", text, flags=re.DOTALL):
|
| 41 |
+
failures.append("character_repetition")
|
| 42 |
+
words = re.findall(r"\w+", text.casefold())
|
| 43 |
+
if len(words) >= 12 and len(set(words)) < 4:
|
| 44 |
+
failures.append("word_repetition")
|
| 45 |
+
expected = item.get("expected_any")
|
| 46 |
+
if expected and not any(normalize(value) in normalize(text) for value in expected):
|
| 47 |
+
failures.append("expected_answer_missing")
|
| 48 |
+
return not failures, failures
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def main() -> None:
|
| 52 |
+
parser = argparse.ArgumentParser()
|
| 53 |
+
parser.add_argument("--model", type=Path, required=True)
|
| 54 |
+
parser.add_argument("--prompts", type=Path, required=True)
|
| 55 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 56 |
+
parser.add_argument("--device", default="mps")
|
| 57 |
+
parser.add_argument("--max-new-tokens", type=int, default=48)
|
| 58 |
+
args = parser.parse_args()
|
| 59 |
+
|
| 60 |
+
items = [json.loads(line) for line in args.prompts.read_text().splitlines() if line]
|
| 61 |
+
tokenizer = AutoTokenizer.from_pretrained(args.model)
|
| 62 |
+
model = load_model(args.model, args.device)
|
| 63 |
+
results = []
|
| 64 |
+
for index, item in enumerate(items, 1):
|
| 65 |
+
batch = tokenizer.apply_chat_template(
|
| 66 |
+
[{"role": "user", "content": item["prompt"]}],
|
| 67 |
+
tokenize=True,
|
| 68 |
+
add_generation_prompt=True,
|
| 69 |
+
enable_thinking=False,
|
| 70 |
+
return_tensors="pt",
|
| 71 |
+
return_dict=True,
|
| 72 |
+
).to(args.device)
|
| 73 |
+
started = time.perf_counter()
|
| 74 |
+
with torch.no_grad():
|
| 75 |
+
output = model.generate(
|
| 76 |
+
**batch,
|
| 77 |
+
max_new_tokens=args.max_new_tokens,
|
| 78 |
+
do_sample=False,
|
| 79 |
+
use_cache=True,
|
| 80 |
+
)
|
| 81 |
+
elapsed = time.perf_counter() - started
|
| 82 |
+
ids = output[0, batch["input_ids"].shape[1]:]
|
| 83 |
+
completion = tokenizer.decode(ids, skip_special_tokens=True)
|
| 84 |
+
passed, failures = score(item, completion)
|
| 85 |
+
result = {
|
| 86 |
+
**item,
|
| 87 |
+
"completion": completion,
|
| 88 |
+
"passed": passed,
|
| 89 |
+
"failures": failures,
|
| 90 |
+
"new_tokens": int(ids.numel()),
|
| 91 |
+
"elapsed_seconds": elapsed,
|
| 92 |
+
}
|
| 93 |
+
results.append(result)
|
| 94 |
+
print(f"[{index:02d}/{len(items)}] {item['id']} {'PASS' if passed else 'FAIL'}", flush=True)
|
| 95 |
+
|
| 96 |
+
by_language = {}
|
| 97 |
+
for language in sorted({item["language"] for item in items}):
|
| 98 |
+
subset = [row for row in results if row["language"] == language]
|
| 99 |
+
by_language[language] = {
|
| 100 |
+
"passed": sum(row["passed"] for row in subset),
|
| 101 |
+
"total": len(subset),
|
| 102 |
+
"pass_rate": sum(row["passed"] for row in subset) / len(subset),
|
| 103 |
+
}
|
| 104 |
+
passed = sum(row["passed"] for row in results)
|
| 105 |
+
report = {
|
| 106 |
+
"model": str(args.model),
|
| 107 |
+
"total": len(results),
|
| 108 |
+
"passed": passed,
|
| 109 |
+
"pass_rate": passed / len(results),
|
| 110 |
+
"gate_threshold": 0.90,
|
| 111 |
+
"gate_passed": passed / len(results) >= 0.90,
|
| 112 |
+
"by_language": by_language,
|
| 113 |
+
"results": results,
|
| 114 |
+
}
|
| 115 |
+
args.output.parent.mkdir(parents=True, exist_ok=True)
|
| 116 |
+
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
|
| 117 |
+
print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
|
| 118 |
+
|
| 119 |
+
|
| 120 |
+
if __name__ == "__main__":
|
| 121 |
+
main()
|
research_code/rescore_chat_gate.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import argparse
|
| 5 |
+
import json
|
| 6 |
+
import sys
|
| 7 |
+
from pathlib import Path
|
| 8 |
+
|
| 9 |
+
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
| 10 |
+
from evaluate_chat_gate import score
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
def main() -> None:
|
| 14 |
+
parser = argparse.ArgumentParser()
|
| 15 |
+
parser.add_argument("--input", type=Path, required=True)
|
| 16 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 17 |
+
args = parser.parse_args()
|
| 18 |
+
|
| 19 |
+
report = json.loads(args.input.read_text())
|
| 20 |
+
for row in report["results"]:
|
| 21 |
+
row["passed"], row["failures"] = score(row, row["completion"])
|
| 22 |
+
for language, summary in report["by_language"].items():
|
| 23 |
+
rows = [row for row in report["results"] if row["language"] == language]
|
| 24 |
+
summary["passed"] = sum(row["passed"] for row in rows)
|
| 25 |
+
summary["pass_rate"] = summary["passed"] / summary["total"]
|
| 26 |
+
report["passed"] = sum(row["passed"] for row in report["results"])
|
| 27 |
+
report["pass_rate"] = report["passed"] / report["total"]
|
| 28 |
+
report["gate_passed"] = report["pass_rate"] >= report["gate_threshold"]
|
| 29 |
+
report["rescored_from"] = str(args.input)
|
| 30 |
+
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
|
| 31 |
+
print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
|
| 32 |
+
|
| 33 |
+
|
| 34 |
+
if __name__ == "__main__":
|
| 35 |
+
main()
|
research_code/service/README.md
ADDED
|
@@ -0,0 +1,20 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Local OpenAI-compatible service
|
| 2 |
+
|
| 3 |
+
Run with the service directory as the working directory:
|
| 4 |
+
|
| 5 |
+
```bash
|
| 6 |
+
cd research/qwen35-moe-a3b/service
|
| 7 |
+
PYTORCH_ENABLE_MPS_FALLBACK=1 \
|
| 8 |
+
MODEL_PATH=../../../models/Qwen/Qwen3.5-4B-A3B-Student-v2 \
|
| 9 |
+
python3 -m uvicorn app:app --host 127.0.0.1 --port 8088
|
| 10 |
+
```
|
| 11 |
+
|
| 12 |
+
Implemented endpoints:
|
| 13 |
+
|
| 14 |
+
- `GET /health`
|
| 15 |
+
- `GET /v1/models`
|
| 16 |
+
- `POST /v1/chat/completions` (non-streaming text requests)
|
| 17 |
+
|
| 18 |
+
This is a single-process research service. It serializes generation calls to
|
| 19 |
+
protect the shared MPS model. Authentication, TLS, streaming, tool calls, and
|
| 20 |
+
multi-worker deployment are intentionally out of scope for this local gate.
|
research_code/service/app.py
ADDED
|
@@ -0,0 +1,167 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
from __future__ import annotations
|
| 2 |
+
|
| 3 |
+
import os
|
| 4 |
+
import threading
|
| 5 |
+
import time
|
| 6 |
+
import uuid
|
| 7 |
+
from contextlib import asynccontextmanager
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
from typing import Literal
|
| 10 |
+
|
| 11 |
+
import torch
|
| 12 |
+
from fastapi import FastAPI, HTTPException
|
| 13 |
+
from pydantic import BaseModel, Field
|
| 14 |
+
from transformers import AutoTokenizer, Qwen3_5MoeForCausalLM
|
| 15 |
+
|
| 16 |
+
|
| 17 |
+
MODEL_ID = "qwen3.5-4b-a3b-student-v2"
|
| 18 |
+
DEFAULT_MODEL_PATH = Path(__file__).resolve().parents[3] / "models/Qwen/Qwen3.5-4B-A3B-Student-v2"
|
| 19 |
+
state: dict = {"latencies": [], "requests": 0}
|
| 20 |
+
generation_lock = threading.Lock()
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
class Message(BaseModel):
|
| 24 |
+
role: Literal["developer", "system", "user", "assistant"]
|
| 25 |
+
content: str
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
class ChatCompletionRequest(BaseModel):
|
| 29 |
+
model: str
|
| 30 |
+
messages: list[Message] = Field(min_length=1)
|
| 31 |
+
max_tokens: int | None = Field(default=None, ge=1, le=2048)
|
| 32 |
+
max_completion_tokens: int | None = Field(default=None, ge=1, le=2048)
|
| 33 |
+
temperature: float = Field(default=0.0, ge=0.0, le=2.0)
|
| 34 |
+
top_p: float = Field(default=1.0, gt=0.0, le=1.0)
|
| 35 |
+
stream: bool = False
|
| 36 |
+
stop: str | list[str] | None = None
|
| 37 |
+
|
| 38 |
+
|
| 39 |
+
def memory_metrics() -> dict[str, float | None]:
|
| 40 |
+
try:
|
| 41 |
+
import psutil
|
| 42 |
+
rss_mb = psutil.Process().memory_info().rss / 1024**2
|
| 43 |
+
except Exception:
|
| 44 |
+
rss_mb = None
|
| 45 |
+
mps_mb = torch.mps.current_allocated_memory() / 1024**2 if torch.backends.mps.is_available() else None
|
| 46 |
+
return {"rss_mb": rss_mb, "mps_allocated_mb": mps_mb}
|
| 47 |
+
|
| 48 |
+
|
| 49 |
+
@asynccontextmanager
|
| 50 |
+
async def lifespan(app: FastAPI):
|
| 51 |
+
model_path = Path(os.environ.get("MODEL_PATH", DEFAULT_MODEL_PATH))
|
| 52 |
+
device = os.environ.get("MODEL_DEVICE", "mps" if torch.backends.mps.is_available() else "cpu")
|
| 53 |
+
tokenizer = AutoTokenizer.from_pretrained(model_path)
|
| 54 |
+
model = Qwen3_5MoeForCausalLM.from_pretrained(model_path, dtype=torch.bfloat16).to(device).eval()
|
| 55 |
+
state.update({
|
| 56 |
+
"model": model,
|
| 57 |
+
"tokenizer": tokenizer,
|
| 58 |
+
"model_path": str(model_path),
|
| 59 |
+
"device": device,
|
| 60 |
+
"parameters": sum(parameter.numel() for parameter in model.parameters()),
|
| 61 |
+
"started_at": int(time.time()),
|
| 62 |
+
})
|
| 63 |
+
yield
|
| 64 |
+
state.pop("model", None)
|
| 65 |
+
state.pop("tokenizer", None)
|
| 66 |
+
if device == "mps":
|
| 67 |
+
torch.mps.empty_cache()
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
app = FastAPI(title="Qwen3.5 A3B Local API", version="0.1.0", lifespan=lifespan)
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
@app.get("/health")
|
| 74 |
+
def health():
|
| 75 |
+
latencies = state["latencies"]
|
| 76 |
+
ordered = sorted(latencies)
|
| 77 |
+
p95 = ordered[max(0, int(len(ordered) * 0.95) - 1)] if ordered else None
|
| 78 |
+
return {
|
| 79 |
+
"status": "ok" if "model" in state else "starting",
|
| 80 |
+
"model": MODEL_ID,
|
| 81 |
+
"model_path": state.get("model_path"),
|
| 82 |
+
"device": state.get("device"),
|
| 83 |
+
"parameters": state.get("parameters"),
|
| 84 |
+
"requests": state["requests"],
|
| 85 |
+
"latency_mean_seconds": sum(latencies) / len(latencies) if latencies else None,
|
| 86 |
+
"latency_p95_seconds": p95,
|
| 87 |
+
**memory_metrics(),
|
| 88 |
+
}
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
@app.get("/v1/models")
|
| 92 |
+
def list_models():
|
| 93 |
+
return {
|
| 94 |
+
"object": "list",
|
| 95 |
+
"data": [{
|
| 96 |
+
"id": MODEL_ID,
|
| 97 |
+
"object": "model",
|
| 98 |
+
"created": state.get("started_at", int(time.time())),
|
| 99 |
+
"owned_by": "local",
|
| 100 |
+
}],
|
| 101 |
+
}
|
| 102 |
+
|
| 103 |
+
|
| 104 |
+
@app.post("/v1/chat/completions")
|
| 105 |
+
def chat_completions(request: ChatCompletionRequest):
|
| 106 |
+
if request.stream:
|
| 107 |
+
raise HTTPException(status_code=400, detail="stream=true is not implemented in this research server")
|
| 108 |
+
if request.model not in {MODEL_ID, "local", state.get("model_path")}:
|
| 109 |
+
raise HTTPException(status_code=404, detail=f"unknown model: {request.model}")
|
| 110 |
+
|
| 111 |
+
messages = [message.model_dump() for message in request.messages]
|
| 112 |
+
for message in messages:
|
| 113 |
+
if message["role"] == "developer":
|
| 114 |
+
message["role"] = "system"
|
| 115 |
+
tokenizer = state["tokenizer"]
|
| 116 |
+
model = state["model"]
|
| 117 |
+
device = state["device"]
|
| 118 |
+
batch = tokenizer.apply_chat_template(
|
| 119 |
+
messages,
|
| 120 |
+
tokenize=True,
|
| 121 |
+
add_generation_prompt=True,
|
| 122 |
+
enable_thinking=False,
|
| 123 |
+
return_tensors="pt",
|
| 124 |
+
return_dict=True,
|
| 125 |
+
).to(device)
|
| 126 |
+
max_new_tokens = request.max_completion_tokens or request.max_tokens or 128
|
| 127 |
+
generation_kwargs = {
|
| 128 |
+
"max_new_tokens": max_new_tokens,
|
| 129 |
+
"do_sample": request.temperature > 0,
|
| 130 |
+
"use_cache": True,
|
| 131 |
+
}
|
| 132 |
+
if request.temperature > 0:
|
| 133 |
+
generation_kwargs.update(temperature=request.temperature, top_p=request.top_p)
|
| 134 |
+
started = time.perf_counter()
|
| 135 |
+
with generation_lock, torch.no_grad():
|
| 136 |
+
output = model.generate(
|
| 137 |
+
**batch,
|
| 138 |
+
**generation_kwargs,
|
| 139 |
+
)
|
| 140 |
+
elapsed = time.perf_counter() - started
|
| 141 |
+
completion_ids = output[0, batch["input_ids"].shape[1]:]
|
| 142 |
+
content = tokenizer.decode(completion_ids, skip_special_tokens=True)
|
| 143 |
+
stops = [request.stop] if isinstance(request.stop, str) else request.stop or []
|
| 144 |
+
for stop in stops:
|
| 145 |
+
if stop in content:
|
| 146 |
+
content = content.split(stop, 1)[0]
|
| 147 |
+
state["latencies"].append(elapsed)
|
| 148 |
+
state["requests"] += 1
|
| 149 |
+
prompt_tokens = int(batch["attention_mask"].sum())
|
| 150 |
+
completion_tokens = int(completion_ids.numel())
|
| 151 |
+
return {
|
| 152 |
+
"id": f"chatcmpl-{uuid.uuid4().hex}",
|
| 153 |
+
"object": "chat.completion",
|
| 154 |
+
"created": int(time.time()),
|
| 155 |
+
"model": MODEL_ID,
|
| 156 |
+
"choices": [{
|
| 157 |
+
"index": 0,
|
| 158 |
+
"message": {"role": "assistant", "content": content, "refusal": None},
|
| 159 |
+
"finish_reason": "stop" if completion_tokens < max_new_tokens else "length",
|
| 160 |
+
"logprobs": None,
|
| 161 |
+
}],
|
| 162 |
+
"usage": {
|
| 163 |
+
"prompt_tokens": prompt_tokens,
|
| 164 |
+
"completion_tokens": completion_tokens,
|
| 165 |
+
"total_tokens": prompt_tokens + completion_tokens,
|
| 166 |
+
},
|
| 167 |
+
}
|
research_code/service/test_openai_service.py
ADDED
|
@@ -0,0 +1,69 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
from __future__ import annotations
|
| 3 |
+
|
| 4 |
+
import argparse
|
| 5 |
+
import json
|
| 6 |
+
import statistics
|
| 7 |
+
import time
|
| 8 |
+
from pathlib import Path
|
| 9 |
+
|
| 10 |
+
import httpx
|
| 11 |
+
|
| 12 |
+
|
| 13 |
+
PROMPTS = [
|
| 14 |
+
"대한민국의 수도는 어디인가요? 짧게 답하세요.",
|
| 15 |
+
"What is the capital of France? Answer briefly.",
|
| 16 |
+
"7 곱하기 8은 얼마인가요?",
|
| 17 |
+
"Write one friendly greeting.",
|
| 18 |
+
]
|
| 19 |
+
|
| 20 |
+
|
| 21 |
+
def main() -> None:
|
| 22 |
+
parser = argparse.ArgumentParser()
|
| 23 |
+
parser.add_argument("--base-url", default="http://127.0.0.1:8088")
|
| 24 |
+
parser.add_argument("--requests", type=int, default=20)
|
| 25 |
+
parser.add_argument("--output", type=Path, required=True)
|
| 26 |
+
args = parser.parse_args()
|
| 27 |
+
|
| 28 |
+
rows = []
|
| 29 |
+
with httpx.Client(timeout=120) as client:
|
| 30 |
+
health_before = client.get(f"{args.base_url}/health").raise_for_status().json()
|
| 31 |
+
models = client.get(f"{args.base_url}/v1/models").raise_for_status().json()
|
| 32 |
+
for index in range(args.requests):
|
| 33 |
+
started = time.perf_counter()
|
| 34 |
+
response = client.post(f"{args.base_url}/v1/chat/completions", json={
|
| 35 |
+
"model": "qwen3.5-4b-a3b-student-v2",
|
| 36 |
+
"messages": [{"role": "user", "content": PROMPTS[index % len(PROMPTS)]}],
|
| 37 |
+
"max_completion_tokens": 16,
|
| 38 |
+
"temperature": 0,
|
| 39 |
+
})
|
| 40 |
+
elapsed = time.perf_counter() - started
|
| 41 |
+
response.raise_for_status()
|
| 42 |
+
body = response.json()
|
| 43 |
+
content = body["choices"][0]["message"]["content"].strip()
|
| 44 |
+
passed = body["object"] == "chat.completion" and bool(content)
|
| 45 |
+
rows.append({"index": index + 1, "passed": passed, "elapsed_seconds": elapsed, "content": content})
|
| 46 |
+
print(f"[{index + 1:02d}/{args.requests}] {'PASS' if passed else 'FAIL'} {elapsed:.3f}s", flush=True)
|
| 47 |
+
health_after = client.get(f"{args.base_url}/health").raise_for_status().json()
|
| 48 |
+
|
| 49 |
+
latencies = [row["elapsed_seconds"] for row in rows]
|
| 50 |
+
ordered = sorted(latencies)
|
| 51 |
+
report = {
|
| 52 |
+
"requests": args.requests,
|
| 53 |
+
"passed": sum(row["passed"] for row in rows),
|
| 54 |
+
"all_passed": all(row["passed"] for row in rows),
|
| 55 |
+
"latency_mean_seconds": statistics.mean(latencies),
|
| 56 |
+
"latency_p50_seconds": statistics.median(latencies),
|
| 57 |
+
"latency_p95_seconds": ordered[max(0, int(len(ordered) * 0.95) - 1)],
|
| 58 |
+
"health_before": health_before,
|
| 59 |
+
"health_after": health_after,
|
| 60 |
+
"models": models,
|
| 61 |
+
"results": rows,
|
| 62 |
+
}
|
| 63 |
+
args.output.parent.mkdir(parents=True, exist_ok=True)
|
| 64 |
+
args.output.write_text(json.dumps(report, ensure_ascii=False, indent=2))
|
| 65 |
+
print(json.dumps({key: value for key, value in report.items() if key != "results"}, ensure_ascii=False, indent=2))
|
| 66 |
+
|
| 67 |
+
|
| 68 |
+
if __name__ == "__main__":
|
| 69 |
+
main()
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5f9e4d4901a92b997e463c1f46055088b6cca5ca61a6522d1b9f64c4bb81cb42
|
| 3 |
+
size 12807982
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,305 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"248044": {
|
| 5 |
+
"content": "<|endoftext|>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": false,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"248045": {
|
| 13 |
+
"content": "<|im_start|>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": false,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"248046": {
|
| 21 |
+
"content": "<|im_end|>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": false,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"248047": {
|
| 29 |
+
"content": "<|object_ref_start|>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": false,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"248048": {
|
| 37 |
+
"content": "<|object_ref_end|>",
|
| 38 |
+
"lstrip": false,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
},
|
| 44 |
+
"248049": {
|
| 45 |
+
"content": "<|box_start|>",
|
| 46 |
+
"lstrip": false,
|
| 47 |
+
"normalized": false,
|
| 48 |
+
"rstrip": false,
|
| 49 |
+
"single_word": false,
|
| 50 |
+
"special": true
|
| 51 |
+
},
|
| 52 |
+
"248050": {
|
| 53 |
+
"content": "<|box_end|>",
|
| 54 |
+
"lstrip": false,
|
| 55 |
+
"normalized": false,
|
| 56 |
+
"rstrip": false,
|
| 57 |
+
"single_word": false,
|
| 58 |
+
"special": true
|
| 59 |
+
},
|
| 60 |
+
"248051": {
|
| 61 |
+
"content": "<|quad_start|>",
|
| 62 |
+
"lstrip": false,
|
| 63 |
+
"normalized": false,
|
| 64 |
+
"rstrip": false,
|
| 65 |
+
"single_word": false,
|
| 66 |
+
"special": true
|
| 67 |
+
},
|
| 68 |
+
"248052": {
|
| 69 |
+
"content": "<|quad_end|>",
|
| 70 |
+
"lstrip": false,
|
| 71 |
+
"normalized": false,
|
| 72 |
+
"rstrip": false,
|
| 73 |
+
"single_word": false,
|
| 74 |
+
"special": true
|
| 75 |
+
},
|
| 76 |
+
"248053": {
|
| 77 |
+
"content": "<|vision_start|>",
|
| 78 |
+
"lstrip": false,
|
| 79 |
+
"normalized": false,
|
| 80 |
+
"rstrip": false,
|
| 81 |
+
"single_word": false,
|
| 82 |
+
"special": true
|
| 83 |
+
},
|
| 84 |
+
"248054": {
|
| 85 |
+
"content": "<|vision_end|>",
|
| 86 |
+
"lstrip": false,
|
| 87 |
+
"normalized": false,
|
| 88 |
+
"rstrip": false,
|
| 89 |
+
"single_word": false,
|
| 90 |
+
"special": true
|
| 91 |
+
},
|
| 92 |
+
"248055": {
|
| 93 |
+
"content": "<|vision_pad|>",
|
| 94 |
+
"lstrip": false,
|
| 95 |
+
"normalized": false,
|
| 96 |
+
"rstrip": false,
|
| 97 |
+
"single_word": false,
|
| 98 |
+
"special": true
|
| 99 |
+
},
|
| 100 |
+
"248056": {
|
| 101 |
+
"content": "<|image_pad|>",
|
| 102 |
+
"lstrip": false,
|
| 103 |
+
"normalized": false,
|
| 104 |
+
"rstrip": false,
|
| 105 |
+
"single_word": false,
|
| 106 |
+
"special": true
|
| 107 |
+
},
|
| 108 |
+
"248057": {
|
| 109 |
+
"content": "<|video_pad|>",
|
| 110 |
+
"lstrip": false,
|
| 111 |
+
"normalized": false,
|
| 112 |
+
"rstrip": false,
|
| 113 |
+
"single_word": false,
|
| 114 |
+
"special": true
|
| 115 |
+
},
|
| 116 |
+
"248058": {
|
| 117 |
+
"content": "<tool_call>",
|
| 118 |
+
"lstrip": false,
|
| 119 |
+
"normalized": false,
|
| 120 |
+
"rstrip": false,
|
| 121 |
+
"single_word": false,
|
| 122 |
+
"special": false
|
| 123 |
+
},
|
| 124 |
+
"248059": {
|
| 125 |
+
"content": "</tool_call>",
|
| 126 |
+
"lstrip": false,
|
| 127 |
+
"normalized": false,
|
| 128 |
+
"rstrip": false,
|
| 129 |
+
"single_word": false,
|
| 130 |
+
"special": false
|
| 131 |
+
},
|
| 132 |
+
"248060": {
|
| 133 |
+
"content": "<|fim_prefix|>",
|
| 134 |
+
"lstrip": false,
|
| 135 |
+
"normalized": false,
|
| 136 |
+
"rstrip": false,
|
| 137 |
+
"single_word": false,
|
| 138 |
+
"special": false
|
| 139 |
+
},
|
| 140 |
+
"248061": {
|
| 141 |
+
"content": "<|fim_middle|>",
|
| 142 |
+
"lstrip": false,
|
| 143 |
+
"normalized": false,
|
| 144 |
+
"rstrip": false,
|
| 145 |
+
"single_word": false,
|
| 146 |
+
"special": false
|
| 147 |
+
},
|
| 148 |
+
"248062": {
|
| 149 |
+
"content": "<|fim_suffix|>",
|
| 150 |
+
"lstrip": false,
|
| 151 |
+
"normalized": false,
|
| 152 |
+
"rstrip": false,
|
| 153 |
+
"single_word": false,
|
| 154 |
+
"special": false
|
| 155 |
+
},
|
| 156 |
+
"248063": {
|
| 157 |
+
"content": "<|fim_pad|>",
|
| 158 |
+
"lstrip": false,
|
| 159 |
+
"normalized": false,
|
| 160 |
+
"rstrip": false,
|
| 161 |
+
"single_word": false,
|
| 162 |
+
"special": false
|
| 163 |
+
},
|
| 164 |
+
"248064": {
|
| 165 |
+
"content": "<|repo_name|>",
|
| 166 |
+
"lstrip": false,
|
| 167 |
+
"normalized": false,
|
| 168 |
+
"rstrip": false,
|
| 169 |
+
"single_word": false,
|
| 170 |
+
"special": false
|
| 171 |
+
},
|
| 172 |
+
"248065": {
|
| 173 |
+
"content": "<|file_sep|>",
|
| 174 |
+
"lstrip": false,
|
| 175 |
+
"normalized": false,
|
| 176 |
+
"rstrip": false,
|
| 177 |
+
"single_word": false,
|
| 178 |
+
"special": false
|
| 179 |
+
},
|
| 180 |
+
"248066": {
|
| 181 |
+
"content": "<tool_response>",
|
| 182 |
+
"lstrip": false,
|
| 183 |
+
"normalized": false,
|
| 184 |
+
"rstrip": false,
|
| 185 |
+
"single_word": false,
|
| 186 |
+
"special": false
|
| 187 |
+
},
|
| 188 |
+
"248067": {
|
| 189 |
+
"content": "</tool_response>",
|
| 190 |
+
"lstrip": false,
|
| 191 |
+
"normalized": false,
|
| 192 |
+
"rstrip": false,
|
| 193 |
+
"single_word": false,
|
| 194 |
+
"special": false
|
| 195 |
+
},
|
| 196 |
+
"248068": {
|
| 197 |
+
"content": "<think>",
|
| 198 |
+
"lstrip": false,
|
| 199 |
+
"normalized": false,
|
| 200 |
+
"rstrip": false,
|
| 201 |
+
"single_word": false,
|
| 202 |
+
"special": false
|
| 203 |
+
},
|
| 204 |
+
"248069": {
|
| 205 |
+
"content": "</think>",
|
| 206 |
+
"lstrip": false,
|
| 207 |
+
"normalized": false,
|
| 208 |
+
"rstrip": false,
|
| 209 |
+
"single_word": false,
|
| 210 |
+
"special": false
|
| 211 |
+
},
|
| 212 |
+
"248070": {
|
| 213 |
+
"content": "<|audio_start|>",
|
| 214 |
+
"lstrip": false,
|
| 215 |
+
"normalized": false,
|
| 216 |
+
"rstrip": false,
|
| 217 |
+
"single_word": false,
|
| 218 |
+
"special": true
|
| 219 |
+
},
|
| 220 |
+
"248071": {
|
| 221 |
+
"content": "<|audio_end|>",
|
| 222 |
+
"lstrip": false,
|
| 223 |
+
"normalized": false,
|
| 224 |
+
"rstrip": false,
|
| 225 |
+
"single_word": false,
|
| 226 |
+
"special": true
|
| 227 |
+
},
|
| 228 |
+
"248072": {
|
| 229 |
+
"content": "<tts_pad>",
|
| 230 |
+
"lstrip": false,
|
| 231 |
+
"normalized": false,
|
| 232 |
+
"rstrip": false,
|
| 233 |
+
"single_word": false,
|
| 234 |
+
"special": true
|
| 235 |
+
},
|
| 236 |
+
"248073": {
|
| 237 |
+
"content": "<tts_text_bos>",
|
| 238 |
+
"lstrip": false,
|
| 239 |
+
"normalized": false,
|
| 240 |
+
"rstrip": false,
|
| 241 |
+
"single_word": false,
|
| 242 |
+
"special": true
|
| 243 |
+
},
|
| 244 |
+
"248074": {
|
| 245 |
+
"content": "<tts_text_eod>",
|
| 246 |
+
"lstrip": false,
|
| 247 |
+
"normalized": false,
|
| 248 |
+
"rstrip": false,
|
| 249 |
+
"single_word": false,
|
| 250 |
+
"special": true
|
| 251 |
+
},
|
| 252 |
+
"248075": {
|
| 253 |
+
"content": "<tts_text_bos_single>",
|
| 254 |
+
"lstrip": false,
|
| 255 |
+
"normalized": false,
|
| 256 |
+
"rstrip": false,
|
| 257 |
+
"single_word": false,
|
| 258 |
+
"special": true
|
| 259 |
+
},
|
| 260 |
+
"248076": {
|
| 261 |
+
"content": "<|audio_pad|>",
|
| 262 |
+
"lstrip": false,
|
| 263 |
+
"normalized": false,
|
| 264 |
+
"rstrip": false,
|
| 265 |
+
"single_word": false,
|
| 266 |
+
"special": true
|
| 267 |
+
}
|
| 268 |
+
},
|
| 269 |
+
"additional_special_tokens": [
|
| 270 |
+
"<|im_start|>",
|
| 271 |
+
"<|im_end|>",
|
| 272 |
+
"<|object_ref_start|>",
|
| 273 |
+
"<|object_ref_end|>",
|
| 274 |
+
"<|box_start|>",
|
| 275 |
+
"<|box_end|>",
|
| 276 |
+
"<|quad_start|>",
|
| 277 |
+
"<|quad_end|>",
|
| 278 |
+
"<|vision_start|>",
|
| 279 |
+
"<|vision_end|>",
|
| 280 |
+
"<|vision_pad|>",
|
| 281 |
+
"<|image_pad|>",
|
| 282 |
+
"<|video_pad|>"
|
| 283 |
+
],
|
| 284 |
+
"bos_token": null,
|
| 285 |
+
"chat_template": "{%- set image_count = namespace(value=0) %}\n{%- set video_count = namespace(value=0) %}\n{%- macro render_content(content, do_vision_count, is_system_content=false) %}\n {%- if content is string %}\n {{- content }}\n {%- elif content is iterable and content is not mapping %}\n {%- for item in content %}\n {%- if 'image' in item or 'image_url' in item or item.type == 'image' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain images.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set image_count.value = image_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Picture ' ~ image_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|image_pad|><|vision_end|>' }}\n {%- elif 'video' in item or item.type == 'video' %}\n {%- if is_system_content %}\n {{- raise_exception('System message cannot contain videos.') }}\n {%- endif %}\n {%- if do_vision_count %}\n {%- set video_count.value = video_count.value + 1 %}\n {%- endif %}\n {%- if add_vision_id %}\n {{- 'Video ' ~ video_count.value ~ ': ' }}\n {%- endif %}\n {{- '<|vision_start|><|video_pad|><|vision_end|>' }}\n {%- elif 'text' in item %}\n {{- item.text }}\n {%- else %}\n {{- raise_exception('Unexpected item type in content.') }}\n {%- endif %}\n {%- endfor %}\n {%- elif content is none or content is undefined %}\n {{- '' }}\n {%- else %}\n {{- raise_exception('Unexpected content type.') }}\n {%- endif %}\n{%- endmacro %}\n{%- if not messages %}\n {{- raise_exception('No messages provided.') }}\n{%- endif %}\n{%- if tools and tools is iterable and tools is not mapping %}\n {{- '<|im_start|>system\\n' }}\n {{- \"# Tools\\n\\nYou have access to the following functions:\\n\\n<tools>\" }}\n {%- for tool in tools %}\n {{- \"\\n\" }}\n {{- tool | tojson }}\n {%- endfor %}\n {{- \"\\n</tools>\" }}\n {{- '\\n\\nIf you choose to call a function ONLY reply in the following format with NO suffix:\\n\\n<tool_call>\\n<function=example_function_name>\\n<parameter=example_parameter_1>\\nvalue_1\\n</parameter>\\n<parameter=example_parameter_2>\\nThis is the value for the second parameter\\nthat can span\\nmultiple lines\\n</parameter>\\n</function>\\n</tool_call>\\n\\n<IMPORTANT>\\nReminder:\\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\\n- Required parameters MUST be specified\\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\\n</IMPORTANT>' }}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {%- if content %}\n {{- '\\n\\n' + content }}\n {%- endif %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n{%- else %}\n {%- if messages[0].role == 'system' %}\n {%- set content = render_content(messages[0].content, false, true)|trim %}\n {{- '<|im_start|>system\\n' + content + '<|im_end|>\\n' }}\n {%- endif %}\n{%- endif %}\n{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}\n{%- for message in messages[::-1] %}\n {%- set index = (messages|length - 1) - loop.index0 %}\n {%- if ns.multi_step_tool and message.role == \"user\" %}\n {%- set content = render_content(message.content, false)|trim %}\n {%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}\n {%- set ns.multi_step_tool = false %}\n {%- set ns.last_query_index = index %}\n {%- endif %}\n {%- endif %}\n{%- endfor %}\n{%- if ns.multi_step_tool %}\n {{- raise_exception('No user query found in messages.') }}\n{%- endif %}\n{%- for message in messages %}\n {%- set content = render_content(message.content, true)|trim %}\n {%- if message.role == \"system\" %}\n {%- if not loop.first %}\n {{- raise_exception('System message must be at the beginning.') }}\n {%- endif %}\n {%- elif message.role == \"user\" %}\n {{- '<|im_start|>' + message.role + '\\n' + content + '<|im_end|>' + '\\n' }}\n {%- elif message.role == \"assistant\" %}\n {%- set reasoning_content = '' %}\n {%- if message.reasoning_content is string %}\n {%- set reasoning_content = message.reasoning_content %}\n {%- else %}\n {%- if '</think>' in content %}\n {%- set reasoning_content = content.split('</think>')[0].rstrip('\\n').split('<think>')[-1].lstrip('\\n') %}\n {%- set content = content.split('</think>')[-1].lstrip('\\n') %}\n {%- endif %}\n {%- endif %}\n {%- set reasoning_content = reasoning_content|trim %}\n {%- if loop.index0 > ns.last_query_index %}\n {{- '<|im_start|>' + message.role + '\\n<think>\\n' + reasoning_content + '\\n</think>\\n\\n' + content }}\n {%- else %}\n {{- '<|im_start|>' + message.role + '\\n' + content }}\n {%- endif %}\n {%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}\n {%- for tool_call in message.tool_calls %}\n {%- if tool_call.function is defined %}\n {%- set tool_call = tool_call.function %}\n {%- endif %}\n {%- if loop.first %}\n {%- if content|trim %}\n {{- '\\n\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- else %}\n {{- '<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- else %}\n {{- '\\n<tool_call>\\n<function=' + tool_call.name + '>\\n' }}\n {%- endif %}\n {%- if tool_call.arguments is defined %}\n {%- for args_name, args_value in tool_call.arguments|items %}\n {{- '<parameter=' + args_name + '>\\n' }}\n {%- set args_value = args_value | tojson | safe if args_value is mapping or (args_value is sequence and args_value is not string) else args_value | string %}\n {{- args_value }}\n {{- '\\n</parameter>\\n' }}\n {%- endfor %}\n {%- endif %}\n {{- '</function>\\n</tool_call>' }}\n {%- endfor %}\n {%- endif %}\n {{- '<|im_end|>\\n' }}\n {%- elif message.role == \"tool\" %}\n {%- if loop.previtem and loop.previtem.role != \"tool\" %}\n {{- '<|im_start|>user' }}\n {%- endif %}\n {{- '\\n<tool_response>\\n' }}\n {{- content }}\n {{- '\\n</tool_response>' }}\n {%- if not loop.last and loop.nextitem.role != \"tool\" %}\n {{- '<|im_end|>\\n' }}\n {%- elif loop.last %}\n {{- '<|im_end|>\\n' }}\n {%- endif %}\n {%- else %}\n {{- raise_exception('Unexpected message role.') }}\n {%- endif %}\n{%- endfor %}\n{%- if add_generation_prompt %}\n {{- '<|im_start|>assistant\\n' }}\n {%- if enable_thinking is defined and enable_thinking is true %}\n {{- '<think>\\n' }}\n {%- else %}\n {{- '<think>\\n\\n</think>\\n\\n' }}\n {%- endif %}\n{%- endif %}",
|
| 286 |
+
"clean_up_tokenization_spaces": false,
|
| 287 |
+
"eos_token": "<|im_end|>",
|
| 288 |
+
"errors": "replace",
|
| 289 |
+
"model_max_length": 262144,
|
| 290 |
+
"pad_token": "<|endoftext|>",
|
| 291 |
+
"split_special_tokens": false,
|
| 292 |
+
"tokenizer_class": "Qwen2Tokenizer",
|
| 293 |
+
"unk_token": null,
|
| 294 |
+
"add_bos_token": false,
|
| 295 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 296 |
+
"extra_special_tokens": {
|
| 297 |
+
"audio_bos_token": "<|audio_start|>",
|
| 298 |
+
"audio_eos_token": "<|audio_end|>",
|
| 299 |
+
"audio_token": "<|audio_pad|>",
|
| 300 |
+
"image_token": "<|image_pad|>",
|
| 301 |
+
"video_token": "<|video_pad|>",
|
| 302 |
+
"vision_bos_token": "<|vision_start|>",
|
| 303 |
+
"vision_eos_token": "<|vision_end|>"
|
| 304 |
+
}
|
| 305 |
+
}
|
vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|