Image-Text-to-Text
Transformers
Safetensors
English
Chinese
agnes
text-generation
agnes-ai
uncensored
reasoning
multimodal
fp8
w8a8
vllm
sglang
long-context
hybrid-attention
conversational
custom_code
compressed-tensors
Instructions to use cbert33/Agnes-3.0-Flash-FP8-Calibrated with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use cbert33/Agnes-3.0-Flash-FP8-Calibrated with Transformers:
# Use a pipeline as a high-level helper from transformers import pipeline pipe = pipeline("image-text-to-text", model="cbert33/Agnes-3.0-Flash-FP8-Calibrated", trust_remote_code=True) messages = [ { "role": "user", "content": [ {"type": "image", "url": "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/p-blog/candy.JPG"}, {"type": "text", "text": "What animal is on the candy?"} ] }, ] pipe(text=messages)# Load model directly from transformers import AutoModelForCausalLM model = AutoModelForCausalLM.from_pretrained("cbert33/Agnes-3.0-Flash-FP8-Calibrated", trust_remote_code=True, device_map="auto") - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- vLLM
How to use cbert33/Agnes-3.0-Flash-FP8-Calibrated with vLLM:
Install from pip and serve model
# Install vLLM from pip: pip install vllm # Start the vLLM server: vllm serve "cbert33/Agnes-3.0-Flash-FP8-Calibrated" # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "cbert33/Agnes-3.0-Flash-FP8-Calibrated", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker
docker model run hf.co/cbert33/Agnes-3.0-Flash-FP8-Calibrated
- SGLang
How to use cbert33/Agnes-3.0-Flash-FP8-Calibrated with SGLang:
Install from pip and serve model
# Install SGLang from pip: pip install sglang # Start the SGLang server: python3 -m sglang.launch_server \ --model-path "cbert33/Agnes-3.0-Flash-FP8-Calibrated" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "cbert33/Agnes-3.0-Flash-FP8-Calibrated", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }'Use Docker images
docker run --gpus all \ --shm-size 32g \ -p 30000:30000 \ -v ~/.cache/huggingface:/root/.cache/huggingface \ --env "HF_TOKEN=<secret>" \ --ipc=host \ lmsysorg/sglang:latest \ python3 -m sglang.launch_server \ --model-path "cbert33/Agnes-3.0-Flash-FP8-Calibrated" \ --host 0.0.0.0 \ --port 30000 # Call the server using curl (OpenAI-compatible API): curl -X POST "http://localhost:30000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "cbert33/Agnes-3.0-Flash-FP8-Calibrated", "messages": [ { "role": "user", "content": [ { "type": "text", "text": "Describe this image in one sentence." }, { "type": "image_url", "image_url": { "url": "https://cdn.britannica.com/61/93061-050-99147DCE/Statue-of-Liberty-Island-New-York-Bay.jpg" } } ] } ] }' - Docker Model Runner
How to use cbert33/Agnes-3.0-Flash-FP8-Calibrated with Docker Model Runner:
docker model run hf.co/cbert33/Agnes-3.0-Flash-FP8-Calibrated
Upload model card, config, tokenizer, and sglang patch
Browse files- .gitattributes +1 -0
- LICENSE +201 -0
- README.md +248 -0
- README_zh.md +273 -0
- agnes_quantization_manifest.json +1071 -0
- chat_template.jinja +170 -0
- config.json +478 -0
- configuration_agnes.py +203 -0
- generation_config.json +12 -0
- image_processing_agnes.py +190 -0
- model.safetensors.index.json +0 -0
- modeling_agnes.py +1559 -0
- preprocessor_config.json +25 -0
- processing_agnes.py +116 -0
- recipe.yaml +10 -0
- serve.sh +35 -0
- sglang_patch/README.md +42 -0
- sglang_patch/agnes_sglang_config.py +57 -0
- sglang_patch/apply_patch.py +130 -0
- sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/configs/agnes.py +57 -0
- sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/models/qwen3_5.py +0 -0
- sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/utils/hf_transformers/common.py +729 -0
- sglang_patch/v0.5.19/sglang/srt/configs/agnes.py +57 -0
- sglang_patch/v0.5.19/sglang/srt/models/qwen3_5.py +0 -0
- sglang_patch/v0.5.19/sglang/srt/utils/hf_transformers/common.py +670 -0
- tokenizer.json +3 -0
- tokenizer_config.json +33 -0
- video_preprocessor_config.json +27 -0
- video_processing_agnes.py +197 -0
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
LICENSE
ADDED
|
@@ -0,0 +1,201 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
Apache License
|
| 2 |
+
Version 2.0, January 2004
|
| 3 |
+
http://www.apache.org/licenses/
|
| 4 |
+
|
| 5 |
+
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
| 6 |
+
|
| 7 |
+
1. Definitions.
|
| 8 |
+
|
| 9 |
+
"License" shall mean the terms and conditions for use, reproduction,
|
| 10 |
+
and distribution as defined by Sections 1 through 9 of this document.
|
| 11 |
+
|
| 12 |
+
"Licensor" shall mean the copyright owner or entity authorized by
|
| 13 |
+
the copyright owner that is granting the License.
|
| 14 |
+
|
| 15 |
+
"Legal Entity" shall mean the union of the acting entity and all
|
| 16 |
+
other entities that control, are controlled by, or are under common
|
| 17 |
+
control with that entity. For the purposes of this definition,
|
| 18 |
+
"control" means (i) the power, direct or indirect, to cause the
|
| 19 |
+
direction or management of such entity, whether by contract or
|
| 20 |
+
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
| 21 |
+
outstanding shares, or (iii) beneficial ownership of such entity.
|
| 22 |
+
|
| 23 |
+
"You" (or "Your") shall mean an individual or Legal Entity
|
| 24 |
+
exercising permissions granted by this License.
|
| 25 |
+
|
| 26 |
+
"Source" form shall mean the preferred form for making modifications,
|
| 27 |
+
including but not limited to software source code, documentation
|
| 28 |
+
source, and configuration files.
|
| 29 |
+
|
| 30 |
+
"Object" form shall mean any form resulting from mechanical
|
| 31 |
+
transformation or translation of a Source form, including but
|
| 32 |
+
not limited to compiled object code, generated documentation,
|
| 33 |
+
and conversions to other media types.
|
| 34 |
+
|
| 35 |
+
"Work" shall mean the work of authorship, whether in Source or
|
| 36 |
+
Object form, made available under the License, as indicated by a
|
| 37 |
+
copyright notice that is included in or attached to the work
|
| 38 |
+
(an example is provided in the Appendix below).
|
| 39 |
+
|
| 40 |
+
"Derivative Works" shall mean any work, whether in Source or Object
|
| 41 |
+
form, that is based on (or derived from) the Work and for which the
|
| 42 |
+
editorial revisions, annotations, elaborations, or other modifications
|
| 43 |
+
represent, as a whole, an original work of authorship. For the purposes
|
| 44 |
+
of this License, Derivative Works shall not include works that remain
|
| 45 |
+
separable from, or merely link (or bind by name) to the interfaces of,
|
| 46 |
+
the Work and Derivative Works thereof.
|
| 47 |
+
|
| 48 |
+
"Contribution" shall mean any work of authorship, including
|
| 49 |
+
the original version of the Work and any modifications or additions
|
| 50 |
+
to that Work or Derivative Works thereof, that is intentionally
|
| 51 |
+
submitted to Licensor for inclusion in the Work by the copyright owner
|
| 52 |
+
or by an individual or Legal Entity authorized to submit on behalf of
|
| 53 |
+
the copyright owner. For the purposes of this definition, "submitted"
|
| 54 |
+
means any form of electronic, verbal, or written communication sent
|
| 55 |
+
to the Licensor or its representatives, including but not limited to
|
| 56 |
+
communication on electronic mailing lists, source code control systems,
|
| 57 |
+
and issue tracking systems that are managed by, or on behalf of, the
|
| 58 |
+
Licensor for the purpose of discussing and improving the Work, but
|
| 59 |
+
excluding communication that is conspicuously marked or otherwise
|
| 60 |
+
designated in writing by the copyright owner as "Not a Contribution."
|
| 61 |
+
|
| 62 |
+
"Contributor" shall mean Licensor and any individual or Legal Entity
|
| 63 |
+
on behalf of whom a Contribution has been received by Licensor and
|
| 64 |
+
subsequently incorporated within the Work.
|
| 65 |
+
|
| 66 |
+
2. Grant of Copyright License. Subject to the terms and conditions of
|
| 67 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 68 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 69 |
+
copyright license to reproduce, prepare Derivative Works of,
|
| 70 |
+
publicly display, publicly perform, sublicense, and distribute the
|
| 71 |
+
Work and such Derivative Works in Source or Object form.
|
| 72 |
+
|
| 73 |
+
3. Grant of Patent License. Subject to the terms and conditions of
|
| 74 |
+
this License, each Contributor hereby grants to You a perpetual,
|
| 75 |
+
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
| 76 |
+
(except as stated in this section) patent license to make, have made,
|
| 77 |
+
use, offer to sell, sell, import, and otherwise transfer the Work,
|
| 78 |
+
where such license applies only to those patent claims licensable
|
| 79 |
+
by such Contributor that are necessarily infringed by their
|
| 80 |
+
Contribution(s) alone or by combination of their Contribution(s)
|
| 81 |
+
with the Work to which such Contribution(s) was submitted. If You
|
| 82 |
+
institute patent litigation against any entity (including a
|
| 83 |
+
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
| 84 |
+
or a Contribution incorporated within the Work constitutes direct
|
| 85 |
+
or contributory patent infringement, then any patent licenses
|
| 86 |
+
granted to You under this License for that Work shall terminate
|
| 87 |
+
as of the date such litigation is filed.
|
| 88 |
+
|
| 89 |
+
4. Redistribution. You may reproduce and distribute copies of the
|
| 90 |
+
Work or Derivative Works thereof in any medium, with or without
|
| 91 |
+
modifications, and in Source or Object form, provided that You
|
| 92 |
+
meet the following conditions:
|
| 93 |
+
|
| 94 |
+
(a) You must give any other recipients of the Work or
|
| 95 |
+
Derivative Works a copy of this License; and
|
| 96 |
+
|
| 97 |
+
(b) You must cause any modified files to carry prominent notices
|
| 98 |
+
stating that You changed the files; and
|
| 99 |
+
|
| 100 |
+
(c) You must retain, in the Source form of any Derivative Works
|
| 101 |
+
that You distribute, all copyright, patent, trademark, and
|
| 102 |
+
attribution notices from the Source form of the Work,
|
| 103 |
+
excluding those notices that do not pertain to any part of
|
| 104 |
+
the Derivative Works; and
|
| 105 |
+
|
| 106 |
+
(d) If the Work includes a "NOTICE" text file as part of its
|
| 107 |
+
distribution, then any Derivative Works that You distribute must
|
| 108 |
+
include a readable copy of the attribution notices contained
|
| 109 |
+
within such NOTICE file, excluding those notices that do not
|
| 110 |
+
pertain to any part of the Derivative Works, in at least one
|
| 111 |
+
of the following places: within a NOTICE text file distributed
|
| 112 |
+
as part of the Derivative Works; within the Source form or
|
| 113 |
+
documentation, if provided along with the Derivative Works; or,
|
| 114 |
+
within a display generated by the Derivative Works, if and
|
| 115 |
+
wherever such third-party notices normally appear. The contents
|
| 116 |
+
of the NOTICE file are for informational purposes only and
|
| 117 |
+
do not modify the License. You may add Your own attribution
|
| 118 |
+
notices within Derivative Works that You distribute, alongside
|
| 119 |
+
or as an addendum to the NOTICE text from the Work, provided
|
| 120 |
+
that such additional attribution notices cannot be construed
|
| 121 |
+
as modifying the License.
|
| 122 |
+
|
| 123 |
+
You may add Your own copyright statement to Your modifications and
|
| 124 |
+
may provide additional or different license terms and conditions
|
| 125 |
+
for use, reproduction, or distribution of Your modifications, or
|
| 126 |
+
for any such Derivative Works as a whole, provided Your use,
|
| 127 |
+
reproduction, and distribution of the Work otherwise complies with
|
| 128 |
+
the conditions stated in this License.
|
| 129 |
+
|
| 130 |
+
5. Submission of Contributions. Unless You explicitly state otherwise,
|
| 131 |
+
any Contribution intentionally submitted for inclusion in the Work
|
| 132 |
+
by You to the Licensor shall be under the terms and conditions of
|
| 133 |
+
this License, without any additional terms or conditions.
|
| 134 |
+
Notwithstanding the above, nothing herein shall supersede or modify
|
| 135 |
+
the terms of any separate license agreement you may have executed
|
| 136 |
+
with Licensor regarding such Contributions.
|
| 137 |
+
|
| 138 |
+
6. Trademarks. This License does not grant permission to use the trade
|
| 139 |
+
names, trademarks, service marks, or product names of the Licensor,
|
| 140 |
+
except as required for reasonable and customary use in describing the
|
| 141 |
+
origin of the Work and reproducing the content of the NOTICE file.
|
| 142 |
+
|
| 143 |
+
7. Disclaimer of Warranty. Unless required by applicable law or
|
| 144 |
+
agreed to in writing, Licensor provides the Work (and each
|
| 145 |
+
Contributor provides its Contributions) on an "AS IS" BASIS,
|
| 146 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
| 147 |
+
implied, including, without limitation, any warranties or conditions
|
| 148 |
+
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
| 149 |
+
PARTICULAR PURPOSE. You are solely responsible for determining the
|
| 150 |
+
appropriateness of using or redistributing the Work and assume any
|
| 151 |
+
risks associated with Your exercise of permissions under this License.
|
| 152 |
+
|
| 153 |
+
8. Limitation of Liability. In no event and under no legal theory,
|
| 154 |
+
whether in tort (including negligence), contract, or otherwise,
|
| 155 |
+
unless required by applicable law (such as deliberate and grossly
|
| 156 |
+
negligent acts) or agreed to in writing, shall any Contributor be
|
| 157 |
+
liable to You for damages, including any direct, indirect, special,
|
| 158 |
+
incidental, or consequential damages of any character arising as a
|
| 159 |
+
result of this License or out of the use or inability to use the
|
| 160 |
+
Work (including but not limited to damages for loss of goodwill,
|
| 161 |
+
work stoppage, computer failure or malfunction, or any and all
|
| 162 |
+
other commercial damages or losses), even if such Contributor
|
| 163 |
+
has been advised of the possibility of such damages.
|
| 164 |
+
|
| 165 |
+
9. Accepting Warranty or Additional Liability. While redistributing
|
| 166 |
+
the Work or Derivative Works thereof, You may choose to offer,
|
| 167 |
+
and charge a fee for, acceptance of support, warranty, indemnity,
|
| 168 |
+
or other liability obligations and/or rights consistent with this
|
| 169 |
+
License. However, in accepting such obligations, You may act only
|
| 170 |
+
on Your own behalf and on Your sole responsibility, not on behalf
|
| 171 |
+
of any other Contributor, and only if You agree to indemnify,
|
| 172 |
+
defend, and hold each Contributor harmless for any liability
|
| 173 |
+
incurred by, or claims asserted against, such Contributor by reason
|
| 174 |
+
of your accepting any such warranty or additional liability.
|
| 175 |
+
|
| 176 |
+
END OF TERMS AND CONDITIONS
|
| 177 |
+
|
| 178 |
+
APPENDIX: How to apply the Apache License to your work.
|
| 179 |
+
|
| 180 |
+
To apply the Apache License to your work, attach the following
|
| 181 |
+
boilerplate notice, with the fields enclosed by brackets "[]"
|
| 182 |
+
replaced with your own identifying information. (Don't include
|
| 183 |
+
the brackets!) The text should be enclosed in the appropriate
|
| 184 |
+
comment syntax for the file format. We also recommend that a
|
| 185 |
+
file or class name and description of purpose be included on the
|
| 186 |
+
same "printed page" as the copyright notice for easier
|
| 187 |
+
identification within third-party archives.
|
| 188 |
+
|
| 189 |
+
Copyright [yyyy] [name of copyright owner]
|
| 190 |
+
|
| 191 |
+
Licensed under the Apache License, Version 2.0 (the "License");
|
| 192 |
+
you may not use this file except in compliance with the License.
|
| 193 |
+
You may obtain a copy of the License at
|
| 194 |
+
|
| 195 |
+
http://www.apache.org/licenses/LICENSE-2.0
|
| 196 |
+
|
| 197 |
+
Unless required by applicable law or agreed to in writing, software
|
| 198 |
+
distributed under the License is distributed on an "AS IS" BASIS,
|
| 199 |
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 200 |
+
See the License for the specific language governing permissions and
|
| 201 |
+
limitations under the License.
|
README.md
CHANGED
|
@@ -1,3 +1,251 @@
|
|
| 1 |
---
|
|
|
|
|
|
|
|
|
|
| 2 |
license: apache-2.0
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 3 |
---
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
---
|
| 2 |
+
language:
|
| 3 |
+
- en
|
| 4 |
+
- zh
|
| 5 |
license: apache-2.0
|
| 6 |
+
library_name: transformers
|
| 7 |
+
pipeline_tag: image-text-to-text
|
| 8 |
+
tags:
|
| 9 |
+
- agnes-ai
|
| 10 |
+
- reasoning
|
| 11 |
+
- multimodal
|
| 12 |
+
- long-context
|
| 13 |
+
- hybrid-attention
|
| 14 |
---
|
| 15 |
+
<p align="center">
|
| 16 |
+
<img width="132" src="assets/agnes_logo.svg" alt="Agnes AI logo">
|
| 17 |
+
</p>
|
| 18 |
+
|
| 19 |
+
<p align="center">
|
| 20 |
+
<a href="https://agnes-ai.com/"><img src="https://img.shields.io/badge/Agnes_AI-Website-3248AF" alt="Agnes AI website"></a>
|
| 21 |
+
<a href="#quickstart"><img src="https://img.shields.io/badge/Agnes--3.0--Flash_Preview-Open_Weights-3248AF" alt="Open weights"></a>
|
| 22 |
+
<img src="https://img.shields.io/badge/License-Apache_2.0-111827" alt="Apache 2.0">
|
| 23 |
+
</p>
|
| 24 |
+
|
| 25 |
+
|
| 26 |
+
|
| 27 |
+
# Agnes-3.0-Flash Preview
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
## Model version clarification
|
| 31 |
+
This repository contains an earlier open-weight **Preview checkpoint** of Agnes 3.0 Flash. It is distinct from the newer **production/API checkpoint** listed on [Artificial Analysis](https://artificialanalysis.ai/models/agnes-3-0-flash).
|
| 32 |
+
The Preview release has **33B parameters** and a context window of **262,144 tokens**. The production/API model uses a different checkpoint and configuration, with a **1M-token context window**. Its benchmark results should not be attributed to the Preview weights released here.
|
| 33 |
+
This repository was initially published as `Agnes-3.0-Flash` without the `Preview` suffix. The model card now explicitly identifies this release as **Agnes-3.0-Flash Preview** to clarify the distinction between the open-weight release and the production/API model.
|
| 34 |
+
The specifications and Agnes benchmark results below refer to the **Preview checkpoint**.
|
| 35 |
+
Hello! 👋 Today we are introducing **Agnes-3.0-Flash Preview**, an **open-weights multimodal preview model** built for people who want flagship-class reasoning without flagship-class hardware.
|
| 36 |
+
Highlights:
|
| 37 |
+
- **Competitive across core capabilities.** Agnes-3.0-Flash Preview posts competitive results across reasoning, coding, and instruction-following evaluations.
|
| 38 |
+
- **Built for demanding work.** A **262 144-token context window**, adjustable reasoning effort, tool calling, and **text, image and video** understanding.
|
| 39 |
+
|
| 40 |
+
|
| 41 |
+
### Benchmarks
|
| 42 |
+
> **Benchmark scope:** The Agnes results in the chart and table below belong to the **Agnes-3.0-Flash Preview open-weight checkpoint** released in this repository. They are not results for the production/API Agnes 3.0 Flash model listed on Artificial Analysis.
|
| 43 |
+
<p align="center">
|
| 44 |
+
<img style="width:100%;max-width:1100px" src="assets/benchmark-preview.png" alt="Agnes-3.0-Flash Preview benchmark reference results">
|
| 45 |
+
</p>
|
| 46 |
+
|
| 47 |
+
|
| 48 |
+
The Agnes-3.0-Flash Preview scores in the chart correspond to the open-weight checkpoint released in this repository.
|
| 49 |
+
Reference results across contemporary models are shown below. The figures were compiled from different sources, harnesses, and model snapshots and do not constitute a controlled head-to-head comparison.
|
| 50 |
+
<div style="font-family:-apple-system,BlinkMacSystemFont,'Segoe UI',Roboto,sans-serif;width:100%;margin:0 auto;padding:16px 0;overflow-x:auto">
|
| 51 |
+
<table style="width:100%;table-layout:fixed;border-collapse:collapse;font-size:11px;min-width:1180px">
|
| 52 |
+
<thead><tr>
|
| 53 |
+
<th style="width:15%;padding:9px 6px;text-align:left;border-bottom:2px solid #3248AF;color:#3248AF;font-size:11px">Benchmark</th>
|
| 54 |
+
<th style="width:10%;padding:9px 5px;text-align:center;font-weight:700;border-bottom:2px solid #3248AF;color:#3248AF;background:rgba(50,72,175,.09);font-size:10px;line-height:1.3">Agnes-3.0-Flash Preview</th>
|
| 55 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.6-35B-A3B<br><span style="opacity:.65">35B / 3B active</span></th>
|
| 56 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Kimi K2.5<br><span style="opacity:.65">1T / 32B active</span></th>
|
| 57 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Muse Glimmer<br><span style="opacity:.65">30B</span></th>
|
| 58 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.5<br><span style="opacity:.65">27B</span></th>
|
| 59 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">DeepSeek V4 Flash 0731<br><span style="opacity:.65">284B / 13B active</span></th>
|
| 60 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.8<br><span style="opacity:.65">27B</span></th>
|
| 61 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Gemini 3.5 Flash<br><span style="opacity:.65">undisclosed</span></th>
|
| 62 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.8 Flash Next<br><span style="opacity:.65">125B / 6B active</span></th>
|
| 63 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">MiniMax M3<br><span style="opacity:.65">428B / 23B active</span></th>
|
| 64 |
+
</tr></thead><tbody>
|
| 65 |
+
<tr><td style="padding:7px 6px">IFBench</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">74.20</td><td style="padding:7px 5px;text-align:center">64.4</td><td style="padding:7px 5px;text-align:center">43.7</td><td style="padding:7px 5px;text-align:center">77.0</td><td style="padding:7px 5px;text-align:center">75.6</td><td style="padding:7px 5px;text-align:center">75.8</td><td style="padding:7px 5px;text-align:center">79.5</td><td style="padding:7px 5px;text-align:center">76.3</td><td style="padding:7px 5px;text-align:center">81.3</td><td style="padding:7px 5px;text-align:center">82.9</td></tr>
|
| 66 |
+
<tr><td style="padding:7px 6px">SciCode</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">38.08</td><td style="padding:7px 5px;text-align:center">35.8</td><td style="padding:7px 5px;text-align:center">39.6</td><td style="padding:7px 5px;text-align:center">43.6</td><td style="padding:7px 5px;text-align:center">39.5</td><td style="padding:7px 5px;text-align:center">50.3</td><td style="padding:7px 5px;text-align:center">46.6</td><td style="padding:7px 5px;text-align:center">53.1</td><td style="padding:7px 5px;text-align:center">50.6</td><td style="padding:7px 5px;text-align:center">45.4</td></tr>
|
| 67 |
+
<tr><td style="padding:7px 6px">GPQA Diamond</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">85.05</td><td style="padding:7px 5px;text-align:center">84.1</td><td style="padding:7px 5px;text-align:center">78.9</td><td style="padding:7px 5px;text-align:center">83.5</td><td style="padding:7px 5px;text-align:center">85.8</td><td style="padding:7px 5px;text-align:center">90.8</td><td style="padding:7px 5px;text-align:center">90.5</td><td style="padding:7px 5px;text-align:center">92.2</td><td style="padding:7px 5px;text-align:center">92.3</td><td style="padding:7px 5px;text-align:center">92.9</td></tr>
|
| 68 |
+
<tr><td style="padding:7px 6px">AA-LCR</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">68.33</td><td style="padding:7px 5px;text-align:center">66.7</td><td style="padding:7px 5px;text-align:center">59.0</td><td style="padding:7px 5px;text-align:center">80.0</td><td style="padding:7px 5px;text-align:center">72.3</td><td style="padding:7px 5px;text-align:center">79.7</td><td style="padding:7px 5px;text-align:center">82.0</td><td style="padding:7px 5px;text-align:center">81.0</td><td style="padding:7px 5px;text-align:center">79.7</td><td style="padding:7px 5px;text-align:center">74.0</td></tr>
|
| 69 |
+
<tr><td style="padding:7px 6px">AA-Omniscience Accuracy</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">23.00</td><td style="padding:7px 5px;text-align:center">18.8</td><td style="padding:7px 5px;text-align:center">22.9</td><td style="padding:7px 5px;text-align:center">27.0</td><td style="padding:7px 5px;text-align:center">20.7</td><td style="padding:7px 5px;text-align:center">40.4</td><td style="padding:7px 5px;text-align:center">18.4</td><td style="padding:7px 5px;text-align:center">51.4</td><td style="padding:7px 5px;text-align:center">24.5</td><td style="padding:7px 5px;text-align:center">16.7</td></tr>
|
| 70 |
+
</tbody></table></div>
|
| 71 |
+
|
| 72 |
+
<p style="font-size:11px;opacity:.72">
|
| 73 |
+
Higher is better for every row. Header parameter figures mix total and active counts, and harnesses and snapshot dates differ across sources, so treat cross-column comparisons as reference values rather than a controlled head-to-head evaluation.
|
| 74 |
+
</p>
|
| 75 |
+
|
| 76 |
+
|
| 77 |
+
|
| 78 |
+
## Architecture
|
| 79 |
+
Agnes-3.0-Flash Preview is a hybrid-attention decoder: three of every four layers run a gated delta rule (recurrent, with per-layer state independent of sequence length), and the fourth runs standard global attention. Only 18 of the 72 layers therefore hold a KV cache that grows with context.
|
| 80 |
+
| | |
|
| 81 |
+
|---|---|
|
| 82 |
+
| Context length | **262 144** tokens |
|
| 83 |
+
| Decoder layers | 72 = 54 delta-rule recurrent + 18 global attention, alternating 3 : 1 |
|
| 84 |
+
| Hidden size | 5120 |
|
| 85 |
+
| Global attention | 24 query heads / 4 KV heads (6 : 1 GQA), head dim 256; RMS-norm on q and k, sigmoid-gated output |
|
| 86 |
+
| Delta-rule layers | 16 key heads / 48 value heads, head dim 128; causal conv (kernel 4) in front, gated RMS-norm; recurrent state in fp32 |
|
| 87 |
+
| Feed-forward | SwiGLU, intermediate size 17408; plus a parallel SwiGLU 2048 branch in every layer |
|
| 88 |
+
| Positions | 3-axis rotary (text / height / width), interleaved mrope sections 11 : 11 : 10, base 1e7, applied to the first 25 % of each head dim (64 dims) |
|
| 89 |
+
| Vocabulary | 248 320 |
|
| 90 |
+
| Vision tower | 27 layers, hidden 1152, patch 16, 2 × 2 spatial merge, projected to 5120 |
|
| 91 |
+
|
| 92 |
+
## Quickstart
|
| 93 |
+
<div style="border-left:4px solid #3248AF;background:rgba(50,72,175,.08);border-radius:6px;padding:12px 16px;font-family:-apple-system,BlinkMacSystemFont,'Segoe UI',Roboto,sans-serif;font-size:14px;line-height:1.6">
|
| 94 |
+
<div style="font-weight:700;color:#3248AF;margin-bottom:6px">REMOTE CODE REQUIRED</div>
|
| 95 |
+
|
| 96 |
+
<p style="margin:0"><b>Agnes-3.0-Flash Preview</b> ships its own model implementation. Always load it with <code>trust_remote_code=True</code>.</p>
|
| 97 |
+
|
| 98 |
+
</div>
|
| 99 |
+
|
| 100 |
+
|
| 101 |
+
|
| 102 |
+
### Requirements
|
| 103 |
+
```bash
|
| 104 |
+
pip install "transformers>=5.12" torch torchvision accelerate
|
| 105 |
+
```
|
| 106 |
+
Tested on transformers 5.12.1. Image and video inputs go through the bundled processor, which needs `torchvision`.
|
| 107 |
+
|
| 108 |
+
### Transformers
|
| 109 |
+
```python
|
| 110 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 111 |
+
path = "Agnes-AI/Agnes-3.0-Flash"
|
| 112 |
+
tok = AutoTokenizer.from_pretrained(path)
|
| 113 |
+
model = AutoModelForCausalLM.from_pretrained(
|
| 114 |
+
path, dtype="bfloat16", device_map="auto", trust_remote_code=True
|
| 115 |
+
)
|
| 116 |
+
msgs = [{"role": "user", "content": "请用三句话解释什么是人工智能。"}]
|
| 117 |
+
ids = tok.apply_chat_template(msgs, add_generation_prompt=True, return_tensors="pt").to(model.device)
|
| 118 |
+
out = model.generate(ids, max_new_tokens=256)
|
| 119 |
+
print(tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True))
|
| 120 |
+
```
|
| 121 |
+
|
| 122 |
+
### Images and video
|
| 123 |
+
Image and video inputs go through the bundled processor (also remote code):
|
| 124 |
+
```python
|
| 125 |
+
from transformers import AutoProcessor
|
| 126 |
+
proc = AutoProcessor.from_pretrained(path, trust_remote_code=True)
|
| 127 |
+
msgs = [{"role": "user", "content": [{"type": "image", "image": "photo.jpg"},
|
| 128 |
+
{"type": "text", "text": "描述这张图。"}]}]
|
| 129 |
+
inputs = proc.apply_chat_template(msgs, add_generation_prompt=True, tokenize=True,
|
| 130 |
+
return_dict=True, return_tensors="pt").to(model.device)
|
| 131 |
+
out = model.generate(**inputs, max_new_tokens=256)
|
| 132 |
+
print(proc.batch_decode(out[:, inputs["input_ids"].shape[1]:], skip_special_tokens=True)[0])
|
| 133 |
+
```
|
| 134 |
+
|
| 135 |
+
### Reasoning effort
|
| 136 |
+
The chat template exposes three reasoning levels — `high` (default), `medium`, `low` — plus a thinking-off switch:
|
| 137 |
+
```python
|
| 138 |
+
ids = tok.apply_chat_template(msgs, add_generation_prompt=True, return_tensors="pt",
|
| 139 |
+
reasoning_effort="medium") # or enable_thinking=False
|
| 140 |
+
```
|
| 141 |
+
|
| 142 |
+
### Tool calling
|
| 143 |
+
The chat template renders tool definitions for you. The model emits calls as `<tool_call><function=…><parameter=…>`, and you feed results back as a `tool` role message:
|
| 144 |
+
```python
|
| 145 |
+
tools = [{
|
| 146 |
+
"type": "function",
|
| 147 |
+
"function": {
|
| 148 |
+
"name": "get_weather",
|
| 149 |
+
"description": "Look up current weather for a city",
|
| 150 |
+
"parameters": {
|
| 151 |
+
"type": "object",
|
| 152 |
+
"properties": {"city": {"type": "string", "description": "City name"}},
|
| 153 |
+
"required": ["city"],
|
| 154 |
+
},
|
| 155 |
+
},
|
| 156 |
+
}]
|
| 157 |
+
msgs = [{"role": "user", "content": "What's the weather in Beijing right now?"}]
|
| 158 |
+
ids = tok.apply_chat_template(msgs, tools=tools, add_generation_prompt=True,
|
| 159 |
+
return_tensors="pt").to(model.device)
|
| 160 |
+
out = model.generate(ids, max_new_tokens=256)
|
| 161 |
+
reply = tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True)
|
| 162 |
+
|
| 163 |
+
# <tool_call>
|
| 164 |
+
|
| 165 |
+
# <function=get_weather>
|
| 166 |
+
|
| 167 |
+
# <parameter=city>
|
| 168 |
+
|
| 169 |
+
# Beijing
|
| 170 |
+
|
| 171 |
+
# </parameter>
|
| 172 |
+
|
| 173 |
+
# </function>
|
| 174 |
+
|
| 175 |
+
# </tool_call>
|
| 176 |
+
|
| 177 |
+
# run the tool, append the result, generate the final answer
|
| 178 |
+
msgs += [{"role": "assistant", "content": reply},
|
| 179 |
+
{"role": "tool", "content": "Clear, 26°C, light northeasterly wind"}]
|
| 180 |
+
```
|
| 181 |
+
Over the OpenAI API pass `tools=` the same way. The server returns the text above verbatim by default; to get structured `tool_calls`, configure sglang with a tool-call parser matching this format (likewise a reasoning parser, if you want the thinking span in `reasoning_content`).
|
| 182 |
+
|
| 183 |
+
### SGLang
|
| 184 |
+
`serve.sh` starts a server from a stock public image, overlaying three files onto the image's `sglang` package and nothing else. See `sglang_patch/README.md`.
|
| 185 |
+
```bash
|
| 186 |
+
docker run --gpus all --shm-size 64g -p 30001:8080 \
|
| 187 |
+
-v /path/to/agnes-3.0-flash:/model \
|
| 188 |
+
lmsysorg/sglang:nightly-dev-20260908-20ca564b \
|
| 189 |
+
bash /agnes-3.0-flash/serve.sh --served-model-name Agnes-3.0-Flash
|
| 190 |
+
```
|
| 191 |
+
`serve.sh` forwards extra command-line arguments to sglang, which is how `--served-model-name` takes effect; `--tp 2` works the same way. The server listens on port 8080 inside the container:
|
| 192 |
+
```python
|
| 193 |
+
from openai import OpenAI
|
| 194 |
+
client = OpenAI(api_key="EMPTY", base_url="http://localhost:30001/v1")
|
| 195 |
+
response = client.chat.completions.create(
|
| 196 |
+
model="Agnes-3.0-Flash",
|
| 197 |
+
messages=[{"role": "user", "content": "Design a fault-tolerant event processing architecture."}],
|
| 198 |
+
temperature=1.0,
|
| 199 |
+
max_tokens=2000,
|
| 200 |
+
)
|
| 201 |
+
print(response.choices[0].message.content)
|
| 202 |
+
```
|
| 203 |
+
Pass `stream=True` for streaming; `tools=` and `reasoning_effort=` are accepted the same way.
|
| 204 |
+
|
| 205 |
+
## Hardware Requirements
|
| 206 |
+
| Resource | Recommendation |
|
| 207 |
+
|---|---|
|
| 208 |
+
| GPUs | 1 × NVIDIA H200 141 GB or NVIDIA H100 80 GB (or equivalent) at bf16 |
|
| 209 |
+
| Tensor parallel | `--tp 1`; `--tp 2` for maximum context and concurrency |
|
| 210 |
+
| Weights on disk | Approximately 66 GB for the bf16 checkpoint |
|
| 211 |
+
| Host memory | 128 GB or more recommended |
|
| 212 |
+
|
| 213 |
+
Actual context length and concurrency depend on KV-cache allocation, runtime overhead, and tensor-parallel configuration; validate the target workload on the intended hardware.
|
| 214 |
+
|
| 215 |
+
## Recommended Inference Settings
|
| 216 |
+
| Setting | Recommended |
|
| 217 |
+
|---|---|
|
| 218 |
+
| `temperature` | 1.0 |
|
| 219 |
+
| `top_p` | 0.95 |
|
| 220 |
+
| `top_k` | 20 |
|
| 221 |
+
| `reasoning_effort` | `high` for hard reasoning, `low` for latency-sensitive traffic |
|
| 222 |
+
| `max_tokens` | 2000 or higher |
|
| 223 |
+
|
| 224 |
+
These are the checkpoint's own `generation_config.json` defaults.
|
| 225 |
+
|
| 226 |
+
## Model Capabilities
|
| 227 |
+
| Capability | Support |
|
| 228 |
+
|---|---|
|
| 229 |
+
| Advanced reasoning | Yes, with `high` / `medium` / `low` effort levels |
|
| 230 |
+
| Coding and debugging | Yes |
|
| 231 |
+
| Long-context analysis | 262 144 tokens |
|
| 232 |
+
| Image understanding | Yes |
|
| 233 |
+
| Video understanding | Yes |
|
| 234 |
+
| Tool calling | Yes (`<tool_call>` / `<tool_response>`) |
|
| 235 |
+
| Streaming | Yes |
|
| 236 |
+
| OpenAI-compatible APIs | Chat Completions via sglang |
|
| 237 |
+
|
| 238 |
+
## License
|
| 239 |
+
Released under the [Apache License 2.0](LICENSE).
|
| 240 |
+
|
| 241 |
+
## Citation
|
| 242 |
+
```bibtex
|
| 243 |
+
@misc{agnes30flash2026,
|
| 244 |
+
title = {Agnes-3.0-Flash Preview},
|
| 245 |
+
author = {{Agnes AI}},
|
| 246 |
+
year = {2026},
|
| 247 |
+
month = sep,
|
| 248 |
+
howpublished = {Open-weights preview checkpoint},
|
| 249 |
+
url = {https://agnes-ai.com/}
|
| 250 |
+
}
|
| 251 |
+
```
|
README_zh.md
ADDED
|
@@ -0,0 +1,273 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
language:
|
| 3 |
+
- zh
|
| 4 |
+
- en
|
| 5 |
+
license: apache-2.0
|
| 6 |
+
library_name: transformers
|
| 7 |
+
pipeline_tag: image-text-to-text
|
| 8 |
+
tags:
|
| 9 |
+
- agnes-ai
|
| 10 |
+
- reasoning
|
| 11 |
+
- multimodal
|
| 12 |
+
- long-context
|
| 13 |
+
- hybrid-attention
|
| 14 |
+
---
|
| 15 |
+
|
| 16 |
+
<p align="center">
|
| 17 |
+
<img width="132" src="assets/agnes_logo.svg" alt="Agnes AI logo">
|
| 18 |
+
</p>
|
| 19 |
+
|
| 20 |
+
<p align="center">
|
| 21 |
+
<a href="https://agnes-ai.com/"><img src="https://img.shields.io/badge/Agnes_AI-官网-3248AF" alt="Agnes AI 官网"></a>
|
| 22 |
+
<a href="#快速开始"><img src="https://img.shields.io/badge/Agnes--3.0--Flash_Preview-开放权重-3248AF" alt="开放权重"></a>
|
| 23 |
+
<img src="https://img.shields.io/badge/License-Apache_2.0-111827" alt="Apache 2.0">
|
| 24 |
+
</p>
|
| 25 |
+
|
| 26 |
+
# Agnes-3.0-Flash Preview
|
| 27 |
+
|
| 28 |
+
## 模型版本说明
|
| 29 |
+
|
| 30 |
+
本仓库包含 Agnes 3.0 Flash 的早期开放权重 **Preview checkpoint**,它与 [Artificial Analysis](https://artificialanalysis.ai/models/agnes-3-0-flash) 页面所列的新版 **production/API checkpoint** 不同。
|
| 31 |
+
|
| 32 |
+
本 Preview 版本约有 **33B 参数**,上下文窗口为 **262,144 token**。production/API 版本使用不同的 checkpoint 和配置,具有 **1M token 上下文窗口**;production/API 版本的评测结果不应归属于本仓库发布的 Preview 权重。
|
| 33 |
+
|
| 34 |
+
本仓库最初以 `Agnes-3.0-Flash` 发布,名称中遗漏了 `Preview` 后缀。现在通过模型卡明确将其标识为 **Agnes-3.0-Flash Preview**,以区分开放权重 Preview 与 production/API 模型。除非另有说明,本模型卡中的规格和 Agnes 评测成绩均指本 Preview checkpoint。
|
| 35 |
+
|
| 36 |
+
本 Preview 是约 **33B 参数的 dense(稠密)checkpoint**,不采用 MoE 架构,也不存在“3B active parameters”的口径。评测表中的 active parameter 数字仅用于标注部分对比模型。
|
| 37 |
+
|
| 38 |
+
你好!👋 今天我们发布 **Agnes-3.0-Flash Preview** —— 一个**开放权重多模态 Preview 模型**,面向那些想要旗舰级推理质量、但不想付出旗舰级硬件代价的使用者。
|
| 39 |
+
|
| 40 |
+
亮点:
|
| 41 |
+
|
| 42 |
+
- **核心能力表现有竞争力。** Agnes-3.0-Flash Preview 在推理、代码和指令遵循等多项评测中展现出有竞争力的结果。
|
| 43 |
+
- **面向真实负载。** **262 144 token 上下文**、三档可调推理强度、工具调用,以及**文本 / 图像 / 视频**理解。
|
| 44 |
+
|
| 45 |
+
### 评测结果
|
| 46 |
+
|
| 47 |
+
> **评测范围:** 下图和下表中的 Agnes 成绩属于本仓库发布的 **Agnes-3.0-Flash Preview 开放权重 checkpoint**,并非 Artificial Analysis 所列 production/API Agnes 3.0 Flash 模型的成绩。
|
| 48 |
+
|
| 49 |
+
<p align="center">
|
| 50 |
+
<img style="width:100%;max-width:1100px" src="assets/benchmark-preview.png" alt="Agnes-3.0-Flash Preview 评测参考结果">
|
| 51 |
+
</p>
|
| 52 |
+
|
| 53 |
+
图表中的 Agnes-3.0-Flash Preview 成绩对应本仓库发布的开放权重 checkpoint。
|
| 54 |
+
|
| 55 |
+
下表汇总了多个同期模型的参考结果。数据来自不同评测环境、模型快照和 harness,不构成同一设置下的受控对比。
|
| 56 |
+
|
| 57 |
+
<div style="font-family:-apple-system,BlinkMacSystemFont,'Segoe UI',Roboto,sans-serif;width:100%;margin:0 auto;padding:16px 0;overflow-x:auto">
|
| 58 |
+
<table style="width:100%;table-layout:fixed;border-collapse:collapse;font-size:11px;min-width:1180px">
|
| 59 |
+
<thead><tr>
|
| 60 |
+
<th style="width:15%;padding:9px 6px;text-align:left;border-bottom:2px solid #3248AF;color:#3248AF;font-size:11px">评测集</th>
|
| 61 |
+
<th style="width:10%;padding:9px 5px;text-align:center;font-weight:700;border-bottom:2px solid #3248AF;color:#3248AF;background:rgba(50,72,175,.09);font-size:10px;line-height:1.3">Agnes-3.0-Flash Preview</th>
|
| 62 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.6-35B-A3B<br><span style="opacity:.65">35B / 3B 激活</span></th>
|
| 63 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Kimi K2.5<br><span style="opacity:.65">1T / 32B 激活</span></th>
|
| 64 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Muse Glimmer<br><span style="opacity:.65">30B</span></th>
|
| 65 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.5<br><span style="opacity:.65">27B</span></th>
|
| 66 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">DeepSeek V4 Flash 0731<br><span style="opacity:.65">284B / 13B 激活</span></th>
|
| 67 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.8<br><span style="opacity:.65">27B</span></th>
|
| 68 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Gemini 3.5 Flash<br><span style="opacity:.65">参数未公开</span></th>
|
| 69 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">Qwen3.8 Flash Next<br><span style="opacity:.65">125B / 6B 激活</span></th>
|
| 70 |
+
<th style="width:8.33%;padding:9px 5px;text-align:center;border-bottom:2px solid #3248AF;font-size:10px;line-height:1.3">MiniMax M3<br><span style="opacity:.65">428B / 23B 激活</span></th>
|
| 71 |
+
</tr></thead><tbody>
|
| 72 |
+
<tr><td style="padding:7px 6px">IFBench</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">74.20</td><td style="padding:7px 5px;text-align:center">64.4</td><td style="padding:7px 5px;text-align:center">43.7</td><td style="padding:7px 5px;text-align:center">77.0</td><td style="padding:7px 5px;text-align:center">75.6</td><td style="padding:7px 5px;text-align:center">75.8</td><td style="padding:7px 5px;text-align:center">79.5</td><td style="padding:7px 5px;text-align:center">76.3</td><td style="padding:7px 5px;text-align:center">81.3</td><td style="padding:7px 5px;text-align:center">82.9</td></tr>
|
| 73 |
+
<tr><td style="padding:7px 6px">SciCode</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">38.08</td><td style="padding:7px 5px;text-align:center">35.8</td><td style="padding:7px 5px;text-align:center">39.6</td><td style="padding:7px 5px;text-align:center">43.6</td><td style="padding:7px 5px;text-align:center">39.5</td><td style="padding:7px 5px;text-align:center">50.3</td><td style="padding:7px 5px;text-align:center">46.6</td><td style="padding:7px 5px;text-align:center">53.1</td><td style="padding:7px 5px;text-align:center">50.6</td><td style="padding:7px 5px;text-align:center">45.4</td></tr>
|
| 74 |
+
<tr><td style="padding:7px 6px">GPQA Diamond</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">85.05</td><td style="padding:7px 5px;text-align:center">84.1</td><td style="padding:7px 5px;text-align:center">78.9</td><td style="padding:7px 5px;text-align:center">83.5</td><td style="padding:7px 5px;text-align:center">85.8</td><td style="padding:7px 5px;text-align:center">90.8</td><td style="padding:7px 5px;text-align:center">90.5</td><td style="padding:7px 5px;text-align:center">92.2</td><td style="padding:7px 5px;text-align:center">92.3</td><td style="padding:7px 5px;text-align:center">92.9</td></tr>
|
| 75 |
+
<tr><td style="padding:7px 6px">AA-LCR</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">68.33</td><td style="padding:7px 5px;text-align:center">66.7</td><td style="padding:7px 5px;text-align:center">59.0</td><td style="padding:7px 5px;text-align:center">80.0</td><td style="padding:7px 5px;text-align:center">72.3</td><td style="padding:7px 5px;text-align:center">79.7</td><td style="padding:7px 5px;text-align:center">82.0</td><td style="padding:7px 5px;text-align:center">81.0</td><td style="padding:7px 5px;text-align:center">79.7</td><td style="padding:7px 5px;text-align:center">74.0</td></tr>
|
| 76 |
+
<tr><td style="padding:7px 6px">AA-Omniscience 准确率</td><td style="padding:7px 5px;text-align:center;font-weight:700;color:#3248AF;background:rgba(50,72,175,.05)">23.00</td><td style="padding:7px 5px;text-align:center">18.8</td><td style="padding:7px 5px;text-align:center">22.9</td><td style="padding:7px 5px;text-align:center">27.0</td><td style="padding:7px 5px;text-align:center">20.7</td><td style="padding:7px 5px;text-align:center">40.4</td><td style="padding:7px 5px;text-align:center">18.4</td><td style="padding:7px 5px;text-align:center">51.4</td><td style="padding:7px 5px;text-align:center">24.5</td><td style="padding:7px 5px;text-align:center">16.7</td></tr>
|
| 77 |
+
</tbody></table></div>
|
| 78 |
+
|
| 79 |
+
<p style="font-size:11px;opacity:.72">
|
| 80 |
+
所有指标均为越高越好。表头的参数标注口径不完全一致(总参数 / 激活参数),不同来源的评测 harness 与快照时点也不相同,跨列数值仅作参考,不构成受控的横向对比。
|
| 81 |
+
</p>
|
| 82 |
+
|
| 83 |
+
## 架构
|
| 84 |
+
|
| 85 |
+
Agnes-3.0-Flash Preview 是一个混合注意力的解码器:每四层中三层走门控 delta rule(循环式,单层状态大小与序列长度无关),第四层走标准全局注意力。72 层里因此只有 18 层持有随长度增长的 KV cache。
|
| 86 |
+
|
| 87 |
+
| | |
|
| 88 |
+
|---|---|
|
| 89 |
+
| 上下文长度 | **262 144** token |
|
| 90 |
+
| 解码层 | 72 层 = 54 层 delta-rule 递归 + 18 层全局注意力,按 3 : 1 交替 |
|
| 91 |
+
| 隐藏维度 | 5120 |
|
| 92 |
+
| 全局注意力 | 24 query heads / 4 KV heads(GQA 6 : 1),head dim 256;q、k 各带 RMS-norm,输出经 sigmoid 门控 |
|
| 93 |
+
| Delta-rule 层 | 16 key heads / 48 value heads,head dim 128;前置因果卷积(kernel 4),gated RMS-norm;循环状态为 fp32 |
|
| 94 |
+
| 前馈 | SwiGLU,中间维 17408;每层另并联一路 SwiGLU 2048 分支 |
|
| 95 |
+
| 位置编码 | 三轴 rotary(text / height / width),mrope 分段 11 : 11 : 10 交错,base 1e7,作用于 head dim 的前 25%(64 维) |
|
| 96 |
+
| 词表 | 248 320 |
|
| 97 |
+
| 视觉塔 | 27 层,hidden 1152,patch 16,2 × 2 空间合并,投影至 5120 |
|
| 98 |
+
|
| 99 |
+
## 快速开始
|
| 100 |
+
|
| 101 |
+
<div style="border-left:4px solid #3248AF;background:rgba(50,72,175,.08);border-radius:6px;padding:12px 16px;font-family:-apple-system,BlinkMacSystemFont,'Segoe UI',Roboto,sans-serif;font-size:14px;line-height:1.6">
|
| 102 |
+
<div style="font-weight:700;color:#3248AF;margin-bottom:6px">必须启用 REMOTE CODE</div>
|
| 103 |
+
<p style="margin:0"><b>Agnes-3.0-Flash Preview</b> 自带模型实现,加载时务必传 <code>trust_remote_code=True</code>。</p>
|
| 104 |
+
</div>
|
| 105 |
+
|
| 106 |
+
### 环境要求
|
| 107 |
+
|
| 108 |
+
```bash
|
| 109 |
+
pip install "transformers>=5.12" torch torchvision accelerate
|
| 110 |
+
```
|
| 111 |
+
|
| 112 |
+
实测环境为 transformers 5.12.1。图像与视频输入由随附的 processor 处理,依赖 `torchvision`。
|
| 113 |
+
|
| 114 |
+
### Transformers
|
| 115 |
+
|
| 116 |
+
```python
|
| 117 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 118 |
+
|
| 119 |
+
path = "Agnes-AI/Agnes-3.0-Flash"
|
| 120 |
+
tok = AutoTokenizer.from_pretrained(path)
|
| 121 |
+
model = AutoModelForCausalLM.from_pretrained(
|
| 122 |
+
path, dtype="bfloat16", device_map="auto", trust_remote_code=True
|
| 123 |
+
)
|
| 124 |
+
|
| 125 |
+
msgs = [{"role": "user", "content": "请用三句话解释什么是人工智能。"}]
|
| 126 |
+
ids = tok.apply_chat_template(msgs, add_generation_prompt=True, return_tensors="pt").to(model.device)
|
| 127 |
+
out = model.generate(ids, max_new_tokens=256)
|
| 128 |
+
print(tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True))
|
| 129 |
+
```
|
| 130 |
+
|
| 131 |
+
### 图像与视频
|
| 132 |
+
|
| 133 |
+
图像、视频输入走随模型附带的 processor(同样是 remote code):
|
| 134 |
+
|
| 135 |
+
```python
|
| 136 |
+
from transformers import AutoProcessor
|
| 137 |
+
|
| 138 |
+
proc = AutoProcessor.from_pretrained(path, trust_remote_code=True)
|
| 139 |
+
msgs = [{"role": "user", "content": [{"type": "image", "image": "photo.jpg"},
|
| 140 |
+
{"type": "text", "text": "描述这张图。"}]}]
|
| 141 |
+
inputs = proc.apply_chat_template(msgs, add_generation_prompt=True, tokenize=True,
|
| 142 |
+
return_dict=True, return_tensors="pt").to(model.device)
|
| 143 |
+
out = model.generate(**inputs, max_new_tokens=256)
|
| 144 |
+
print(proc.batch_decode(out[:, inputs["input_ids"].shape[1]:], skip_special_tokens=True)[0])
|
| 145 |
+
```
|
| 146 |
+
|
| 147 |
+
### 推理强度
|
| 148 |
+
|
| 149 |
+
chat template 提供三档推理强度 —— `high`(默认)、`medium`、`low`,也可以整体关闭思考:
|
| 150 |
+
|
| 151 |
+
```python
|
| 152 |
+
ids = tok.apply_chat_template(msgs, add_generation_prompt=True, return_tensors="pt",
|
| 153 |
+
reasoning_effort="medium") # 或 enable_thinking=False
|
| 154 |
+
```
|
| 155 |
+
|
| 156 |
+
### 工具调用
|
| 157 |
+
|
| 158 |
+
chat template 会自动渲染工具定义,模型按 `<tool_call><function=…><parameter=…>` 的格式发起调用,工具返回值作为 `tool` 角色消息接回去即可:
|
| 159 |
+
|
| 160 |
+
```python
|
| 161 |
+
tools = [{
|
| 162 |
+
"type": "function",
|
| 163 |
+
"function": {
|
| 164 |
+
"name": "get_weather",
|
| 165 |
+
"description": "查询指定城市的实时天气",
|
| 166 |
+
"parameters": {
|
| 167 |
+
"type": "object",
|
| 168 |
+
"properties": {"city": {"type": "string", "description": "城市名称"}},
|
| 169 |
+
"required": ["city"],
|
| 170 |
+
},
|
| 171 |
+
},
|
| 172 |
+
}]
|
| 173 |
+
|
| 174 |
+
msgs = [{"role": "user", "content": "北京现在天气怎么样?"}]
|
| 175 |
+
ids = tok.apply_chat_template(msgs, tools=tools, add_generation_prompt=True,
|
| 176 |
+
return_tensors="pt").to(model.device)
|
| 177 |
+
out = model.generate(ids, max_new_tokens=256)
|
| 178 |
+
reply = tok.decode(out[0][ids.shape[1]:], skip_special_tokens=True)
|
| 179 |
+
# <tool_call>
|
| 180 |
+
# <function=get_weather>
|
| 181 |
+
# <parameter=city>
|
| 182 |
+
# 北京
|
| 183 |
+
# </parameter>
|
| 184 |
+
# </function>
|
| 185 |
+
# </tool_call>
|
| 186 |
+
|
| 187 |
+
# 执行工具后把结果接回对话,继续生成最终回复
|
| 188 |
+
msgs += [{"role": "assistant", "content": reply},
|
| 189 |
+
{"role": "tool", "content": "晴,26°C,东北风 2 级"}]
|
| 190 |
+
```
|
| 191 |
+
|
| 192 |
+
走 OpenAI 接口时同样传 `tools=`。服务端默认原样返回上面这段文本;要拿到结构化的 `tool_calls`,需给 sglang 配置与该格式匹配的 tool-call parser(思考段同理,需配置 reasoning parser 才会落入 `reasoning_content`)。
|
| 193 |
+
|
| 194 |
+
### SGLang
|
| 195 |
+
|
| 196 |
+
`serve.sh` 用官方公开镜像起服务,只覆盖 `sglang` 包里的三个文件,镜像内其他内容一概不动。详见 `sglang_patch/README.md`。
|
| 197 |
+
|
| 198 |
+
```bash
|
| 199 |
+
docker run --gpus all --shm-size 64g -p 30001:8080 \
|
| 200 |
+
-v /path/to/agnes-3.0-flash:/model \
|
| 201 |
+
lmsysorg/sglang:nightly-dev-20260908-20ca564b \
|
| 202 |
+
bash /agnes-3.0-flash/serve.sh --served-model-name Agnes-3.0-Flash
|
| 203 |
+
```
|
| 204 |
+
|
| 205 |
+
`serve.sh` 会把命令行上的额外参数透传给 sglang,`--served-model-name` 即由此生效,同理可以追加 `--tp 2`。服务在容器内 8080 端口启动:
|
| 206 |
+
|
| 207 |
+
```python
|
| 208 |
+
from openai import OpenAI
|
| 209 |
+
|
| 210 |
+
client = OpenAI(api_key="EMPTY", base_url="http://localhost:30001/v1")
|
| 211 |
+
response = client.chat.completions.create(
|
| 212 |
+
model="Agnes-3.0-Flash",
|
| 213 |
+
messages=[{"role": "user", "content": "设计一个容错的事件处理架构。"}],
|
| 214 |
+
temperature=1.0,
|
| 215 |
+
max_tokens=2000,
|
| 216 |
+
)
|
| 217 |
+
print(response.choices[0].message.content)
|
| 218 |
+
```
|
| 219 |
+
|
| 220 |
+
流式输出传 `stream=True` 即可;`tools=`、`reasoning_effort=` 等参数同样按 OpenAI 协议传递。
|
| 221 |
+
|
| 222 |
+
## 硬件需求
|
| 223 |
+
|
| 224 |
+
| 资源 | 建议 |
|
| 225 |
+
|---|---|
|
| 226 |
+
| GPU | 1 × NVIDIA H200 141 GB 或 NVIDIA H100 80 GB(或同等),bf16 |
|
| 227 |
+
| 张量并行 | `--tp 1`;追求最大上下文与并发时用 `--tp 2` |
|
| 228 |
+
| 权重磁盘占用 | bf16 检查点约 66 GB |
|
| 229 |
+
| 主机内存 | 建议 128 GB 以上 |
|
| 230 |
+
|
| 231 |
+
实际可用上下文长度和并发能力取决于 KV cache 分配、运行时开销和张量并行配置;请在目标硬件上验证实际负载。
|
| 232 |
+
|
| 233 |
+
## 推荐推理参数
|
| 234 |
+
|
| 235 |
+
| 参数 | 推荐值 |
|
| 236 |
+
|---|---|
|
| 237 |
+
| `temperature` | 1.0 |
|
| 238 |
+
| `top_p` | 0.95 |
|
| 239 |
+
| `top_k` | 20 |
|
| 240 |
+
| `reasoning_effort` | 难推理任务用 `high`,延迟敏感场景用 `low` |
|
| 241 |
+
| `max_tokens` | 2000 起 |
|
| 242 |
+
|
| 243 |
+
以上即检查点 `generation_config.json` 自带的默认值。
|
| 244 |
+
|
| 245 |
+
## 能力一览
|
| 246 |
+
|
| 247 |
+
| 能力 | 支持情况 |
|
| 248 |
+
|---|---|
|
| 249 |
+
| 深度推理 | 支持,可调 `high` / `medium` / `low` 三档 |
|
| 250 |
+
| 代码与调试 | 支持 |
|
| 251 |
+
| 长上下文分析 | 262 144 token |
|
| 252 |
+
| 图像理解 | 支持 |
|
| 253 |
+
| 视频理解 | 支持 |
|
| 254 |
+
| 工具调用 | 支持(`<tool_call>` / `<tool_response>`) |
|
| 255 |
+
| 流式输出 | 支持 |
|
| 256 |
+
| OpenAI 兼容接口 | 通过 sglang 提供 Chat Completions |
|
| 257 |
+
|
| 258 |
+
## 许可证
|
| 259 |
+
|
| 260 |
+
本项目采用 [Apache License 2.0](LICENSE)。
|
| 261 |
+
|
| 262 |
+
## 引用
|
| 263 |
+
|
| 264 |
+
```bibtex
|
| 265 |
+
@misc{agnes30flash2026,
|
| 266 |
+
title = {Agnes-3.0-Flash Preview},
|
| 267 |
+
author = {{Agnes AI}},
|
| 268 |
+
year = {2026},
|
| 269 |
+
month = sep,
|
| 270 |
+
howpublished = {Open-weights preview checkpoint},
|
| 271 |
+
url = {https://agnes-ai.com/}
|
| 272 |
+
}
|
| 273 |
+
```
|
agnes_quantization_manifest.json
ADDED
|
@@ -0,0 +1,1071 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"source": "/models/agnes-3.0-flash-fused-bf16",
|
| 3 |
+
"source_revision": "891ce4f9ffb89b22888aa7fcc2bb2f3618867684",
|
| 4 |
+
"llm_compressor_version": "0.13.0",
|
| 5 |
+
"compressed_tensors_version": "0.18.0",
|
| 6 |
+
"scheme": "FP8",
|
| 7 |
+
"targets": "Linear",
|
| 8 |
+
"ignore": [
|
| 9 |
+
"lm_head",
|
| 10 |
+
"model.language_model.embed_tokens",
|
| 11 |
+
"model.visual",
|
| 12 |
+
"re:^model\\.visual\\..*",
|
| 13 |
+
"re:.*delta_attn\\.conv1d$",
|
| 14 |
+
"re:.*delta_attn\\.in_proj_a$",
|
| 15 |
+
"re:.*delta_attn\\.in_proj_b$",
|
| 16 |
+
"re:^mtp.*"
|
| 17 |
+
],
|
| 18 |
+
"tracing_ignore": [
|
| 19 |
+
"_update_causal_mask",
|
| 20 |
+
"create_causal_mask",
|
| 21 |
+
"_update_mamba_mask",
|
| 22 |
+
"make_causal_mask",
|
| 23 |
+
"get_causal_mask",
|
| 24 |
+
"mask_interface",
|
| 25 |
+
"mask_function",
|
| 26 |
+
"_prepare_4d_causal_attention_mask",
|
| 27 |
+
"_prepare_fsmt_decoder_inputs",
|
| 28 |
+
"_prepare_4d_causal_attention_mask_with_cache_position",
|
| 29 |
+
"_update_linear_attn_mask",
|
| 30 |
+
"project_per_layer_inputs",
|
| 31 |
+
"_recurrent_mask"
|
| 32 |
+
],
|
| 33 |
+
"dataset": "HuggingFaceH4/ultrachat_200k",
|
| 34 |
+
"dataset_revision": "8049631c405ae6576f93f445c6b8166f76f5505a",
|
| 35 |
+
"dataset_split": "train_sft",
|
| 36 |
+
"seed": 42,
|
| 37 |
+
"num_calibration_samples": 512,
|
| 38 |
+
"max_sequence_length": 2048,
|
| 39 |
+
"batch_size": 1,
|
| 40 |
+
"native_chat_template": true,
|
| 41 |
+
"fix_mistral_regex": true,
|
| 42 |
+
"sample_indices": [
|
| 43 |
+
167621,
|
| 44 |
+
29184,
|
| 45 |
+
6556,
|
| 46 |
+
194393,
|
| 47 |
+
72097,
|
| 48 |
+
64196,
|
| 49 |
+
58513,
|
| 50 |
+
36579,
|
| 51 |
+
193061,
|
| 52 |
+
26868,
|
| 53 |
+
177392,
|
| 54 |
+
194161,
|
| 55 |
+
142964,
|
| 56 |
+
22790,
|
| 57 |
+
154794,
|
| 58 |
+
110604,
|
| 59 |
+
8331,
|
| 60 |
+
7811,
|
| 61 |
+
24561,
|
| 62 |
+
57314,
|
| 63 |
+
60990,
|
| 64 |
+
132475,
|
| 65 |
+
157815,
|
| 66 |
+
6956,
|
| 67 |
+
147127,
|
| 68 |
+
52124,
|
| 69 |
+
187700,
|
| 70 |
+
170363,
|
| 71 |
+
183848,
|
| 72 |
+
142853,
|
| 73 |
+
109974,
|
| 74 |
+
57787,
|
| 75 |
+
117757,
|
| 76 |
+
154472,
|
| 77 |
+
72926,
|
| 78 |
+
1703,
|
| 79 |
+
198916,
|
| 80 |
+
41853,
|
| 81 |
+
183013,
|
| 82 |
+
110785,
|
| 83 |
+
89194,
|
| 84 |
+
72842,
|
| 85 |
+
40758,
|
| 86 |
+
56443,
|
| 87 |
+
200145,
|
| 88 |
+
88236,
|
| 89 |
+
26793,
|
| 90 |
+
24312,
|
| 91 |
+
99595,
|
| 92 |
+
25353,
|
| 93 |
+
94104,
|
| 94 |
+
90165,
|
| 95 |
+
158263,
|
| 96 |
+
69342,
|
| 97 |
+
11390,
|
| 98 |
+
191294,
|
| 99 |
+
120435,
|
| 100 |
+
140568,
|
| 101 |
+
32722,
|
| 102 |
+
99230,
|
| 103 |
+
20656,
|
| 104 |
+
144714,
|
| 105 |
+
76854,
|
| 106 |
+
164794,
|
| 107 |
+
162141,
|
| 108 |
+
94800,
|
| 109 |
+
151349,
|
| 110 |
+
50407,
|
| 111 |
+
184699,
|
| 112 |
+
18233,
|
| 113 |
+
12012,
|
| 114 |
+
173346,
|
| 115 |
+
59742,
|
| 116 |
+
202655,
|
| 117 |
+
75861,
|
| 118 |
+
20916,
|
| 119 |
+
61024,
|
| 120 |
+
26476,
|
| 121 |
+
99647,
|
| 122 |
+
72869,
|
| 123 |
+
118858,
|
| 124 |
+
166640,
|
| 125 |
+
95638,
|
| 126 |
+
42638,
|
| 127 |
+
97040,
|
| 128 |
+
93132,
|
| 129 |
+
54921,
|
| 130 |
+
175682,
|
| 131 |
+
69986,
|
| 132 |
+
183977,
|
| 133 |
+
179187,
|
| 134 |
+
169878,
|
| 135 |
+
18717,
|
| 136 |
+
159680,
|
| 137 |
+
166455,
|
| 138 |
+
44862,
|
| 139 |
+
140021,
|
| 140 |
+
191136,
|
| 141 |
+
64175,
|
| 142 |
+
42834,
|
| 143 |
+
121178,
|
| 144 |
+
99471,
|
| 145 |
+
70765,
|
| 146 |
+
167772,
|
| 147 |
+
180397,
|
| 148 |
+
146001,
|
| 149 |
+
57570,
|
| 150 |
+
179467,
|
| 151 |
+
85008,
|
| 152 |
+
201408,
|
| 153 |
+
203423,
|
| 154 |
+
14663,
|
| 155 |
+
60043,
|
| 156 |
+
8414,
|
| 157 |
+
82694,
|
| 158 |
+
105162,
|
| 159 |
+
70186,
|
| 160 |
+
17350,
|
| 161 |
+
55307,
|
| 162 |
+
148682,
|
| 163 |
+
188196,
|
| 164 |
+
82490,
|
| 165 |
+
55738,
|
| 166 |
+
171819,
|
| 167 |
+
130870,
|
| 168 |
+
103712,
|
| 169 |
+
168519,
|
| 170 |
+
120285,
|
| 171 |
+
37452,
|
| 172 |
+
69436,
|
| 173 |
+
36603,
|
| 174 |
+
64651,
|
| 175 |
+
195294,
|
| 176 |
+
147159,
|
| 177 |
+
141289,
|
| 178 |
+
68876,
|
| 179 |
+
195825,
|
| 180 |
+
153245,
|
| 181 |
+
112311,
|
| 182 |
+
152969,
|
| 183 |
+
104700,
|
| 184 |
+
94895,
|
| 185 |
+
57493,
|
| 186 |
+
36262,
|
| 187 |
+
133569,
|
| 188 |
+
129372,
|
| 189 |
+
23831,
|
| 190 |
+
198123,
|
| 191 |
+
12351,
|
| 192 |
+
28743,
|
| 193 |
+
40066,
|
| 194 |
+
164481,
|
| 195 |
+
41938,
|
| 196 |
+
207638,
|
| 197 |
+
178384,
|
| 198 |
+
110666,
|
| 199 |
+
156345,
|
| 200 |
+
16653,
|
| 201 |
+
100864,
|
| 202 |
+
100039,
|
| 203 |
+
156208,
|
| 204 |
+
122696,
|
| 205 |
+
138704,
|
| 206 |
+
65906,
|
| 207 |
+
145024,
|
| 208 |
+
3009,
|
| 209 |
+
178332,
|
| 210 |
+
188932,
|
| 211 |
+
30029,
|
| 212 |
+
178706,
|
| 213 |
+
140763,
|
| 214 |
+
196838,
|
| 215 |
+
69946,
|
| 216 |
+
201483,
|
| 217 |
+
168024,
|
| 218 |
+
89174,
|
| 219 |
+
29242,
|
| 220 |
+
76939,
|
| 221 |
+
113971,
|
| 222 |
+
41460,
|
| 223 |
+
118940,
|
| 224 |
+
850,
|
| 225 |
+
189292,
|
| 226 |
+
188659,
|
| 227 |
+
69045,
|
| 228 |
+
131225,
|
| 229 |
+
199743,
|
| 230 |
+
46832,
|
| 231 |
+
133085,
|
| 232 |
+
27894,
|
| 233 |
+
163918,
|
| 234 |
+
78235,
|
| 235 |
+
167496,
|
| 236 |
+
133080,
|
| 237 |
+
159637,
|
| 238 |
+
52143,
|
| 239 |
+
40065,
|
| 240 |
+
98019,
|
| 241 |
+
199887,
|
| 242 |
+
42349,
|
| 243 |
+
141394,
|
| 244 |
+
204112,
|
| 245 |
+
139029,
|
| 246 |
+
149,
|
| 247 |
+
157009,
|
| 248 |
+
84975,
|
| 249 |
+
128085,
|
| 250 |
+
5105,
|
| 251 |
+
29325,
|
| 252 |
+
95153,
|
| 253 |
+
80612,
|
| 254 |
+
62770,
|
| 255 |
+
15184,
|
| 256 |
+
63143,
|
| 257 |
+
148729,
|
| 258 |
+
20645,
|
| 259 |
+
22453,
|
| 260 |
+
191865,
|
| 261 |
+
127399,
|
| 262 |
+
18143,
|
| 263 |
+
199387,
|
| 264 |
+
139645,
|
| 265 |
+
200758,
|
| 266 |
+
32967,
|
| 267 |
+
33657,
|
| 268 |
+
172949,
|
| 269 |
+
124592,
|
| 270 |
+
144127,
|
| 271 |
+
43287,
|
| 272 |
+
69483,
|
| 273 |
+
138326,
|
| 274 |
+
159014,
|
| 275 |
+
110923,
|
| 276 |
+
55521,
|
| 277 |
+
141373,
|
| 278 |
+
197988,
|
| 279 |
+
191347,
|
| 280 |
+
180844,
|
| 281 |
+
52730,
|
| 282 |
+
186895,
|
| 283 |
+
81714,
|
| 284 |
+
104593,
|
| 285 |
+
176078,
|
| 286 |
+
170361,
|
| 287 |
+
97889,
|
| 288 |
+
114845,
|
| 289 |
+
135679,
|
| 290 |
+
118354,
|
| 291 |
+
31720,
|
| 292 |
+
64986,
|
| 293 |
+
58903,
|
| 294 |
+
16784,
|
| 295 |
+
88627,
|
| 296 |
+
5514,
|
| 297 |
+
154221,
|
| 298 |
+
145207,
|
| 299 |
+
60323,
|
| 300 |
+
154256,
|
| 301 |
+
57728,
|
| 302 |
+
1885,
|
| 303 |
+
18610,
|
| 304 |
+
185556,
|
| 305 |
+
165439,
|
| 306 |
+
15433,
|
| 307 |
+
60015,
|
| 308 |
+
17668,
|
| 309 |
+
8234,
|
| 310 |
+
86619,
|
| 311 |
+
18574,
|
| 312 |
+
134782,
|
| 313 |
+
62391,
|
| 314 |
+
73001,
|
| 315 |
+
175368,
|
| 316 |
+
127248,
|
| 317 |
+
56160,
|
| 318 |
+
141356,
|
| 319 |
+
34684,
|
| 320 |
+
189622,
|
| 321 |
+
149695,
|
| 322 |
+
151050,
|
| 323 |
+
123907,
|
| 324 |
+
63700,
|
| 325 |
+
205683,
|
| 326 |
+
123987,
|
| 327 |
+
106708,
|
| 328 |
+
49914,
|
| 329 |
+
24726,
|
| 330 |
+
25409,
|
| 331 |
+
172748,
|
| 332 |
+
112997,
|
| 333 |
+
92876,
|
| 334 |
+
111038,
|
| 335 |
+
107767,
|
| 336 |
+
122427,
|
| 337 |
+
191122,
|
| 338 |
+
14200,
|
| 339 |
+
176518,
|
| 340 |
+
171299,
|
| 341 |
+
169392,
|
| 342 |
+
25799,
|
| 343 |
+
15889,
|
| 344 |
+
105544,
|
| 345 |
+
190896,
|
| 346 |
+
88946,
|
| 347 |
+
28644,
|
| 348 |
+
65183,
|
| 349 |
+
50224,
|
| 350 |
+
49862,
|
| 351 |
+
140584,
|
| 352 |
+
117601,
|
| 353 |
+
36747,
|
| 354 |
+
110593,
|
| 355 |
+
48100,
|
| 356 |
+
73018,
|
| 357 |
+
121275,
|
| 358 |
+
65485,
|
| 359 |
+
19761,
|
| 360 |
+
116164,
|
| 361 |
+
144264,
|
| 362 |
+
25666,
|
| 363 |
+
13261,
|
| 364 |
+
170955,
|
| 365 |
+
141711,
|
| 366 |
+
3868,
|
| 367 |
+
24448,
|
| 368 |
+
197542,
|
| 369 |
+
61965,
|
| 370 |
+
43597,
|
| 371 |
+
106539,
|
| 372 |
+
127307,
|
| 373 |
+
126185,
|
| 374 |
+
56032,
|
| 375 |
+
105130,
|
| 376 |
+
15370,
|
| 377 |
+
43158,
|
| 378 |
+
99345,
|
| 379 |
+
565,
|
| 380 |
+
102346,
|
| 381 |
+
69521,
|
| 382 |
+
205539,
|
| 383 |
+
205816,
|
| 384 |
+
119277,
|
| 385 |
+
74776,
|
| 386 |
+
110888,
|
| 387 |
+
182607,
|
| 388 |
+
191497,
|
| 389 |
+
205353,
|
| 390 |
+
145691,
|
| 391 |
+
173505,
|
| 392 |
+
188326,
|
| 393 |
+
127577,
|
| 394 |
+
40579,
|
| 395 |
+
49780,
|
| 396 |
+
77780,
|
| 397 |
+
57068,
|
| 398 |
+
15331,
|
| 399 |
+
151828,
|
| 400 |
+
192869,
|
| 401 |
+
142133,
|
| 402 |
+
15979,
|
| 403 |
+
196077,
|
| 404 |
+
82209,
|
| 405 |
+
14985,
|
| 406 |
+
13144,
|
| 407 |
+
153138,
|
| 408 |
+
124987,
|
| 409 |
+
131819,
|
| 410 |
+
139231,
|
| 411 |
+
41270,
|
| 412 |
+
14910,
|
| 413 |
+
133124,
|
| 414 |
+
21000,
|
| 415 |
+
48712,
|
| 416 |
+
17962,
|
| 417 |
+
155984,
|
| 418 |
+
17815,
|
| 419 |
+
177002,
|
| 420 |
+
61657,
|
| 421 |
+
105847,
|
| 422 |
+
31427,
|
| 423 |
+
149336,
|
| 424 |
+
64543,
|
| 425 |
+
151760,
|
| 426 |
+
155849,
|
| 427 |
+
10418,
|
| 428 |
+
162367,
|
| 429 |
+
21491,
|
| 430 |
+
109897,
|
| 431 |
+
172326,
|
| 432 |
+
153006,
|
| 433 |
+
148170,
|
| 434 |
+
137044,
|
| 435 |
+
82934,
|
| 436 |
+
68358,
|
| 437 |
+
53545,
|
| 438 |
+
175564,
|
| 439 |
+
187745,
|
| 440 |
+
82361,
|
| 441 |
+
62570,
|
| 442 |
+
69629,
|
| 443 |
+
103752,
|
| 444 |
+
34308,
|
| 445 |
+
176079,
|
| 446 |
+
169214,
|
| 447 |
+
78642,
|
| 448 |
+
119858,
|
| 449 |
+
82883,
|
| 450 |
+
197096,
|
| 451 |
+
19016,
|
| 452 |
+
2441,
|
| 453 |
+
120136,
|
| 454 |
+
162833,
|
| 455 |
+
147585,
|
| 456 |
+
26209,
|
| 457 |
+
19204,
|
| 458 |
+
140937,
|
| 459 |
+
55877,
|
| 460 |
+
132614,
|
| 461 |
+
69520,
|
| 462 |
+
34722,
|
| 463 |
+
91490,
|
| 464 |
+
18033,
|
| 465 |
+
64037,
|
| 466 |
+
96869,
|
| 467 |
+
74707,
|
| 468 |
+
41352,
|
| 469 |
+
114867,
|
| 470 |
+
142401,
|
| 471 |
+
184428,
|
| 472 |
+
79303,
|
| 473 |
+
160347,
|
| 474 |
+
171435,
|
| 475 |
+
138658,
|
| 476 |
+
2050,
|
| 477 |
+
175076,
|
| 478 |
+
145385,
|
| 479 |
+
78480,
|
| 480 |
+
173903,
|
| 481 |
+
27154,
|
| 482 |
+
35203,
|
| 483 |
+
69328,
|
| 484 |
+
30258,
|
| 485 |
+
28058,
|
| 486 |
+
194620,
|
| 487 |
+
40749,
|
| 488 |
+
71394,
|
| 489 |
+
73860,
|
| 490 |
+
158552,
|
| 491 |
+
55215,
|
| 492 |
+
188117,
|
| 493 |
+
89884,
|
| 494 |
+
53371,
|
| 495 |
+
180223,
|
| 496 |
+
166261,
|
| 497 |
+
69201,
|
| 498 |
+
132489,
|
| 499 |
+
128065,
|
| 500 |
+
65829,
|
| 501 |
+
13316,
|
| 502 |
+
24195,
|
| 503 |
+
166273,
|
| 504 |
+
111037,
|
| 505 |
+
72530,
|
| 506 |
+
11557,
|
| 507 |
+
929,
|
| 508 |
+
87439,
|
| 509 |
+
202144,
|
| 510 |
+
34293,
|
| 511 |
+
167015,
|
| 512 |
+
68670,
|
| 513 |
+
42357,
|
| 514 |
+
194309,
|
| 515 |
+
115824,
|
| 516 |
+
144619,
|
| 517 |
+
184986,
|
| 518 |
+
112115,
|
| 519 |
+
147038,
|
| 520 |
+
2534,
|
| 521 |
+
29327,
|
| 522 |
+
19724,
|
| 523 |
+
181146,
|
| 524 |
+
39073,
|
| 525 |
+
143023,
|
| 526 |
+
9444,
|
| 527 |
+
96787,
|
| 528 |
+
152701,
|
| 529 |
+
144841,
|
| 530 |
+
38821,
|
| 531 |
+
112666,
|
| 532 |
+
33409,
|
| 533 |
+
10965,
|
| 534 |
+
80808,
|
| 535 |
+
95591,
|
| 536 |
+
10458,
|
| 537 |
+
93797,
|
| 538 |
+
55070,
|
| 539 |
+
178799,
|
| 540 |
+
65412,
|
| 541 |
+
174832,
|
| 542 |
+
26946,
|
| 543 |
+
92714,
|
| 544 |
+
204502,
|
| 545 |
+
146770,
|
| 546 |
+
106529,
|
| 547 |
+
162702,
|
| 548 |
+
196471,
|
| 549 |
+
40515,
|
| 550 |
+
62059,
|
| 551 |
+
42598,
|
| 552 |
+
46413,
|
| 553 |
+
108080,
|
| 554 |
+
6497
|
| 555 |
+
],
|
| 556 |
+
"sample_prompt_ids": [
|
| 557 |
+
"3b231c9ce21c9c7cbdf095ce42eda84ff3c05c9000e5da42f8bfcc1080388ca7",
|
| 558 |
+
"8435894d92d1643dc3367bdcb4aa2574c8ed4c153233f6d625028eada7cedb06",
|
| 559 |
+
"d1f0ebf77b98794e415d77421a366b2dd432546701acdc3146669497562a6eff",
|
| 560 |
+
"55beaf2fb762b07ec9fd3861a18a82be00d29ededa05ace0aa3456ec71f10cef",
|
| 561 |
+
"b2acccac34c624c5ce1ff88449741d77a83d47a3e2d78d2381be24e3ecc61b8b",
|
| 562 |
+
"b225002b639bb026c8dc8936178fa3b23cdab02a8fd7b5a870c9ef3143d1b138",
|
| 563 |
+
"3891bfb1d616c2b3b22f388f1cc5e8d0833703531d603632d1cf297d65f077c2",
|
| 564 |
+
"dd74093aadd7ad1a349a7554e0c49b9a69ce6659c19ce9e6cbdbe19381b2089f",
|
| 565 |
+
"20553ed878929adea682516496a36b665a7e6c6e9136bb4702f50ec75c46a23a",
|
| 566 |
+
"b5ca804253807b96c368b5af4df5740ff2782b2c44643a1a3813b4f70a56ce03",
|
| 567 |
+
"0e17fde64b053048ed468309663e6ff6319d85dd02a029b9b0814a95fe79fd24",
|
| 568 |
+
"4384b506bdd3c3345bc864279c946fc7d86397adc0e6130ae89d425461e5b030",
|
| 569 |
+
"aca3950e665796da2937d4e1949c1c7b008cb8a7feb5125a4b561a379b34b7aa",
|
| 570 |
+
"d0b0644e60d867f4b11b5e28805596c6967a0568b9602a06fcf04c44b0834a2d",
|
| 571 |
+
"a9207ad5b09ee948dc22332b2cedc5a7b9abaf0a09b4897a42bcfbb3394d7eb1",
|
| 572 |
+
"ee667f6f50c4cd6efe87e999fb37786d32ecbf874a37305b60b2c067e32dcddd",
|
| 573 |
+
"69706549bd0566ce93d482016449b3f46fedf9a4a334dbc7c0b73fe92d574d78",
|
| 574 |
+
"b64dd8f46d60c82cf7af198e6ec9d2be17e5646671db866809b60d41d7f2e587",
|
| 575 |
+
"87a51531aca4ec4dc49ebf8ace1d3d33d6831430688b24a0a5cb255d27e1fd19",
|
| 576 |
+
"93093c453005fe64b6554ef7628846970e959b08ee3c0b2b1284b859743fead0",
|
| 577 |
+
"f385df90823211763fe2c9162f80ea3481d8cf1fe5f3ed9c70623fceda33688e",
|
| 578 |
+
"86f90a033e7189a3201d3abfbf2ad40a5fb1c31aa9b33cf8f7f1bb3cb6e8cab8",
|
| 579 |
+
"c6b923fa4aec3295ae8fa1b389fc2d1920408a40b1115da808ebf7b4e7f0217c",
|
| 580 |
+
"c470812c2aeea7d7b45476ca32b4542b6704b95f9a70fca745417ee6c215461e",
|
| 581 |
+
"d153af4dba3fd44d4792a2b4460b1fcf88d8acbc6326ccfd9a8b1ee4bdd16474",
|
| 582 |
+
"61f58fefee23aa549b8cc4903d034c60d493fe0c2c1d17435aff1b3c0a9356c2",
|
| 583 |
+
"d2bbd2455d02c58552e55a544eb9eaeec146ddfcb5777653ac3e81c042a70fe1",
|
| 584 |
+
"ce30231c02c6497e31729e35f1e4cba9d153da6c494ef109b0fe6faa4fbbab1f",
|
| 585 |
+
"44ff734cf9bbc072a06ffcffb250d26504dcea4e1644822decfc8d0e75be6b91",
|
| 586 |
+
"5f6accdb8f7425ad6b5ff307f431bcd592093749e4430a69132077cfcff55c86",
|
| 587 |
+
"994f7ae8a2c48283157aaa48ec0e94d4882b830c06bc6eaf24f218857e04f155",
|
| 588 |
+
"05eab8512ba0ce88093c15b38baf4623b2b9d422282dfbaa7b61107ae9371e36",
|
| 589 |
+
"031aa3882fdbd8fc13988d3cd8aaad2cef5853a7721bbeb0f77b4cb6430543ae",
|
| 590 |
+
"8a1c7c2db912fd5037414f3b9a686a3979e6ffbe79aeb115b84fcfc6a6e5cd5f",
|
| 591 |
+
"10c9e718308a38a702e9cee09c901a115dcc6c77c37de5d807eff306a5ccabec",
|
| 592 |
+
"017f398d6d97c52ff8c88d1a59bebeb6552fce96383faba33a6028eb240a61f2",
|
| 593 |
+
"011dc221af013e38fc8c0b1378241cbc41b9d46866587bc3a5f752e0671577d1",
|
| 594 |
+
"6c81d88e92ebfa019c26172af0d645b9d2dc34dffe34c0beeeae2ef347553767",
|
| 595 |
+
"b7a0c14b467162ff640b47a42898186a94633fc87f0443cd85eeb429a8affbc1",
|
| 596 |
+
"8180255ad3c1572487fe7cbb16fce9e052f886e5ac56256a56e95542f2e2a758",
|
| 597 |
+
"8cf0ff1117e1674a843fa3cfe7a79d6b753b7e70451e83a90ef91a901a0bae4d",
|
| 598 |
+
"b91f1317ca0a23fb50f335362e91643ea3530fa86cd05caf9f360c4d6d253260",
|
| 599 |
+
"dea9dbafd940be4a95e273ac382c677ce2ad93b60f29df3974e1f31776b1ecb7",
|
| 600 |
+
"b20be72c8053378c91c6c6d0b42cfab98c35fc37f98d47f35d3c65cd73bf673f",
|
| 601 |
+
"495ff066df0cf422be51619a8304feca285cb09e0a961ec2f07629e2bc22db24",
|
| 602 |
+
"355b720d002262a88b756ba0038e420742f1ae7c88e5ce79ecb44e6de6f09511",
|
| 603 |
+
"7889bafb74d28f9fedc706f0f3bb860a1df36b3cb4fad68f40baae75711db87c",
|
| 604 |
+
"36afceacb2e2ba3371fb9bbf8a0536259207cfe9956d90f16dc889912cb68e76",
|
| 605 |
+
"264f337ea8eb4fc95cbd6b2f8f001c1c841a588e57aeb5ea5b70cceea067fba4",
|
| 606 |
+
"18de5c0c9056000159ea74881c90e1289a7b3ff5bd0091e958225787f861dbdf",
|
| 607 |
+
"8f4ddf1144b8d58a7fbf967f163c6c4e8aa455573f53726294ca831cf249c3fd",
|
| 608 |
+
"63896cf642f886fd26f9415b6614708463882965f962f6fe09b03f3e27538904",
|
| 609 |
+
"4232092ba8e89af57b7ac0c6d7bcb2f89ae469e6f43b9976b5ee0379df262396",
|
| 610 |
+
"227e7903aa0db6f33c2906d36c9241f53191744ce1b21adda2aed1883e8aa5b6",
|
| 611 |
+
"d6d04613e41990f1e2a922ee973bc8047a434c9670fe4749495b77093cf66c36",
|
| 612 |
+
"0597420931e6ba9a4e7009623c8eacfd4e8a86e19ab30bd3932a588c3b17df39",
|
| 613 |
+
"2bd85a0cc149eefc5c1d18581b7d32902a334fcff70a3a5275130889b6c11f66",
|
| 614 |
+
"39dff85e9f367995ed7d249e119c1978c268c27633dfe635d266cd3f5e6e7923",
|
| 615 |
+
"487421c6f7159db1b429649a10d233f258b306b25cbac69363dc81da6ec91d00",
|
| 616 |
+
"a487e092200d50ea681f7af186ba8e476ba3e7885236601398dc3496f6fa6db1",
|
| 617 |
+
"903bf7373f507c0234f2c27e8782baadecb9e795aee7494dfbbb28048bcd45ff",
|
| 618 |
+
"f4984b87b66b94afc5f4bbb128c23d2450410db7473a2d431c884edeb7c3f61b",
|
| 619 |
+
"6918dee7fa7cdcb9356cac90e76a64677accf186cf9dcc9681642f41c6a69247",
|
| 620 |
+
"6277dfc49cf56fd0e730d31be88db1e1d6a4f4fcb9a4fd054083654361e1ffce",
|
| 621 |
+
"7b64dd0daa79a522341e75f10dbfd34afcfcede3c6e1c066bbd26778b00addb5",
|
| 622 |
+
"c5bb5b57fd72e686397c9f57f6ab613be43a01659b1b10334c3fe8137ca028e5",
|
| 623 |
+
"18f365e144ddebeeec63d0630fe57345a0b18ef77f9e3856aa2634053b8cf8da",
|
| 624 |
+
"304b2b5b51c175fff01814b223aff6ba8a4a1af1499d299458ff81da29629ecb",
|
| 625 |
+
"42a659dea56a7a7c9e526727afdae0432a40ab4feb5d8f40044410a4b323cb70",
|
| 626 |
+
"1be16211f01fab5b000a1a0a77fda9111e12cf1901a3a884a935caa3c704052b",
|
| 627 |
+
"4bbf2b0e7eaefa42e24cf9ecf50eecc80edcaf9562084c5580c6949869d409a0",
|
| 628 |
+
"09e57b7dbeb2431ae9dda28fdb1553ce3df43ed29e741058b073cf1af09999ad",
|
| 629 |
+
"102517e5bbc6c0feba32c1d28036192699c0ab650c9efc48a31861ad123fca76",
|
| 630 |
+
"ede3edfe0364652d7f906d05997717dba7657b255bee0eb1cb47aa5032a25d71",
|
| 631 |
+
"c0b491a96a0f8c5bfc7a62c92e15b430d21d63b45598508d4b7f1dc6246ea280",
|
| 632 |
+
"24003e66ccbfdceb0f1089c511f3a6b2abed1f81e3836809d1496e54e31b82a7",
|
| 633 |
+
"b444bda87da4c1d929d658d1d483d54cc70831910a9b8e9f1512457c87aa006d",
|
| 634 |
+
"d5261bd6cdb9d964ae4fb564518d7177f806a27e2716ccaf5f29bd7ea8d43474",
|
| 635 |
+
"07d9f3d57c0ad16a1abf346e424c18985bad48bcc0dae2279d316d4fde01db24",
|
| 636 |
+
"bac4eec9705a42543f663b58b182cf1836665f345682c91099247ff37d4d8259",
|
| 637 |
+
"9d1355ac63cd0badfab382f9b66c519ecc69419601f638ce9f704515cf078759",
|
| 638 |
+
"b8423b8368a503c0da6c211e13b2143a98b4fd7f9e1996d5b3b29cdd416d444e",
|
| 639 |
+
"0c3edea4d85b40002f6df0494d891e5d3960c02c44a74a5933473c9072a24d84",
|
| 640 |
+
"ca9dcc8c50c3a684c4e93d82ff8b3763b29c3db802a7a2b344d568fe59404cfc",
|
| 641 |
+
"0e1748412355b8721f6bfd536c3b1abf51e86232f48853166f422a3d06dfcebd",
|
| 642 |
+
"35218109de423f9022a19d09fb7c41aac7e35cc47994e914acba9a0db41f05ec",
|
| 643 |
+
"9e1d3835236eed7f67dec291900d87960afe7c9e98dade9a1b15deb974e0211c",
|
| 644 |
+
"5ee602a36590d41f12936b1784bea5740e82c044170a34b7fd82383393c8e637",
|
| 645 |
+
"3fa2797b91764ecbf3082b26a5dd7d3ce16487f7bbb56c1dd3c1e3a502be7a84",
|
| 646 |
+
"fa85002238e36a2cbb271a439ebbaa682b0c22f156bce9c5ae6728db6cc0e6ff",
|
| 647 |
+
"a10fb2f6f8836fb52ce0b08c05654de30031eef5f4fae135b33b94ddd9683db3",
|
| 648 |
+
"c94f12e9f88eba9ae61ce5baa5b91b30cfc23508a4f76fc14f9d0e6864dc7703",
|
| 649 |
+
"52a8495974981aa7d035997acf3aa9186b6f7114f7f51225add8f3f01587839d",
|
| 650 |
+
"c869b4843339c985297b6303ce1d5d1e2587a4805ba7b7d0176b15a4805d1d66",
|
| 651 |
+
"babcb877c113ae3cc7ce052217310620e64fd1ba9f70759a1f1e9dddc76a38fb",
|
| 652 |
+
"1175693d5a383475916d00b4e67221a94827d10421247e251e8642526f33ad24",
|
| 653 |
+
"c36fbf017b546830ddda3296f489e50703e2b847097d6869dfa7dc09cef9e0ad",
|
| 654 |
+
"312a2f7a77821104a9ac2742888f26cf613ca34c67c87cf14d18c9632232e7a2",
|
| 655 |
+
"5fb03ce79fdb0cd9acada965c93e11f592d4a49dba803f103d8045012d81e018",
|
| 656 |
+
"3929c32a3e181e3c1c5cc8c1385ab7967d12acd8961ff1e5c15924285af2572c",
|
| 657 |
+
"6222208e8056248d76b6eb0828774fc8453d9e930b33304c2894fb3fb3b783e0",
|
| 658 |
+
"d6bf41e86a3ae13fe37b22c02348b222dda06bb7c1843be4a39fcd6baefa6b0c",
|
| 659 |
+
"713efb4db7e753985dcb10238517c7ef58abcac7d2bc58525480ccb2ecf98dca",
|
| 660 |
+
"2c3620279e8e938f56dffccad98b97309bb912a82e771c49cf83222730a66ebb",
|
| 661 |
+
"54b96a925aca1e1088112623c4430689b6759eb36a0ccc4947fdd9c80adb6c22",
|
| 662 |
+
"41d8cb3d5f76a8313b8ec326ac52a65568d1d3690cfdad2680b9441aefd4a75c",
|
| 663 |
+
"fcf08c8befcaf61b62d05f9ec7e4fd1fb35ba84b99f5bd47e98b090aaadfb47b",
|
| 664 |
+
"919429fc64861357aca22f4e777c0e1681fe7f30ad2e1b4f2df531519dd73c69",
|
| 665 |
+
"255fd701d7b67751a6d60c30dd2be2c8f90e95548d1d2055ff2e85b18916b208",
|
| 666 |
+
"3784a80e0b8a9c6c1c9f30296f1d9ba7185985f30fe7b518d9233acf5047e23e",
|
| 667 |
+
"05b7c274868c699c24e224d7dffb744444462171679ead406f7e22cdb3473b45",
|
| 668 |
+
"12d9f48203b8166328941e839b977e1e96b5d8e4dd97851fb5f66cf8e206cb46",
|
| 669 |
+
"881810088feae54fd2d973274dce67c5beb7ecca905be5e95c99cb3d9d830fcd",
|
| 670 |
+
"c80ae90ba4942d87da6e94cf7ee0caaa67044453b45e4cebed33804cc4e83d0e",
|
| 671 |
+
"98af8412983c0b94e41722c580bf8ce90473d9842ae023e505f7a1d8427d5176",
|
| 672 |
+
"0d315ec6bc42579b21e91e5f77e78b9d1095385fe5f7e493da8df47e820af0cd",
|
| 673 |
+
"9ddc7ac86c1329f7221d03f6e8aec1c90f23e163d8448158da81b084a89ceb98",
|
| 674 |
+
"4098cba9b89f557b35480cf9531a3bf46296c3683bc052593e3ff107fd3cf73d",
|
| 675 |
+
"b5e0cab4248682b46675a23fa4e04336a29a44447fdf4e60a684a2b24b33948e",
|
| 676 |
+
"e02d8907b26c30c76ff2879853dba5756f6c1d97adfc83241e879730b78144fd",
|
| 677 |
+
"cd6ccb0b27f5f1f508f7fe604c0f8d5f32e3c31f40707e1ed5f576ffe60121be",
|
| 678 |
+
"fb8f053c586283eddfbbe09d3d3d94ab792e89d29530c74874a720c370d59786",
|
| 679 |
+
"fe122fc14df152523c8a56343960daee849c2d8185691969db2e874ebf3d9895",
|
| 680 |
+
"0b89ee0e83a2f68631c3a883421f321c107c5dcfdb48752af0225533fbff494f",
|
| 681 |
+
"9feaddd99a61ba94945648be7446f52330ddcf242925ba84419026bb760e22a8",
|
| 682 |
+
"9d0b7376737ea00316fcb6b5244514da82923f660a1d09f6c294af7c386b6876",
|
| 683 |
+
"38380b658f851d508979075f2671dd5820eaa53f04df45a5dcadc34709eac308",
|
| 684 |
+
"64e29b67970d114d495065a695587dfd26aae1acf7980e21a404fc7328a81222",
|
| 685 |
+
"3abb756250af16416d51aa30b79f5a5cff47143f74c1b5f8b763351cfc43158d",
|
| 686 |
+
"384fb4babf54594bca54aadaf629df346fa72d149070e0cfb8c8faa337c281c2",
|
| 687 |
+
"84f4e4acf0b1841925ea00ad64fad581c8ec37979433b7276687c86b028f6ca1",
|
| 688 |
+
"ee03bd57af07341dab88ed5822f73e19b8345354d29fd6adebeecae2a940c243",
|
| 689 |
+
"a3f4eabe2b57a5b9adaec2d11ab8fdad1d20564d462366a58ec2b2d1b300fc86",
|
| 690 |
+
"ae41dfe1e7d54146c11eceefcd062b0a941b14002cffd3c14d8391cbef2a60ce",
|
| 691 |
+
"733ad6e87e4862eacf03c83cbf38c3982b8d1d7f089cf54ad1dfc66811d9c3ac",
|
| 692 |
+
"fd8652b034e7c84e92bbb9dbdcfc8ac2d31deced6490b33983a6b11c59812341",
|
| 693 |
+
"47a7000f72c5d5c895abd9f51eb493a1cc7d7f360bd6240008b20036df4dc48c",
|
| 694 |
+
"9b2fef9c3aa4972ac0a08fccabf81a6f1eb5d658748031ba625716e80944a0e7",
|
| 695 |
+
"3cb12789a7b9b0d2b24c4ea8cd01094843c263b042b517c947a63ac39f2e739b",
|
| 696 |
+
"e922b1bd2d44434b19efa32dbd3004d1e204ca1e004cd1b89ebbfd3e517b9170",
|
| 697 |
+
"04e4d61a3d0dfa6aa7c4e522a79905e406796d29fc91c1cfb63a45ea0cf1493e",
|
| 698 |
+
"8d3c93254474c69454a271972e91c13e49ce4aedf81c30cee1c9edf4ac0f2b1c",
|
| 699 |
+
"7c063b24d460297e7f6c07d301aae70529d8c8b8d250475153308c1cad2a0ac0",
|
| 700 |
+
"7393bbee369df93d4e5c3e584fa112c3727a3ee607fc111a8e96eb94af277cdd",
|
| 701 |
+
"811af0a09d370a63fd15f1014f660677283d327ddb28c06e949c87baa05fd607",
|
| 702 |
+
"9d7a7d956d3effc5e318ec8823ce99fbf7e09ef251a6b19a0314550a9dfa7109",
|
| 703 |
+
"b0f96e41083413840594ba34f8243c404b44094d00e369ac1e355b52ee15d697",
|
| 704 |
+
"9cdd0fd9c302db7450159c0b73712e183feea933abd4d46c7f8a20dd502342a4",
|
| 705 |
+
"54fe0fc6eda2ace6633c182b381df624ea60395e8a6ec21855d3df2774d3b09d",
|
| 706 |
+
"872d97436a8b5b008ab85ee61b4d722b92102cb78cd8089ed21de4e261cf789c",
|
| 707 |
+
"ab073929e1cc031de0fb85e872f805ecd4ce0d6ebe66b0a269ff05fd26b9d37f",
|
| 708 |
+
"d5804e590c895f621703a54f1432844abdbfc5e86e7f7ef10514fd676e8c4db9",
|
| 709 |
+
"5cdd8b62ca319fccd2138cf926fbce423420efa7001ae10640b279ab8c3e3d35",
|
| 710 |
+
"455849f34d3873a4b39bb580c4acb9da9f09cc0a31325d695840624c60c42572",
|
| 711 |
+
"0966787122afc82e83d0d9e192ce4453e4bf9898bec084128a202dcd14644ab1",
|
| 712 |
+
"c617e5aed8d27f742be69d9ea1166e270e7b8bc7cbae66753c28e30cfb5be426",
|
| 713 |
+
"236ab91ed27e38174776c3e2c38df6ec4977b0e3422d2e5013e3fa00e6a648ba",
|
| 714 |
+
"dd75115ec8a1ed803209cab7f055ca96674156e700f87eda9332ed02e40e4997",
|
| 715 |
+
"54f39bb695a63277d0ea640780f63199c411610243bb968b10fac64d34db12e9",
|
| 716 |
+
"d9cdfaa8615c5367040c2596f5ec1532b0b66a375a4718df4fe79710fbb15bf0",
|
| 717 |
+
"df2a95059321d68f931f84f02663ad734c2b00e7aba10b210e9cebd9c12deecb",
|
| 718 |
+
"45d10b369f69d3af5ad1afa194d6310ce241258e8189d17bb128692f01405541",
|
| 719 |
+
"efe2aaca9937a43030289ed4085a8a94ed538ee4ce16649fdf731bf5a1de5d9e",
|
| 720 |
+
"190349a4f5729faf2212f9bdeb9e7cb6af8b4469a25e1bdd26ff0ccbc65cf8a9",
|
| 721 |
+
"626ae1f358751263ff17cae60d6d2be647969b04aa162fcdbdfd582f5e2d443d",
|
| 722 |
+
"b8b94b5b4921794f391057f017aa10ef1f0c2ecdf52bbcd73659531a03a52bae",
|
| 723 |
+
"be51f0a7eb7d4f70c87b42bc94035ff150eb7214b424af824560971e5cc52a09",
|
| 724 |
+
"12d11f5e79e4cf728d91e1de0c7270de126bcada8b0a8b7c77ca5bb8ba4bd9fd",
|
| 725 |
+
"3c77e3c0c57b479f356e1ac2f3864485aede7e620c3c78788ccafbee18932751",
|
| 726 |
+
"fc6c33f2733339b10f1ad56e6c53153072eb1bfaa7be7f252dac3eec22936b01",
|
| 727 |
+
"e6e1d618076051a1c9f46059362bba46572805423ac0a5385e85b49de936af83",
|
| 728 |
+
"5242fcdc9ba65a46416b5dd03f53235d362a6de67661a30798f3db247d1d9fa4",
|
| 729 |
+
"73cd4bfe782bc493b6ef25d9bc77bf58179297bbc9d6c050cf0a2dee05fc939d",
|
| 730 |
+
"8bc3d59209629baa68b88437c6fc876004ba1b3af13673fc3ce37fed81123f13",
|
| 731 |
+
"1adfcb94805931703b2079a286fca1698a3862d37a4ce8e5e336404188e3992a",
|
| 732 |
+
"896653f8b95b3655b950da302eeb74c80d280ccc394bdb10850f0b62244ef3b3",
|
| 733 |
+
"1b924b53bc1cdbb029be38550dae9222628c6c82d8a4dd64a270fbfcad7065a7",
|
| 734 |
+
"a0d3962209b44136310bbddb6ebbb84e909bc869eec53c82dcc04eb862d0a44a",
|
| 735 |
+
"0f3144a63e75b73be93341ee0e81f8110ac99d20cda885f15ad518455f65a76b",
|
| 736 |
+
"ad07fd2a3ab146f20f7ca187fc195f7f0bb3f85ad728e7ad0255407f5e406231",
|
| 737 |
+
"cb5bc4963f3e796cb3d1c652724471bfffd0a7b32aed0180f6b2a9088178e8fd",
|
| 738 |
+
"43991752f48def329aa85b7716a489f00a37c616f72358f8f6ef46063132101e",
|
| 739 |
+
"275de2fc62902fb584f514f767196b2907248456280150c1c83266e8577be785",
|
| 740 |
+
"a3bc93b42be9b9ad91c7f4053e4535482a23e131fdce5b10a5d43493bc927673",
|
| 741 |
+
"b22b4faa683cc18ed2a9d0e5b453722356c0e67ee73efc5df1382d4406a1daa3",
|
| 742 |
+
"b2dccb4ab25337ad74fbf707423e3aaabe9a38ca681e3b3f9489a4beac148d00",
|
| 743 |
+
"00a75f67d7069ef3814157e68f9f9b411260aea9875be1eac36061c8244da1e4",
|
| 744 |
+
"01335a0c8c71020cdf74fa524c2c2f1c7f8d9a29d7d90e71fa07d939de59a5cb",
|
| 745 |
+
"e4d3b86df606966a13e8e6fc0477ee4ac83b686eec77d8573ce3157a684d421a",
|
| 746 |
+
"d9b868e179b38a7ddbe096eae3a72eee7e0385213d3130d9caa0fc1acc20c7c6",
|
| 747 |
+
"4c538b03edb481bde831b2434791a9f1fc82d7bcf9a70cd793889ab9506898a9",
|
| 748 |
+
"a7834eaa5d1f326611c3de1f06a1bca95cd89713232fc00cfde4ccc38d57a14e",
|
| 749 |
+
"1bab7b8973042868ee50f4540a554dac22e9dcb90c6cb0af8dbf20e3dd96e5e2",
|
| 750 |
+
"07eb53e9516ad9a84237a15b6f2839e35bb24649e1b183a21f3eb8ce103aa242",
|
| 751 |
+
"bcb3aaa6748f9b6d8eaee5becbd7f606cffc9130cbc1871c831794aaf625c517",
|
| 752 |
+
"b5feb6ba862da1f1977ce9a47a3401b8e4190ceae34461747a15a88090518bfa",
|
| 753 |
+
"8fdf2ca351ada7c683d4bc711ce5f5a2e8693b70fdcef6a48e237a7b53e0b3a8",
|
| 754 |
+
"8282b33fc9a27761c68be5fcce06c051085ead5eafadcb460595111c9c99b968",
|
| 755 |
+
"056d9b054e8685c6d4fcea92681ae32471ef6c7cdae59fa65a1ad79990baaec2",
|
| 756 |
+
"4b97cd6188ef1b5c497821a2dc668bd0a2a44f0f08536898b3dd78670a53c863",
|
| 757 |
+
"243b9b01762766e758a201f087299c9a7ce8a494d9c3035a481494c8193e5ad6",
|
| 758 |
+
"c0d85fc30b1d5fbf82f42766a5be07c543a7ef6f7b0433a22abc58722a17f8ba",
|
| 759 |
+
"6b28a4aa46ad409336357927c5026a441ba64129d337a72725eaf2169e52cf94",
|
| 760 |
+
"eb2a92b84936037e94c409a2b169164ff46ea252d27662b39c40b75a28baa7f7",
|
| 761 |
+
"a6aa305d5e75d0f9cd28257f76116dbe74d998f97bbc437fa5238803eab2e771",
|
| 762 |
+
"8b83abe9a0d683c89b631b36788b83c0dcddd0419595634b33fabe1db0b97d1a",
|
| 763 |
+
"4a739d7842a0cef721b409412ad71e6aa98b2c284a81b89580ee1e03ed6b0a7a",
|
| 764 |
+
"7065299337ce4a27c6e85a9057aa072464429d1c50d5e96ae731e979692f4a39",
|
| 765 |
+
"a09b3f7c6592c1bd8b71d7c03e4735e4c6090132aedd003507c79e20cf9da923",
|
| 766 |
+
"2954a62f550654918e116b7d0cd5e40b65dbbef4e58d9e08157dd856395ca97e",
|
| 767 |
+
"e8e46be02a32303a88f60654595f56382719011756e7b25216658ef038727cd7",
|
| 768 |
+
"3bfb004c1bcba4fed983d59e871924bc7e4205744bd9d4cc9792c71081941e45",
|
| 769 |
+
"8f4c302ca56c583efa1d1405bd4d38bfd13063f0c043ac8a34728ccbee123573",
|
| 770 |
+
"6ad79359a8b688e49fb9a5dee317d3e59de8d5ece30cc0faffbfc249266feb92",
|
| 771 |
+
"1e34e7d65646fca04bc8119981edb503cd2a5baf89024f04d0a6b991baf96d73",
|
| 772 |
+
"be3193e19e9174914adf56e66eefbc990df9a0b4f1175c1bbb0d6e9b98edc78d",
|
| 773 |
+
"d11d323719cce29ba7ee12dfebf585669e7086e239ceaa6f2077571a672f1a71",
|
| 774 |
+
"b13f2c112be428c2b0340a7c366c12f4fc9f8b8d726e7954c55acd116c38813d",
|
| 775 |
+
"d48e713f1ad2262e972b895c37b57e999decaf2daffcb2cf748497518eb0136b",
|
| 776 |
+
"49f67acf6ad8eaeb0f134b13354cb39c60527983bd162a855e7657c06d92a15e",
|
| 777 |
+
"d39a39373538ccc112323834053d97f5e207b20eb58a934516a6f58611fe7b09",
|
| 778 |
+
"d379c0b31e1ecb6ef8ead8eddd12e6686363b105b52eb25b00ea08bced1dff2b",
|
| 779 |
+
"40947b78ad49dad8ca6fbd1bc157b994345ec9902c3fcb8c62d4dd39ee31464d",
|
| 780 |
+
"587182fda52db03fd5b80a5a8e0e850f29602de632ba0391553399387aebf146",
|
| 781 |
+
"1b95aead2c1dd9ab41e914e5ba5480fabdcb959f3838c6d6a4f8aeb352e48fa2",
|
| 782 |
+
"b022a29e20b737567f2e5ec16b0bc24c81d3deb60b9a66a4796ef211fdf399f2",
|
| 783 |
+
"345c51fdafa16130c34806a3a901d081ae22f00d61f0d683eaa0999315bad6ef",
|
| 784 |
+
"bfbcb89268e28e4eb2f3eaa5ca5ac54dacbcce96dbe604ce4268b77966521f96",
|
| 785 |
+
"8b693f6179ae7ba51c2a328f855b07fc4ccbd352bc972f99bdc6088cfe7cd730",
|
| 786 |
+
"0565832e319c2295f282a84f9f7c0cb2fec8e0ba1cce9edd19150fa5341d82ab",
|
| 787 |
+
"1a3e52fa145d5cbab8ce8ac24ee0eabd0d54ba992f4d816895c52a672fc946f4",
|
| 788 |
+
"f6e1d856e22e20660229f9d7393ce1e5b08385e0960655446b9001037fb1e7bb",
|
| 789 |
+
"4d9ada26d0333268c53a7f30ea5fe61012b027c3cb65d0a0f19f58f35688d1d4",
|
| 790 |
+
"cbccdc7430b0ed4166ae37020c758fce793e958e93fc51bb2d6597926054288f",
|
| 791 |
+
"0db04821fabe631c9c36ce391180e09f9186d203cc11cc4237022b4053d864ad",
|
| 792 |
+
"767c6bfa5c180b5a9fe445835febfdaf0e0b743369d5b108dd2849b62aeae5f2",
|
| 793 |
+
"fb59e2e25096c0f04d3f7cc4224fa2ec604103debb26a34146b209f54c276cd1",
|
| 794 |
+
"864c1f70d2cf4f5139f46a33c6d750183912159920bf3a2fa0c33c9a209d3f59",
|
| 795 |
+
"af1d123402c483114984ab4512a7c91dd71aca3af2e24f5efd80e35ac0977226",
|
| 796 |
+
"da749f410ea48fe364b31f9b058362b7010292ddf2606498e2ba484645ed8f98",
|
| 797 |
+
"8d0a7b174085fb44d6644060a9d6336498ad93241fb54ecb86b9202b4084f5cc",
|
| 798 |
+
"72d33b01b89c002cfd5ae1ffdc7b197394517708a4ea462e61055c192ba6bd95",
|
| 799 |
+
"e0e123f4d576147b2df2c1dc2d827c737cbcd82709e32aa70dc5ef9f5e295dff",
|
| 800 |
+
"cd6dda1743e74f4e3a779164347ebd5dadc944ba13c8c234a687bb7b6d8b22e9",
|
| 801 |
+
"9daa73b1afb48fdd1632b3076733d547a59f37fb72d2107a1560e6c7ed8fc751",
|
| 802 |
+
"880e86e164dd729bb3758abf5c49e86726571b7c9e785187d217d3dc7e93cfd9",
|
| 803 |
+
"58b1fbb10ed009eede3d5e30015a997ca13c9eecfb0d8b2ba3dc4ebcb8eabd53",
|
| 804 |
+
"9aeb19291bac3d6c2446ab3ca0c9d2496c7d854d28310f5397c2b5b621717438",
|
| 805 |
+
"c0efeb65384dbb8cd7dbd4dc8d320d58f4f68b3b1284bafe79ceffd8c56b050f",
|
| 806 |
+
"a4c5ab101aa0384c39a7d52f48e231e5b0c0f51ffc4ac3141d3379ce90c45cca",
|
| 807 |
+
"1f4af68687d5c32dfbdd1443eb483b31a5853a1f24319fe93ab2c882a087c824",
|
| 808 |
+
"8aacaf3a947b827497a07470ac4d50fa6eb93745b7415fecc51c720a9b4a7f82",
|
| 809 |
+
"1f66f79b8d9bcbbb488b58264a3b37f2fc526cd6bbe5341f719cab4ad79f0669",
|
| 810 |
+
"b770820e82517ed2c643acdb58c3e18ebc7a2b782c003b55a2a07b5d53f146d1",
|
| 811 |
+
"87f221248c39c3778c4cee73c04bbf4fcdc3051fc481898ae6c45b47f2c5ff81",
|
| 812 |
+
"f865735bb145f13f116f69365f96265fbf3c74af0649898ce4e724e1d9c827ea",
|
| 813 |
+
"1db2b203692778d14af586791bf8fb6143ba128ea7ac00959811bf6abf5a5fe6",
|
| 814 |
+
"f52111d0a7ab75f36fb9298b46b60b16dfa9d028fe556fd6485bfc34f216928c",
|
| 815 |
+
"5d46263353f9f4e1f5e9c75acb1e5b2db3682c84b291f06cb69c00fa36e28d69",
|
| 816 |
+
"dfc99dc35f67ffad0e4c074318e8ff7fd0dca2cdab76ff36e37842d47375f8b4",
|
| 817 |
+
"97637671147bc1cc1fb68aea5e796d208b7052821f6ba11d6d35f419102e016b",
|
| 818 |
+
"63dbcd1ce94de046ab5964d54f55bfe7a107fb6622693fb6f19fc5395c580e36",
|
| 819 |
+
"e2144cfc08a1ab1a701205ad27bdce031bade2c72873710826baab33b8df9c4b",
|
| 820 |
+
"58fa65778005cc9af1ab44e67f124e98ca7aace1659f55625d71b934114f22dc",
|
| 821 |
+
"86c1b1219187c3737796eaa2918d0259b7048ebcf0822d0eb6f4c4523a2aa21a",
|
| 822 |
+
"f6056be65f1f07edd9954f946f65238cefd9a32e97d39ebcde6f2674894971b5",
|
| 823 |
+
"2ead1df49a76a397562ab9576619a7be4f8b0e6e2919c754a8f381e6dbc70bcb",
|
| 824 |
+
"c7085e178166ea7686d92967af66501428fb6947dffa3bab14c0104e01042c00",
|
| 825 |
+
"15b73164a01f5e26b17d636f705749faeaf0bfcb21da4f40430df5a22bf956eb",
|
| 826 |
+
"29b97e60a528906de65afd9244209865a4fd9d3f6176b330cc2474c95b428e1a",
|
| 827 |
+
"345c58c84c4bd502244cee3169209db9fafb7af20e3d9712e67f137fe6d19700",
|
| 828 |
+
"347006590396fa607d5917e6e7fd5232a4d418163a073e60c7b00e146b912af0",
|
| 829 |
+
"4595dbe4aa59cc5175205d4e0656e13f9631d13337aa93c904c56619c30f8fb2",
|
| 830 |
+
"4f5c0c0d396631c6d5a29ec9fca2177d9edfee7cdc0ce1d011f085ea167f7ee5",
|
| 831 |
+
"27c8bcff659afe49b92dff78a6ab0de671962ce20e0621e0659e5639e52514b6",
|
| 832 |
+
"3199cd279a00f9205a70b4e33f1c50e9b42bcee25f96317a0ed1908e733059cf",
|
| 833 |
+
"78bd523e07cbe1d2f2ac78dfddf533ad6ff3fc4a7f76a4d5bdc3bb343aa09efd",
|
| 834 |
+
"bb01d2d594414efae12b2ce89048ab2b9119f5fbd03d51615dee7218d46e7c40",
|
| 835 |
+
"725e8eb7140879a559edbe7b34f1d7080bc141841811414222e5c03e336fe894",
|
| 836 |
+
"00df96f3a96fdfbdf812728eca0826755e3c899e4cafa169eb0db5a682e864b8",
|
| 837 |
+
"af80f11bd4c27afb2ac3e41d7180bbe9c9550565b0833994d6f2d33b91898a3e",
|
| 838 |
+
"f5359d43385fff9a4e3b8914ac6df431ab0c9d16c3802dee532fbab141d9692f",
|
| 839 |
+
"4f533d1364c8ac79c05753c059a97fc9ebc3771e44247b357c456ef1a93443b0",
|
| 840 |
+
"0b2616b99611389b2d6e48816d78c4037022908cd933e7fe06934abb1435e996",
|
| 841 |
+
"dc1b9410be53e2c29f25c823d1b390e2553b2d5ebda566226bf14233f7d85003",
|
| 842 |
+
"39b90de85e463a622b3a2e320bdc593469c0a8ed983e98aacc7982be30ae7415",
|
| 843 |
+
"ce6758dae252e5aed6b4da6c486cc0fbc23605409e3d1610a9582e0730f976d9",
|
| 844 |
+
"8d30c7fcd8786cd870349a1a49ef69864552d9601e14f4f3e2e221b1972ce830",
|
| 845 |
+
"7634249745e7ea4df3a25be050ce2107c3f0311c1140e20a7cdf7c081f3edbaf",
|
| 846 |
+
"83a0e1e46dd7f941f96f71383d99514688a9667584ff88fb7773e7ae826af69d",
|
| 847 |
+
"16a8d5a707e69a66d55be2af747cc556aaeac97ccbd11ec3c65411f675fc6c5c",
|
| 848 |
+
"f843bdf5a1be3cd79b23acc15eab2e4cf8bbaa1a19358bfa81c015ad7738f6fe",
|
| 849 |
+
"f302f45dd9a7bf33363b7038a4f906784c43a265c6155fdd7381181e1eb34907",
|
| 850 |
+
"fa027e5676fe3d74281e627c010cff5cc5161ecc468686349d56c108a5fa31e6",
|
| 851 |
+
"f5777bc9e9110f52dbebdcf36a47a94134b3eecae77c14ad72efdff86c94ac8a",
|
| 852 |
+
"46b712c8f0486a85653b929e0dffe2ac40aa2773b55ca786a8d58c123e70acfd",
|
| 853 |
+
"b641aa9011fa40c2fa85b027888ad743ab5babbd16d153f1e3d555284f166d73",
|
| 854 |
+
"63dc9593f0d2b701d5b8533550acd25548f22f0e9f311c31f70ba9657aa02dac",
|
| 855 |
+
"a280ab109bcd771133e23842654b1c86fba8c0b7a77f406f3f2193095011ac93",
|
| 856 |
+
"cd4f508ea5ea75f161c5570fc427c9bf6e9647d393ee887bf00a670a9401885e",
|
| 857 |
+
"18c589419ba388499df89231be1e1f9bc494da2be899c7a530d21049ddbb8d8b",
|
| 858 |
+
"528b1819278cf73ea2b27f63158cc3c9a878a007196e930c25d3b0cfd0b29794",
|
| 859 |
+
"4b8967f88d5cb5f62272b308e5995ce983e0e06ad0fa7fdee21774080f2b4ca8",
|
| 860 |
+
"954ed3380b80db20d047386d9b68b4fd43268d160c89de9363dbda8e7a8f9d4d",
|
| 861 |
+
"353f0e5557ecf5aa89f53fecdb96fac70b800473c11da0f347d63f7f8180178e",
|
| 862 |
+
"dac9aa3f41869a1b5fef0e97e5d3dad29d256c13aab05534551f765f8e78e990",
|
| 863 |
+
"592cb9b8e165fcf1be74c8e25992e562397085c2c5673040fa23583d7d1a4ee5",
|
| 864 |
+
"93c1f03e73e5cd820e8378dc871d845acf523c4f2bbe5df9934f1f2134db4993",
|
| 865 |
+
"86d4e5274255d30ae9ed89af6cd2729c444f805dc260a9db36494d199ef0e7e4",
|
| 866 |
+
"d821bfc182662cc5d834d4f56398c61df4a3ac4a43d157e193f5bb09a5744894",
|
| 867 |
+
"c3fd0d521e73464e1e844dfe43455bc2a6a204ca6688e8b300bd794048300e08",
|
| 868 |
+
"e0faaf4327fe24bcc5144db078b143a3f797ace257c817d816756903c726fe38",
|
| 869 |
+
"c1beb034479e20e5b12e180a219aa2789cce748c3b1bcb7e559ec269645b7b89",
|
| 870 |
+
"1571e429e636d8ed70c9a5b5c1236f424ef8505c9085d14ef6a5979ec8a9737d",
|
| 871 |
+
"c337d43922c7fe4bb6484bd6b08e923aa39b2e7ce855ef24697ae28bca414c68",
|
| 872 |
+
"1b5ac6446b44654ffa9035933bb6d1149fe87d66083e4a41619086e0a2a2e327",
|
| 873 |
+
"9d21c5ab8cb11c07dd869982ed58836f14bc88bf7879909e10fa6d14646f7140",
|
| 874 |
+
"4f7ceb1c078309e071e668c4fb1436a73a158bc2f8b2139fdac512c181572b3c",
|
| 875 |
+
"81eb9ba6d70daccee578ced2fc1475c86c1147c5689567d1c24644fc9dbe24fb",
|
| 876 |
+
"87d22a8f102af8cb029db876407b24f81bf1eb8ccbdfe40bd95ee183721f74c9",
|
| 877 |
+
"ce0ef78fa844fa7e0bdef2033afce520e1839825593ac0eef48c282ca4d84898",
|
| 878 |
+
"de49cf4e7ecb5c9922a8fea64bfe0b5788dfe2d962852da179695ee93118794b",
|
| 879 |
+
"10758ffc48c6f692be64dccc47e302f460cb777a614e190bc5571391767a4485",
|
| 880 |
+
"438e70a39f4a5e8f1489fd3523cb05a2104d9ad8d7c78b01c8f4eb73f3658247",
|
| 881 |
+
"973bb27905a28761f02a6e697783cdec89e83c14d06e174c5730cce6f431993a",
|
| 882 |
+
"30bc75737623717d2dab06472f4e723104092785463cd905f3b864e6632ff84c",
|
| 883 |
+
"57142574857b6a1340201c6cbfdbd9cd9cec347a6337174a9b4a38a49b79b410",
|
| 884 |
+
"c650c0ef36011bd28edc977824a6351cd25afc4f22687baf0e93ed976357d90b",
|
| 885 |
+
"c3be353460343304076105683fe7660da39fe3666d09945f0a3dc3eb843e4d78",
|
| 886 |
+
"830daadd1e415b722d96607c5eb6e22e3fad778d21ca7ae652949d4c9373465f",
|
| 887 |
+
"6656e9e86b8619cd2dd77ac64ce6fab12a646a53a7f4eb7b93e21e2d36386c63",
|
| 888 |
+
"87534fdf4e79c522434f9fe734a94aef2a049ef7511bab704bb936c5db47ab51",
|
| 889 |
+
"9a7e72aa7d5a093d6dc52b297b1c49cb6930c2555eaaf517efddf7943f05f2aa",
|
| 890 |
+
"913bf74b85fdcef5bb941911273f80a6f0a1bf7baf058876c53a2d87105a30db",
|
| 891 |
+
"a1d6d8d765a7d72dd3d9553bdc6cfa2c6ddbd81ba1228cce8e83191e6b216b9e",
|
| 892 |
+
"09a8e4280bf6c2875192e792b4ec9e1d6bc9e569d2b069f38654470351106f99",
|
| 893 |
+
"240c98be1af97caa8088f9547a31199fc7eec73f74c553495c30b0c7b50e7b0f",
|
| 894 |
+
"482e9885b81b95d56571fa72d4b65ce946f107e7c9620248f6d753abddfa1e72",
|
| 895 |
+
"a1acf6e344149793964077bd77cec49db98bad61aacb15be89a2e73a103c12e7",
|
| 896 |
+
"82ff4257661021506a64cac9445aaf6ea437ce2847aa8a23b711bf958dcfc6ed",
|
| 897 |
+
"aec802a52fc9523d14e0061184823ac22e4d78e24d95d8d90a1b5a5e48b83734",
|
| 898 |
+
"7d30edf41be089bee9ca49ad09a8968902b647eb93f0bc0e124021febeb1fe43",
|
| 899 |
+
"f5a990115f11232531a8e88b2749d18bbab0ebfd6d7f6f08096bf8767e6f78e2",
|
| 900 |
+
"9e31c03ea34c04d96fb8f7b8cc019754cc97737b4c4fb8015bbf9b679a304ec0",
|
| 901 |
+
"67a0fbfa4ec18e22565549791ff1939c4bb2e9da03b45c877784107b4e7de53d",
|
| 902 |
+
"4096f6ef8f3bfa1712132b152a7931c6371d84bca9fba5dc2b072fa1208eb80e",
|
| 903 |
+
"1bdd3dfb2401ca0ed4aab6125078e18c300abc4928b8ef7e7cd8dd87da166b00",
|
| 904 |
+
"88d49a42fc11d9d6c539ce6df3acc581cb404f21f9e5211328d643103f6aff69",
|
| 905 |
+
"b46e8b1febd00d7387396952a7012410735cd82d9931b1636ee9bca6b29a24c4",
|
| 906 |
+
"59e39ef2732cc002585a896a182f5aee5583aa699e6d4e6e60065eb6fa7d9cd5",
|
| 907 |
+
"aa5af98decc5e1bf9075a33075cb15f7325abaeb5a71a1886e93f6522a87ce8b",
|
| 908 |
+
"687497742af2f7d662ab8199bd0c91ae45c88aa6a2e158ef01fbf00106ce6e7b",
|
| 909 |
+
"f17b81c6840f554feee26124f9e0d4943a102436eff7712921ba4fd1c4ac5ff8",
|
| 910 |
+
"4389b0918168399e10dfa48e5a25297f4a7b656b4284aa21f4d989d80aad1a7e",
|
| 911 |
+
"36322f6750f4c4ac403e8c8c1acc704790012be251aa92bc8a9d1f1c19a7dcab",
|
| 912 |
+
"1b34d0357971bb73317a2108ea75d37b730331a07a7666b37479c23bef31f8ce",
|
| 913 |
+
"ac2df93e106fd8137ba91d9cf6c279db831f9837ae8421d3af93833c9aa9f4cd",
|
| 914 |
+
"4d90fb4d69ac364e0fad20def7e7db056bd081478e4ae133fadedc8e06467a66",
|
| 915 |
+
"7d0295520d86c972ff662f238b503328b85d87aa281d4b164a5d36d4e9429663",
|
| 916 |
+
"094151b8be884d2a7e603bf61c4a59de7374d2d9fb88656fcd013678902767ca",
|
| 917 |
+
"65d6a6cd6573d4f0d818a197b44e1a005e5132cceebd67dcb86e6430b348eafb",
|
| 918 |
+
"d01e34e6581308d3ee366cc946a61fd8ad18f89cf42f4cc6c005ae2419ad0116",
|
| 919 |
+
"183eb97c86fa335e63fed0f0c8be01ecad63b8e39f0ad00011176d0aa182f056",
|
| 920 |
+
"6f7cdc1071a66c87654d7732a5c6c7577f9d59253ded7f6a2c38dc9f5e33aa44",
|
| 921 |
+
"15e99d5df047b8c4be9a2c40c682264a66cb13ab34291db95f4bc72112393a17",
|
| 922 |
+
"f42ffd4f752371a9e983df3132cdd1e6ce3feabea38cf16726fdc4bce7dfa247",
|
| 923 |
+
"9871464d40e41cbe92b9d5a8368f69b00f686d692c4546f2604de976e3ba2911",
|
| 924 |
+
"855d129fd1c0008cdce3dc98b54b320cff188380708becefbe3bc11ceb42254d",
|
| 925 |
+
"ca9c54d915e2507065b81189249b025ccf5d054d4dee1ab8cccc124e3c53d3ef",
|
| 926 |
+
"b81cfdeacb353f06a1a7ed92b60175f5021a6d942b678adc440ebe100f04c2c8",
|
| 927 |
+
"901f7888a0a4a2e6225685df062252ebc62ac7998d772a37fc140fbadc30e9ab",
|
| 928 |
+
"28949bb2c594edfee0e9c05ad1a4329ffc9ae095db9555ab61f831ccc5d7acee",
|
| 929 |
+
"96d35536ebdbc3e753c7d443c5e64d22d22f08553293d1cd89ace9cc0aaf9633",
|
| 930 |
+
"fd8e768bb08913476783351e12f4c9aeaa729b1251c63a2e85f6b8ee350bd8f8",
|
| 931 |
+
"8b6a4152c3ea8bddb47bf16002e132b012633ebc8518bf427cf903e37365c065",
|
| 932 |
+
"ebac85e15050f14e8e0251354cf9565c4cfdd05a8b48a4096af4ab954f49e0ea",
|
| 933 |
+
"f3ed9a247f7c44577768eebc383d80618c77be3c7d4820dfb15b8d775614c66f",
|
| 934 |
+
"83c88f5e9c2ef708c36a3efdcc83070ffb690ce7f69683286379f38e03b80583",
|
| 935 |
+
"80e0292702d8a22c10e6d227e7ec08ad264683c7f7d554b8f071cc1338922ca4",
|
| 936 |
+
"4e355bb9e0c1cf48e93a0236d15f2b801ad61bad167f0ce1436f4b44e565bb3c",
|
| 937 |
+
"3d653e6d390d6baa20eb8b4fc3b3f739d335eecf5991aa2a2cc12fa209b14458",
|
| 938 |
+
"b94d3440e825f9b707f98352fb12295cb068a3603e9993f688297507d563a319",
|
| 939 |
+
"9e9f87005369381eb36b7800b97932fce55d1d15c50965b908b47814851a77ca",
|
| 940 |
+
"dcf9987d380c1d7fcf3f9e46996024fac7539655bf432db39f9ead3c31bf8319",
|
| 941 |
+
"58f833496f21165b898454f0703bdd11d534f6fa85bf742a7b702728ab1b3b19",
|
| 942 |
+
"9ab47dee2be2b0d0eb7d53099a3a60dbaf2ae8febef373883797cd8041b804b9",
|
| 943 |
+
"b71740bbef8384a467b1c889dd12cb6f4fa78464a9ee8ab9f13f5a13e1fc0b2e",
|
| 944 |
+
"d0d26c085e469c688739e1c06a7c3aee2da177ec204090c9371d3e83f6a5e20b",
|
| 945 |
+
"6e638e472425dbbaf129ad371bd73c34356ca666fcaab8ccb65b3517503d81f3",
|
| 946 |
+
"296413ef430e95349e91c9439b7534e15bf9021be7b4c18742a0596258fbd7a8",
|
| 947 |
+
"ae31b4b6af14c1043a243ca665a7d9dc2a87cdf9b2a11d9b0ca565b0b2dc9907",
|
| 948 |
+
"b06f6dc8d4d8e8518ceed2a311ee1e45c1efdcbdfbbe01c234e069cadfcefed1",
|
| 949 |
+
"3116ee4c6db480fa7ec4b1ef356bde16f6696836690c6a0897e133ceb166d99a",
|
| 950 |
+
"d5a8dfb68deca5e527233dd5dc41900fb8fe788319f1f429c708a0837ab9214d",
|
| 951 |
+
"09d3b590ef63e0cf912fb45fd9e6bda52eba9993a034425344d4f0838c420d61",
|
| 952 |
+
"ac43511bc20d81c5dea6ff22096289cdfa5b4a13a89e44c2d8b96ee6be7be791",
|
| 953 |
+
"baa8dd0b0cf94f11420a58b2f5f804457217f53ca958c919473acde80baecaf9",
|
| 954 |
+
"184889e81191d431fcddd6087e057152412652f3ff5d3493c88b007085086a08",
|
| 955 |
+
"3fa530553bdc7f7bfd0b74fc93ec3c3a92d2386680c2a4a91cd0e3fc3ba6bf58",
|
| 956 |
+
"a028ef0b88d18273d3e8da7f4c4a9ff994561a990b09bed8ac4292092f343958",
|
| 957 |
+
"beb53d88d0aef3e262cd2dbf0a53aa64f17fa388e935aa8bbf80ae024bb9cfbd",
|
| 958 |
+
"f18c61446647080e23e0b9fab53f27277e84bf342872dc2cd2e76a11a300f505",
|
| 959 |
+
"2dc80742aa98a2e22a4654f32a34042d1ddfce52ed21b1b9853477d4cb66293b",
|
| 960 |
+
"8a32dbd61e7cd3149ada4de44d17409f6af5913fcb42da07c1dd0af0da7c831e",
|
| 961 |
+
"c2e611f332c149e4bdb8d310cd7836c004af75560de3e66453c62150e766275a",
|
| 962 |
+
"cf0fa31d341b16a3b040c834d1e84338740547e360d15716fe180418721d924f",
|
| 963 |
+
"bcf661ede332e80b41614d717e76b4af865efa10f9b8c372dacf7d667d373da6",
|
| 964 |
+
"d10578c50a08c8c80e144b9a298925b20638f5e7ba5e89997b1a37298ed655e3",
|
| 965 |
+
"e4f0bafc3bbf3b43af6a2c6605ceebc76d0b6d3a9df1aa5fede946e3b8bc3cf3",
|
| 966 |
+
"3f5626b38709ee6d9e3abe474264ccd462b6eba5d3f49a4b32c0dfb41428ad43",
|
| 967 |
+
"ad2023bf5a6ab24be29a47c5f03631502cafa07442197ca8eb55ef0e1ff489e7",
|
| 968 |
+
"d89084d3fa4df37fa8c4387e8ad3d196fcba70bc586894a37b32068a775413e3",
|
| 969 |
+
"3a6b7a44e01d09d0f500e8beb5217decf0403df8f4678af124c1ddfb97cfae6b",
|
| 970 |
+
"6e712e125e175f95219ba0fe91c3352903dcaa4835eb158135d730edcd149af8",
|
| 971 |
+
"5d000ac16613e5d5af947722e72759df873aafda945c7f5114153031ff13d99c",
|
| 972 |
+
"380d335716afcbfd56f581d694b8a2c6f867ac62e0e9052e123ce731d6f9e089",
|
| 973 |
+
"35147e48229956c07b43595a6d9093ab687e0445c0b17090657d0075d1aad64a",
|
| 974 |
+
"c869cdc10f6632bb8d4ece7cd36f88211b1a962e9801ee80a5b1a87de3a34368",
|
| 975 |
+
"8b9d1ec3fbf543ea577204b4d790e21a816c169bcd6f3b5a730542c3de924567",
|
| 976 |
+
"e9739657e9fe956175c3680827b93db254dc08c46a03de9b39318817f88d346a",
|
| 977 |
+
"a5515c8d3cfcc19ac55f7fba8d8b1b2d418ec2eceb471575323378596fff6287",
|
| 978 |
+
"7a500e4edf6f53083868a5f58de94bee933516ccfcd998525f8580e07e1a903d",
|
| 979 |
+
"60412c3d2110cc82360c9f71ce18c60a352643f427e827585ddc6a918acb5521",
|
| 980 |
+
"a567e99f733f35d1331e11a5f8df6dde81156fe18ff96f3c68d0b266a092332b",
|
| 981 |
+
"4b98206cbb362ff6d5efc7b1ae4a7a2d861d0b6e72232cc24bacac3a94a1e130",
|
| 982 |
+
"0b3dddad4bb36cc6a28865237d85a5a68fcddd697e80f5ead63fc5dc9f6fb135",
|
| 983 |
+
"e3862220fcb5f16707b04fd797c107043a1b5844d3ffe874ef0485d9421f22a5",
|
| 984 |
+
"fb5dbe501eb3c43e8e58f97af70e8c1602480576597a950a21a7cd73951ffd09",
|
| 985 |
+
"086c45ccd14513e09b657f019f958652526a14bec3b1c5caf907aa2f20d580cb",
|
| 986 |
+
"f20656535835b37dbe273759c224c73eff4e8972a68c4dc2b4a5f06b9ce35c05",
|
| 987 |
+
"f27cac0e853214ca8f622c2de95f457ad74ca36fa79b32458ce3aedceb2e20ac",
|
| 988 |
+
"841f68ca58bce5587c3280cddedbf5f3d67e553ff783f64b52319a4047b237b6",
|
| 989 |
+
"d7c849172a0fb6fce456c3d0cf0e0f31b979ad1e9de16f1c1327a96897e521a6",
|
| 990 |
+
"fa1b8469f5954cd4efa382a7742663face7774d667548294156d1e7b745816bf",
|
| 991 |
+
"8b8f0d103949c9b96e824483ec13e4cc29f8aca374b9e0bfe3d30f51ac7804d4",
|
| 992 |
+
"7bd1dacf3d88813194458899887ddd537452947fbf3f0a12f1d098f0e0a64ad1",
|
| 993 |
+
"b725078f06625a3f8c29869ca7c222b33cdf21abba416de3acd0f278e637cf78",
|
| 994 |
+
"c9cce7e8bda7a09daac6656b581ff08557797ba9d4cf35c67e1e741762de0c83",
|
| 995 |
+
"df875ac1aca180ba592fcb8b140d73d37a3eae11fd5598ec18e6d7ebbc81e1f7",
|
| 996 |
+
"7f19060c04ded6918d7bbe07ef5c2f1db6a9086601a3ebfe67649415ea6b9fe9",
|
| 997 |
+
"627da6a10c85cc54548db786db814c1710b7230efe1bc49e5ab17fafc775f332",
|
| 998 |
+
"93d3fa70b7186c58b0c8207afa157d84f352475197b6166e2166a5748cc3e7ba",
|
| 999 |
+
"40e911531decd7f83645b73e64f24767ee743f021430694e122e3898d7791359",
|
| 1000 |
+
"ce21d069f12b621b2ef6f28f82cbdecef08e2fb0f60463f97b1b8395ba610b0f",
|
| 1001 |
+
"9622915764cd99bfa0b9a08a001ff5d9f547cbdc6cbfcd2357fbcf7476d659fb",
|
| 1002 |
+
"9f817e9183b95e42244cdcf699d619e027e160eb1f10f179a0f4f8293eead9db",
|
| 1003 |
+
"41f74b89c9df9846983ba44ef9a769901b9948b9ef092491232d0fbbc72c109c",
|
| 1004 |
+
"f9fecf3840bc716ff1d535b1720fd0797661eeab4ed141351093a6d98f3ced9c",
|
| 1005 |
+
"8c73882fb4b28e42b8457d7c547f14316d733c2d18061658af1386d64c0e630f",
|
| 1006 |
+
"9f92fb40fc7ba293cb55c8eb4b104d202b9cbaa0d7502c4deac8d9450069141c",
|
| 1007 |
+
"bf3b19d157ef6eba478f0f370f40b2a92deea3423e1b4602912c098c547589fe",
|
| 1008 |
+
"4b2cd5c5bdd9bf8e3603ed4a490f703c306dc68de69d4c5f225a59c503df7e7c",
|
| 1009 |
+
"40e4fe2a4369bbbe03faf80170e1f9375f8829f7b49a4d17f0c8bf8f63542923",
|
| 1010 |
+
"38ba7942ab9ddc6d26edbbcc4d6ff2bacac0d3605b8d59ae7852d88032889300",
|
| 1011 |
+
"d1edc72d49b7b86c8b4aa0a1ea9785dd0c5bb9f9270c56f0f6138a8b06d2408d",
|
| 1012 |
+
"1ef57e7806f2e9b50188a1eae7d8f7b962b99064b101c5663fd27aa7c8cd9bbc",
|
| 1013 |
+
"a2b39a552871f145225a605b83b41f4a9e145723ecd5fd8045413ecc11fa90bc",
|
| 1014 |
+
"d1eecc20ec1b387a40bc696c62f0446deea97aa5b95c69026691b07710af1f14",
|
| 1015 |
+
"800dc0a9b89d0ac4a24387e8960278c5b7e9295149411f54770b1c0be46bdb26",
|
| 1016 |
+
"0ba29f0f8597c921ef90fbb58d651d2e0a2de0d7cc87749ccf669f96c41771d1",
|
| 1017 |
+
"72904fb7dedbbce513dbdb5b71c136d3b4fc58be867b86b7ca6ec8889515884b",
|
| 1018 |
+
"333f81bc843a42161b60e4faf98e9bf7f3dc0d530bbd329a484036b73aa8b5ff",
|
| 1019 |
+
"7886f754ed847ca698b2da49464b8cf028d574f067ab6c6a44ad5703ee0955ef",
|
| 1020 |
+
"8df56b75baf77ba783f43797e9b68b159188c0ab29c993a4ae08d02e7e85fdbe",
|
| 1021 |
+
"05c12cfdb73f7a1c89e28355fb4fb8711260c80a086d28e383f74c9981ec0751",
|
| 1022 |
+
"01bf934b2318daf21d27dcfdd06ddbf15de5c33fd3d983a957ffdd91a5da0a5c",
|
| 1023 |
+
"e1605881822078495282c211e28ef49b9cba1f47066a6f8b7cbc24e35d6ba176",
|
| 1024 |
+
"42013b23681213ca8b1513ae68910d4c3ca32f8931202b32c669e80f6c816b35",
|
| 1025 |
+
"6ff12998007bd4a9f9ce9241bde767f465b4a227011a0cd9bdb01d551413781c",
|
| 1026 |
+
"02840c2b2b00042374b7ce77d699e2d393986ffcb125e5456ba6450fbf5a7628",
|
| 1027 |
+
"b30a0d8af3ae051d7add37bf5efa1ae78f35fbbe034800e28f9e9149b8dae514",
|
| 1028 |
+
"c80ae4142f6b472e73f682ddac94fb009adab3acff19650d420edd61e57cecd7",
|
| 1029 |
+
"4d8db0fd2087291718d11eb47c4e20f1a79cc0ad69282c928a80e6089b26f819",
|
| 1030 |
+
"30ccdaeb3817919cd42c4941f616db5f2bcf1ff345998e49b4f714e3763cd608",
|
| 1031 |
+
"a44c1ce98821f6479f2b0f900589594f9829d62683b27207df79e2ee05f76686",
|
| 1032 |
+
"d4d0aa1b2b07cee3c3085ca4ca371e81fab75ae79043774b2fefd31445b9bc0e",
|
| 1033 |
+
"cb76adaffe4e8006f7d68999b3ed0d499b2b29c730f7d8e64e4b1ddec903d0ee",
|
| 1034 |
+
"8b940e9465f687e83d35da824ff40b82fc1f1de5e2fc7633c8dee101190d297a",
|
| 1035 |
+
"352256d237df3069e48170d46ab66f5c3d34455b8397d356449429bab2fe5ab7",
|
| 1036 |
+
"06e90c90d3ffcaa6a1f63cb6992d9384ced2c9d573cbe1fafd1bcf69853f4ef1",
|
| 1037 |
+
"8dd07f5d8c71f4ab35791e432199679d4a340e7605f56a67e5395f3e1f414ef4",
|
| 1038 |
+
"ca0ff28f7fb8ae67b40cb4d265df00409e0c221af08620ef19bcbffc367d5602",
|
| 1039 |
+
"0cb110df276fd116844bf9167b23b835db172a099d4abe915ef488500d2667cb",
|
| 1040 |
+
"7a4a535b7f422afe60a3c0f279b06b9cad7104787b2cd5ed364ca6dbeb2bd4e2",
|
| 1041 |
+
"377a17cfb37f3897bec5259129287432444b708e10a8efa64d2e708d50eb6902",
|
| 1042 |
+
"0b8013b5775f619a2beab8336fc3da2d5c68d3d92fcaae92b0d2f2e991ae4dba",
|
| 1043 |
+
"50ca4038bdd0c194a6b18bf505a53f9f705b2d2975e188d79f983e71a16486f4",
|
| 1044 |
+
"558201f669b611e69b38f5adf6076578e56eac5fe51a0564ffd68240b8de0e5e",
|
| 1045 |
+
"1ec64875da1b1fb02bfe00c26264ed82c4ceacff4b92b4b5992f891953e8f9af",
|
| 1046 |
+
"c18a6c7a81f9c721d0c0173973788cfea6d4b812b9be8141499709bd4551a734",
|
| 1047 |
+
"5ac2f12c410ea7412088e143e4b88be61ac32f49d2e2abbd629ea49c9c72cd45",
|
| 1048 |
+
"fcbf1aa685f65e8e788810f6595decc002c4d9d6ce9eedb901031c8abf9e5c0e",
|
| 1049 |
+
"8b0aaefcbf19070ef4fef6a35ce5afd2c601d7fcd221b6e0d7c3ac7fbbbdc523",
|
| 1050 |
+
"cf0e085796a02c256aa9dd426f033519d2c142b7795fd41d65746a6a160f5ddb",
|
| 1051 |
+
"40589d336903f06f0ac2ae5a6088405921dcf2d4251dcaf140696f528425a8c1",
|
| 1052 |
+
"02d8717f861f7d72135a166934776371621677b934458ba2a848d7a2d8cf6751",
|
| 1053 |
+
"c615e816a5b705c4e700f4350f391c478f3ccc35255c8b04a3a5e5e70d62c64d",
|
| 1054 |
+
"b86475040ad6d175551887ff44a472cc99cd3036273e95e443ad062a18787d46",
|
| 1055 |
+
"a8de623badfdeeba072765b9724a5015fa3988a1fe38fea93181e50d94b1df4c",
|
| 1056 |
+
"ccb6a7cad62bf653d3a699db734fe813fbb64c274e47cfe9cff4d8283eb6aa99",
|
| 1057 |
+
"f693119c35fae02b0265d7400c7491a6f2e69d2a8f843cddef004e6bd8260c8f",
|
| 1058 |
+
"4347e563c3999c80f38c0ac90bc5b0181ab43cfb4eabca51e4069aec1c8592e2",
|
| 1059 |
+
"744aee1aaabbb2b3d6db7962c53c9e1c976a68ce63397d82b65b63e244dff296",
|
| 1060 |
+
"d240db39848443d48b9caaf3842f9ec238424a0b7c49364ea22e1fc38a7db564",
|
| 1061 |
+
"69b33b9c58d7e286973cff2d883d38d2f89e94d30fcff9d7254de4f8f404c9e7",
|
| 1062 |
+
"74fdc60bb7b8ae34b33835d09b64bebc2e1ac3602d87dbd7abb980b09be8ae05",
|
| 1063 |
+
"baab1add1ebb9033002228328093733830c35711c2066a437c938f4792a2c304",
|
| 1064 |
+
"3318545166204cff07be0b637c5b1ef6a9ea237ae942f8a8b4c3ffd88370d3af",
|
| 1065 |
+
"aab1f91322a2c805b84bf1cd289e1efccb6ecad435935c1b193aedcfc5305d4e",
|
| 1066 |
+
"4b9fd0e6842096a25b799a0425326a806469c38bfdf7b0c44c0fb1dd1a8aa0a0",
|
| 1067 |
+
"1cbbce5ef2dbfc746ee3884670ea795470fe7d8c963f69762a12706ea608d61e",
|
| 1068 |
+
"2155e7ae920cd32564c5800699b8ddfcf91ef7d41ada35e484cc2b95f9d4c51e"
|
| 1069 |
+
],
|
| 1070 |
+
"sample_prompt_ids_sha256": "a1fed5a2ab838c2978333616fd6038d7a14bdc937c93535d7d190a72fb627fa3"
|
| 1071 |
+
}
|
chat_template.jinja
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{%- set image_count = namespace(value=0) %}
|
| 2 |
+
{%- set video_count = namespace(value=0) %}
|
| 3 |
+
{%- macro render_content(content, do_vision_count, is_system_content=false) %}
|
| 4 |
+
{%- if content is string %}
|
| 5 |
+
{{- content }}
|
| 6 |
+
{%- elif content is iterable and content is not mapping %}
|
| 7 |
+
{%- for item in content %}
|
| 8 |
+
{%- if 'image' in item or 'image_url' in item or item.type == 'image' %}
|
| 9 |
+
{%- if is_system_content %}
|
| 10 |
+
{{- raise_exception('System message cannot contain images.') }}
|
| 11 |
+
{%- endif %}
|
| 12 |
+
{%- if do_vision_count %}
|
| 13 |
+
{%- set image_count.value = image_count.value + 1 %}
|
| 14 |
+
{%- endif %}
|
| 15 |
+
{%- if add_vision_id %}
|
| 16 |
+
{{- 'Picture ' ~ image_count.value ~ ': ' }}
|
| 17 |
+
{%- endif %}
|
| 18 |
+
{{- '<|vision_start|><|image_pad|><|vision_end|>' }}
|
| 19 |
+
{%- elif 'video' in item or item.type == 'video' %}
|
| 20 |
+
{%- if is_system_content %}
|
| 21 |
+
{{- raise_exception('System message cannot contain videos.') }}
|
| 22 |
+
{%- endif %}
|
| 23 |
+
{%- if do_vision_count %}
|
| 24 |
+
{%- set video_count.value = video_count.value + 1 %}
|
| 25 |
+
{%- endif %}
|
| 26 |
+
{%- if add_vision_id %}
|
| 27 |
+
{{- 'Video ' ~ video_count.value ~ ': ' }}
|
| 28 |
+
{%- endif %}
|
| 29 |
+
{{- '<|vision_start|><|video_pad|><|vision_end|>' }}
|
| 30 |
+
{%- elif 'text' in item %}
|
| 31 |
+
{{- item.text }}
|
| 32 |
+
{%- else %}
|
| 33 |
+
{{- raise_exception('Unexpected item type in content.') }}
|
| 34 |
+
{%- endif %}
|
| 35 |
+
{%- endfor %}
|
| 36 |
+
{%- elif content is none or content is undefined %}
|
| 37 |
+
{{- '' }}
|
| 38 |
+
{%- else %}
|
| 39 |
+
{{- raise_exception('Unexpected content type.') }}
|
| 40 |
+
{%- endif %}
|
| 41 |
+
{%- endmacro %}
|
| 42 |
+
{%- if not messages %}
|
| 43 |
+
{{- raise_exception('No messages provided.') }}
|
| 44 |
+
{%- endif %}
|
| 45 |
+
{%- set reasoning_instructions = '' %}
|
| 46 |
+
{%- if enable_thinking is undefined or enable_thinking is true %}
|
| 47 |
+
{%- set resolved_reasoning_effort = reasoning_effort|default('xhigh') %}
|
| 48 |
+
{%- if resolved_reasoning_effort not in ('xhigh', 'medium', 'low') %}
|
| 49 |
+
{{- raise_exception('Unexpected reasoning effort ' ~ reasoning_effort ~ '. Supported types are xhigh (default), medium, and low.') }}
|
| 50 |
+
{%- endif %}
|
| 51 |
+
{%- if resolved_reasoning_effort == 'xhigh' %}
|
| 52 |
+
{%- set reasoning_instructions = 'Reasoning effort is set to xhigh. Please think carefully through the task, validate key assumptions, consider plausible alternatives, and prioritize correctness, consistency, and clarity in the final answer.' %}
|
| 53 |
+
{%- elif resolved_reasoning_effort == 'low' %}
|
| 54 |
+
{%- set reasoning_instructions = 'Reasoning effort is set to low. Keep your thinking brief and focused, moving directly to the conclusion without unnecessary elaboration.' %}
|
| 55 |
+
{%- endif %}
|
| 56 |
+
{%- endif %}
|
| 57 |
+
{%- if tools and tools is iterable and tools is not mapping %}
|
| 58 |
+
{{- '<|im_start|>system\n' }}
|
| 59 |
+
{%- if reasoning_instructions %}
|
| 60 |
+
{{- reasoning_instructions + '\n\n' }}
|
| 61 |
+
{%- endif %}
|
| 62 |
+
{{- "# Tools\n\nYou have access to the following functions:\n\n<tools>" }}
|
| 63 |
+
{%- for tool in tools %}
|
| 64 |
+
{{- "\n" }}
|
| 65 |
+
{{- tool | tojson }}
|
| 66 |
+
{%- endfor %}
|
| 67 |
+
{{- "\n</tools>" }}
|
| 68 |
+
{{- '\n\nIf you choose to call a function ONLY reply in the following format with NO suffix:\n\n<tool_call>\n<function=example_function_name>\n<parameter=example_parameter_1>\nvalue_1\n</parameter>\n<parameter=example_parameter_2>\nThis is the value for the second parameter\nthat can span\nmultiple lines\n</parameter>\n</function>\n</tool_call>\n\n<IMPORTANT>\nReminder:\n- Function calls MUST follow the specified format: an inner <function=...></function> block must be nested within <tool_call></tool_call> XML tags\n- Required parameters MUST be specified\n- You may provide optional reasoning for your function call in natural language BEFORE the function call, but NOT after\n- If there is no function call available, answer the question like normal with your current knowledge and do not tell the user about function calls\n</IMPORTANT>' }}
|
| 69 |
+
{%- if messages[0].role == 'system' %}
|
| 70 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 71 |
+
{%- if content %}
|
| 72 |
+
{{- '\n\n' + content }}
|
| 73 |
+
{%- endif %}
|
| 74 |
+
{%- endif %}
|
| 75 |
+
{{- '<|im_end|>\n' }}
|
| 76 |
+
{%- else %}
|
| 77 |
+
{%- if messages[0].role == 'system' %}
|
| 78 |
+
{%- set content = render_content(messages[0].content, false, true)|trim %}
|
| 79 |
+
{%- if content %}
|
| 80 |
+
{{- '<|im_start|>system\n' + (reasoning_instructions + '\n\n' if reasoning_instructions else '') + content + '<|im_end|>\n' }}
|
| 81 |
+
{%- elif reasoning_instructions %}
|
| 82 |
+
{{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
|
| 83 |
+
{%- endif %}
|
| 84 |
+
{%- elif reasoning_instructions %}
|
| 85 |
+
{{- '<|im_start|>system\n' + reasoning_instructions + '<|im_end|>\n' }}
|
| 86 |
+
{%- endif %}
|
| 87 |
+
{%- endif %}
|
| 88 |
+
{%- set ns = namespace(multi_step_tool=true, last_query_index=messages|length - 1) %}
|
| 89 |
+
{%- for message in messages[::-1] %}
|
| 90 |
+
{%- set index = (messages|length - 1) - loop.index0 %}
|
| 91 |
+
{%- if ns.multi_step_tool and message.role == "user" %}
|
| 92 |
+
{%- set content = render_content(message.content, false)|trim %}
|
| 93 |
+
{%- if not(content.startswith('<tool_response>') and content.endswith('</tool_response>')) %}
|
| 94 |
+
{%- set ns.multi_step_tool = false %}
|
| 95 |
+
{%- set ns.last_query_index = index %}
|
| 96 |
+
{%- endif %}
|
| 97 |
+
{%- endif %}
|
| 98 |
+
{%- endfor %}
|
| 99 |
+
{%- if ns.multi_step_tool %}
|
| 100 |
+
{{- raise_exception('No user query found in messages.') }}
|
| 101 |
+
{%- endif %}
|
| 102 |
+
{%- for message in messages %}
|
| 103 |
+
{%- set content = render_content(message.content, true)|trim %}
|
| 104 |
+
{%- if message.role == "system" %}
|
| 105 |
+
{%- if not loop.first %}
|
| 106 |
+
{{- raise_exception('System message must be at the beginning.') }}
|
| 107 |
+
{%- endif %}
|
| 108 |
+
{%- elif message.role == "user" %}
|
| 109 |
+
{{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
|
| 110 |
+
{%- elif message.role == "assistant" %}
|
| 111 |
+
{%- set reasoning_content = '' %}
|
| 112 |
+
{%- if message.reasoning_content is string %}
|
| 113 |
+
{%- set reasoning_content = message.reasoning_content %}
|
| 114 |
+
{%- endif %}
|
| 115 |
+
{%- set reasoning_content = reasoning_content|trim %}
|
| 116 |
+
{%- if preserve_thinking is undefined or preserve_thinking is true or loop.index0 > ns.last_query_index %}
|
| 117 |
+
{{- '<|im_start|>' + message.role + '\n<think>\n' + reasoning_content + '\n</think>\n\n' + content }}
|
| 118 |
+
{%- else %}
|
| 119 |
+
{{- '<|im_start|>' + message.role + '\n' + content }}
|
| 120 |
+
{%- endif %}
|
| 121 |
+
{%- if message.tool_calls and message.tool_calls is iterable and message.tool_calls is not mapping %}
|
| 122 |
+
{%- for tool_call in message.tool_calls %}
|
| 123 |
+
{%- if tool_call.function is defined %}
|
| 124 |
+
{%- set tool_call = tool_call.function %}
|
| 125 |
+
{%- endif %}
|
| 126 |
+
{%- if loop.first %}
|
| 127 |
+
{%- if content|trim %}
|
| 128 |
+
{{- '\n\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 129 |
+
{%- else %}
|
| 130 |
+
{{- '<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 131 |
+
{%- endif %}
|
| 132 |
+
{%- else %}
|
| 133 |
+
{{- '\n<tool_call>\n<function=' + tool_call.name + '>\n' }}
|
| 134 |
+
{%- endif %}
|
| 135 |
+
{%- if tool_call.arguments is defined and tool_call.arguments != '' %}
|
| 136 |
+
{%- for args_name, args_value in tool_call.arguments|items %}
|
| 137 |
+
{{- '<parameter=' + args_name + '>\n' }}
|
| 138 |
+
{%- set args_value = args_value | string if args_value is string else args_value | tojson | safe %}
|
| 139 |
+
{{- args_value }}
|
| 140 |
+
{{- '\n</parameter>\n' }}
|
| 141 |
+
{%- endfor %}
|
| 142 |
+
{%- endif %}
|
| 143 |
+
{{- '</function>\n</tool_call>' }}
|
| 144 |
+
{%- endfor %}
|
| 145 |
+
{%- endif %}
|
| 146 |
+
{{- '<|im_end|>\n' }}
|
| 147 |
+
{%- elif message.role == "tool" %}
|
| 148 |
+
{%- if loop.previtem and loop.previtem.role != "tool" %}
|
| 149 |
+
{{- '<|im_start|>user' }}
|
| 150 |
+
{%- endif %}
|
| 151 |
+
{{- '\n<tool_response>\n' }}
|
| 152 |
+
{{- content }}
|
| 153 |
+
{{- '\n</tool_response>' }}
|
| 154 |
+
{%- if not loop.last and loop.nextitem.role != "tool" %}
|
| 155 |
+
{{- '<|im_end|>\n' }}
|
| 156 |
+
{%- elif loop.last %}
|
| 157 |
+
{{- '<|im_end|>\n' }}
|
| 158 |
+
{%- endif %}
|
| 159 |
+
{%- else %}
|
| 160 |
+
{{- raise_exception('Unexpected message role.') }}
|
| 161 |
+
{%- endif %}
|
| 162 |
+
{%- endfor %}
|
| 163 |
+
{%- if add_generation_prompt %}
|
| 164 |
+
{{- '<|im_start|>assistant\n' }}
|
| 165 |
+
{%- if enable_thinking is defined and enable_thinking is false %}
|
| 166 |
+
{{- '<think>\n\n</think>\n\n' }}
|
| 167 |
+
{%- else %}
|
| 168 |
+
{{- '<think>\n' }}
|
| 169 |
+
{%- endif %}
|
| 170 |
+
{%- endif %}
|
config.json
ADDED
|
@@ -0,0 +1,478 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"AgnesForConditionalGeneration"
|
| 4 |
+
],
|
| 5 |
+
"auto_map": {
|
| 6 |
+
"AutoConfig": "configuration_agnes.AgnesConfig",
|
| 7 |
+
"AutoModel": "modeling_agnes.AgnesModel",
|
| 8 |
+
"AutoModelForCausalLM": "modeling_agnes.AgnesForConditionalGeneration",
|
| 9 |
+
"AutoModelForImageTextToText": "modeling_agnes.AgnesForConditionalGeneration"
|
| 10 |
+
},
|
| 11 |
+
"dtype": "bfloat16",
|
| 12 |
+
"image_token_id": 248056,
|
| 13 |
+
"model_type": "agnes",
|
| 14 |
+
"quantization_config": {
|
| 15 |
+
"config_groups": {
|
| 16 |
+
"group_0": {
|
| 17 |
+
"format": "float-quantized",
|
| 18 |
+
"input_activations": {
|
| 19 |
+
"actorder": null,
|
| 20 |
+
"block_structure": null,
|
| 21 |
+
"dynamic": false,
|
| 22 |
+
"group_size": null,
|
| 23 |
+
"num_bits": 8,
|
| 24 |
+
"observer": "static_minmax",
|
| 25 |
+
"observer_kwargs": {},
|
| 26 |
+
"scale_dtype": null,
|
| 27 |
+
"strategy": "tensor",
|
| 28 |
+
"symmetric": true,
|
| 29 |
+
"type": "float",
|
| 30 |
+
"zp_dtype": null
|
| 31 |
+
},
|
| 32 |
+
"output_activations": null,
|
| 33 |
+
"targets": [
|
| 34 |
+
"Linear"
|
| 35 |
+
],
|
| 36 |
+
"weights": {
|
| 37 |
+
"actorder": null,
|
| 38 |
+
"block_structure": null,
|
| 39 |
+
"dynamic": false,
|
| 40 |
+
"group_size": null,
|
| 41 |
+
"num_bits": 8,
|
| 42 |
+
"observer": "memoryless_minmax",
|
| 43 |
+
"observer_kwargs": {},
|
| 44 |
+
"scale_dtype": null,
|
| 45 |
+
"strategy": "tensor",
|
| 46 |
+
"symmetric": true,
|
| 47 |
+
"type": "float",
|
| 48 |
+
"zp_dtype": null
|
| 49 |
+
}
|
| 50 |
+
}
|
| 51 |
+
},
|
| 52 |
+
"format": "float-quantized",
|
| 53 |
+
"global_compression_ratio": null,
|
| 54 |
+
"ignore": [
|
| 55 |
+
"model.visual.blocks.0.attn.qkv",
|
| 56 |
+
"model.visual.blocks.0.attn.proj",
|
| 57 |
+
"model.visual.blocks.0.mlp.linear_fc1",
|
| 58 |
+
"model.visual.blocks.0.mlp.linear_fc2",
|
| 59 |
+
"model.visual.blocks.1.attn.qkv",
|
| 60 |
+
"model.visual.blocks.1.attn.proj",
|
| 61 |
+
"model.visual.blocks.1.mlp.linear_fc1",
|
| 62 |
+
"model.visual.blocks.1.mlp.linear_fc2",
|
| 63 |
+
"model.visual.blocks.2.attn.qkv",
|
| 64 |
+
"model.visual.blocks.2.attn.proj",
|
| 65 |
+
"model.visual.blocks.2.mlp.linear_fc1",
|
| 66 |
+
"model.visual.blocks.2.mlp.linear_fc2",
|
| 67 |
+
"model.visual.blocks.3.attn.qkv",
|
| 68 |
+
"model.visual.blocks.3.attn.proj",
|
| 69 |
+
"model.visual.blocks.3.mlp.linear_fc1",
|
| 70 |
+
"model.visual.blocks.3.mlp.linear_fc2",
|
| 71 |
+
"model.visual.blocks.4.attn.qkv",
|
| 72 |
+
"model.visual.blocks.4.attn.proj",
|
| 73 |
+
"model.visual.blocks.4.mlp.linear_fc1",
|
| 74 |
+
"model.visual.blocks.4.mlp.linear_fc2",
|
| 75 |
+
"model.visual.blocks.5.attn.qkv",
|
| 76 |
+
"model.visual.blocks.5.attn.proj",
|
| 77 |
+
"model.visual.blocks.5.mlp.linear_fc1",
|
| 78 |
+
"model.visual.blocks.5.mlp.linear_fc2",
|
| 79 |
+
"model.visual.blocks.6.attn.qkv",
|
| 80 |
+
"model.visual.blocks.6.attn.proj",
|
| 81 |
+
"model.visual.blocks.6.mlp.linear_fc1",
|
| 82 |
+
"model.visual.blocks.6.mlp.linear_fc2",
|
| 83 |
+
"model.visual.blocks.7.attn.qkv",
|
| 84 |
+
"model.visual.blocks.7.attn.proj",
|
| 85 |
+
"model.visual.blocks.7.mlp.linear_fc1",
|
| 86 |
+
"model.visual.blocks.7.mlp.linear_fc2",
|
| 87 |
+
"model.visual.blocks.8.attn.qkv",
|
| 88 |
+
"model.visual.blocks.8.attn.proj",
|
| 89 |
+
"model.visual.blocks.8.mlp.linear_fc1",
|
| 90 |
+
"model.visual.blocks.8.mlp.linear_fc2",
|
| 91 |
+
"model.visual.blocks.9.attn.qkv",
|
| 92 |
+
"model.visual.blocks.9.attn.proj",
|
| 93 |
+
"model.visual.blocks.9.mlp.linear_fc1",
|
| 94 |
+
"model.visual.blocks.9.mlp.linear_fc2",
|
| 95 |
+
"model.visual.blocks.10.attn.qkv",
|
| 96 |
+
"model.visual.blocks.10.attn.proj",
|
| 97 |
+
"model.visual.blocks.10.mlp.linear_fc1",
|
| 98 |
+
"model.visual.blocks.10.mlp.linear_fc2",
|
| 99 |
+
"model.visual.blocks.11.attn.qkv",
|
| 100 |
+
"model.visual.blocks.11.attn.proj",
|
| 101 |
+
"model.visual.blocks.11.mlp.linear_fc1",
|
| 102 |
+
"model.visual.blocks.11.mlp.linear_fc2",
|
| 103 |
+
"model.visual.blocks.12.attn.qkv",
|
| 104 |
+
"model.visual.blocks.12.attn.proj",
|
| 105 |
+
"model.visual.blocks.12.mlp.linear_fc1",
|
| 106 |
+
"model.visual.blocks.12.mlp.linear_fc2",
|
| 107 |
+
"model.visual.blocks.13.attn.qkv",
|
| 108 |
+
"model.visual.blocks.13.attn.proj",
|
| 109 |
+
"model.visual.blocks.13.mlp.linear_fc1",
|
| 110 |
+
"model.visual.blocks.13.mlp.linear_fc2",
|
| 111 |
+
"model.visual.blocks.14.attn.qkv",
|
| 112 |
+
"model.visual.blocks.14.attn.proj",
|
| 113 |
+
"model.visual.blocks.14.mlp.linear_fc1",
|
| 114 |
+
"model.visual.blocks.14.mlp.linear_fc2",
|
| 115 |
+
"model.visual.blocks.15.attn.qkv",
|
| 116 |
+
"model.visual.blocks.15.attn.proj",
|
| 117 |
+
"model.visual.blocks.15.mlp.linear_fc1",
|
| 118 |
+
"model.visual.blocks.15.mlp.linear_fc2",
|
| 119 |
+
"model.visual.blocks.16.attn.qkv",
|
| 120 |
+
"model.visual.blocks.16.attn.proj",
|
| 121 |
+
"model.visual.blocks.16.mlp.linear_fc1",
|
| 122 |
+
"model.visual.blocks.16.mlp.linear_fc2",
|
| 123 |
+
"model.visual.blocks.17.attn.qkv",
|
| 124 |
+
"model.visual.blocks.17.attn.proj",
|
| 125 |
+
"model.visual.blocks.17.mlp.linear_fc1",
|
| 126 |
+
"model.visual.blocks.17.mlp.linear_fc2",
|
| 127 |
+
"model.visual.blocks.18.attn.qkv",
|
| 128 |
+
"model.visual.blocks.18.attn.proj",
|
| 129 |
+
"model.visual.blocks.18.mlp.linear_fc1",
|
| 130 |
+
"model.visual.blocks.18.mlp.linear_fc2",
|
| 131 |
+
"model.visual.blocks.19.attn.qkv",
|
| 132 |
+
"model.visual.blocks.19.attn.proj",
|
| 133 |
+
"model.visual.blocks.19.mlp.linear_fc1",
|
| 134 |
+
"model.visual.blocks.19.mlp.linear_fc2",
|
| 135 |
+
"model.visual.blocks.20.attn.qkv",
|
| 136 |
+
"model.visual.blocks.20.attn.proj",
|
| 137 |
+
"model.visual.blocks.20.mlp.linear_fc1",
|
| 138 |
+
"model.visual.blocks.20.mlp.linear_fc2",
|
| 139 |
+
"model.visual.blocks.21.attn.qkv",
|
| 140 |
+
"model.visual.blocks.21.attn.proj",
|
| 141 |
+
"model.visual.blocks.21.mlp.linear_fc1",
|
| 142 |
+
"model.visual.blocks.21.mlp.linear_fc2",
|
| 143 |
+
"model.visual.blocks.22.attn.qkv",
|
| 144 |
+
"model.visual.blocks.22.attn.proj",
|
| 145 |
+
"model.visual.blocks.22.mlp.linear_fc1",
|
| 146 |
+
"model.visual.blocks.22.mlp.linear_fc2",
|
| 147 |
+
"model.visual.blocks.23.attn.qkv",
|
| 148 |
+
"model.visual.blocks.23.attn.proj",
|
| 149 |
+
"model.visual.blocks.23.mlp.linear_fc1",
|
| 150 |
+
"model.visual.blocks.23.mlp.linear_fc2",
|
| 151 |
+
"model.visual.blocks.24.attn.qkv",
|
| 152 |
+
"model.visual.blocks.24.attn.proj",
|
| 153 |
+
"model.visual.blocks.24.mlp.linear_fc1",
|
| 154 |
+
"model.visual.blocks.24.mlp.linear_fc2",
|
| 155 |
+
"model.visual.blocks.25.attn.qkv",
|
| 156 |
+
"model.visual.blocks.25.attn.proj",
|
| 157 |
+
"model.visual.blocks.25.mlp.linear_fc1",
|
| 158 |
+
"model.visual.blocks.25.mlp.linear_fc2",
|
| 159 |
+
"model.visual.blocks.26.attn.qkv",
|
| 160 |
+
"model.visual.blocks.26.attn.proj",
|
| 161 |
+
"model.visual.blocks.26.mlp.linear_fc1",
|
| 162 |
+
"model.visual.blocks.26.mlp.linear_fc2",
|
| 163 |
+
"model.visual.merger.linear_fc1",
|
| 164 |
+
"model.visual.merger.linear_fc2",
|
| 165 |
+
"model.language_model.layers.0.delta_attn.norm",
|
| 166 |
+
"model.language_model.layers.0.delta_attn.in_proj_b",
|
| 167 |
+
"model.language_model.layers.0.delta_attn.in_proj_a",
|
| 168 |
+
"model.language_model.layers.1.delta_attn.norm",
|
| 169 |
+
"model.language_model.layers.1.delta_attn.in_proj_b",
|
| 170 |
+
"model.language_model.layers.1.delta_attn.in_proj_a",
|
| 171 |
+
"model.language_model.layers.2.delta_attn.norm",
|
| 172 |
+
"model.language_model.layers.2.delta_attn.in_proj_b",
|
| 173 |
+
"model.language_model.layers.2.delta_attn.in_proj_a",
|
| 174 |
+
"model.language_model.layers.4.delta_attn.norm",
|
| 175 |
+
"model.language_model.layers.4.delta_attn.in_proj_b",
|
| 176 |
+
"model.language_model.layers.4.delta_attn.in_proj_a",
|
| 177 |
+
"model.language_model.layers.5.delta_attn.norm",
|
| 178 |
+
"model.language_model.layers.5.delta_attn.in_proj_b",
|
| 179 |
+
"model.language_model.layers.5.delta_attn.in_proj_a",
|
| 180 |
+
"model.language_model.layers.6.delta_attn.norm",
|
| 181 |
+
"model.language_model.layers.6.delta_attn.in_proj_b",
|
| 182 |
+
"model.language_model.layers.6.delta_attn.in_proj_a",
|
| 183 |
+
"model.language_model.layers.8.delta_attn.norm",
|
| 184 |
+
"model.language_model.layers.8.delta_attn.in_proj_b",
|
| 185 |
+
"model.language_model.layers.8.delta_attn.in_proj_a",
|
| 186 |
+
"model.language_model.layers.9.delta_attn.norm",
|
| 187 |
+
"model.language_model.layers.9.delta_attn.in_proj_b",
|
| 188 |
+
"model.language_model.layers.9.delta_attn.in_proj_a",
|
| 189 |
+
"model.language_model.layers.10.delta_attn.norm",
|
| 190 |
+
"model.language_model.layers.10.delta_attn.in_proj_b",
|
| 191 |
+
"model.language_model.layers.10.delta_attn.in_proj_a",
|
| 192 |
+
"model.language_model.layers.12.delta_attn.norm",
|
| 193 |
+
"model.language_model.layers.12.delta_attn.in_proj_b",
|
| 194 |
+
"model.language_model.layers.12.delta_attn.in_proj_a",
|
| 195 |
+
"model.language_model.layers.13.delta_attn.norm",
|
| 196 |
+
"model.language_model.layers.13.delta_attn.in_proj_b",
|
| 197 |
+
"model.language_model.layers.13.delta_attn.in_proj_a",
|
| 198 |
+
"model.language_model.layers.14.delta_attn.norm",
|
| 199 |
+
"model.language_model.layers.14.delta_attn.in_proj_b",
|
| 200 |
+
"model.language_model.layers.14.delta_attn.in_proj_a",
|
| 201 |
+
"model.language_model.layers.16.delta_attn.norm",
|
| 202 |
+
"model.language_model.layers.16.delta_attn.in_proj_b",
|
| 203 |
+
"model.language_model.layers.16.delta_attn.in_proj_a",
|
| 204 |
+
"model.language_model.layers.17.delta_attn.norm",
|
| 205 |
+
"model.language_model.layers.17.delta_attn.in_proj_b",
|
| 206 |
+
"model.language_model.layers.17.delta_attn.in_proj_a",
|
| 207 |
+
"model.language_model.layers.18.delta_attn.norm",
|
| 208 |
+
"model.language_model.layers.18.delta_attn.in_proj_b",
|
| 209 |
+
"model.language_model.layers.18.delta_attn.in_proj_a",
|
| 210 |
+
"model.language_model.layers.20.delta_attn.norm",
|
| 211 |
+
"model.language_model.layers.20.delta_attn.in_proj_b",
|
| 212 |
+
"model.language_model.layers.20.delta_attn.in_proj_a",
|
| 213 |
+
"model.language_model.layers.21.delta_attn.norm",
|
| 214 |
+
"model.language_model.layers.21.delta_attn.in_proj_b",
|
| 215 |
+
"model.language_model.layers.21.delta_attn.in_proj_a",
|
| 216 |
+
"model.language_model.layers.22.delta_attn.norm",
|
| 217 |
+
"model.language_model.layers.22.delta_attn.in_proj_b",
|
| 218 |
+
"model.language_model.layers.22.delta_attn.in_proj_a",
|
| 219 |
+
"model.language_model.layers.24.delta_attn.norm",
|
| 220 |
+
"model.language_model.layers.24.delta_attn.in_proj_b",
|
| 221 |
+
"model.language_model.layers.24.delta_attn.in_proj_a",
|
| 222 |
+
"model.language_model.layers.25.delta_attn.norm",
|
| 223 |
+
"model.language_model.layers.25.delta_attn.in_proj_b",
|
| 224 |
+
"model.language_model.layers.25.delta_attn.in_proj_a",
|
| 225 |
+
"model.language_model.layers.26.delta_attn.norm",
|
| 226 |
+
"model.language_model.layers.26.delta_attn.in_proj_b",
|
| 227 |
+
"model.language_model.layers.26.delta_attn.in_proj_a",
|
| 228 |
+
"model.language_model.layers.28.delta_attn.norm",
|
| 229 |
+
"model.language_model.layers.28.delta_attn.in_proj_b",
|
| 230 |
+
"model.language_model.layers.28.delta_attn.in_proj_a",
|
| 231 |
+
"model.language_model.layers.29.delta_attn.norm",
|
| 232 |
+
"model.language_model.layers.29.delta_attn.in_proj_b",
|
| 233 |
+
"model.language_model.layers.29.delta_attn.in_proj_a",
|
| 234 |
+
"model.language_model.layers.30.delta_attn.norm",
|
| 235 |
+
"model.language_model.layers.30.delta_attn.in_proj_b",
|
| 236 |
+
"model.language_model.layers.30.delta_attn.in_proj_a",
|
| 237 |
+
"model.language_model.layers.32.delta_attn.norm",
|
| 238 |
+
"model.language_model.layers.32.delta_attn.in_proj_b",
|
| 239 |
+
"model.language_model.layers.32.delta_attn.in_proj_a",
|
| 240 |
+
"model.language_model.layers.33.delta_attn.norm",
|
| 241 |
+
"model.language_model.layers.33.delta_attn.in_proj_b",
|
| 242 |
+
"model.language_model.layers.33.delta_attn.in_proj_a",
|
| 243 |
+
"model.language_model.layers.34.delta_attn.norm",
|
| 244 |
+
"model.language_model.layers.34.delta_attn.in_proj_b",
|
| 245 |
+
"model.language_model.layers.34.delta_attn.in_proj_a",
|
| 246 |
+
"model.language_model.layers.36.delta_attn.norm",
|
| 247 |
+
"model.language_model.layers.36.delta_attn.in_proj_b",
|
| 248 |
+
"model.language_model.layers.36.delta_attn.in_proj_a",
|
| 249 |
+
"model.language_model.layers.37.delta_attn.norm",
|
| 250 |
+
"model.language_model.layers.37.delta_attn.in_proj_b",
|
| 251 |
+
"model.language_model.layers.37.delta_attn.in_proj_a",
|
| 252 |
+
"model.language_model.layers.38.delta_attn.norm",
|
| 253 |
+
"model.language_model.layers.38.delta_attn.in_proj_b",
|
| 254 |
+
"model.language_model.layers.38.delta_attn.in_proj_a",
|
| 255 |
+
"model.language_model.layers.40.delta_attn.norm",
|
| 256 |
+
"model.language_model.layers.40.delta_attn.in_proj_b",
|
| 257 |
+
"model.language_model.layers.40.delta_attn.in_proj_a",
|
| 258 |
+
"model.language_model.layers.41.delta_attn.norm",
|
| 259 |
+
"model.language_model.layers.41.delta_attn.in_proj_b",
|
| 260 |
+
"model.language_model.layers.41.delta_attn.in_proj_a",
|
| 261 |
+
"model.language_model.layers.42.delta_attn.norm",
|
| 262 |
+
"model.language_model.layers.42.delta_attn.in_proj_b",
|
| 263 |
+
"model.language_model.layers.42.delta_attn.in_proj_a",
|
| 264 |
+
"model.language_model.layers.44.delta_attn.norm",
|
| 265 |
+
"model.language_model.layers.44.delta_attn.in_proj_b",
|
| 266 |
+
"model.language_model.layers.44.delta_attn.in_proj_a",
|
| 267 |
+
"model.language_model.layers.45.delta_attn.norm",
|
| 268 |
+
"model.language_model.layers.45.delta_attn.in_proj_b",
|
| 269 |
+
"model.language_model.layers.45.delta_attn.in_proj_a",
|
| 270 |
+
"model.language_model.layers.46.delta_attn.norm",
|
| 271 |
+
"model.language_model.layers.46.delta_attn.in_proj_b",
|
| 272 |
+
"model.language_model.layers.46.delta_attn.in_proj_a",
|
| 273 |
+
"model.language_model.layers.48.delta_attn.norm",
|
| 274 |
+
"model.language_model.layers.48.delta_attn.in_proj_b",
|
| 275 |
+
"model.language_model.layers.48.delta_attn.in_proj_a",
|
| 276 |
+
"model.language_model.layers.49.delta_attn.norm",
|
| 277 |
+
"model.language_model.layers.49.delta_attn.in_proj_b",
|
| 278 |
+
"model.language_model.layers.49.delta_attn.in_proj_a",
|
| 279 |
+
"model.language_model.layers.50.delta_attn.norm",
|
| 280 |
+
"model.language_model.layers.50.delta_attn.in_proj_b",
|
| 281 |
+
"model.language_model.layers.50.delta_attn.in_proj_a",
|
| 282 |
+
"model.language_model.layers.52.delta_attn.norm",
|
| 283 |
+
"model.language_model.layers.52.delta_attn.in_proj_b",
|
| 284 |
+
"model.language_model.layers.52.delta_attn.in_proj_a",
|
| 285 |
+
"model.language_model.layers.53.delta_attn.norm",
|
| 286 |
+
"model.language_model.layers.53.delta_attn.in_proj_b",
|
| 287 |
+
"model.language_model.layers.53.delta_attn.in_proj_a",
|
| 288 |
+
"model.language_model.layers.54.delta_attn.norm",
|
| 289 |
+
"model.language_model.layers.54.delta_attn.in_proj_b",
|
| 290 |
+
"model.language_model.layers.54.delta_attn.in_proj_a",
|
| 291 |
+
"model.language_model.layers.56.delta_attn.norm",
|
| 292 |
+
"model.language_model.layers.56.delta_attn.in_proj_b",
|
| 293 |
+
"model.language_model.layers.56.delta_attn.in_proj_a",
|
| 294 |
+
"model.language_model.layers.57.delta_attn.norm",
|
| 295 |
+
"model.language_model.layers.57.delta_attn.in_proj_b",
|
| 296 |
+
"model.language_model.layers.57.delta_attn.in_proj_a",
|
| 297 |
+
"model.language_model.layers.58.delta_attn.norm",
|
| 298 |
+
"model.language_model.layers.58.delta_attn.in_proj_b",
|
| 299 |
+
"model.language_model.layers.58.delta_attn.in_proj_a",
|
| 300 |
+
"model.language_model.layers.60.delta_attn.norm",
|
| 301 |
+
"model.language_model.layers.60.delta_attn.in_proj_b",
|
| 302 |
+
"model.language_model.layers.60.delta_attn.in_proj_a",
|
| 303 |
+
"model.language_model.layers.61.delta_attn.norm",
|
| 304 |
+
"model.language_model.layers.61.delta_attn.in_proj_b",
|
| 305 |
+
"model.language_model.layers.61.delta_attn.in_proj_a",
|
| 306 |
+
"model.language_model.layers.62.delta_attn.norm",
|
| 307 |
+
"model.language_model.layers.62.delta_attn.in_proj_b",
|
| 308 |
+
"model.language_model.layers.62.delta_attn.in_proj_a",
|
| 309 |
+
"model.language_model.layers.64.delta_attn.norm",
|
| 310 |
+
"model.language_model.layers.64.delta_attn.in_proj_b",
|
| 311 |
+
"model.language_model.layers.64.delta_attn.in_proj_a",
|
| 312 |
+
"model.language_model.layers.65.delta_attn.norm",
|
| 313 |
+
"model.language_model.layers.65.delta_attn.in_proj_b",
|
| 314 |
+
"model.language_model.layers.65.delta_attn.in_proj_a",
|
| 315 |
+
"model.language_model.layers.66.delta_attn.norm",
|
| 316 |
+
"model.language_model.layers.66.delta_attn.in_proj_b",
|
| 317 |
+
"model.language_model.layers.66.delta_attn.in_proj_a",
|
| 318 |
+
"model.language_model.layers.68.delta_attn.norm",
|
| 319 |
+
"model.language_model.layers.68.delta_attn.in_proj_b",
|
| 320 |
+
"model.language_model.layers.68.delta_attn.in_proj_a",
|
| 321 |
+
"model.language_model.layers.69.delta_attn.norm",
|
| 322 |
+
"model.language_model.layers.69.delta_attn.in_proj_b",
|
| 323 |
+
"model.language_model.layers.69.delta_attn.in_proj_a",
|
| 324 |
+
"model.language_model.layers.70.delta_attn.norm",
|
| 325 |
+
"model.language_model.layers.70.delta_attn.in_proj_b",
|
| 326 |
+
"model.language_model.layers.70.delta_attn.in_proj_a",
|
| 327 |
+
"lm_head"
|
| 328 |
+
],
|
| 329 |
+
"kv_cache_scheme": null,
|
| 330 |
+
"quant_method": "compressed-tensors",
|
| 331 |
+
"quantization_status": "compressed",
|
| 332 |
+
"sparsity_config": {},
|
| 333 |
+
"transform_config": {},
|
| 334 |
+
"version": "0.18.0"
|
| 335 |
+
},
|
| 336 |
+
"text_config": {
|
| 337 |
+
"attention_bias": false,
|
| 338 |
+
"attention_dropout": 0.0,
|
| 339 |
+
"attn_output_gate": true,
|
| 340 |
+
"bos_token_id": 248044,
|
| 341 |
+
"dtype": "bfloat16",
|
| 342 |
+
"eos_token_id": 248044,
|
| 343 |
+
"global_attention_interval": 4,
|
| 344 |
+
"head_dim": 256,
|
| 345 |
+
"hidden_act": "silu",
|
| 346 |
+
"hidden_size": 5120,
|
| 347 |
+
"initializer_range": 0.02,
|
| 348 |
+
"intermediate_size": 19456,
|
| 349 |
+
"layer_types": [
|
| 350 |
+
"agnes_delta_attention",
|
| 351 |
+
"agnes_delta_attention",
|
| 352 |
+
"agnes_delta_attention",
|
| 353 |
+
"agnes_global_attention",
|
| 354 |
+
"agnes_delta_attention",
|
| 355 |
+
"agnes_delta_attention",
|
| 356 |
+
"agnes_delta_attention",
|
| 357 |
+
"agnes_global_attention",
|
| 358 |
+
"agnes_delta_attention",
|
| 359 |
+
"agnes_delta_attention",
|
| 360 |
+
"agnes_delta_attention",
|
| 361 |
+
"agnes_global_attention",
|
| 362 |
+
"agnes_delta_attention",
|
| 363 |
+
"agnes_delta_attention",
|
| 364 |
+
"agnes_delta_attention",
|
| 365 |
+
"agnes_global_attention",
|
| 366 |
+
"agnes_delta_attention",
|
| 367 |
+
"agnes_delta_attention",
|
| 368 |
+
"agnes_delta_attention",
|
| 369 |
+
"agnes_global_attention",
|
| 370 |
+
"agnes_delta_attention",
|
| 371 |
+
"agnes_delta_attention",
|
| 372 |
+
"agnes_delta_attention",
|
| 373 |
+
"agnes_global_attention",
|
| 374 |
+
"agnes_delta_attention",
|
| 375 |
+
"agnes_delta_attention",
|
| 376 |
+
"agnes_delta_attention",
|
| 377 |
+
"agnes_global_attention",
|
| 378 |
+
"agnes_delta_attention",
|
| 379 |
+
"agnes_delta_attention",
|
| 380 |
+
"agnes_delta_attention",
|
| 381 |
+
"agnes_global_attention",
|
| 382 |
+
"agnes_delta_attention",
|
| 383 |
+
"agnes_delta_attention",
|
| 384 |
+
"agnes_delta_attention",
|
| 385 |
+
"agnes_global_attention",
|
| 386 |
+
"agnes_delta_attention",
|
| 387 |
+
"agnes_delta_attention",
|
| 388 |
+
"agnes_delta_attention",
|
| 389 |
+
"agnes_global_attention",
|
| 390 |
+
"agnes_delta_attention",
|
| 391 |
+
"agnes_delta_attention",
|
| 392 |
+
"agnes_delta_attention",
|
| 393 |
+
"agnes_global_attention",
|
| 394 |
+
"agnes_delta_attention",
|
| 395 |
+
"agnes_delta_attention",
|
| 396 |
+
"agnes_delta_attention",
|
| 397 |
+
"agnes_global_attention",
|
| 398 |
+
"agnes_delta_attention",
|
| 399 |
+
"agnes_delta_attention",
|
| 400 |
+
"agnes_delta_attention",
|
| 401 |
+
"agnes_global_attention",
|
| 402 |
+
"agnes_delta_attention",
|
| 403 |
+
"agnes_delta_attention",
|
| 404 |
+
"agnes_delta_attention",
|
| 405 |
+
"agnes_global_attention",
|
| 406 |
+
"agnes_delta_attention",
|
| 407 |
+
"agnes_delta_attention",
|
| 408 |
+
"agnes_delta_attention",
|
| 409 |
+
"agnes_global_attention",
|
| 410 |
+
"agnes_delta_attention",
|
| 411 |
+
"agnes_delta_attention",
|
| 412 |
+
"agnes_delta_attention",
|
| 413 |
+
"agnes_global_attention",
|
| 414 |
+
"agnes_delta_attention",
|
| 415 |
+
"agnes_delta_attention",
|
| 416 |
+
"agnes_delta_attention",
|
| 417 |
+
"agnes_global_attention",
|
| 418 |
+
"agnes_delta_attention",
|
| 419 |
+
"agnes_delta_attention",
|
| 420 |
+
"agnes_delta_attention",
|
| 421 |
+
"agnes_global_attention"
|
| 422 |
+
],
|
| 423 |
+
"linear_conv_kernel_dim": 4,
|
| 424 |
+
"linear_key_head_dim": 128,
|
| 425 |
+
"linear_num_key_heads": 16,
|
| 426 |
+
"linear_num_value_heads": 48,
|
| 427 |
+
"linear_value_head_dim": 128,
|
| 428 |
+
"mamba_ssm_dtype": "float32",
|
| 429 |
+
"max_position_embeddings": 262144,
|
| 430 |
+
"model_type": "agnes_text",
|
| 431 |
+
"mtp_num_hidden_layers": 1,
|
| 432 |
+
"mtp_use_dedicated_embeddings": false,
|
| 433 |
+
"num_attention_heads": 24,
|
| 434 |
+
"num_hidden_layers": 72,
|
| 435 |
+
"num_key_value_heads": 4,
|
| 436 |
+
"output_gate_type": "swish",
|
| 437 |
+
"pad_token_id": null,
|
| 438 |
+
"parallel_ffn_intermediate_size": 0,
|
| 439 |
+
"partial_rotary_factor": 0.25,
|
| 440 |
+
"rms_norm_eps": 1e-06,
|
| 441 |
+
"rope_parameters": {
|
| 442 |
+
"mrope_interleaved": true,
|
| 443 |
+
"mrope_section": [
|
| 444 |
+
11,
|
| 445 |
+
11,
|
| 446 |
+
10
|
| 447 |
+
],
|
| 448 |
+
"partial_rotary_factor": 0.25,
|
| 449 |
+
"rope_theta": 10000000,
|
| 450 |
+
"rope_type": "default"
|
| 451 |
+
},
|
| 452 |
+
"tie_word_embeddings": false,
|
| 453 |
+
"use_cache": true,
|
| 454 |
+
"vocab_size": 248320
|
| 455 |
+
},
|
| 456 |
+
"tie_word_embeddings": false,
|
| 457 |
+
"transformers_version": "5.13.1",
|
| 458 |
+
"video_token_id": 248057,
|
| 459 |
+
"vision_config": {
|
| 460 |
+
"deepstack_visual_indexes": [],
|
| 461 |
+
"depth": 27,
|
| 462 |
+
"dtype": "bfloat16",
|
| 463 |
+
"hidden_act": "gelu_pytorch_tanh",
|
| 464 |
+
"hidden_size": 1152,
|
| 465 |
+
"in_channels": 3,
|
| 466 |
+
"initializer_range": 0.02,
|
| 467 |
+
"intermediate_size": 4304,
|
| 468 |
+
"model_type": "agnes_vision",
|
| 469 |
+
"num_heads": 16,
|
| 470 |
+
"num_position_embeddings": 2304,
|
| 471 |
+
"out_hidden_size": 5120,
|
| 472 |
+
"patch_size": 16,
|
| 473 |
+
"spatial_merge_size": 2,
|
| 474 |
+
"temporal_patch_size": 2
|
| 475 |
+
},
|
| 476 |
+
"vision_end_token_id": 248054,
|
| 477 |
+
"vision_start_token_id": 248053
|
| 478 |
+
}
|
configuration_agnes.py
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2026 Agnes AI. All rights reserved.
|
| 2 |
+
#
|
| 3 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 4 |
+
# you may not use this file except in compliance with the License.
|
| 5 |
+
# You may obtain a copy of the License at
|
| 6 |
+
#
|
| 7 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
#
|
| 9 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
# See the License for the specific language governing permissions and
|
| 13 |
+
# limitations under the License.
|
| 14 |
+
"""Agnes 3.0 Flash model configuration."""
|
| 15 |
+
|
| 16 |
+
from huggingface_hub.dataclasses import strict
|
| 17 |
+
|
| 18 |
+
from transformers.configuration_utils import PreTrainedConfig
|
| 19 |
+
from transformers.modeling_rope_utils import RopeParameters
|
| 20 |
+
from transformers.utils import auto_docstring
|
| 21 |
+
|
| 22 |
+
|
| 23 |
+
LAYER_GLOBAL = "agnes_global_attention"
|
| 24 |
+
LAYER_DELTA = "agnes_delta_attention"
|
| 25 |
+
LAYER_TYPES = (LAYER_GLOBAL, LAYER_DELTA)
|
| 26 |
+
|
| 27 |
+
|
| 28 |
+
@auto_docstring
|
| 29 |
+
@strict
|
| 30 |
+
class AgnesTextConfig(PreTrainedConfig):
|
| 31 |
+
r"""
|
| 32 |
+
linear_num_key_heads (`int`, *optional*, defaults to 16):
|
| 33 |
+
Number of key heads in the delta-rule (recurrent) attention layers.
|
| 34 |
+
linear_num_value_heads (`int`, *optional*, defaults to 32):
|
| 35 |
+
Number of value heads in the delta-rule attention layers.
|
| 36 |
+
linear_key_head_dim (`int`, *optional*, defaults to 128):
|
| 37 |
+
Per-head key width in the delta-rule attention layers.
|
| 38 |
+
linear_value_head_dim (`int`, *optional*, defaults to 128):
|
| 39 |
+
Per-head value width in the delta-rule attention layers.
|
| 40 |
+
linear_conv_kernel_dim (`int`, *optional*, defaults to 4):
|
| 41 |
+
Kernel width of the causal depthwise convolution feeding the delta-rule layers.
|
| 42 |
+
parallel_ffn_intermediate_size (`int`, *optional*, defaults to 0):
|
| 43 |
+
Width of the dense SwiGLU branch that runs beside the main MLP in every
|
| 44 |
+
decoder layer and is summed into the same residual. 0 disables the branch.
|
| 45 |
+
"""
|
| 46 |
+
|
| 47 |
+
model_type = "agnes_text"
|
| 48 |
+
base_config_key = "text_config"
|
| 49 |
+
keys_to_ignore_at_inference = ["past_key_values"]
|
| 50 |
+
ignore_keys_at_rope_validation = {"mrope_section", "mrope_interleaved"}
|
| 51 |
+
|
| 52 |
+
# core dimensions
|
| 53 |
+
hidden_size: int = 4096
|
| 54 |
+
num_hidden_layers: int = 32
|
| 55 |
+
intermediate_size: int = 12288
|
| 56 |
+
parallel_ffn_intermediate_size: int = 0
|
| 57 |
+
vocab_size: int = 248320
|
| 58 |
+
|
| 59 |
+
# layer plan
|
| 60 |
+
layer_types: list[str] | None = None
|
| 61 |
+
|
| 62 |
+
# global attention
|
| 63 |
+
num_attention_heads: int = 16
|
| 64 |
+
num_key_value_heads: int = 4
|
| 65 |
+
head_dim: int = 256
|
| 66 |
+
attention_bias: bool = False
|
| 67 |
+
attention_dropout: float | int = 0.0
|
| 68 |
+
|
| 69 |
+
# delta-rule attention
|
| 70 |
+
linear_num_key_heads: int = 16
|
| 71 |
+
linear_num_value_heads: int = 32
|
| 72 |
+
linear_key_head_dim: int = 128
|
| 73 |
+
linear_value_head_dim: int = 128
|
| 74 |
+
linear_conv_kernel_dim: int = 4
|
| 75 |
+
|
| 76 |
+
# ffn / norm / init
|
| 77 |
+
hidden_act: str = "silu"
|
| 78 |
+
rms_norm_eps: float = 1e-6
|
| 79 |
+
initializer_range: float = 0.02
|
| 80 |
+
|
| 81 |
+
# positions
|
| 82 |
+
rope_parameters: RopeParameters | dict | None = None
|
| 83 |
+
max_position_embeddings: int = 32768
|
| 84 |
+
|
| 85 |
+
# tokens / runtime
|
| 86 |
+
bos_token_id: int | None = None
|
| 87 |
+
eos_token_id: int | list[int] | None = None
|
| 88 |
+
pad_token_id: int | None = None
|
| 89 |
+
tie_word_embeddings: bool = False
|
| 90 |
+
use_cache: bool = True
|
| 91 |
+
|
| 92 |
+
base_model_pp_plan = {
|
| 93 |
+
"embed_tokens": (["input_ids"], ["inputs_embeds"]),
|
| 94 |
+
"layers": (["hidden_states", "attention_mask"], ["hidden_states"]),
|
| 95 |
+
"norm": (["hidden_states"], ["hidden_states"]),
|
| 96 |
+
}
|
| 97 |
+
base_model_tp_plan = {
|
| 98 |
+
"layers.*.global_attn.q_proj": "colwise",
|
| 99 |
+
"layers.*.global_attn.k_proj": "colwise",
|
| 100 |
+
"layers.*.global_attn.v_proj": "colwise",
|
| 101 |
+
"layers.*.global_attn.o_proj": "rowwise",
|
| 102 |
+
"layers.*.global_attn.q_norm": "replicated_with_grad_allreduce",
|
| 103 |
+
"layers.*.global_attn.k_norm": "replicated_with_grad_allreduce",
|
| 104 |
+
"layers.*.mlp.gate_proj": "colwise",
|
| 105 |
+
"layers.*.mlp.up_proj": "colwise",
|
| 106 |
+
"layers.*.mlp.down_proj": "rowwise",
|
| 107 |
+
"layers.*.mlp.parallel_ffn.gate_proj": "colwise",
|
| 108 |
+
"layers.*.mlp.parallel_ffn.up_proj": "colwise",
|
| 109 |
+
"layers.*.mlp.parallel_ffn.down_proj": "rowwise",
|
| 110 |
+
}
|
| 111 |
+
|
| 112 |
+
def __post_init__(self, **kwargs):
|
| 113 |
+
# rotary embedding covers a quarter of each head unless told otherwise
|
| 114 |
+
kwargs.setdefault("partial_rotary_factor", 0.25)
|
| 115 |
+
if self.layer_types is None:
|
| 116 |
+
period = kwargs.pop("global_attention_interval", kwargs.pop("full_attention_interval", 4))
|
| 117 |
+
plan = []
|
| 118 |
+
for i in range(self.num_hidden_layers):
|
| 119 |
+
is_global = (i + 1) % period == 0
|
| 120 |
+
plan.append(LAYER_GLOBAL if is_global else LAYER_DELTA)
|
| 121 |
+
self.layer_types = plan
|
| 122 |
+
super().__post_init__(**kwargs)
|
| 123 |
+
|
| 124 |
+
def validate_layer_type(self):
|
| 125 |
+
"""The layer plan uses this model's own type names, so the generic
|
| 126 |
+
allow-list in the base class does not apply."""
|
| 127 |
+
plan = self.layer_types
|
| 128 |
+
if plan is None:
|
| 129 |
+
return
|
| 130 |
+
unknown = sorted({t for t in plan if t not in LAYER_TYPES})
|
| 131 |
+
if unknown:
|
| 132 |
+
raise ValueError(f"`layer_types` entries must be in {LAYER_TYPES}, got {unknown}")
|
| 133 |
+
if self.num_hidden_layers is not None and self.num_hidden_layers != len(plan):
|
| 134 |
+
raise ValueError(
|
| 135 |
+
f"`num_hidden_layers` ({self.num_hidden_layers}) must equal the number of `layer_types` ({len(plan)})"
|
| 136 |
+
)
|
| 137 |
+
|
| 138 |
+
|
| 139 |
+
@auto_docstring
|
| 140 |
+
@strict
|
| 141 |
+
class AgnesVisionConfig(PreTrainedConfig):
|
| 142 |
+
r"""
|
| 143 |
+
out_hidden_size (`int`, *optional*, defaults to 3584):
|
| 144 |
+
Width of the merged visual tokens handed to the language model.
|
| 145 |
+
num_position_embeddings (`int`, *optional*, defaults to 2304):
|
| 146 |
+
Size of the learned 2-D position table (a square grid, side = sqrt of this).
|
| 147 |
+
"""
|
| 148 |
+
|
| 149 |
+
model_type = "agnes_vision"
|
| 150 |
+
base_config_key = "vision_config"
|
| 151 |
+
|
| 152 |
+
depth: int = 27
|
| 153 |
+
hidden_size: int = 1152
|
| 154 |
+
intermediate_size: int = 4304
|
| 155 |
+
num_heads: int = 16
|
| 156 |
+
out_hidden_size: int = 3584
|
| 157 |
+
patch_size: int | list[int] | tuple[int, int] = 16
|
| 158 |
+
spatial_merge_size: int = 2
|
| 159 |
+
temporal_patch_size: int | list[int] | tuple[int, int] = 2
|
| 160 |
+
in_channels: int = 3
|
| 161 |
+
num_position_embeddings: int = 2304
|
| 162 |
+
hidden_act: str = "gelu_pytorch_tanh"
|
| 163 |
+
initializer_range: float = 0.02
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
@auto_docstring
|
| 167 |
+
@strict
|
| 168 |
+
class AgnesConfig(PreTrainedConfig):
|
| 169 |
+
r"""
|
| 170 |
+
Example:
|
| 171 |
+
|
| 172 |
+
```python
|
| 173 |
+
>>> from transformers import AutoConfig
|
| 174 |
+
|
| 175 |
+
>>> configuration = AutoConfig.from_pretrained("<model dir>", trust_remote_code=True)
|
| 176 |
+
>>> configuration.text_config.num_hidden_layers
|
| 177 |
+
72
|
| 178 |
+
```"""
|
| 179 |
+
|
| 180 |
+
model_type = "agnes"
|
| 181 |
+
sub_configs = {"text_config": AgnesTextConfig, "vision_config": AgnesVisionConfig}
|
| 182 |
+
keys_to_ignore_at_inference = ["past_key_values"]
|
| 183 |
+
|
| 184 |
+
text_config: dict | PreTrainedConfig | None = None
|
| 185 |
+
vision_config: dict | PreTrainedConfig | None = None
|
| 186 |
+
|
| 187 |
+
image_token_id: int = 248056
|
| 188 |
+
video_token_id: int = 248057
|
| 189 |
+
vision_start_token_id: int = 248053
|
| 190 |
+
vision_end_token_id: int = 248054
|
| 191 |
+
tie_word_embeddings: bool = False
|
| 192 |
+
|
| 193 |
+
def __post_init__(self, **kwargs):
|
| 194 |
+
for key, cls in self.sub_configs.items():
|
| 195 |
+
value = getattr(self, key)
|
| 196 |
+
if isinstance(value, dict):
|
| 197 |
+
setattr(self, key, cls(**value))
|
| 198 |
+
elif value is None:
|
| 199 |
+
setattr(self, key, cls())
|
| 200 |
+
super().__post_init__(**kwargs)
|
| 201 |
+
|
| 202 |
+
|
| 203 |
+
__all__ = ["AgnesConfig", "AgnesTextConfig", "AgnesVisionConfig", "LAYER_GLOBAL", "LAYER_DELTA", "LAYER_TYPES"]
|
generation_config.json
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token_id": 248044,
|
| 3 |
+
"do_sample": true,
|
| 4 |
+
"eos_token_id": [
|
| 5 |
+
248046,
|
| 6 |
+
248044
|
| 7 |
+
],
|
| 8 |
+
"pad_token_id": 248044,
|
| 9 |
+
"temperature": 1.0,
|
| 10 |
+
"top_k": 20,
|
| 11 |
+
"top_p": 0.95
|
| 12 |
+
}
|
image_processing_agnes.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2026 Agnes AI. All rights reserved.
|
| 2 |
+
#
|
| 3 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 4 |
+
# you may not use this file except in compliance with the License.
|
| 5 |
+
# You may obtain a copy of the License at
|
| 6 |
+
#
|
| 7 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
#
|
| 9 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
# See the License for the specific language governing permissions and
|
| 13 |
+
# limitations under the License.
|
| 14 |
+
"""Image processor for Agnes 3.0 Flash: dynamic-resolution patching."""
|
| 15 |
+
|
| 16 |
+
import math
|
| 17 |
+
from collections.abc import Iterable
|
| 18 |
+
|
| 19 |
+
import torch
|
| 20 |
+
from torchvision.transforms.v2 import functional as tvF
|
| 21 |
+
|
| 22 |
+
from transformers.image_processing_backends import TorchvisionBackend
|
| 23 |
+
from transformers.image_processing_utils import BatchFeature
|
| 24 |
+
from transformers.image_transforms import group_images_by_shape, reorder_images
|
| 25 |
+
from transformers.image_utils import ImageInput, PILImageResampling, SizeDict
|
| 26 |
+
from transformers.processing_utils import ImagesKwargs, Unpack
|
| 27 |
+
from transformers.utils import TensorType, auto_docstring
|
| 28 |
+
|
| 29 |
+
|
| 30 |
+
class AgnesImageProcessorKwargs(ImagesKwargs, total=False):
|
| 31 |
+
r"""
|
| 32 |
+
min_pixels (`int`, *optional*, defaults to `256 * 256`):
|
| 33 |
+
Lower bound on the pixel count after resizing.
|
| 34 |
+
max_pixels (`int`, *optional*, defaults to `4096 * 4096`):
|
| 35 |
+
Upper bound on the pixel count after resizing.
|
| 36 |
+
patch_size (`int`, *optional*, defaults to 16):
|
| 37 |
+
Spatial patch size of the vision tower.
|
| 38 |
+
temporal_patch_size (`int`, *optional*, defaults to 2):
|
| 39 |
+
Temporal patch size of the vision tower (images are duplicated to fill it).
|
| 40 |
+
merge_size (`int`, *optional*, defaults to 2):
|
| 41 |
+
Side of the patch square merged into one language-model token.
|
| 42 |
+
"""
|
| 43 |
+
|
| 44 |
+
min_pixels: int
|
| 45 |
+
max_pixels: int
|
| 46 |
+
patch_size: int
|
| 47 |
+
temporal_patch_size: int
|
| 48 |
+
merge_size: int
|
| 49 |
+
|
| 50 |
+
|
| 51 |
+
def fit_to_grid(height: int, width: int, factor: int = 32, min_pixels: int = 256 * 256, max_pixels: int = 4096 * 4096):
|
| 52 |
+
"""Pick a (height, width) that is a multiple of `factor` on both sides,
|
| 53 |
+
keeps the pixel count inside [min_pixels, max_pixels] and stays as close
|
| 54 |
+
as possible to the original aspect ratio."""
|
| 55 |
+
if max(height, width) / min(height, width) > 200:
|
| 56 |
+
raise ValueError(f"absolute aspect ratio must be smaller than 200, got {max(height, width) / min(height, width)}")
|
| 57 |
+
h = round(height / factor) * factor
|
| 58 |
+
w = round(width / factor) * factor
|
| 59 |
+
if h * w > max_pixels:
|
| 60 |
+
scale = math.sqrt((height * width) / max_pixels)
|
| 61 |
+
h = max(factor, math.floor(height / scale / factor) * factor)
|
| 62 |
+
w = max(factor, math.floor(width / scale / factor) * factor)
|
| 63 |
+
elif h * w < min_pixels:
|
| 64 |
+
scale = math.sqrt(min_pixels / (height * width))
|
| 65 |
+
h = math.ceil(height * scale / factor) * factor
|
| 66 |
+
w = math.ceil(width * scale / factor) * factor
|
| 67 |
+
return h, w
|
| 68 |
+
|
| 69 |
+
|
| 70 |
+
@auto_docstring
|
| 71 |
+
class AgnesImageProcessor(TorchvisionBackend):
|
| 72 |
+
do_resize = True
|
| 73 |
+
resample = PILImageResampling.BICUBIC
|
| 74 |
+
size = {"shortest_edge": 256 * 256, "longest_edge": 4096 * 4096}
|
| 75 |
+
default_to_square = False
|
| 76 |
+
do_rescale = True
|
| 77 |
+
do_normalize = True
|
| 78 |
+
image_mean = [0.5, 0.5, 0.5]
|
| 79 |
+
image_std = [0.5, 0.5, 0.5]
|
| 80 |
+
do_convert_rgb = True
|
| 81 |
+
patch_size = 16
|
| 82 |
+
temporal_patch_size = 2
|
| 83 |
+
merge_size = 2
|
| 84 |
+
valid_kwargs = AgnesImageProcessorKwargs
|
| 85 |
+
model_input_names = ["pixel_values", "image_grid_thw"]
|
| 86 |
+
|
| 87 |
+
def __init__(self, **kwargs: Unpack[AgnesImageProcessorKwargs]):
|
| 88 |
+
size = kwargs.pop("size", None)
|
| 89 |
+
min_pixels = kwargs.pop("min_pixels", None)
|
| 90 |
+
max_pixels = kwargs.pop("max_pixels", None)
|
| 91 |
+
size = self.size if size is None else size
|
| 92 |
+
# min_pixels / max_pixels are the older spelling of the two size keys
|
| 93 |
+
if min_pixels is not None:
|
| 94 |
+
size["shortest_edge"] = min_pixels
|
| 95 |
+
size.pop("min_pixels", None)
|
| 96 |
+
if max_pixels is not None:
|
| 97 |
+
size["longest_edge"] = max_pixels
|
| 98 |
+
size.pop("max_pixels", None)
|
| 99 |
+
if "shortest_edge" not in size or "longest_edge" not in size:
|
| 100 |
+
raise ValueError("size must contain 'shortest_edge' and 'longest_edge' keys.")
|
| 101 |
+
super().__init__(size=size, **kwargs)
|
| 102 |
+
|
| 103 |
+
def _standardize_kwargs(
|
| 104 |
+
self,
|
| 105 |
+
size: int | Iterable[int] | dict[str, int] | SizeDict | None = None,
|
| 106 |
+
min_pixels: int | None = None,
|
| 107 |
+
max_pixels: int | None = None,
|
| 108 |
+
**kwargs,
|
| 109 |
+
) -> dict:
|
| 110 |
+
if min_pixels is not None and max_pixels is not None:
|
| 111 |
+
size = SizeDict(shortest_edge=min_pixels, longest_edge=max_pixels)
|
| 112 |
+
kwargs = super()._standardize_kwargs(size=size, **kwargs)
|
| 113 |
+
size = kwargs.get("size", self.size)
|
| 114 |
+
if not size.shortest_edge or not size.longest_edge:
|
| 115 |
+
raise ValueError("size must contain 'shortest_edge' and 'longest_edge' keys.")
|
| 116 |
+
return kwargs
|
| 117 |
+
|
| 118 |
+
@auto_docstring
|
| 119 |
+
def preprocess(self, images: ImageInput, **kwargs: Unpack[AgnesImageProcessorKwargs]) -> BatchFeature:
|
| 120 |
+
return super().preprocess(images, **kwargs)
|
| 121 |
+
|
| 122 |
+
def _preprocess(
|
| 123 |
+
self,
|
| 124 |
+
images: list["torch.Tensor"],
|
| 125 |
+
do_resize: bool,
|
| 126 |
+
size: SizeDict,
|
| 127 |
+
resample: "PILImageResampling | tvF.InterpolationMode | int | None",
|
| 128 |
+
do_rescale: bool,
|
| 129 |
+
rescale_factor: float,
|
| 130 |
+
do_normalize: bool,
|
| 131 |
+
image_mean: float | list[float] | None,
|
| 132 |
+
image_std: float | list[float] | None,
|
| 133 |
+
patch_size: int,
|
| 134 |
+
temporal_patch_size: int,
|
| 135 |
+
merge_size: int,
|
| 136 |
+
disable_grouping: bool | None,
|
| 137 |
+
return_tensors: str | TensorType | None,
|
| 138 |
+
**kwargs,
|
| 139 |
+
) -> BatchFeature:
|
| 140 |
+
# 1. resize, batched per input shape
|
| 141 |
+
by_shape, order = group_images_by_shape(images, disable_grouping=disable_grouping)
|
| 142 |
+
resized = {}
|
| 143 |
+
for shape, batch in by_shape.items():
|
| 144 |
+
height, width = batch.shape[-2:]
|
| 145 |
+
if do_resize:
|
| 146 |
+
new_h, new_w = fit_to_grid(
|
| 147 |
+
height, width, factor=patch_size * merge_size,
|
| 148 |
+
min_pixels=size.shortest_edge, max_pixels=size.longest_edge,
|
| 149 |
+
)
|
| 150 |
+
batch = self.resize(image=batch, size=SizeDict(height=new_h, width=new_w), resample=resample)
|
| 151 |
+
resized[shape] = batch
|
| 152 |
+
images = reorder_images(resized, order)
|
| 153 |
+
|
| 154 |
+
# 2. normalise and cut into merge-ordered patches, batched per resized shape
|
| 155 |
+
by_shape, order = group_images_by_shape(images, disable_grouping=disable_grouping)
|
| 156 |
+
flat = {}
|
| 157 |
+
grids = {}
|
| 158 |
+
for shape, batch in by_shape.items():
|
| 159 |
+
new_h, new_w = batch.shape[-2:]
|
| 160 |
+
px = self.rescale_and_normalize(batch, do_rescale, rescale_factor, do_normalize, image_mean, image_std)
|
| 161 |
+
n, c = px.shape[:2]
|
| 162 |
+
gh, gw = new_h // patch_size, new_w // patch_size
|
| 163 |
+
px = px.reshape(n, c, gh // merge_size, merge_size, patch_size, gw // merge_size, merge_size, patch_size)
|
| 164 |
+
# -> [n, gh/merge, gw/merge, merge, merge, c, patch, patch]: patches of one
|
| 165 |
+
# merge square end up adjacent in the flattened sequence
|
| 166 |
+
px = px.permute(0, 2, 5, 3, 6, 1, 4, 7)
|
| 167 |
+
px = (
|
| 168 |
+
px.unsqueeze(6)
|
| 169 |
+
.expand(-1, -1, -1, -1, -1, -1, temporal_patch_size, -1, -1)
|
| 170 |
+
.reshape(n, gh * gw, c * temporal_patch_size * patch_size * patch_size)
|
| 171 |
+
)
|
| 172 |
+
flat[shape] = px
|
| 173 |
+
grids[shape] = [[1, gh, gw]] * n
|
| 174 |
+
|
| 175 |
+
pixel_values = torch.cat(reorder_images(flat, order), dim=0)
|
| 176 |
+
image_grid_thw = torch.tensor(reorder_images(grids, order), dtype=torch.long)
|
| 177 |
+
return BatchFeature(data={"pixel_values": pixel_values, "image_grid_thw": image_grid_thw}, tensor_type=return_tensors)
|
| 178 |
+
|
| 179 |
+
def get_number_of_image_patches(self, height: int, width: int, images_kwargs=None):
|
| 180 |
+
"""Number of vision patches an image of this size produces (used by
|
| 181 |
+
serving engines to lay out placeholders without running the processor)."""
|
| 182 |
+
min_pixels = images_kwargs["min_pixels"] if "min_pixels" in images_kwargs else self.size["shortest_edge"]
|
| 183 |
+
max_pixels = images_kwargs["max_pixels"] if "max_pixels" in images_kwargs else self.size["longest_edge"]
|
| 184 |
+
patch_size = images_kwargs.get("patch_size", self.patch_size)
|
| 185 |
+
merge_size = images_kwargs.get("merge_size", self.merge_size)
|
| 186 |
+
new_h, new_w = fit_to_grid(height, width, patch_size * merge_size, min_pixels=min_pixels, max_pixels=max_pixels)
|
| 187 |
+
return (new_h // patch_size) * (new_w // patch_size)
|
| 188 |
+
|
| 189 |
+
|
| 190 |
+
__all__ = ["AgnesImageProcessor"]
|
model.safetensors.index.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
modeling_agnes.py
ADDED
|
@@ -0,0 +1,1559 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2026 Agnes AI. All rights reserved.
|
| 2 |
+
#
|
| 3 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 4 |
+
# you may not use this file except in compliance with the License.
|
| 5 |
+
# You may obtain a copy of the License at
|
| 6 |
+
#
|
| 7 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
#
|
| 9 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
# See the License for the specific language governing permissions and
|
| 13 |
+
# limitations under the License.
|
| 14 |
+
"""Agnes 3.0 Flash.
|
| 15 |
+
|
| 16 |
+
A hybrid decoder: three delta-rule recurrent layers followed by one global
|
| 17 |
+
attention layer, repeated; every layer carries a main SwiGLU MLP plus a
|
| 18 |
+
parallel SwiGLU branch summed into the same residual. A vision tower feeds
|
| 19 |
+
merged patch tokens into the language model through placeholder tokens.
|
| 20 |
+
"""
|
| 21 |
+
|
| 22 |
+
import itertools
|
| 23 |
+
from collections.abc import Callable
|
| 24 |
+
from dataclasses import dataclass
|
| 25 |
+
from typing import Any, Optional
|
| 26 |
+
|
| 27 |
+
import torch
|
| 28 |
+
import torch.nn.functional as F
|
| 29 |
+
from torch import nn
|
| 30 |
+
|
| 31 |
+
from transformers import initialization as init
|
| 32 |
+
from transformers.activations import ACT2FN
|
| 33 |
+
from transformers.cache_utils import LAYER_TYPE_CACHE_MAPPING, Cache, DynamicCache, DynamicLayer, LinearAttentionLayer
|
| 34 |
+
from transformers.generation import GenerationMixin
|
| 35 |
+
from transformers.masking_utils import LAYER_PATTERN_TO_MASK_FUNCTION_MAPPING, create_causal_mask
|
| 36 |
+
from transformers.modeling_flash_attention_utils import FlashAttentionKwargs
|
| 37 |
+
from transformers.modeling_layers import GradientCheckpointingLayer
|
| 38 |
+
from transformers.modeling_outputs import BaseModelOutputWithPast, BaseModelOutputWithPooling, CausalLMOutputWithPast
|
| 39 |
+
from transformers.modeling_rope_utils import ROPE_INIT_FUNCTIONS, dynamic_rope_update
|
| 40 |
+
from transformers.modeling_utils import ALL_ATTENTION_FUNCTIONS, PreTrainedModel
|
| 41 |
+
from transformers.processing_utils import Unpack
|
| 42 |
+
from transformers.utils import TransformersKwargs, auto_docstring, can_return_tuple, logging, torch_compilable_check
|
| 43 |
+
from transformers.utils.generic import (
|
| 44 |
+
accepts_precomputed_kwargs,
|
| 45 |
+
is_flash_attention_requested,
|
| 46 |
+
maybe_autocast,
|
| 47 |
+
merge_with_config_defaults,
|
| 48 |
+
)
|
| 49 |
+
from transformers.utils.import_utils import is_causal_conv1d_available, is_flash_linear_attention_available
|
| 50 |
+
from transformers.utils.output_capturing import capture_outputs
|
| 51 |
+
from transformers.vision_utils import get_vision_bilinear_indices_and_weights, get_vision_cu_seqlens, get_vision_position_ids
|
| 52 |
+
|
| 53 |
+
from .configuration_agnes import LAYER_DELTA, LAYER_GLOBAL, AgnesConfig, AgnesTextConfig, AgnesVisionConfig
|
| 54 |
+
|
| 55 |
+
|
| 56 |
+
logger = logging.get_logger(__name__)
|
| 57 |
+
|
| 58 |
+
# Optional fused kernels. When absent the pure-torch paths below are used;
|
| 59 |
+
# both paths are numerically interchangeable.
|
| 60 |
+
if is_causal_conv1d_available():
|
| 61 |
+
from causal_conv1d import causal_conv1d_fn, causal_conv1d_update
|
| 62 |
+
else:
|
| 63 |
+
causal_conv1d_fn = causal_conv1d_update = None
|
| 64 |
+
|
| 65 |
+
if is_flash_linear_attention_available():
|
| 66 |
+
from fla.modules import FusedRMSNormGated
|
| 67 |
+
from fla.ops.gated_delta_rule import chunk_gated_delta_rule, fused_recurrent_gated_delta_rule
|
| 68 |
+
else:
|
| 69 |
+
FusedRMSNormGated = None
|
| 70 |
+
chunk_gated_delta_rule = fused_recurrent_gated_delta_rule = None
|
| 71 |
+
|
| 72 |
+
_FUSED_DELTA_PATH = all((causal_conv1d_fn, causal_conv1d_update, chunk_gated_delta_rule, fused_recurrent_gated_delta_rule))
|
| 73 |
+
|
| 74 |
+
|
| 75 |
+
# ---------------------------------------------------------------------------
|
| 76 |
+
# Layer-type plumbing: the cache builds one cache layer per entry of
|
| 77 |
+
# config.layer_types through a registry keyed by type name, and generation
|
| 78 |
+
# picks a mask function the same way. Both are told about this model's two
|
| 79 |
+
# layer types here, at import time.
|
| 80 |
+
# ---------------------------------------------------------------------------
|
| 81 |
+
|
| 82 |
+
|
| 83 |
+
class AgnesGlobalCacheLayer(DynamicLayer):
|
| 84 |
+
layer_type = LAYER_GLOBAL
|
| 85 |
+
|
| 86 |
+
|
| 87 |
+
class AgnesDeltaCacheLayer(LinearAttentionLayer):
|
| 88 |
+
layer_type = LAYER_DELTA
|
| 89 |
+
|
| 90 |
+
|
| 91 |
+
# DynamicLayer subclasses self-register through CacheLayerMixin; the linear-attention
|
| 92 |
+
# base does not share that hook, and an unknown type silently falls back to a plain
|
| 93 |
+
# KV layer, so both entries are written explicitly.
|
| 94 |
+
LAYER_TYPE_CACHE_MAPPING[LAYER_GLOBAL] = AgnesGlobalCacheLayer
|
| 95 |
+
LAYER_TYPE_CACHE_MAPPING[LAYER_DELTA] = AgnesDeltaCacheLayer
|
| 96 |
+
LAYER_PATTERN_TO_MASK_FUNCTION_MAPPING.setdefault(LAYER_GLOBAL, create_causal_mask)
|
| 97 |
+
LAYER_PATTERN_TO_MASK_FUNCTION_MAPPING.setdefault(LAYER_DELTA, create_causal_mask)
|
| 98 |
+
|
| 99 |
+
|
| 100 |
+
# ---------------------------------------------------------------------------
|
| 101 |
+
# Normalisation
|
| 102 |
+
# ---------------------------------------------------------------------------
|
| 103 |
+
|
| 104 |
+
|
| 105 |
+
class _RMSNormFn(torch.autograd.Function):
|
| 106 |
+
"""RMS normalisation with a one-centred scale, y = x / rms(x) * (1 + w).
|
| 107 |
+
|
| 108 |
+
The forward matches the reference op-for-op (fp32 statistics, cast back to
|
| 109 |
+
the input dtype at the end); the backward is written out analytically
|
| 110 |
+
instead of being traced through the forward graph.
|
| 111 |
+
"""
|
| 112 |
+
|
| 113 |
+
@staticmethod
|
| 114 |
+
def forward(ctx, x, weight, eps):
|
| 115 |
+
xf = x.float()
|
| 116 |
+
inv = torch.rsqrt(xf.pow(2).mean(-1, keepdim=True) + eps)
|
| 117 |
+
unit = xf * inv
|
| 118 |
+
out = unit * (1.0 + weight.float())
|
| 119 |
+
ctx.save_for_backward(unit, inv, weight)
|
| 120 |
+
ctx.in_dtype = x.dtype
|
| 121 |
+
return out.type_as(x)
|
| 122 |
+
|
| 123 |
+
@staticmethod
|
| 124 |
+
def backward(ctx, grad_out):
|
| 125 |
+
unit, inv, weight = ctx.saved_tensors
|
| 126 |
+
g = grad_out.float()
|
| 127 |
+
g_unit = g * (1.0 + weight.float())
|
| 128 |
+
grad_x = inv * (g_unit - unit * (g_unit * unit).mean(-1, keepdim=True))
|
| 129 |
+
grad_w = (g * unit).reshape(-1, unit.shape[-1]).sum(0)
|
| 130 |
+
return grad_x.to(ctx.in_dtype), grad_w.to(weight.dtype), None
|
| 131 |
+
|
| 132 |
+
|
| 133 |
+
class AgnesRMSNorm(nn.Module):
|
| 134 |
+
def __init__(self, dim: int, eps: float = 1e-6):
|
| 135 |
+
super().__init__()
|
| 136 |
+
self.eps = eps
|
| 137 |
+
self.weight = nn.Parameter(torch.zeros(dim))
|
| 138 |
+
|
| 139 |
+
def forward(self, x):
|
| 140 |
+
return _RMSNormFn.apply(x, self.weight, self.eps)
|
| 141 |
+
|
| 142 |
+
def extra_repr(self):
|
| 143 |
+
return f"{tuple(self.weight.shape)}, eps={self.eps}"
|
| 144 |
+
|
| 145 |
+
|
| 146 |
+
class _GatedRMSNormFn(torch.autograd.Function):
|
| 147 |
+
"""RMS normalisation followed by a learned scale and a SiLU gate.
|
| 148 |
+
|
| 149 |
+
Forward order (matters for bit-exactness): fp32 statistics, cast to the
|
| 150 |
+
input dtype, multiply by the weight in that dtype, then multiply by the
|
| 151 |
+
fp32 gate activation and cast back.
|
| 152 |
+
"""
|
| 153 |
+
|
| 154 |
+
@staticmethod
|
| 155 |
+
def forward(ctx, x, weight, gate, eps):
|
| 156 |
+
in_dtype = x.dtype
|
| 157 |
+
xf = x.to(torch.float32)
|
| 158 |
+
inv = torch.rsqrt(xf.pow(2).mean(-1, keepdim=True) + eps)
|
| 159 |
+
unit = xf * inv
|
| 160 |
+
scaled = weight * unit.to(in_dtype)
|
| 161 |
+
gate_f = gate.to(torch.float32)
|
| 162 |
+
act = F.silu(gate_f)
|
| 163 |
+
out = scaled * act
|
| 164 |
+
ctx.save_for_backward(unit, inv, weight, gate_f, scaled)
|
| 165 |
+
ctx.in_dtype = in_dtype
|
| 166 |
+
return out.to(in_dtype)
|
| 167 |
+
|
| 168 |
+
@staticmethod
|
| 169 |
+
def backward(ctx, grad_out):
|
| 170 |
+
unit, inv, weight, gate_f, scaled = ctx.saved_tensors
|
| 171 |
+
g = grad_out.float()
|
| 172 |
+
sig = torch.sigmoid(gate_f)
|
| 173 |
+
act = gate_f * sig
|
| 174 |
+
g_scaled = g * act
|
| 175 |
+
grad_w = (g_scaled * unit).reshape(-1, unit.shape[-1]).sum(0)
|
| 176 |
+
g_unit = g_scaled * weight.float()
|
| 177 |
+
grad_x = inv * (g_unit - unit * (g_unit * unit).mean(-1, keepdim=True))
|
| 178 |
+
grad_gate = g * scaled.float() * (sig * (1.0 + gate_f * (1.0 - sig)))
|
| 179 |
+
return grad_x.to(ctx.in_dtype), grad_w.to(weight.dtype), grad_gate.to(ctx.in_dtype), None
|
| 180 |
+
|
| 181 |
+
|
| 182 |
+
class AgnesGatedNorm(nn.Module):
|
| 183 |
+
def __init__(self, hidden_size, eps=1e-6, **kwargs):
|
| 184 |
+
super().__init__()
|
| 185 |
+
self.weight = nn.Parameter(torch.ones(hidden_size))
|
| 186 |
+
self.variance_epsilon = eps
|
| 187 |
+
|
| 188 |
+
def forward(self, hidden_states, gate=None):
|
| 189 |
+
return _GatedRMSNormFn.apply(hidden_states, self.weight, gate, self.variance_epsilon)
|
| 190 |
+
|
| 191 |
+
|
| 192 |
+
# ---------------------------------------------------------------------------
|
| 193 |
+
# Feed-forward
|
| 194 |
+
# ---------------------------------------------------------------------------
|
| 195 |
+
|
| 196 |
+
|
| 197 |
+
class AgnesMLP(nn.Module):
|
| 198 |
+
"""SwiGLU block. With `parallel_size > 0` a second, narrower SwiGLU runs
|
| 199 |
+
on the same input and its output is added to the main one."""
|
| 200 |
+
|
| 201 |
+
def __init__(self, config, intermediate_size: int, parallel_size: int = 0):
|
| 202 |
+
super().__init__()
|
| 203 |
+
self.config = config
|
| 204 |
+
self.hidden_size = config.hidden_size
|
| 205 |
+
self.intermediate_size = intermediate_size
|
| 206 |
+
self.gate_proj = nn.Linear(self.hidden_size, intermediate_size, bias=False)
|
| 207 |
+
self.up_proj = nn.Linear(self.hidden_size, intermediate_size, bias=False)
|
| 208 |
+
self.down_proj = nn.Linear(intermediate_size, self.hidden_size, bias=False)
|
| 209 |
+
self.act_fn = ACT2FN[config.hidden_act]
|
| 210 |
+
self.parallel_ffn = AgnesMLP(config, parallel_size) if parallel_size > 0 else None
|
| 211 |
+
|
| 212 |
+
def forward(self, x):
|
| 213 |
+
y = self.down_proj(self.act_fn(self.gate_proj(x)) * self.up_proj(x))
|
| 214 |
+
if self.parallel_ffn is not None:
|
| 215 |
+
y = y + self.parallel_ffn(x)
|
| 216 |
+
return y
|
| 217 |
+
|
| 218 |
+
|
| 219 |
+
# ---------------------------------------------------------------------------
|
| 220 |
+
# Rotary position encodings
|
| 221 |
+
# ---------------------------------------------------------------------------
|
| 222 |
+
|
| 223 |
+
|
| 224 |
+
def _swap_halves(x):
|
| 225 |
+
"""(a, b) -> (-b, a) along the last axis."""
|
| 226 |
+
half = x.shape[-1] // 2
|
| 227 |
+
return torch.cat((-x[..., half:], x[..., :half]), dim=-1)
|
| 228 |
+
|
| 229 |
+
|
| 230 |
+
def _apply_rope(q, k, cos, sin, unsqueeze_dim=1):
|
| 231 |
+
"""Rotate the leading `cos.shape[-1]` channels of q and k; the rest pass through."""
|
| 232 |
+
cos = cos.unsqueeze(unsqueeze_dim)
|
| 233 |
+
sin = sin.unsqueeze(unsqueeze_dim)
|
| 234 |
+
n_rot = cos.shape[-1]
|
| 235 |
+
q_rot, q_rest = q[..., :n_rot], q[..., n_rot:]
|
| 236 |
+
k_rot, k_rest = k[..., :n_rot], k[..., n_rot:]
|
| 237 |
+
q_rot = (q_rot * cos) + (_swap_halves(q_rot) * sin)
|
| 238 |
+
k_rot = (k_rot * cos) + (_swap_halves(k_rot) * sin)
|
| 239 |
+
return torch.cat([q_rot, q_rest], dim=-1), torch.cat([k_rot, k_rest], dim=-1)
|
| 240 |
+
|
| 241 |
+
|
| 242 |
+
class AgnesRotaryEmbedding(nn.Module):
|
| 243 |
+
"""Three-axis (text / height / width) rotary tables with interleaved sections."""
|
| 244 |
+
|
| 245 |
+
inv_freq: torch.Tensor
|
| 246 |
+
|
| 247 |
+
def __init__(self, config: AgnesTextConfig, device=None):
|
| 248 |
+
super().__init__()
|
| 249 |
+
self.config = config
|
| 250 |
+
self.max_seq_len_cached = config.max_position_embeddings
|
| 251 |
+
self.original_max_seq_len = config.max_position_embeddings
|
| 252 |
+
self.rope_type = config.rope_parameters["rope_type"]
|
| 253 |
+
builder: Callable = self.compute_default_rope_parameters
|
| 254 |
+
if self.rope_type != "default":
|
| 255 |
+
builder = ROPE_INIT_FUNCTIONS[self.rope_type]
|
| 256 |
+
inv_freq, self.attention_scaling = builder(config, device)
|
| 257 |
+
self.register_buffer("inv_freq", inv_freq, persistent=False)
|
| 258 |
+
self.register_buffer("original_inv_freq", inv_freq.clone(), persistent=False)
|
| 259 |
+
self.mrope_section = config.rope_parameters.get("mrope_section", [11, 11, 10])
|
| 260 |
+
|
| 261 |
+
@staticmethod
|
| 262 |
+
def compute_default_rope_parameters(
|
| 263 |
+
config: AgnesTextConfig | None = None,
|
| 264 |
+
device: Optional["torch.device"] = None,
|
| 265 |
+
seq_len: int | None = None,
|
| 266 |
+
) -> tuple["torch.Tensor", float]:
|
| 267 |
+
theta = config.rope_parameters["rope_theta"]
|
| 268 |
+
fraction = config.rope_parameters.get("partial_rotary_factor", 1.0)
|
| 269 |
+
head_dim = getattr(config, "head_dim", None) or config.hidden_size // config.num_attention_heads
|
| 270 |
+
n_rot = int(head_dim * fraction)
|
| 271 |
+
exponents = torch.arange(0, n_rot, 2, dtype=torch.int64).to(device=device, dtype=torch.float) / n_rot
|
| 272 |
+
inv_freq = 1.0 / (theta**exponents)
|
| 273 |
+
return inv_freq, 1.0
|
| 274 |
+
|
| 275 |
+
@torch.no_grad()
|
| 276 |
+
@dynamic_rope_update
|
| 277 |
+
def forward(self, x, position_ids):
|
| 278 |
+
if position_ids.ndim == 2:
|
| 279 |
+
position_ids = position_ids[None, ...].expand(3, position_ids.shape[0], -1)
|
| 280 |
+
freq_col = self.inv_freq[None, None, :, None].float().expand(3, position_ids.shape[1], -1, 1).to(x.device)
|
| 281 |
+
pos_row = position_ids[:, :, None, :].float()
|
| 282 |
+
dev = x.device.type if isinstance(x.device.type, str) and x.device.type != "mps" else "cpu"
|
| 283 |
+
with maybe_autocast(device_type=dev, enabled=False):
|
| 284 |
+
angles = (freq_col.float() @ pos_row.float()).transpose(2, 3)
|
| 285 |
+
angles = self._interleave_axes(angles, self.mrope_section)
|
| 286 |
+
table = torch.cat((angles, angles), dim=-1)
|
| 287 |
+
cos = table.cos() * self.attention_scaling
|
| 288 |
+
sin = table.sin() * self.attention_scaling
|
| 289 |
+
return cos.to(dtype=x.dtype), sin.to(dtype=x.dtype)
|
| 290 |
+
|
| 291 |
+
@staticmethod
|
| 292 |
+
def _interleave_axes(angles, section):
|
| 293 |
+
"""Fold the H and W axes into the T table at every 2nd / 3rd frequency slot."""
|
| 294 |
+
merged = angles[0]
|
| 295 |
+
for axis, offset in ((1, 1), (2, 2)):
|
| 296 |
+
take = slice(offset, section[axis] * 3, 3)
|
| 297 |
+
merged[..., take] = angles[axis, ..., take]
|
| 298 |
+
return merged
|
| 299 |
+
|
| 300 |
+
|
| 301 |
+
class AgnesVisionRotary(nn.Module):
|
| 302 |
+
inv_freq: torch.Tensor
|
| 303 |
+
|
| 304 |
+
def __init__(self, dim: int, theta: float = 10000.0) -> None:
|
| 305 |
+
super().__init__()
|
| 306 |
+
self.dim = dim
|
| 307 |
+
self.theta = theta
|
| 308 |
+
inv_freq = 1.0 / (theta ** (torch.arange(0, dim, 2, dtype=torch.float) / dim))
|
| 309 |
+
self.register_buffer("inv_freq", inv_freq, persistent=False)
|
| 310 |
+
|
| 311 |
+
def forward(self, position_ids: torch.Tensor) -> torch.Tensor:
|
| 312 |
+
return (position_ids.unsqueeze(-1) * self.inv_freq).flatten(1)
|
| 313 |
+
|
| 314 |
+
|
| 315 |
+
def _apply_vision_rope(q, k, cos, sin):
|
| 316 |
+
q_dtype, k_dtype = q.dtype, k.dtype
|
| 317 |
+
q, k = q.float(), k.float()
|
| 318 |
+
cos, sin = cos.unsqueeze(-2).float(), sin.unsqueeze(-2).float()
|
| 319 |
+
q = (q * cos) + (_swap_halves(q) * sin)
|
| 320 |
+
k = (k * cos) + (_swap_halves(k) * sin)
|
| 321 |
+
return q.to(q_dtype), k.to(k_dtype)
|
| 322 |
+
|
| 323 |
+
|
| 324 |
+
# ---------------------------------------------------------------------------
|
| 325 |
+
# Global (softmax) attention
|
| 326 |
+
# ---------------------------------------------------------------------------
|
| 327 |
+
|
| 328 |
+
|
| 329 |
+
def _broadcast_kv(x: torch.Tensor, groups: int) -> torch.Tensor:
|
| 330 |
+
"""Repeat each kv head `groups` times along the head axis."""
|
| 331 |
+
b, h, t, d = x.shape
|
| 332 |
+
if groups == 1:
|
| 333 |
+
return x
|
| 334 |
+
return x[:, :, None, :, :].expand(b, h, groups, t, d).reshape(b, h * groups, t, d)
|
| 335 |
+
|
| 336 |
+
|
| 337 |
+
def _dense_attention(
|
| 338 |
+
module: nn.Module,
|
| 339 |
+
query: torch.Tensor,
|
| 340 |
+
key: torch.Tensor,
|
| 341 |
+
value: torch.Tensor,
|
| 342 |
+
attention_mask: torch.Tensor | None,
|
| 343 |
+
scaling: float,
|
| 344 |
+
dropout: float = 0.0,
|
| 345 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 346 |
+
):
|
| 347 |
+
k = _broadcast_kv(key, module.num_key_value_groups)
|
| 348 |
+
v = _broadcast_kv(value, module.num_key_value_groups)
|
| 349 |
+
scores = torch.matmul(query, k.transpose(2, 3)) * scaling
|
| 350 |
+
if attention_mask is not None:
|
| 351 |
+
scores = scores + attention_mask
|
| 352 |
+
probs = nn.functional.softmax(scores, dim=-1, dtype=torch.float32).to(query.dtype)
|
| 353 |
+
probs = nn.functional.dropout(probs, p=dropout, training=module.training)
|
| 354 |
+
out = torch.matmul(probs, v).transpose(1, 2).contiguous()
|
| 355 |
+
return out, probs
|
| 356 |
+
|
| 357 |
+
|
| 358 |
+
class AgnesGlobalAttention(nn.Module):
|
| 359 |
+
"""Grouped-query softmax attention with per-head q/k normalisation and a
|
| 360 |
+
sigmoid output gate produced alongside the queries."""
|
| 361 |
+
|
| 362 |
+
def __init__(self, config: AgnesTextConfig, layer_idx: int):
|
| 363 |
+
super().__init__()
|
| 364 |
+
self.config = config
|
| 365 |
+
self.layer_idx = layer_idx
|
| 366 |
+
self.head_dim = getattr(config, "head_dim", config.hidden_size // config.num_attention_heads)
|
| 367 |
+
self.num_key_value_groups = config.num_attention_heads // config.num_key_value_heads
|
| 368 |
+
self.scaling = self.head_dim**-0.5
|
| 369 |
+
self.attention_dropout = config.attention_dropout
|
| 370 |
+
self.is_causal = True
|
| 371 |
+
q_width = config.num_attention_heads * self.head_dim
|
| 372 |
+
kv_width = config.num_key_value_heads * self.head_dim
|
| 373 |
+
self.q_proj = nn.Linear(config.hidden_size, q_width * 2, bias=config.attention_bias)
|
| 374 |
+
self.k_proj = nn.Linear(config.hidden_size, kv_width, bias=config.attention_bias)
|
| 375 |
+
self.v_proj = nn.Linear(config.hidden_size, kv_width, bias=config.attention_bias)
|
| 376 |
+
self.o_proj = nn.Linear(q_width, config.hidden_size, bias=config.attention_bias)
|
| 377 |
+
self.q_norm = AgnesRMSNorm(self.head_dim, eps=config.rms_norm_eps)
|
| 378 |
+
self.k_norm = AgnesRMSNorm(self.head_dim, eps=config.rms_norm_eps)
|
| 379 |
+
|
| 380 |
+
def forward(
|
| 381 |
+
self,
|
| 382 |
+
hidden_states: torch.Tensor,
|
| 383 |
+
position_embeddings: tuple[torch.Tensor, torch.Tensor],
|
| 384 |
+
attention_mask: torch.Tensor | None,
|
| 385 |
+
past_key_values: Cache | None = None,
|
| 386 |
+
**kwargs: Unpack[FlashAttentionKwargs],
|
| 387 |
+
) -> tuple[torch.Tensor, torch.Tensor | None]:
|
| 388 |
+
lead = hidden_states.shape[:-1]
|
| 389 |
+
per_head = (*lead, -1, self.head_dim)
|
| 390 |
+
|
| 391 |
+
q, gate = torch.chunk(self.q_proj(hidden_states).view(*lead, -1, self.head_dim * 2), 2, dim=-1)
|
| 392 |
+
gate = gate.reshape(*lead, -1)
|
| 393 |
+
q = self.q_norm(q.view(per_head)).transpose(1, 2)
|
| 394 |
+
k = self.k_norm(self.k_proj(hidden_states).view(per_head)).transpose(1, 2)
|
| 395 |
+
v = self.v_proj(hidden_states).view(per_head).transpose(1, 2)
|
| 396 |
+
|
| 397 |
+
cos, sin = position_embeddings
|
| 398 |
+
q, k = _apply_rope(q, k, cos, sin)
|
| 399 |
+
if past_key_values is not None:
|
| 400 |
+
k, v = past_key_values.update(k, v, self.layer_idx)
|
| 401 |
+
|
| 402 |
+
attend: Callable = ALL_ATTENTION_FUNCTIONS.get_interface(self.config._attn_implementation, _dense_attention)
|
| 403 |
+
out, probs = attend(
|
| 404 |
+
self, q, k, v, attention_mask,
|
| 405 |
+
dropout=0.0 if not self.training else self.attention_dropout,
|
| 406 |
+
scaling=self.scaling,
|
| 407 |
+
**kwargs,
|
| 408 |
+
)
|
| 409 |
+
out = out.reshape(*lead, -1).contiguous()
|
| 410 |
+
out = out * torch.sigmoid(gate)
|
| 411 |
+
return self.o_proj(out), probs
|
| 412 |
+
|
| 413 |
+
|
| 414 |
+
# ---------------------------------------------------------------------------
|
| 415 |
+
# Delta-rule (recurrent) attention
|
| 416 |
+
# ---------------------------------------------------------------------------
|
| 417 |
+
|
| 418 |
+
|
| 419 |
+
def _zero_padded_positions(hidden_states, attention_mask):
|
| 420 |
+
"""Mask out padding tokens so they leave no trace in the recurrent state."""
|
| 421 |
+
if attention_mask is not None and attention_mask.shape[1] > 1 and attention_mask.shape[0] > 1:
|
| 422 |
+
dt = hidden_states.dtype
|
| 423 |
+
hidden_states = (hidden_states * attention_mask[:, :, None]).to(dt)
|
| 424 |
+
return hidden_states
|
| 425 |
+
|
| 426 |
+
|
| 427 |
+
def _causal_conv_step(hidden_states, conv_state, weight, bias=None, activation=None):
|
| 428 |
+
"""One decode step of the depthwise causal conv, updating `conv_state` in place."""
|
| 429 |
+
_, channels, steps = hidden_states.shape
|
| 430 |
+
window = conv_state.shape[-1]
|
| 431 |
+
joined = torch.cat([conv_state, hidden_states], dim=-1).to(weight.dtype)
|
| 432 |
+
conv_state.copy_(joined[:, :, -window:])
|
| 433 |
+
y = F.conv1d(joined, weight.unsqueeze(1), bias, padding=0, groups=channels)
|
| 434 |
+
y = F.silu(y[:, :, -steps:])
|
| 435 |
+
return y.to(hidden_states.dtype)
|
| 436 |
+
|
| 437 |
+
|
| 438 |
+
def _unit_normalize(x: torch.FloatTensor, dim: int = -1, eps: float = 1e-6):
|
| 439 |
+
return x * torch.rsqrt((x * x).sum(dim=dim, keepdim=True) + eps)
|
| 440 |
+
|
| 441 |
+
|
| 442 |
+
def _delta_rule_chunked(
|
| 443 |
+
query, key, value, g, beta, chunk_size=64, initial_state=None, output_final_state=False,
|
| 444 |
+
use_qk_l2norm_in_kernel=False, **kwargs,
|
| 445 |
+
):
|
| 446 |
+
"""Chunk-parallel gated delta rule (prefill path)."""
|
| 447 |
+
out_dtype = query.dtype
|
| 448 |
+
if use_qk_l2norm_in_kernel:
|
| 449 |
+
query = _unit_normalize(query, dim=-1, eps=1e-6)
|
| 450 |
+
key = _unit_normalize(key, dim=-1, eps=1e-6)
|
| 451 |
+
query, key, value, beta, g = [t.transpose(1, 2).contiguous().to(torch.float32) for t in (query, key, value, beta, g)]
|
| 452 |
+
|
| 453 |
+
bsz, heads, seq, dk = key.shape
|
| 454 |
+
dv = value.shape[-1]
|
| 455 |
+
pad = (chunk_size - seq % chunk_size) % chunk_size
|
| 456 |
+
query = F.pad(query, (0, 0, 0, pad))
|
| 457 |
+
key = F.pad(key, (0, 0, 0, pad))
|
| 458 |
+
value = F.pad(value, (0, 0, 0, pad))
|
| 459 |
+
beta = F.pad(beta, (0, pad))
|
| 460 |
+
g = F.pad(g, (0, pad))
|
| 461 |
+
padded = seq + pad
|
| 462 |
+
query = query * (1 / (query.shape[-1] ** 0.5))
|
| 463 |
+
|
| 464 |
+
v_beta = value * beta.unsqueeze(-1)
|
| 465 |
+
k_beta = key * beta.unsqueeze(-1)
|
| 466 |
+
query, key, value, k_beta, v_beta = [
|
| 467 |
+
t.reshape(t.shape[0], t.shape[1], -1, chunk_size, t.shape[-1]) for t in (query, key, value, k_beta, v_beta)
|
| 468 |
+
]
|
| 469 |
+
g = g.reshape(g.shape[0], g.shape[1], -1, chunk_size)
|
| 470 |
+
upper = torch.triu(torch.ones(chunk_size, chunk_size, dtype=torch.bool, device=query.device), diagonal=0)
|
| 471 |
+
|
| 472 |
+
g = g.cumsum(dim=-1)
|
| 473 |
+
decay = ((g.unsqueeze(-1) - g.unsqueeze(-2)).tril().exp().float()).tril()
|
| 474 |
+
solve = -((k_beta @ key.transpose(-1, -2)) * decay).masked_fill(upper, 0)
|
| 475 |
+
for i in range(1, chunk_size):
|
| 476 |
+
row = solve[..., i, :i].clone()
|
| 477 |
+
block = solve[..., :i, :i].clone()
|
| 478 |
+
solve[..., i, :i] = row + (row.unsqueeze(-1) * block).sum(-2)
|
| 479 |
+
solve = solve + torch.eye(chunk_size, dtype=solve.dtype, device=solve.device)
|
| 480 |
+
value = solve @ v_beta
|
| 481 |
+
k_decayed = solve @ (k_beta * g.exp().unsqueeze(-1))
|
| 482 |
+
state = (
|
| 483 |
+
torch.zeros(bsz, heads, dk, dv, dtype=value.dtype, device=value.device)
|
| 484 |
+
if initial_state is None
|
| 485 |
+
else initial_state.to(value)
|
| 486 |
+
)
|
| 487 |
+
out = torch.zeros_like(value)
|
| 488 |
+
strict_upper = torch.triu(torch.ones(chunk_size, chunk_size, dtype=torch.bool, device=query.device), diagonal=1)
|
| 489 |
+
|
| 490 |
+
for i in range(0, padded // chunk_size):
|
| 491 |
+
q_i, k_i, v_i = query[:, :, i], key[:, :, i], value[:, :, i]
|
| 492 |
+
local = q_i @ k_i.transpose(-1, -2) * decay[:, :, i]
|
| 493 |
+
v_pred = (k_decayed[:, :, i]) @ state
|
| 494 |
+
v_res = v_i - v_pred
|
| 495 |
+
carried = (q_i * g[:, :, i, :, None].exp()) @ state
|
| 496 |
+
out[:, :, i] = carried + local @ v_res
|
| 497 |
+
state = (
|
| 498 |
+
state * g[:, :, i, -1, None, None].exp()
|
| 499 |
+
+ (k_i * (g[:, :, i, -1, None] - g[:, :, i]).exp()[..., None]).transpose(-1, -2) @ v_res
|
| 500 |
+
)
|
| 501 |
+
|
| 502 |
+
if not output_final_state:
|
| 503 |
+
state = None
|
| 504 |
+
out = out.reshape(out.shape[0], out.shape[1], -1, out.shape[-1])
|
| 505 |
+
out = out[:, :, :seq]
|
| 506 |
+
return out.transpose(1, 2).contiguous().to(out_dtype), state
|
| 507 |
+
|
| 508 |
+
|
| 509 |
+
def _delta_rule_stepwise(query, key, value, g, beta, initial_state, output_final_state, use_qk_l2norm_in_kernel=False):
|
| 510 |
+
"""Token-by-token gated delta rule (decode path)."""
|
| 511 |
+
out_dtype = query.dtype
|
| 512 |
+
if use_qk_l2norm_in_kernel:
|
| 513 |
+
query = _unit_normalize(query, dim=-1, eps=1e-6)
|
| 514 |
+
key = _unit_normalize(key, dim=-1, eps=1e-6)
|
| 515 |
+
query, key, value, beta, g = [t.transpose(1, 2).contiguous().to(torch.float32) for t in (query, key, value, beta, g)]
|
| 516 |
+
|
| 517 |
+
bsz, heads, seq, dk = key.shape
|
| 518 |
+
dv = value.shape[-1]
|
| 519 |
+
query = query * (1 / (query.shape[-1] ** 0.5))
|
| 520 |
+
|
| 521 |
+
out = torch.zeros(bsz, heads, seq, dv, dtype=value.dtype, device=value.device)
|
| 522 |
+
state = (
|
| 523 |
+
torch.zeros(bsz, heads, dk, dv, dtype=value.dtype, device=value.device)
|
| 524 |
+
if initial_state is None
|
| 525 |
+
else initial_state.to(value)
|
| 526 |
+
)
|
| 527 |
+
for t in range(seq):
|
| 528 |
+
q_t, k_t, v_t = query[:, :, t], key[:, :, t], value[:, :, t]
|
| 529 |
+
decay_t = g[:, :, t].exp().unsqueeze(-1).unsqueeze(-1)
|
| 530 |
+
beta_t = beta[:, :, t].unsqueeze(-1)
|
| 531 |
+
state = state * decay_t
|
| 532 |
+
recalled = (state * k_t.unsqueeze(-1)).sum(dim=-2)
|
| 533 |
+
correction = (v_t - recalled) * beta_t
|
| 534 |
+
state = state + k_t.unsqueeze(-1) * correction.unsqueeze(-2)
|
| 535 |
+
out[:, :, t] = (state * q_t.unsqueeze(-1)).sum(dim=-2)
|
| 536 |
+
|
| 537 |
+
if not output_final_state:
|
| 538 |
+
state = None
|
| 539 |
+
return out.transpose(1, 2).contiguous().to(out_dtype), state
|
| 540 |
+
|
| 541 |
+
|
| 542 |
+
class AgnesDeltaAttention(nn.Module):
|
| 543 |
+
"""Gated delta-rule recurrent attention with a causal depthwise conv on
|
| 544 |
+
the projected q/k/v and a gated RMS norm on the output."""
|
| 545 |
+
|
| 546 |
+
def __init__(self, config: AgnesTextConfig, layer_idx: int):
|
| 547 |
+
super().__init__()
|
| 548 |
+
self.hidden_size = config.hidden_size
|
| 549 |
+
self.num_v_heads = config.linear_num_value_heads
|
| 550 |
+
self.num_k_heads = config.linear_num_key_heads
|
| 551 |
+
self.head_k_dim = config.linear_key_head_dim
|
| 552 |
+
self.head_v_dim = config.linear_value_head_dim
|
| 553 |
+
self.key_dim = self.head_k_dim * self.num_k_heads
|
| 554 |
+
self.value_dim = self.head_v_dim * self.num_v_heads
|
| 555 |
+
self.conv_kernel_size = config.linear_conv_kernel_dim
|
| 556 |
+
self.layer_idx = layer_idx
|
| 557 |
+
self.activation = config.hidden_act
|
| 558 |
+
self.act = ACT2FN[config.hidden_act]
|
| 559 |
+
self.layer_norm_epsilon = config.rms_norm_eps
|
| 560 |
+
|
| 561 |
+
self.conv_dim = self.key_dim * 2 + self.value_dim
|
| 562 |
+
self.conv1d = nn.Conv1d(
|
| 563 |
+
in_channels=self.conv_dim, out_channels=self.conv_dim, bias=False,
|
| 564 |
+
kernel_size=self.conv_kernel_size, groups=self.conv_dim, padding=self.conv_kernel_size - 1,
|
| 565 |
+
)
|
| 566 |
+
self.dt_bias = nn.Parameter(torch.ones(self.num_v_heads))
|
| 567 |
+
self.A_log = nn.Parameter(torch.log(torch.empty(self.num_v_heads).uniform_(0, 16)))
|
| 568 |
+
|
| 569 |
+
if FusedRMSNormGated is None:
|
| 570 |
+
self.norm = AgnesGatedNorm(self.head_v_dim, eps=self.layer_norm_epsilon)
|
| 571 |
+
else:
|
| 572 |
+
self.norm = FusedRMSNormGated(
|
| 573 |
+
self.head_v_dim, eps=self.layer_norm_epsilon, activation=self.activation,
|
| 574 |
+
device=torch.cuda.current_device(),
|
| 575 |
+
dtype=config.dtype if config.dtype is not None else torch.get_default_dtype(),
|
| 576 |
+
)
|
| 577 |
+
self.out_proj = nn.Linear(self.value_dim, self.hidden_size, bias=False)
|
| 578 |
+
|
| 579 |
+
self.causal_conv1d_fn = causal_conv1d_fn
|
| 580 |
+
self.causal_conv1d_update = causal_conv1d_update or _causal_conv_step
|
| 581 |
+
self.chunk_gated_delta_rule = chunk_gated_delta_rule or _delta_rule_chunked
|
| 582 |
+
self.recurrent_gated_delta_rule = fused_recurrent_gated_delta_rule or _delta_rule_stepwise
|
| 583 |
+
if not _FUSED_DELTA_PATH:
|
| 584 |
+
logger.warning_once(
|
| 585 |
+
"Fused delta-rule kernels not found (flash-linear-attention / causal-conv1d); "
|
| 586 |
+
"using the pure-torch implementation."
|
| 587 |
+
)
|
| 588 |
+
|
| 589 |
+
self.in_proj_qkv = nn.Linear(self.hidden_size, self.key_dim * 2 + self.value_dim, bias=False)
|
| 590 |
+
self.in_proj_z = nn.Linear(self.hidden_size, self.value_dim, bias=False)
|
| 591 |
+
self.in_proj_b = nn.Linear(self.hidden_size, self.num_v_heads, bias=False)
|
| 592 |
+
self.in_proj_a = nn.Linear(self.hidden_size, self.num_v_heads, bias=False)
|
| 593 |
+
|
| 594 |
+
def forward(
|
| 595 |
+
self,
|
| 596 |
+
hidden_states: torch.Tensor,
|
| 597 |
+
cache_params: Cache | None = None,
|
| 598 |
+
attention_mask: torch.Tensor | None = None,
|
| 599 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 600 |
+
):
|
| 601 |
+
hidden_states = _zero_padded_positions(hidden_states, attention_mask)
|
| 602 |
+
bsz, seq, _ = hidden_states.shape
|
| 603 |
+
|
| 604 |
+
resume = cache_params is not None and cache_params.has_previous_state(self.layer_idx)
|
| 605 |
+
if resume:
|
| 606 |
+
conv_state = cache_params.layers[self.layer_idx].conv_states
|
| 607 |
+
rec_state = cache_params.layers[self.layer_idx].recurrent_states
|
| 608 |
+
|
| 609 |
+
qkv = self.in_proj_qkv(hidden_states).transpose(1, 2)
|
| 610 |
+
z = self.in_proj_z(hidden_states).reshape(bsz, seq, -1, self.head_v_dim)
|
| 611 |
+
b = self.in_proj_b(hidden_states)
|
| 612 |
+
a = self.in_proj_a(hidden_states)
|
| 613 |
+
|
| 614 |
+
if resume and seq == 1:
|
| 615 |
+
qkv = self.causal_conv1d_update(qkv, conv_state, self.conv1d.weight.squeeze(1), self.conv1d.bias, self.activation)
|
| 616 |
+
else:
|
| 617 |
+
if resume:
|
| 618 |
+
qkv = torch.cat([conv_state, qkv], dim=-1)
|
| 619 |
+
if cache_params is not None:
|
| 620 |
+
cache_params.update_conv_state(F.pad(qkv, (self.conv_kernel_size - qkv.shape[-1], 0)), self.layer_idx)
|
| 621 |
+
if self.causal_conv1d_fn is not None:
|
| 622 |
+
qkv = self.causal_conv1d_fn(
|
| 623 |
+
x=qkv, weight=self.conv1d.weight.squeeze(1), bias=self.conv1d.bias,
|
| 624 |
+
activation=self.activation, seq_idx=kwargs.get("seq_idx"),
|
| 625 |
+
)
|
| 626 |
+
else:
|
| 627 |
+
qkv = F.silu(self.conv1d(qkv)[:, :, : qkv.shape[-1]])
|
| 628 |
+
if resume:
|
| 629 |
+
qkv = qkv[:, :, -seq:]
|
| 630 |
+
|
| 631 |
+
qkv = qkv.transpose(1, 2)
|
| 632 |
+
query, key, value = torch.split(qkv, [self.key_dim, self.key_dim, self.value_dim], dim=-1)
|
| 633 |
+
query = query.reshape(bsz, seq, -1, self.head_k_dim)
|
| 634 |
+
key = key.reshape(bsz, seq, -1, self.head_k_dim)
|
| 635 |
+
value = value.reshape(bsz, seq, -1, self.head_v_dim)
|
| 636 |
+
|
| 637 |
+
beta = b.sigmoid()
|
| 638 |
+
# fp32 keeps exp(A_log) finite under fp16 weights
|
| 639 |
+
g = -self.A_log.float().exp() * F.softplus(a.float() + self.dt_bias)
|
| 640 |
+
groups = self.num_v_heads // self.num_k_heads
|
| 641 |
+
if groups > 1:
|
| 642 |
+
query = query.repeat_interleave(groups, dim=2)
|
| 643 |
+
key = key.repeat_interleave(groups, dim=2)
|
| 644 |
+
|
| 645 |
+
if resume and seq == 1:
|
| 646 |
+
mixed, rec_state = self.recurrent_gated_delta_rule(
|
| 647 |
+
query, key, value, g=g, beta=beta, initial_state=rec_state,
|
| 648 |
+
output_final_state=cache_params is not None, use_qk_l2norm_in_kernel=True,
|
| 649 |
+
)
|
| 650 |
+
else:
|
| 651 |
+
mixed, rec_state = self.chunk_gated_delta_rule(
|
| 652 |
+
query, key, value, g=g, beta=beta,
|
| 653 |
+
initial_state=rec_state if resume else None,
|
| 654 |
+
output_final_state=cache_params is not None, use_qk_l2norm_in_kernel=True,
|
| 655 |
+
cu_seqlens=kwargs.get("cu_seq_lens_q"),
|
| 656 |
+
)
|
| 657 |
+
if cache_params is not None:
|
| 658 |
+
cache_params.update_recurrent_state(rec_state, self.layer_idx)
|
| 659 |
+
|
| 660 |
+
mixed = self.norm(mixed.reshape(-1, self.head_v_dim), z.reshape(-1, self.head_v_dim))
|
| 661 |
+
return self.out_proj(mixed.reshape(bsz, seq, -1))
|
| 662 |
+
|
| 663 |
+
|
| 664 |
+
# ---------------------------------------------------------------------------
|
| 665 |
+
# Decoder layer / base class
|
| 666 |
+
# ---------------------------------------------------------------------------
|
| 667 |
+
|
| 668 |
+
|
| 669 |
+
class AgnesDecoderLayer(GradientCheckpointingLayer):
|
| 670 |
+
def __init__(self, config: AgnesTextConfig, layer_idx: int):
|
| 671 |
+
super().__init__()
|
| 672 |
+
self.hidden_size = config.hidden_size
|
| 673 |
+
self.layer_type = config.layer_types[layer_idx]
|
| 674 |
+
if self.layer_type == LAYER_DELTA:
|
| 675 |
+
self.delta_attn = AgnesDeltaAttention(config, layer_idx)
|
| 676 |
+
elif self.layer_type == LAYER_GLOBAL:
|
| 677 |
+
self.global_attn = AgnesGlobalAttention(config, layer_idx)
|
| 678 |
+
self.mlp = AgnesMLP(config, config.intermediate_size, getattr(config, "parallel_ffn_intermediate_size", 0) or 0)
|
| 679 |
+
self.input_layernorm = AgnesRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 680 |
+
self.post_attention_layernorm = AgnesRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 681 |
+
|
| 682 |
+
def forward(
|
| 683 |
+
self,
|
| 684 |
+
hidden_states: torch.Tensor,
|
| 685 |
+
position_embeddings: tuple[torch.Tensor, torch.Tensor],
|
| 686 |
+
attention_mask: torch.Tensor | None = None,
|
| 687 |
+
position_ids: torch.LongTensor | None = None,
|
| 688 |
+
past_key_values: Cache | None = None,
|
| 689 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 690 |
+
) -> torch.FloatTensor:
|
| 691 |
+
skip = hidden_states
|
| 692 |
+
h = self.input_layernorm(hidden_states)
|
| 693 |
+
if self.layer_type == LAYER_DELTA:
|
| 694 |
+
h = self.delta_attn(hidden_states=h, cache_params=past_key_values, attention_mask=attention_mask, **kwargs)
|
| 695 |
+
elif self.layer_type == LAYER_GLOBAL:
|
| 696 |
+
h, _ = self.global_attn(
|
| 697 |
+
hidden_states=h, attention_mask=attention_mask, position_ids=position_ids,
|
| 698 |
+
past_key_values=past_key_values, position_embeddings=position_embeddings, **kwargs,
|
| 699 |
+
)
|
| 700 |
+
h = skip + h
|
| 701 |
+
|
| 702 |
+
skip = h
|
| 703 |
+
h = self.mlp(self.post_attention_layernorm(h))
|
| 704 |
+
return skip + h
|
| 705 |
+
|
| 706 |
+
|
| 707 |
+
class AgnesPreTrainedModel(PreTrainedModel):
|
| 708 |
+
config: AgnesConfig
|
| 709 |
+
base_model_prefix = "model"
|
| 710 |
+
supports_gradient_checkpointing = True
|
| 711 |
+
_no_split_modules = ["AgnesDecoderLayer", "AgnesVisionBlock"]
|
| 712 |
+
_skip_keys_device_placement = ["past_key_values"]
|
| 713 |
+
_supports_flash_attn = True
|
| 714 |
+
_supports_sdpa = True
|
| 715 |
+
_keys_to_ignore_on_load_unexpected = [r"^mtp.*"]
|
| 716 |
+
_can_record_outputs = {
|
| 717 |
+
"hidden_states": AgnesDecoderLayer,
|
| 718 |
+
"attentions": AgnesGlobalAttention,
|
| 719 |
+
}
|
| 720 |
+
_is_stateful = True
|
| 721 |
+
|
| 722 |
+
@torch.no_grad()
|
| 723 |
+
def _init_weights(self, module):
|
| 724 |
+
super()._init_weights(module)
|
| 725 |
+
if isinstance(module, AgnesDeltaAttention):
|
| 726 |
+
init.ones_(module.dt_bias)
|
| 727 |
+
init.copy_(module.A_log, torch.empty_like(module.A_log).uniform_(0, 16).log_())
|
| 728 |
+
elif isinstance(module, AgnesRMSNorm):
|
| 729 |
+
# the norm scales by (1 + weight), so zero is the identity
|
| 730 |
+
init.zeros_(module.weight)
|
| 731 |
+
elif isinstance(module, AgnesVisionRotary):
|
| 732 |
+
inv_freq = 1.0 / (module.theta ** (torch.arange(0, module.dim, 2, dtype=torch.float) / module.dim))
|
| 733 |
+
init.copy_(module.inv_freq, inv_freq)
|
| 734 |
+
|
| 735 |
+
|
| 736 |
+
# ---------------------------------------------------------------------------
|
| 737 |
+
# Vision tower
|
| 738 |
+
# ---------------------------------------------------------------------------
|
| 739 |
+
|
| 740 |
+
|
| 741 |
+
class AgnesVisionPatchEmbed(nn.Module):
|
| 742 |
+
def __init__(self, config) -> None:
|
| 743 |
+
super().__init__()
|
| 744 |
+
self.patch_size = config.patch_size
|
| 745 |
+
self.temporal_patch_size = config.temporal_patch_size
|
| 746 |
+
self.in_channels = config.in_channels
|
| 747 |
+
self.embed_dim = config.hidden_size
|
| 748 |
+
kernel = [self.temporal_patch_size, self.patch_size, self.patch_size]
|
| 749 |
+
self.proj = nn.Conv3d(self.in_channels, self.embed_dim, kernel_size=kernel, stride=kernel, bias=True)
|
| 750 |
+
|
| 751 |
+
def forward(self, hidden_states: torch.Tensor) -> torch.Tensor:
|
| 752 |
+
dt = self.proj.weight.dtype
|
| 753 |
+
patches = hidden_states.view(-1, self.in_channels, self.temporal_patch_size, self.patch_size, self.patch_size)
|
| 754 |
+
return self.proj(patches.to(dtype=dt)).view(-1, self.embed_dim)
|
| 755 |
+
|
| 756 |
+
|
| 757 |
+
class AgnesVisionMLP(nn.Module):
|
| 758 |
+
def __init__(self, config):
|
| 759 |
+
super().__init__()
|
| 760 |
+
self.hidden_size = config.hidden_size
|
| 761 |
+
self.intermediate_size = config.intermediate_size
|
| 762 |
+
self.linear_fc1 = nn.Linear(self.hidden_size, self.intermediate_size, bias=True)
|
| 763 |
+
self.linear_fc2 = nn.Linear(self.intermediate_size, self.hidden_size, bias=True)
|
| 764 |
+
self.act_fn = ACT2FN[config.hidden_act]
|
| 765 |
+
|
| 766 |
+
def forward(self, hidden_state):
|
| 767 |
+
return self.linear_fc2(self.act_fn(self.linear_fc1(hidden_state)))
|
| 768 |
+
|
| 769 |
+
|
| 770 |
+
class AgnesVisionAttention(nn.Module):
|
| 771 |
+
def __init__(self, config: AgnesVisionConfig) -> None:
|
| 772 |
+
super().__init__()
|
| 773 |
+
self.dim = config.hidden_size
|
| 774 |
+
self.num_heads = config.num_heads
|
| 775 |
+
self.head_dim = self.dim // self.num_heads
|
| 776 |
+
self.num_key_value_groups = 1
|
| 777 |
+
self.qkv = nn.Linear(self.dim, self.dim * 3, bias=True)
|
| 778 |
+
self.proj = nn.Linear(self.dim, self.dim)
|
| 779 |
+
self.scaling = self.head_dim**-0.5
|
| 780 |
+
self.config = config
|
| 781 |
+
self.attention_dropout = 0.0
|
| 782 |
+
self.is_causal = False
|
| 783 |
+
|
| 784 |
+
def forward(
|
| 785 |
+
self,
|
| 786 |
+
hidden_states: torch.Tensor,
|
| 787 |
+
cu_seqlens: torch.Tensor,
|
| 788 |
+
position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
|
| 789 |
+
**kwargs,
|
| 790 |
+
) -> torch.Tensor:
|
| 791 |
+
n_tok = hidden_states.shape[0]
|
| 792 |
+
q, k, v = self.qkv(hidden_states).reshape(n_tok, 3, self.num_heads, -1).permute(1, 0, 2, 3).unbind(0)
|
| 793 |
+
cos, sin = position_embeddings
|
| 794 |
+
q, k = _apply_vision_rope(q, k, cos, sin)
|
| 795 |
+
q = q.transpose(0, 1).unsqueeze(0)
|
| 796 |
+
k = k.transpose(0, 1).unsqueeze(0)
|
| 797 |
+
v = v.transpose(0, 1).unsqueeze(0)
|
| 798 |
+
|
| 799 |
+
attend: Callable = ALL_ATTENTION_FUNCTIONS.get_interface(self.config._attn_implementation, _dense_attention)
|
| 800 |
+
drop = 0.0 if not self.training else self.attention_dropout
|
| 801 |
+
if is_flash_attention_requested(self.config):
|
| 802 |
+
longest = (cu_seqlens[1:] - cu_seqlens[:-1]).max()
|
| 803 |
+
out, _ = attend(
|
| 804 |
+
self, q, k, v, attention_mask=None, scaling=self.scaling, dropout=drop,
|
| 805 |
+
cu_seq_lens_q=cu_seqlens, cu_seq_lens_k=cu_seqlens, max_length_q=longest, max_length_k=longest,
|
| 806 |
+
is_causal=False, **kwargs,
|
| 807 |
+
)
|
| 808 |
+
else:
|
| 809 |
+
spans = (cu_seqlens[1:] - cu_seqlens[:-1]).tolist()
|
| 810 |
+
pieces = [torch.split(t, spans, dim=2) for t in (q, k, v)]
|
| 811 |
+
out = torch.cat(
|
| 812 |
+
[
|
| 813 |
+
attend(self, qi, ki, vi, attention_mask=None, scaling=self.scaling, dropout=drop, is_causal=False, **kwargs)[0]
|
| 814 |
+
for qi, ki, vi in zip(*pieces)
|
| 815 |
+
],
|
| 816 |
+
dim=1,
|
| 817 |
+
)
|
| 818 |
+
return self.proj(out.reshape(n_tok, -1).contiguous())
|
| 819 |
+
|
| 820 |
+
|
| 821 |
+
class AgnesVisionBlock(GradientCheckpointingLayer):
|
| 822 |
+
def __init__(self, config, attn_implementation: str = "sdpa") -> None:
|
| 823 |
+
super().__init__()
|
| 824 |
+
self.norm1 = nn.LayerNorm(config.hidden_size, eps=1e-6)
|
| 825 |
+
self.norm2 = nn.LayerNorm(config.hidden_size, eps=1e-6)
|
| 826 |
+
self.attn = AgnesVisionAttention(config=config)
|
| 827 |
+
self.mlp = AgnesVisionMLP(config=config)
|
| 828 |
+
|
| 829 |
+
@auto_docstring
|
| 830 |
+
def forward(
|
| 831 |
+
self,
|
| 832 |
+
hidden_states: torch.Tensor,
|
| 833 |
+
cu_seqlens: torch.Tensor,
|
| 834 |
+
position_embeddings: tuple[torch.Tensor, torch.Tensor] | None = None,
|
| 835 |
+
**kwargs,
|
| 836 |
+
) -> torch.Tensor:
|
| 837 |
+
r"""
|
| 838 |
+
cu_seqlens (`torch.Tensor`):
|
| 839 |
+
Cumulative sequence lengths used for packed variable-length attention.
|
| 840 |
+
"""
|
| 841 |
+
hidden_states = hidden_states + self.attn(
|
| 842 |
+
self.norm1(hidden_states), cu_seqlens=cu_seqlens, position_embeddings=position_embeddings, **kwargs
|
| 843 |
+
)
|
| 844 |
+
return hidden_states + self.mlp(self.norm2(hidden_states))
|
| 845 |
+
|
| 846 |
+
|
| 847 |
+
class AgnesVisionMerger(nn.Module):
|
| 848 |
+
def __init__(self, config: AgnesVisionConfig, use_postshuffle_norm=False) -> None:
|
| 849 |
+
super().__init__()
|
| 850 |
+
self.hidden_size = config.hidden_size * (config.spatial_merge_size**2)
|
| 851 |
+
self.use_postshuffle_norm = use_postshuffle_norm
|
| 852 |
+
self.norm = nn.LayerNorm(self.hidden_size if use_postshuffle_norm else config.hidden_size, eps=1e-6)
|
| 853 |
+
self.linear_fc1 = nn.Linear(self.hidden_size, self.hidden_size)
|
| 854 |
+
self.act_fn = nn.GELU()
|
| 855 |
+
self.linear_fc2 = nn.Linear(self.hidden_size, config.out_hidden_size)
|
| 856 |
+
|
| 857 |
+
def forward(self, x: torch.Tensor) -> torch.Tensor:
|
| 858 |
+
x = self.norm(x.view(-1, self.hidden_size) if self.use_postshuffle_norm else x).view(-1, self.hidden_size)
|
| 859 |
+
return self.linear_fc2(self.act_fn(self.linear_fc1(x)))
|
| 860 |
+
|
| 861 |
+
|
| 862 |
+
class AgnesVisionModel(AgnesPreTrainedModel):
|
| 863 |
+
config: AgnesVisionConfig
|
| 864 |
+
config_class = AgnesVisionConfig
|
| 865 |
+
input_modalities = ("image", "video")
|
| 866 |
+
_can_record_outputs = {
|
| 867 |
+
"hidden_states": AgnesVisionBlock,
|
| 868 |
+
"attentions": AgnesVisionAttention,
|
| 869 |
+
}
|
| 870 |
+
_no_split_modules = ["AgnesVisionBlock"]
|
| 871 |
+
|
| 872 |
+
def __init__(self, config, *inputs, **kwargs) -> None:
|
| 873 |
+
super().__init__(config, *inputs, **kwargs)
|
| 874 |
+
self.spatial_merge_size = config.spatial_merge_size
|
| 875 |
+
self.patch_size = config.patch_size
|
| 876 |
+
self.spatial_merge_unit = self.spatial_merge_size * self.spatial_merge_size
|
| 877 |
+
self.patch_embed = AgnesVisionPatchEmbed(config=config)
|
| 878 |
+
self.pos_embed = nn.Embedding(config.num_position_embeddings, config.hidden_size)
|
| 879 |
+
self.num_grid_per_side = int(config.num_position_embeddings**0.5)
|
| 880 |
+
self.rotary_pos_emb = AgnesVisionRotary((config.hidden_size // config.num_heads) // 2)
|
| 881 |
+
self.blocks = nn.ModuleList([AgnesVisionBlock(config) for _ in range(config.depth)])
|
| 882 |
+
self.merger = AgnesVisionMerger(config=config, use_postshuffle_norm=False)
|
| 883 |
+
self.gradient_checkpointing = False
|
| 884 |
+
self.post_init()
|
| 885 |
+
|
| 886 |
+
@merge_with_config_defaults
|
| 887 |
+
@capture_outputs
|
| 888 |
+
def forward(self, hidden_states: torch.Tensor, grid_thw: torch.Tensor, **kwargs) -> torch.Tensor:
|
| 889 |
+
"""
|
| 890 |
+
Args:
|
| 891 |
+
hidden_states (`torch.Tensor` of shape `(seq_len, hidden_size)`):
|
| 892 |
+
Flattened patches.
|
| 893 |
+
grid_thw (`torch.Tensor` of shape `(num_images_or_videos, 3)`):
|
| 894 |
+
Temporal / height / width extent of every image or video.
|
| 895 |
+
"""
|
| 896 |
+
idx, w = get_vision_bilinear_indices_and_weights(
|
| 897 |
+
grid_thw, num_grid_per_side=self.num_grid_per_side, spatial_merge_size=self.config.spatial_merge_size, kwargs=kwargs
|
| 898 |
+
)
|
| 899 |
+
pos_ids = get_vision_position_ids(grid_thw, self.spatial_merge_size, kwargs=kwargs)
|
| 900 |
+
cu_seqlens = get_vision_cu_seqlens(grid_thw, kwargs=kwargs)
|
| 901 |
+
|
| 902 |
+
tokens = self.patch_embed(hidden_states)
|
| 903 |
+
tokens = tokens + (self.pos_embed(idx) * w[:, :, None]).sum(0).to(tokens.dtype)
|
| 904 |
+
angles = self.rotary_pos_emb(pos_ids)
|
| 905 |
+
|
| 906 |
+
n_tok, _ = tokens.size()
|
| 907 |
+
tokens = tokens.reshape(n_tok, -1)
|
| 908 |
+
angles = angles.reshape(n_tok, -1)
|
| 909 |
+
table = torch.cat((angles, angles), dim=-1)
|
| 910 |
+
rope = (table.cos(), table.sin())
|
| 911 |
+
|
| 912 |
+
for block in self.blocks:
|
| 913 |
+
tokens = block(tokens, cu_seqlens=cu_seqlens, position_embeddings=rope, **kwargs)
|
| 914 |
+
|
| 915 |
+
return BaseModelOutputWithPooling(last_hidden_state=tokens, pooler_output=self.merger(tokens))
|
| 916 |
+
|
| 917 |
+
|
| 918 |
+
# ---------------------------------------------------------------------------
|
| 919 |
+
# Outputs
|
| 920 |
+
# ---------------------------------------------------------------------------
|
| 921 |
+
|
| 922 |
+
|
| 923 |
+
@auto_docstring
|
| 924 |
+
@dataclass
|
| 925 |
+
class AgnesModelOutput(BaseModelOutputWithPast):
|
| 926 |
+
r"""
|
| 927 |
+
rope_deltas (`torch.LongTensor` of shape `(batch_size, )`, *optional*):
|
| 928 |
+
Offset between the multimodal rotary positions and plain sequence positions.
|
| 929 |
+
"""
|
| 930 |
+
|
| 931 |
+
rope_deltas: torch.LongTensor | None = None
|
| 932 |
+
|
| 933 |
+
|
| 934 |
+
@auto_docstring
|
| 935 |
+
@dataclass
|
| 936 |
+
class AgnesCausalLMOutput(CausalLMOutputWithPast):
|
| 937 |
+
r"""
|
| 938 |
+
rope_deltas (`torch.LongTensor` of shape `(batch_size, )`, *optional*):
|
| 939 |
+
Offset between the multimodal rotary positions and plain sequence positions.
|
| 940 |
+
"""
|
| 941 |
+
|
| 942 |
+
rope_deltas: torch.LongTensor | None = None
|
| 943 |
+
|
| 944 |
+
|
| 945 |
+
# ---------------------------------------------------------------------------
|
| 946 |
+
# Language model
|
| 947 |
+
# ---------------------------------------------------------------------------
|
| 948 |
+
|
| 949 |
+
|
| 950 |
+
class AgnesTextModel(AgnesPreTrainedModel):
|
| 951 |
+
config: AgnesTextConfig
|
| 952 |
+
config_class = AgnesTextConfig
|
| 953 |
+
|
| 954 |
+
def __init__(self, config: AgnesTextConfig):
|
| 955 |
+
super().__init__(config)
|
| 956 |
+
self.embed_tokens = nn.Embedding(config.vocab_size, config.hidden_size, config.pad_token_id)
|
| 957 |
+
self.layers = nn.ModuleList([AgnesDecoderLayer(config, i) for i in range(config.num_hidden_layers)])
|
| 958 |
+
self.norm = AgnesRMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
| 959 |
+
self.rotary_emb = AgnesRotaryEmbedding(config=config)
|
| 960 |
+
self.gradient_checkpointing = False
|
| 961 |
+
self.post_init()
|
| 962 |
+
|
| 963 |
+
@merge_with_config_defaults
|
| 964 |
+
@capture_outputs
|
| 965 |
+
@auto_docstring
|
| 966 |
+
def forward(
|
| 967 |
+
self,
|
| 968 |
+
input_ids: torch.LongTensor | None = None,
|
| 969 |
+
attention_mask: torch.Tensor | None = None,
|
| 970 |
+
position_ids: torch.LongTensor | None = None,
|
| 971 |
+
past_key_values: Cache | None = None,
|
| 972 |
+
inputs_embeds: torch.FloatTensor | None = None,
|
| 973 |
+
use_cache: bool | None = None,
|
| 974 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 975 |
+
) -> BaseModelOutputWithPast:
|
| 976 |
+
if (input_ids is None) ^ (inputs_embeds is not None):
|
| 977 |
+
raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
|
| 978 |
+
if inputs_embeds is None:
|
| 979 |
+
inputs_embeds = self.embed_tokens(input_ids)
|
| 980 |
+
if use_cache and past_key_values is None:
|
| 981 |
+
past_key_values = DynamicCache(config=self.config)
|
| 982 |
+
|
| 983 |
+
# position ids carry four rows: plain text positions, then T / H / W
|
| 984 |
+
if position_ids is None:
|
| 985 |
+
offset = past_key_values.get_seq_length() if past_key_values is not None else 0
|
| 986 |
+
position_ids = torch.arange(inputs_embeds.shape[1], device=inputs_embeds.device) + offset
|
| 987 |
+
position_ids = position_ids.view(1, 1, -1).expand(4, inputs_embeds.shape[0], -1)
|
| 988 |
+
elif position_ids.ndim == 2:
|
| 989 |
+
position_ids = position_ids[None, ...].expand(4, position_ids.shape[0], -1)
|
| 990 |
+
|
| 991 |
+
if position_ids.ndim == 3 and position_ids.shape[0] == 4:
|
| 992 |
+
text_positions, position_ids = position_ids[0], position_ids[1:]
|
| 993 |
+
else:
|
| 994 |
+
text_positions = None
|
| 995 |
+
|
| 996 |
+
global_mask = create_causal_mask(
|
| 997 |
+
config=self.config, inputs_embeds=inputs_embeds, attention_mask=attention_mask,
|
| 998 |
+
past_key_values=past_key_values, position_ids=text_positions,
|
| 999 |
+
)
|
| 1000 |
+
delta_mask = self._recurrent_mask(attention_mask, past_key_values)
|
| 1001 |
+
|
| 1002 |
+
h = inputs_embeds
|
| 1003 |
+
rope = self.rotary_emb(h, position_ids)
|
| 1004 |
+
for i, layer in enumerate(self.layers[: self.config.num_hidden_layers]):
|
| 1005 |
+
mask = delta_mask if self.config.layer_types[i] == LAYER_DELTA else global_mask
|
| 1006 |
+
h = layer(
|
| 1007 |
+
h, position_embeddings=rope, attention_mask=mask, position_ids=text_positions,
|
| 1008 |
+
past_key_values=past_key_values, use_cache=use_cache, **kwargs,
|
| 1009 |
+
)
|
| 1010 |
+
h = self.norm(h)
|
| 1011 |
+
return AgnesModelOutput(last_hidden_state=h, past_key_values=past_key_values)
|
| 1012 |
+
|
| 1013 |
+
@staticmethod
|
| 1014 |
+
def _recurrent_mask(attention_mask, past_key_values):
|
| 1015 |
+
"""Padding mask for the recurrent layers (left padding). Not needed
|
| 1016 |
+
once a cache holds prior state, or when nothing is masked."""
|
| 1017 |
+
if past_key_values is not None and past_key_values.has_previous_state():
|
| 1018 |
+
return None
|
| 1019 |
+
if attention_mask is not None and torch.all(attention_mask == 1):
|
| 1020 |
+
return None
|
| 1021 |
+
return attention_mask
|
| 1022 |
+
|
| 1023 |
+
|
| 1024 |
+
@auto_docstring
|
| 1025 |
+
class AgnesModel(AgnesPreTrainedModel):
|
| 1026 |
+
config: AgnesConfig
|
| 1027 |
+
config_class = AgnesConfig
|
| 1028 |
+
base_model_prefix = "model"
|
| 1029 |
+
accepts_loss_kwargs = False
|
| 1030 |
+
_no_split_modules = ["AgnesDecoderLayer", "AgnesVisionBlock"]
|
| 1031 |
+
|
| 1032 |
+
def __init__(self, config):
|
| 1033 |
+
super().__init__(config)
|
| 1034 |
+
self.visual = AgnesVisionModel(config.vision_config)
|
| 1035 |
+
self.language_model = AgnesTextModel(config.text_config)
|
| 1036 |
+
self.rope_deltas = None
|
| 1037 |
+
self.post_init()
|
| 1038 |
+
|
| 1039 |
+
def get_vision_position_ids(
|
| 1040 |
+
self,
|
| 1041 |
+
start_position: int,
|
| 1042 |
+
grid_thw: list[int, int, int] | torch.Tensor,
|
| 1043 |
+
temp_merge_size: int = 1,
|
| 1044 |
+
spatial_merge_size: int = 1,
|
| 1045 |
+
time_interval: int = 1,
|
| 1046 |
+
device: str | torch.device | None = None,
|
| 1047 |
+
):
|
| 1048 |
+
"""Three-axis positions for the tokens of one image / video, offset by `start_position`."""
|
| 1049 |
+
n_t = grid_thw[0].item() // temp_merge_size
|
| 1050 |
+
n_h = grid_thw[1].item() // spatial_merge_size
|
| 1051 |
+
n_w = grid_thw[2].item() // spatial_merge_size
|
| 1052 |
+
|
| 1053 |
+
t_axis = torch.arange(n_t, device=device) * time_interval
|
| 1054 |
+
w_axis = torch.arange(n_w, device=device) + start_position
|
| 1055 |
+
h_axis = torch.arange(n_h, device=device) + start_position
|
| 1056 |
+
|
| 1057 |
+
# repeat patterns define the raster order; keep them as is
|
| 1058 |
+
w_axis = w_axis.repeat(n_h * n_t)
|
| 1059 |
+
h_axis = h_axis.repeat_interleave(n_w).repeat(n_t)
|
| 1060 |
+
t_axis = t_axis.repeat_interleave(n_h * n_w) + start_position
|
| 1061 |
+
return torch.stack([t_axis, h_axis, w_axis], dim=0)
|
| 1062 |
+
|
| 1063 |
+
def get_rope_index(
|
| 1064 |
+
self,
|
| 1065 |
+
input_ids: torch.LongTensor,
|
| 1066 |
+
mm_token_type_ids: torch.IntTensor,
|
| 1067 |
+
image_grid_thw: torch.LongTensor | None = None,
|
| 1068 |
+
video_grid_thw: torch.LongTensor | None = None,
|
| 1069 |
+
attention_mask: torch.Tensor | None = None,
|
| 1070 |
+
**kwargs,
|
| 1071 |
+
) -> tuple[torch.Tensor, torch.Tensor]:
|
| 1072 |
+
"""Build (3, batch, seq) rotary positions: text runs advance all three
|
| 1073 |
+
axes together, vision runs get grid positions. Videos are split per
|
| 1074 |
+
frame because frames are separated by timestamp tokens."""
|
| 1075 |
+
if video_grid_thw is not None:
|
| 1076 |
+
video_grid_thw = torch.repeat_interleave(video_grid_thw, video_grid_thw[:, 0], dim=0)
|
| 1077 |
+
video_grid_thw[:, 0] = 1
|
| 1078 |
+
merge = self.config.vision_config.spatial_merge_size
|
| 1079 |
+
|
| 1080 |
+
deltas = []
|
| 1081 |
+
position_ids = torch.zeros(3, input_ids.shape[0], input_ids.shape[1], dtype=input_ids.dtype, device=input_ids.device)
|
| 1082 |
+
grids = {
|
| 1083 |
+
1: iter(image_grid_thw) if image_grid_thw is not None else None,
|
| 1084 |
+
2: iter(video_grid_thw) if video_grid_thw is not None else None,
|
| 1085 |
+
}
|
| 1086 |
+
|
| 1087 |
+
for b, ids in enumerate(input_ids):
|
| 1088 |
+
kinds = mm_token_type_ids[b]
|
| 1089 |
+
if attention_mask is not None:
|
| 1090 |
+
keep = attention_mask[b].bool()
|
| 1091 |
+
ids, kinds = ids[keep], kinds[keep]
|
| 1092 |
+
|
| 1093 |
+
runs = []
|
| 1094 |
+
for kind, members in itertools.groupby(enumerate(kinds.tolist()), lambda x: x[1]):
|
| 1095 |
+
members = list(members)
|
| 1096 |
+
runs.append((kind, members[0][0], members[-1][0] + 1))
|
| 1097 |
+
|
| 1098 |
+
cursor = 0
|
| 1099 |
+
pieces = []
|
| 1100 |
+
for kind, lo, hi in runs:
|
| 1101 |
+
if kind == 0:
|
| 1102 |
+
n = hi - lo
|
| 1103 |
+
pieces.append(torch.arange(n, device=input_ids.device).view(1, -1).expand(3, -1) + cursor)
|
| 1104 |
+
cursor += n
|
| 1105 |
+
else:
|
| 1106 |
+
thw = next(grids[kind])
|
| 1107 |
+
pieces.append(self.get_vision_position_ids(cursor, thw, 1, merge, device=input_ids.device))
|
| 1108 |
+
cursor += max(thw[1], thw[2]) // merge
|
| 1109 |
+
pos = torch.cat(pieces, dim=1).reshape(3, -1)
|
| 1110 |
+
if attention_mask is not None:
|
| 1111 |
+
position_ids[:, b, attention_mask[b].bool()] = pos.to(position_ids.device)
|
| 1112 |
+
else:
|
| 1113 |
+
position_ids[:, b] = pos.to(position_ids.device)
|
| 1114 |
+
deltas.append(pos.max() + 1 - len(ids))
|
| 1115 |
+
return position_ids, torch.tensor(deltas, device=input_ids.device).unsqueeze(1)
|
| 1116 |
+
|
| 1117 |
+
@accepts_precomputed_kwargs(modality="video")
|
| 1118 |
+
@can_return_tuple
|
| 1119 |
+
@auto_docstring
|
| 1120 |
+
def get_video_features(
|
| 1121 |
+
self,
|
| 1122 |
+
pixel_values_videos: torch.FloatTensor,
|
| 1123 |
+
video_grid_thw: torch.LongTensor | None = None,
|
| 1124 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1125 |
+
) -> tuple | BaseModelOutputWithPooling:
|
| 1126 |
+
r"""
|
| 1127 |
+
pixel_values_videos (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
|
| 1128 |
+
Video frames as patches.
|
| 1129 |
+
video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
|
| 1130 |
+
Temporal / height / width extent of every video.
|
| 1131 |
+
"""
|
| 1132 |
+
return self.get_image_features(pixel_values_videos, video_grid_thw, **kwargs)
|
| 1133 |
+
|
| 1134 |
+
@accepts_precomputed_kwargs(modality="image")
|
| 1135 |
+
@can_return_tuple
|
| 1136 |
+
@auto_docstring
|
| 1137 |
+
def get_image_features(
|
| 1138 |
+
self,
|
| 1139 |
+
pixel_values: torch.FloatTensor,
|
| 1140 |
+
image_grid_thw: torch.LongTensor | None = None,
|
| 1141 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1142 |
+
) -> tuple | BaseModelOutputWithPooling:
|
| 1143 |
+
r"""
|
| 1144 |
+
pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
|
| 1145 |
+
Images as patches.
|
| 1146 |
+
image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
|
| 1147 |
+
Temporal / height / width extent of every image.
|
| 1148 |
+
"""
|
| 1149 |
+
pixel_values = pixel_values.type(self.visual.dtype)
|
| 1150 |
+
vis: BaseModelOutputWithPooling = self.visual(pixel_values, grid_thw=image_grid_thw, return_dict=True, **kwargs)
|
| 1151 |
+
per_image = (image_grid_thw.prod(-1) // self.visual.spatial_merge_size**2).tolist()
|
| 1152 |
+
vis.pooler_output = torch.split(vis.pooler_output, per_image)
|
| 1153 |
+
return vis
|
| 1154 |
+
|
| 1155 |
+
def get_placeholder_mask(
|
| 1156 |
+
self,
|
| 1157 |
+
input_ids: torch.LongTensor,
|
| 1158 |
+
inputs_embeds: torch.FloatTensor,
|
| 1159 |
+
image_features: torch.FloatTensor | None = None,
|
| 1160 |
+
video_features: torch.FloatTensor | None = None,
|
| 1161 |
+
):
|
| 1162 |
+
"""Boolean masks over the image / video placeholder tokens, checked
|
| 1163 |
+
against the number of visual features supplied."""
|
| 1164 |
+
if input_ids is None:
|
| 1165 |
+
embed = self.get_input_embeddings()
|
| 1166 |
+
dev = inputs_embeds.device
|
| 1167 |
+
image_mask = (inputs_embeds == embed(torch.tensor(self.config.image_token_id, dtype=torch.long, device=dev))).all(-1)
|
| 1168 |
+
video_mask = (inputs_embeds == embed(torch.tensor(self.config.video_token_id, dtype=torch.long, device=dev))).all(-1)
|
| 1169 |
+
else:
|
| 1170 |
+
image_mask = input_ids == self.config.image_token_id
|
| 1171 |
+
video_mask = input_ids == self.config.video_token_id
|
| 1172 |
+
|
| 1173 |
+
n_img = image_mask.sum()
|
| 1174 |
+
image_mask = image_mask.unsqueeze(-1).to(inputs_embeds.device)
|
| 1175 |
+
if image_features is not None:
|
| 1176 |
+
torch_compilable_check(
|
| 1177 |
+
n_img * inputs_embeds.shape[-1] == image_features.numel(),
|
| 1178 |
+
f"Image features and image tokens do not match, tokens: {n_img}, features: {image_features.shape[0]}",
|
| 1179 |
+
)
|
| 1180 |
+
n_vid = video_mask.sum()
|
| 1181 |
+
video_mask = video_mask.unsqueeze(-1).to(inputs_embeds.device)
|
| 1182 |
+
if video_features is not None:
|
| 1183 |
+
torch_compilable_check(
|
| 1184 |
+
n_vid * inputs_embeds.shape[-1] == video_features.numel(),
|
| 1185 |
+
f"Video features and video tokens do not match, tokens: {n_vid}, features: {video_features.shape[0]}",
|
| 1186 |
+
)
|
| 1187 |
+
return image_mask, video_mask
|
| 1188 |
+
|
| 1189 |
+
def compute_3d_position_ids(
|
| 1190 |
+
self,
|
| 1191 |
+
input_ids: torch.Tensor | None,
|
| 1192 |
+
inputs_embeds: torch.Tensor | None,
|
| 1193 |
+
image_grid_thw: torch.Tensor | None = None,
|
| 1194 |
+
video_grid_thw: torch.Tensor | None = None,
|
| 1195 |
+
attention_mask: torch.Tensor | None = None,
|
| 1196 |
+
past_key_values: torch.Tensor | None = None,
|
| 1197 |
+
mm_token_type_ids: torch.IntTensor | None = None,
|
| 1198 |
+
) -> torch.Tensor | None:
|
| 1199 |
+
past_len = 0 if past_key_values is None else past_key_values.get_seq_length()
|
| 1200 |
+
has_vision = image_grid_thw is not None or video_grid_thw is not None
|
| 1201 |
+
if has_vision and mm_token_type_ids is None and input_ids is not None:
|
| 1202 |
+
raise ValueError(
|
| 1203 |
+
"Multimodal data was passed (via `image_grid_thw` or `video_grid_thw`) but `mm_token_type_ids` is "
|
| 1204 |
+
"missing. Pass `mm_token_type_ids` (returned by the processor) so the 3-D rotary positions can be built."
|
| 1205 |
+
)
|
| 1206 |
+
fresh = input_ids is not None and mm_token_type_ids is not None and has_vision
|
| 1207 |
+
|
| 1208 |
+
if fresh and (self.rope_deltas is None or past_len == 0):
|
| 1209 |
+
position_ids, self.rope_deltas = self.get_rope_index(
|
| 1210 |
+
input_ids, image_grid_thw=image_grid_thw, video_grid_thw=video_grid_thw,
|
| 1211 |
+
attention_mask=attention_mask, mm_token_type_ids=mm_token_type_ids,
|
| 1212 |
+
)
|
| 1213 |
+
return position_ids
|
| 1214 |
+
# continuing generation, or embeds-only input: reuse the stored deltas
|
| 1215 |
+
if self.rope_deltas is not None and (past_len > 0 or input_ids is None):
|
| 1216 |
+
bsz, seq, _ = inputs_embeds.shape
|
| 1217 |
+
if attention_mask is not None:
|
| 1218 |
+
position_ids = attention_mask.long().cumsum(-1) - 1
|
| 1219 |
+
position_ids = position_ids.masked_fill(attention_mask == 0, 0)
|
| 1220 |
+
position_ids = position_ids.view(1, bsz, -1).repeat(3, 1, 1).to(inputs_embeds.device)
|
| 1221 |
+
else:
|
| 1222 |
+
position_ids = torch.arange(past_len, past_len + seq)
|
| 1223 |
+
position_ids = position_ids.view(1, 1, -1).expand(3, bsz, -1).to(inputs_embeds.device)
|
| 1224 |
+
delta = self.rope_deltas.repeat_interleave(bsz // self.rope_deltas.shape[0], dim=0)
|
| 1225 |
+
return position_ids + delta.to(device=inputs_embeds.device)
|
| 1226 |
+
return None
|
| 1227 |
+
|
| 1228 |
+
@auto_docstring
|
| 1229 |
+
@can_return_tuple
|
| 1230 |
+
def forward(
|
| 1231 |
+
self,
|
| 1232 |
+
input_ids: torch.LongTensor = None,
|
| 1233 |
+
attention_mask: torch.Tensor | None = None,
|
| 1234 |
+
position_ids: torch.LongTensor | None = None,
|
| 1235 |
+
past_key_values: Cache | None = None,
|
| 1236 |
+
inputs_embeds: torch.FloatTensor | None = None,
|
| 1237 |
+
pixel_values: torch.Tensor | None = None,
|
| 1238 |
+
pixel_values_videos: torch.FloatTensor | None = None,
|
| 1239 |
+
image_grid_thw: torch.LongTensor | None = None,
|
| 1240 |
+
video_grid_thw: torch.LongTensor | None = None,
|
| 1241 |
+
mm_token_type_ids: torch.IntTensor | None = None,
|
| 1242 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1243 |
+
) -> tuple | AgnesModelOutput:
|
| 1244 |
+
r"""
|
| 1245 |
+
image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
|
| 1246 |
+
Temporal / height / width extent of every image.
|
| 1247 |
+
video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
|
| 1248 |
+
Temporal / height / width extent of every video.
|
| 1249 |
+
"""
|
| 1250 |
+
if (input_ids is None) ^ (inputs_embeds is not None):
|
| 1251 |
+
raise ValueError("You must specify exactly one of input_ids or inputs_embeds")
|
| 1252 |
+
if inputs_embeds is None:
|
| 1253 |
+
inputs_embeds = self.get_input_embeddings()(input_ids)
|
| 1254 |
+
|
| 1255 |
+
if pixel_values is not None:
|
| 1256 |
+
feats = torch.cat(
|
| 1257 |
+
self.get_image_features(pixel_values, image_grid_thw, return_dict=True, **kwargs).pooler_output, dim=0
|
| 1258 |
+
).to(inputs_embeds.device, inputs_embeds.dtype)
|
| 1259 |
+
mask, _ = self.get_placeholder_mask(input_ids, inputs_embeds=inputs_embeds, image_features=feats)
|
| 1260 |
+
inputs_embeds = inputs_embeds.masked_scatter(mask, feats)
|
| 1261 |
+
|
| 1262 |
+
if pixel_values_videos is not None:
|
| 1263 |
+
feats = torch.cat(
|
| 1264 |
+
self.get_video_features(pixel_values_videos, video_grid_thw, return_dict=True, **kwargs).pooler_output, dim=0
|
| 1265 |
+
).to(inputs_embeds.device, inputs_embeds.dtype)
|
| 1266 |
+
_, mask = self.get_placeholder_mask(input_ids, inputs_embeds=inputs_embeds, video_features=feats)
|
| 1267 |
+
inputs_embeds = inputs_embeds.masked_scatter(mask, feats)
|
| 1268 |
+
|
| 1269 |
+
if position_ids is None:
|
| 1270 |
+
position_ids = self.compute_3d_position_ids(
|
| 1271 |
+
input_ids=input_ids, image_grid_thw=image_grid_thw, video_grid_thw=video_grid_thw,
|
| 1272 |
+
inputs_embeds=inputs_embeds, attention_mask=attention_mask, past_key_values=past_key_values,
|
| 1273 |
+
mm_token_type_ids=mm_token_type_ids,
|
| 1274 |
+
)
|
| 1275 |
+
|
| 1276 |
+
out = self.language_model(
|
| 1277 |
+
input_ids=None, position_ids=position_ids, attention_mask=attention_mask,
|
| 1278 |
+
past_key_values=past_key_values, inputs_embeds=inputs_embeds, **kwargs,
|
| 1279 |
+
)
|
| 1280 |
+
return AgnesModelOutput(**out, rope_deltas=self.rope_deltas)
|
| 1281 |
+
|
| 1282 |
+
|
| 1283 |
+
@auto_docstring
|
| 1284 |
+
class AgnesForCausalLM(AgnesPreTrainedModel, GenerationMixin):
|
| 1285 |
+
"""Text-only head over `AgnesTextModel`."""
|
| 1286 |
+
|
| 1287 |
+
_tied_weights_keys = {"lm_head.weight": "model.embed_tokens.weight"}
|
| 1288 |
+
_tp_plan = {"lm_head": "colwise_gather_output"}
|
| 1289 |
+
_pp_plan = {"lm_head": (["hidden_states"], ["logits"])}
|
| 1290 |
+
config: AgnesTextConfig
|
| 1291 |
+
config_class = AgnesTextConfig
|
| 1292 |
+
_keys_to_ignore_on_load_unexpected = [r"^mtp.*", r"^model.visual.*"]
|
| 1293 |
+
|
| 1294 |
+
def __init__(self, config):
|
| 1295 |
+
super().__init__(config)
|
| 1296 |
+
self.model = AgnesTextModel(config)
|
| 1297 |
+
self.vocab_size = config.vocab_size
|
| 1298 |
+
self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False)
|
| 1299 |
+
self.post_init()
|
| 1300 |
+
|
| 1301 |
+
@can_return_tuple
|
| 1302 |
+
@auto_docstring
|
| 1303 |
+
def forward(
|
| 1304 |
+
self,
|
| 1305 |
+
input_ids: torch.LongTensor | None = None,
|
| 1306 |
+
attention_mask: torch.Tensor | None = None,
|
| 1307 |
+
position_ids: torch.LongTensor | None = None,
|
| 1308 |
+
past_key_values: Cache | None = None,
|
| 1309 |
+
inputs_embeds: torch.FloatTensor | None = None,
|
| 1310 |
+
labels: torch.LongTensor | None = None,
|
| 1311 |
+
use_cache: bool | None = None,
|
| 1312 |
+
logits_to_keep: int | torch.Tensor = 0,
|
| 1313 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1314 |
+
) -> CausalLMOutputWithPast:
|
| 1315 |
+
r"""
|
| 1316 |
+
labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1317 |
+
Targets for the language-modelling loss; -100 marks ignored positions.
|
| 1318 |
+
"""
|
| 1319 |
+
out: BaseModelOutputWithPast = self.model(
|
| 1320 |
+
input_ids=input_ids, attention_mask=attention_mask, position_ids=position_ids,
|
| 1321 |
+
past_key_values=past_key_values, inputs_embeds=inputs_embeds, use_cache=use_cache, **kwargs,
|
| 1322 |
+
)
|
| 1323 |
+
keep = slice(-logits_to_keep, None) if isinstance(logits_to_keep, int) else logits_to_keep
|
| 1324 |
+
logits = self.lm_head(out.last_hidden_state[:, keep, :])
|
| 1325 |
+
loss = None
|
| 1326 |
+
if labels is not None:
|
| 1327 |
+
loss = self.loss_function(logits=logits, labels=labels, vocab_size=self.config.vocab_size, **kwargs)
|
| 1328 |
+
return CausalLMOutputWithPast(
|
| 1329 |
+
loss=loss, logits=logits, past_key_values=out.past_key_values,
|
| 1330 |
+
hidden_states=out.hidden_states, attentions=out.attentions,
|
| 1331 |
+
)
|
| 1332 |
+
|
| 1333 |
+
|
| 1334 |
+
class AgnesForConditionalGeneration(AgnesPreTrainedModel, GenerationMixin):
|
| 1335 |
+
_tied_weights_keys = {"lm_head.weight": "model.language_model.embed_tokens.weight"}
|
| 1336 |
+
accepts_loss_kwargs = False
|
| 1337 |
+
config: AgnesConfig
|
| 1338 |
+
config_class = AgnesConfig
|
| 1339 |
+
|
| 1340 |
+
def __init__(self, config):
|
| 1341 |
+
super().__init__(config)
|
| 1342 |
+
self.model = AgnesModel(config)
|
| 1343 |
+
self.lm_head = nn.Linear(config.text_config.hidden_size, config.text_config.vocab_size, bias=False)
|
| 1344 |
+
self.post_init()
|
| 1345 |
+
|
| 1346 |
+
@auto_docstring
|
| 1347 |
+
def get_video_features(
|
| 1348 |
+
self,
|
| 1349 |
+
pixel_values_videos: torch.FloatTensor,
|
| 1350 |
+
video_grid_thw: torch.LongTensor | None = None,
|
| 1351 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1352 |
+
) -> tuple | BaseModelOutputWithPooling:
|
| 1353 |
+
r"""
|
| 1354 |
+
pixel_values_videos (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
|
| 1355 |
+
Video frames as patches.
|
| 1356 |
+
video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
|
| 1357 |
+
Temporal / height / width extent of every video.
|
| 1358 |
+
"""
|
| 1359 |
+
return self.model.get_video_features(pixel_values_videos, video_grid_thw, **kwargs)
|
| 1360 |
+
|
| 1361 |
+
@auto_docstring
|
| 1362 |
+
def get_image_features(
|
| 1363 |
+
self,
|
| 1364 |
+
pixel_values: torch.FloatTensor,
|
| 1365 |
+
image_grid_thw: torch.LongTensor | None = None,
|
| 1366 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1367 |
+
) -> tuple | BaseModelOutputWithPooling:
|
| 1368 |
+
r"""
|
| 1369 |
+
pixel_values (`torch.FloatTensor` of shape `(batch_size, num_channels, image_size, image_size)`):
|
| 1370 |
+
Images as patches.
|
| 1371 |
+
image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
|
| 1372 |
+
Temporal / height / width extent of every image.
|
| 1373 |
+
"""
|
| 1374 |
+
return self.model.get_image_features(pixel_values, image_grid_thw, **kwargs)
|
| 1375 |
+
|
| 1376 |
+
@can_return_tuple
|
| 1377 |
+
def forward(
|
| 1378 |
+
self,
|
| 1379 |
+
input_ids: torch.LongTensor = None,
|
| 1380 |
+
attention_mask: torch.Tensor | None = None,
|
| 1381 |
+
position_ids: torch.LongTensor | None = None,
|
| 1382 |
+
past_key_values: Cache | None = None,
|
| 1383 |
+
inputs_embeds: torch.FloatTensor | None = None,
|
| 1384 |
+
labels: torch.LongTensor | None = None,
|
| 1385 |
+
pixel_values: torch.Tensor | None = None,
|
| 1386 |
+
pixel_values_videos: torch.FloatTensor | None = None,
|
| 1387 |
+
image_grid_thw: torch.LongTensor | None = None,
|
| 1388 |
+
video_grid_thw: torch.LongTensor | None = None,
|
| 1389 |
+
mm_token_type_ids: torch.IntTensor | None = None,
|
| 1390 |
+
logits_to_keep: int | torch.Tensor = 0,
|
| 1391 |
+
**kwargs: Unpack[TransformersKwargs],
|
| 1392 |
+
) -> tuple | AgnesCausalLMOutput:
|
| 1393 |
+
r"""
|
| 1394 |
+
labels (`torch.LongTensor` of shape `(batch_size, sequence_length)`, *optional*):
|
| 1395 |
+
Targets for the language-modelling loss; -100 marks ignored positions.
|
| 1396 |
+
image_grid_thw (`torch.LongTensor` of shape `(num_images, 3)`, *optional*):
|
| 1397 |
+
Temporal / height / width extent of every image.
|
| 1398 |
+
video_grid_thw (`torch.LongTensor` of shape `(num_videos, 3)`, *optional*):
|
| 1399 |
+
Temporal / height / width extent of every video.
|
| 1400 |
+
"""
|
| 1401 |
+
out = self.model(
|
| 1402 |
+
input_ids=input_ids, pixel_values=pixel_values, pixel_values_videos=pixel_values_videos,
|
| 1403 |
+
image_grid_thw=image_grid_thw, video_grid_thw=video_grid_thw, position_ids=position_ids,
|
| 1404 |
+
attention_mask=attention_mask, past_key_values=past_key_values, inputs_embeds=inputs_embeds,
|
| 1405 |
+
mm_token_type_ids=mm_token_type_ids, **kwargs,
|
| 1406 |
+
)
|
| 1407 |
+
keep = slice(-logits_to_keep, None) if isinstance(logits_to_keep, int) else logits_to_keep
|
| 1408 |
+
logits = self.lm_head(out[0][:, keep, :])
|
| 1409 |
+
loss = None
|
| 1410 |
+
if labels is not None:
|
| 1411 |
+
loss = self.loss_function(logits=logits, labels=labels, vocab_size=self.config.text_config.vocab_size)
|
| 1412 |
+
return AgnesCausalLMOutput(
|
| 1413 |
+
loss=loss, logits=logits, past_key_values=out.past_key_values,
|
| 1414 |
+
hidden_states=out.hidden_states, attentions=out.attentions, rope_deltas=out.rope_deltas,
|
| 1415 |
+
)
|
| 1416 |
+
|
| 1417 |
+
def prepare_inputs_for_generation(
|
| 1418 |
+
self,
|
| 1419 |
+
input_ids,
|
| 1420 |
+
past_key_values=None,
|
| 1421 |
+
attention_mask=None,
|
| 1422 |
+
inputs_embeds=None,
|
| 1423 |
+
position_ids=None,
|
| 1424 |
+
use_cache=True,
|
| 1425 |
+
pixel_values=None,
|
| 1426 |
+
pixel_values_videos=None,
|
| 1427 |
+
image_grid_thw=None,
|
| 1428 |
+
video_grid_thw=None,
|
| 1429 |
+
is_first_iteration=False,
|
| 1430 |
+
**kwargs,
|
| 1431 |
+
):
|
| 1432 |
+
# pixels are consumed on the first step only; later steps run on the cache
|
| 1433 |
+
inputs = super().prepare_inputs_for_generation(
|
| 1434 |
+
input_ids, past_key_values=past_key_values, attention_mask=attention_mask, inputs_embeds=inputs_embeds,
|
| 1435 |
+
position_ids=position_ids, pixel_values=pixel_values, pixel_values_videos=pixel_values_videos,
|
| 1436 |
+
image_grid_thw=image_grid_thw, video_grid_thw=video_grid_thw, use_cache=use_cache,
|
| 1437 |
+
is_first_iteration=is_first_iteration, **kwargs,
|
| 1438 |
+
)
|
| 1439 |
+
if not is_first_iteration and use_cache:
|
| 1440 |
+
inputs["pixel_values"] = None
|
| 1441 |
+
inputs["pixel_values_videos"] = None
|
| 1442 |
+
return inputs
|
| 1443 |
+
|
| 1444 |
+
def _prepare_position_ids_for_generation(self, inputs_tensor, model_kwargs):
|
| 1445 |
+
# four-row positions: text row on top of the three rotary axes
|
| 1446 |
+
text_pos = super()._prepare_position_ids_for_generation(inputs_tensor, model_kwargs)
|
| 1447 |
+
|
| 1448 |
+
past_len = 0
|
| 1449 |
+
if (cache := model_kwargs.get("past_key_values")) is not None:
|
| 1450 |
+
past_len = cache.get_seq_length()
|
| 1451 |
+
if past_len != 0 and self.model.rope_deltas is not None:
|
| 1452 |
+
return text_pos[None, ...] + self.model.rope_deltas
|
| 1453 |
+
|
| 1454 |
+
if "input_ids" in model_kwargs and model_kwargs["input_ids"].shape[1] > 0:
|
| 1455 |
+
inputs_tensor = model_kwargs["input_ids"]
|
| 1456 |
+
is_ids = len(inputs_tensor.shape) == 2 and inputs_tensor.dtype in [torch.int, torch.long]
|
| 1457 |
+
has_vision = model_kwargs.get("image_grid_thw") is not None or model_kwargs.get("video_grid_thw") is not None
|
| 1458 |
+
if is_ids and model_kwargs.get("mm_token_type_ids") is not None and has_vision:
|
| 1459 |
+
rest = {k: v for k, v in model_kwargs.items() if k != "input_ids"}
|
| 1460 |
+
axes, self.model.rope_deltas = self.model.get_rope_index(inputs_tensor, **rest)
|
| 1461 |
+
else:
|
| 1462 |
+
axes = text_pos.unsqueeze(0).expand(3, -1, -1)
|
| 1463 |
+
self.model.rope_deltas = torch.zeros(inputs_tensor.shape[0], 1, dtype=torch.long, device=inputs_tensor.device)
|
| 1464 |
+
return torch.cat([text_pos[None, ...], axes], dim=0)
|
| 1465 |
+
|
| 1466 |
+
def _count_images_and_videos(
|
| 1467 |
+
self,
|
| 1468 |
+
input_ids: torch.LongTensor | None,
|
| 1469 |
+
inputs_embeds: torch.Tensor | None = None,
|
| 1470 |
+
) -> tuple[torch.Tensor, torch.Tensor]:
|
| 1471 |
+
"""Per-sample number of images and of video frames, read off the
|
| 1472 |
+
vision-start / placeholder tokens."""
|
| 1473 |
+
img_id = self.config.image_token_id
|
| 1474 |
+
vid_id = self.config.video_token_id
|
| 1475 |
+
start_id = self.config.vision_start_token_id
|
| 1476 |
+
if inputs_embeds is not None:
|
| 1477 |
+
embed = self.get_input_embeddings()
|
| 1478 |
+
dev = inputs_embeds.device
|
| 1479 |
+
start = (inputs_embeds == embed(torch.tensor(start_id, dtype=torch.long, device=dev)))[..., 0]
|
| 1480 |
+
img = (inputs_embeds == embed(torch.tensor(img_id, dtype=torch.long, device=dev)))[..., 0]
|
| 1481 |
+
vid = (inputs_embeds == embed(torch.tensor(vid_id, dtype=torch.long, device=dev)))[..., 0]
|
| 1482 |
+
else:
|
| 1483 |
+
start = input_ids == start_id
|
| 1484 |
+
img = input_ids == img_id
|
| 1485 |
+
vid = input_ids == vid_id
|
| 1486 |
+
after_start = torch.roll(start, shifts=1, dims=1)
|
| 1487 |
+
return torch.sum(after_start & img, dim=1), torch.sum(after_start & vid, dim=1)
|
| 1488 |
+
|
| 1489 |
+
def _expand_inputs_for_generation(
|
| 1490 |
+
self,
|
| 1491 |
+
expand_size: int = 1,
|
| 1492 |
+
is_encoder_decoder: bool = False,
|
| 1493 |
+
input_ids: torch.LongTensor | None = None,
|
| 1494 |
+
**model_kwargs,
|
| 1495 |
+
) -> tuple[torch.LongTensor, dict[str, Any]]:
|
| 1496 |
+
# visual tensors have no batch axis (they are concatenated over samples),
|
| 1497 |
+
# so they are expanded sample by sample using the per-sample counts
|
| 1498 |
+
if expand_size == 1:
|
| 1499 |
+
return input_ids, model_kwargs
|
| 1500 |
+
|
| 1501 |
+
visual_keys = ["pixel_values", "image_grid_thw", "pixel_values_videos", "video_grid_thw"]
|
| 1502 |
+
|
| 1503 |
+
def expand_visual(d):
|
| 1504 |
+
image_grid_thw = model_kwargs.get("image_grid_thw", None)
|
| 1505 |
+
video_grid_thw = model_kwargs.get("video_grid_thw", None)
|
| 1506 |
+
n_img, n_vid = self._count_images_and_videos(input_ids, inputs_embeds=model_kwargs.get("inputs_embeds", None))
|
| 1507 |
+
|
| 1508 |
+
# n_vid counts frames (each frame carries a vision-start token); fold back to videos
|
| 1509 |
+
if video_grid_thw is not None:
|
| 1510 |
+
frame_cum = torch.cumsum(video_grid_thw[:, 0], dim=0)
|
| 1511 |
+
token_cum = torch.cumsum(n_vid, dim=0)
|
| 1512 |
+
edges = torch.searchsorted(frame_cum, token_cum)
|
| 1513 |
+
n_vid = torch.diff(torch.cat([-edges.new_ones(1), edges]))
|
| 1514 |
+
|
| 1515 |
+
def tile(x, lengths, times):
|
| 1516 |
+
parts = torch.split(x, lengths)
|
| 1517 |
+
reps = [times] + [1] * (x.dim() - 1)
|
| 1518 |
+
return torch.cat([p.repeat(*reps) for p in parts], dim=0)
|
| 1519 |
+
|
| 1520 |
+
for key in d:
|
| 1521 |
+
if key == "pixel_values":
|
| 1522 |
+
per = torch.split(image_grid_thw, list(n_img))
|
| 1523 |
+
d[key] = tile(d[key], [torch.prod(s, dim=1).sum() for s in per], expand_size)
|
| 1524 |
+
elif key == "image_grid_thw":
|
| 1525 |
+
d[key] = tile(d[key], list(n_img), expand_size)
|
| 1526 |
+
elif key == "pixel_values_videos":
|
| 1527 |
+
per = torch.split(video_grid_thw, list(n_vid))
|
| 1528 |
+
d[key] = tile(d[key], [torch.prod(s, dim=1).sum() for s in per], expand_size)
|
| 1529 |
+
elif key == "video_grid_thw":
|
| 1530 |
+
d[key] = tile(d[key], list(n_vid), expand_size)
|
| 1531 |
+
return d
|
| 1532 |
+
|
| 1533 |
+
def expand_rest(d):
|
| 1534 |
+
for key in d:
|
| 1535 |
+
if key == "position_ids" and d[key].ndim == 3:
|
| 1536 |
+
d[key] = d[key].repeat_interleave(expand_size, dim=1)
|
| 1537 |
+
elif d[key] is not None and isinstance(d[key], torch.Tensor) and key not in visual_keys:
|
| 1538 |
+
d[key] = d[key].repeat_interleave(expand_size, dim=0)
|
| 1539 |
+
return d
|
| 1540 |
+
|
| 1541 |
+
model_kwargs = expand_visual(model_kwargs)
|
| 1542 |
+
if input_ids is not None:
|
| 1543 |
+
input_ids = input_ids.repeat_interleave(expand_size, dim=0)
|
| 1544 |
+
model_kwargs = expand_rest(model_kwargs)
|
| 1545 |
+
if is_encoder_decoder:
|
| 1546 |
+
if model_kwargs.get("encoder_outputs") is None:
|
| 1547 |
+
raise ValueError("If `is_encoder_decoder` is True, make sure that `encoder_outputs` is defined.")
|
| 1548 |
+
model_kwargs["encoder_outputs"] = expand_rest(model_kwargs["encoder_outputs"])
|
| 1549 |
+
return input_ids, model_kwargs
|
| 1550 |
+
|
| 1551 |
+
|
| 1552 |
+
__all__ = [
|
| 1553 |
+
"AgnesPreTrainedModel",
|
| 1554 |
+
"AgnesVisionModel",
|
| 1555 |
+
"AgnesTextModel",
|
| 1556 |
+
"AgnesModel",
|
| 1557 |
+
"AgnesForCausalLM",
|
| 1558 |
+
"AgnesForConditionalGeneration",
|
| 1559 |
+
]
|
preprocessor_config.json
ADDED
|
@@ -0,0 +1,25 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"processor_class": "AgnesProcessor",
|
| 3 |
+
"image_processor_type": "AgnesImageProcessor",
|
| 4 |
+
"auto_map": {
|
| 5 |
+
"AutoProcessor": "processing_agnes.AgnesProcessor",
|
| 6 |
+
"AutoImageProcessor": "image_processing_agnes.AgnesImageProcessor"
|
| 7 |
+
},
|
| 8 |
+
"size": {
|
| 9 |
+
"longest_edge": 16777216,
|
| 10 |
+
"shortest_edge": 65536
|
| 11 |
+
},
|
| 12 |
+
"patch_size": 16,
|
| 13 |
+
"temporal_patch_size": 2,
|
| 14 |
+
"merge_size": 2,
|
| 15 |
+
"image_mean": [
|
| 16 |
+
0.5,
|
| 17 |
+
0.5,
|
| 18 |
+
0.5
|
| 19 |
+
],
|
| 20 |
+
"image_std": [
|
| 21 |
+
0.5,
|
| 22 |
+
0.5,
|
| 23 |
+
0.5
|
| 24 |
+
]
|
| 25 |
+
}
|
processing_agnes.py
ADDED
|
@@ -0,0 +1,116 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2026 Agnes AI. All rights reserved.
|
| 2 |
+
#
|
| 3 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 4 |
+
# you may not use this file except in compliance with the License.
|
| 5 |
+
# You may obtain a copy of the License at
|
| 6 |
+
#
|
| 7 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
#
|
| 9 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
# See the License for the specific language governing permissions and
|
| 13 |
+
# limitations under the License.
|
| 14 |
+
"""Processor for Agnes 3.0 Flash: tokenizer + image processor + video processor."""
|
| 15 |
+
|
| 16 |
+
import numpy as np
|
| 17 |
+
|
| 18 |
+
from transformers.processing_utils import MultiModalData, ProcessingKwargs, ProcessorMixin
|
| 19 |
+
from transformers.utils import auto_docstring, logging
|
| 20 |
+
|
| 21 |
+
|
| 22 |
+
logger = logging.get_logger(__name__)
|
| 23 |
+
|
| 24 |
+
|
| 25 |
+
class AgnesProcessorKwargs(ProcessingKwargs, total=False):
|
| 26 |
+
_defaults = {
|
| 27 |
+
"text_kwargs": {"padding": False, "return_token_type_ids": False, "return_mm_token_type_ids": True},
|
| 28 |
+
"videos_kwargs": {"return_metadata": True},
|
| 29 |
+
}
|
| 30 |
+
|
| 31 |
+
|
| 32 |
+
@auto_docstring
|
| 33 |
+
class AgnesProcessor(ProcessorMixin):
|
| 34 |
+
valid_processor_kwargs = AgnesProcessorKwargs
|
| 35 |
+
|
| 36 |
+
def __init__(self, image_processor=None, tokenizer=None, video_processor=None, chat_template=None, **kwargs):
|
| 37 |
+
self.image_token = getattr(tokenizer, "image_token", "<|image_pad|>")
|
| 38 |
+
self.video_token = getattr(tokenizer, "video_token", "<|video_pad|>")
|
| 39 |
+
self.image_token_id = getattr(tokenizer, "image_token_id", None) or tokenizer.convert_tokens_to_ids(self.image_token)
|
| 40 |
+
self.video_token_id = getattr(tokenizer, "video_token_id", None) or tokenizer.convert_tokens_to_ids(self.video_token)
|
| 41 |
+
super().__init__(image_processor, tokenizer, video_processor, chat_template=chat_template)
|
| 42 |
+
self.vision_start_token = getattr(tokenizer, "vision_start_token", "<|vision_start|>")
|
| 43 |
+
self.vision_end_token = getattr(tokenizer, "vision_end_token", "<|vision_end|>")
|
| 44 |
+
self.vision_start_token_id = getattr(tokenizer, "vision_start_token_id", None) or tokenizer.convert_tokens_to_ids(
|
| 45 |
+
self.vision_start_token
|
| 46 |
+
)
|
| 47 |
+
self.vision_end_token_id = getattr(tokenizer, "vision_end_token_id", None) or tokenizer.convert_tokens_to_ids(
|
| 48 |
+
self.vision_end_token
|
| 49 |
+
)
|
| 50 |
+
|
| 51 |
+
def replace_image_token(self, image_inputs: dict, image_idx: int) -> str:
|
| 52 |
+
per_token = self.image_processor.merge_size**2
|
| 53 |
+
n = image_inputs["image_grid_thw"][image_idx].prod() // per_token
|
| 54 |
+
return self.image_token * n
|
| 55 |
+
|
| 56 |
+
def replace_video_token(self, video_inputs: dict, video_idx: int) -> str:
|
| 57 |
+
per_token = self.video_processor.merge_size**2
|
| 58 |
+
thw = video_inputs["video_grid_thw"][video_idx]
|
| 59 |
+
n_frames = thw[0]
|
| 60 |
+
per_frame = thw[1:].prod() // per_token
|
| 61 |
+
meta = video_inputs["video_metadata"][video_idx]
|
| 62 |
+
if meta.fps is None:
|
| 63 |
+
logger.warning_once(
|
| 64 |
+
"Frame timestamps are needed to build the video prompt but the `fps` of the input video could not be "
|
| 65 |
+
"inferred (no `video_metadata`, pre-sampled frames?). Defaulting to `fps=24`."
|
| 66 |
+
)
|
| 67 |
+
meta.fps = 24 if meta.fps is None else meta.fps
|
| 68 |
+
stamps = self._frame_timestamps(meta.frames_indices, meta.fps, self.video_processor.temporal_patch_size)
|
| 69 |
+
text = ""
|
| 70 |
+
for f in range(n_frames):
|
| 71 |
+
text += f"<{stamps[f]:.1f} seconds>"
|
| 72 |
+
text += self.vision_start_token + self.video_token * per_frame + self.vision_end_token
|
| 73 |
+
return text
|
| 74 |
+
|
| 75 |
+
def _get_num_multimodal_tokens(self, image_sizes=None, video_sizes=None, **kwargs):
|
| 76 |
+
"""Placeholder counts for inputs of the given sizes, without running the
|
| 77 |
+
processors on real pixels."""
|
| 78 |
+
data = {}
|
| 79 |
+
if image_sizes is not None:
|
| 80 |
+
ik = AgnesProcessorKwargs._defaults.get("images_kwargs", {})
|
| 81 |
+
ik.update(kwargs)
|
| 82 |
+
merge = ik.get("merge_size", None) or self.image_processor.merge_size
|
| 83 |
+
patches = [self.image_processor.get_number_of_image_patches(*s, ik) for s in image_sizes]
|
| 84 |
+
data.update({"num_image_tokens": [p // merge**2 for p in patches], "num_image_patches": patches})
|
| 85 |
+
if video_sizes is not None:
|
| 86 |
+
vk = AgnesProcessorKwargs._defaults.get("videos_kwargs", {})
|
| 87 |
+
vk.update(kwargs)
|
| 88 |
+
merge = vk.get("merge_size", None) or self.video_processor.merge_size
|
| 89 |
+
patches = [self.video_processor.get_number_of_video_patches(*s, vk) for s in video_sizes]
|
| 90 |
+
data["num_video_tokens"] = [p // merge**2 for p in patches]
|
| 91 |
+
return MultiModalData(**data)
|
| 92 |
+
|
| 93 |
+
def post_process_image_text_to_text(self, generated_outputs, skip_special_tokens=True, clean_up_tokenization_spaces=False, **kwargs):
|
| 94 |
+
"""Decode generated ids to text."""
|
| 95 |
+
return self.tokenizer.batch_decode(
|
| 96 |
+
generated_outputs, skip_special_tokens=skip_special_tokens,
|
| 97 |
+
clean_up_tokenization_spaces=clean_up_tokenization_spaces, **kwargs,
|
| 98 |
+
)
|
| 99 |
+
|
| 100 |
+
@property
|
| 101 |
+
def model_input_names(self):
|
| 102 |
+
return super().model_input_names + ["mm_token_type_ids"]
|
| 103 |
+
|
| 104 |
+
@staticmethod
|
| 105 |
+
def _frame_timestamps(indices: list[int] | np.ndarray, video_fps: float, merge_size: int = 2):
|
| 106 |
+
"""One timestamp per temporal patch: the mean of the first and last
|
| 107 |
+
frame time inside the patch."""
|
| 108 |
+
if not isinstance(indices, list):
|
| 109 |
+
indices = indices.tolist()
|
| 110 |
+
if len(indices) % merge_size != 0:
|
| 111 |
+
indices.extend(indices[-1] for _ in range(merge_size - len(indices) % merge_size))
|
| 112 |
+
times = [i / video_fps for i in indices]
|
| 113 |
+
return [(times[i] + times[i + merge_size - 1]) / 2 for i in range(0, len(times), merge_size)]
|
| 114 |
+
|
| 115 |
+
|
| 116 |
+
__all__ = ["AgnesProcessor"]
|
recipe.yaml
ADDED
|
@@ -0,0 +1,10 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
default_stage:
|
| 2 |
+
default_modifiers:
|
| 3 |
+
QuantizationModifier:
|
| 4 |
+
targets: [Linear]
|
| 5 |
+
ignore: [lm_head, model.language_model.embed_tokens, model.visual, 're:^model\.visual\..*',
|
| 6 |
+
're:.*delta_attn\.conv1d$', 're:.*delta_attn\.in_proj_a$', 're:.*delta_attn\.in_proj_b$',
|
| 7 |
+
're:^mtp.*']
|
| 8 |
+
scheme: FP8
|
| 9 |
+
bypass_divisibility_checks: false
|
| 10 |
+
requires_calibration_data: true
|
serve.sh
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/bin/bash
|
| 2 |
+
# Serve Agnes 3.0 Flash with a stock sglang image.
|
| 3 |
+
#
|
| 4 |
+
# Tested image: lmsysorg/sglang:nightly-dev-20260908-20ca564b
|
| 5 |
+
#
|
| 6 |
+
# docker run --gpus all --shm-size 64g -p 30001:30002 \
|
| 7 |
+
# -v /path/to/agnes-3.0-flash:/model \
|
| 8 |
+
# lmsysorg/sglang:nightly-dev-20260908-20ca564b \
|
| 9 |
+
# bash /model/serve.sh [extra sglang arguments, e.g. --tp 2]
|
| 10 |
+
#
|
| 11 |
+
# The script overlays three files from sglang_patch/ onto the image's sglang
|
| 12 |
+
# package (a prebuilt variant when one matches the installed version, else
|
| 13 |
+
# apply_patch.py patches the installed files in place) and starts the server
|
| 14 |
+
# on port 30002 inside the container.
|
| 15 |
+
set -euo pipefail
|
| 16 |
+
D="$(cd "$(dirname "$0")" && pwd)"
|
| 17 |
+
PKG="$(python3 -c 'import sglang, os; print(os.path.dirname(sglang.__file__))')"
|
| 18 |
+
VER="$(python3 -c 'import sglang; print(getattr(sglang, "__version__", ""))' 2>/dev/null || true)"
|
| 19 |
+
|
| 20 |
+
case "$VER" in
|
| 21 |
+
*20ca564b*) VARIANT="nightly-dev-20260908-20ca564b" ;;
|
| 22 |
+
0.5.19*) VARIANT="v0.5.19" ;;
|
| 23 |
+
*) VARIANT="" ;;
|
| 24 |
+
esac
|
| 25 |
+
|
| 26 |
+
if [ -n "$VARIANT" ] && [ -d "$D/sglang_patch/$VARIANT/sglang" ]; then
|
| 27 |
+
echo "[serve.sh] sglang $VER -> prebuilt patch $VARIANT"
|
| 28 |
+
cp -r "$D/sglang_patch/$VARIANT/sglang/." "$PKG/"
|
| 29 |
+
else
|
| 30 |
+
echo "[serve.sh] sglang $VER -> patching installed package in place"
|
| 31 |
+
python3 "$D/sglang_patch/apply_patch.py" "$PKG"
|
| 32 |
+
fi
|
| 33 |
+
|
| 34 |
+
export AGNES_MODEL_PATH="$D"
|
| 35 |
+
exec python3 -m sglang.launch_server --model-path "$D" --trust-remote-code --host 0.0.0.0 --port 8080 "$@"
|
sglang_patch/README.md
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# sglang patch for Agnes 3.0 Flash
|
| 2 |
+
|
| 3 |
+
Serve with a stock sglang image; `serve.sh` overlays three files onto the image's
|
| 4 |
+
`sglang` package and starts the server.
|
| 5 |
+
|
| 6 |
+
| Variant | Source | Status |
|
| 7 |
+
|---|---|---|
|
| 8 |
+
| `nightly-dev-20260908-20ca564b/` | `lmsysorg/sglang:nightly-dev-20260908-20ca564b` | served and measured (see below) |
|
| 9 |
+
| `v0.5.19/` | `sglang==0.5.19` | generated from the release source, not served-tested |
|
| 10 |
+
| other versions | `apply_patch.py <path/to/sglang>` | patched in place by `serve.sh` when no variant matches |
|
| 11 |
+
|
| 12 |
+
```
|
| 13 |
+
docker run --gpus all --shm-size 64g -p 30001:30002 \
|
| 14 |
+
-v /path/to/agnes-3.0-flash:/model \
|
| 15 |
+
lmsysorg/sglang:nightly-dev-20260908-20ca564b \
|
| 16 |
+
bash /model/serve.sh # extra sglang args may follow, e.g. --tp 2
|
| 17 |
+
```
|
| 18 |
+
|
| 19 |
+
## The three files
|
| 20 |
+
|
| 21 |
+
| File | Change |
|
| 22 |
+
|---|---|
|
| 23 |
+
| `sglang/srt/configs/agnes.py` | new. Reads the `model_type: agnes` config and presents it to the server in the terms of its built-in hybrid (delta-rule + global attention) implementation: layer plan from `global_attention_interval`, the parallel FFN width added to `intermediate_size`, the checkpoint directory recorded as a config field for the loader. |
|
| 24 |
+
| `sglang/srt/utils/hf_transformers/common.py` | +3 lines at the end: registers `AgnesConfig` for `model_type` `agnes`. |
|
| 25 |
+
| `sglang/srt/models/qwen3_5.py` | one generator in front of the weight stream in `load_weights`: `delta_attn.*` → `linear_attn.*`, `global_attn.*` → `self_attn.*`, and each layer's `mlp.parallel_ffn.{gate,up,down}_proj` concatenated onto the main projections (gate/up along the output dim, down along the input dim). Checkpoints without a parallel branch pass through untouched. |
|
| 26 |
+
|
| 27 |
+
Nothing else in the image is modified. `--trust-remote-code` is required because sglang
|
| 28 |
+
resolves the model configuration through transformers first, which reads the
|
| 29 |
+
`configuration_agnes.py` shipped with the checkpoint. `serve.sh` also exports
|
| 30 |
+
`AGNES_MODEL_PATH` as a fallback for the loader.
|
| 31 |
+
|
| 32 |
+
## Numerics
|
| 33 |
+
|
| 34 |
+
Folding the parallel branch into the main MLP changes the reduction length of the down
|
| 35 |
+
projection, so served logits are not bit-identical to the transformers implementation.
|
| 36 |
+
Measured on the nightly image (TP1, H200, 2144 teacher-forced positions, full 248 320-way
|
| 37 |
+
softmax): full-vocabulary KL 5.9e-4 against the same weights served without the branch by the
|
| 38 |
+
unpatched engine, below the 6.5e-4 measured between the transformers and sglang
|
| 39 |
+
implementations of one and the same checkpoint. Perplexity 17.07 vs 17.05.
|
| 40 |
+
|
| 41 |
+
`apply_patch.py` anchors on `QWEN3_5_KV_SCALE_MAPPER` in `models/qwen3_5.py` and on the end of
|
| 42 |
+
`utils/hf_transformers/common.py`; it is idempotent.
|
sglang_patch/agnes_sglang_config.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Agnes 3.0 Flash configuration for sglang.
|
| 2 |
+
#
|
| 3 |
+
# The HF checkpoint says model_type "agnes", names its layer types
|
| 4 |
+
# agnes_delta_attention / agnes_global_attention and carries a parallel FFN
|
| 5 |
+
# branch per layer. The server runs it on its built-in hybrid
|
| 6 |
+
# (delta-rule + global attention) implementation, so this class maps those
|
| 7 |
+
# onto the fields that implementation reads; the checkpoint's tensor names
|
| 8 |
+
# are translated while loading (see the model file patched by apply_patch.py).
|
| 9 |
+
from sglang.srt.configs.qwen3_5 import Qwen3_5Config
|
| 10 |
+
|
| 11 |
+
AGNES_DELTA = "agnes_delta_attention"
|
| 12 |
+
AGNES_GLOBAL = "agnes_global_attention"
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class AgnesConfig(Qwen3_5Config):
|
| 16 |
+
model_type = "agnes"
|
| 17 |
+
|
| 18 |
+
def __init__(self, text_config=None, vision_config=None, **kwargs):
|
| 19 |
+
kwargs.pop("auto_map", None) # the transformers remote code is not used in the server
|
| 20 |
+
if isinstance(text_config, dict):
|
| 21 |
+
text_config = dict(text_config)
|
| 22 |
+
text_config["model_type"] = "qwen3_5_text"
|
| 23 |
+
width = int(text_config.pop("parallel_ffn_intermediate_size", 0) or 0)
|
| 24 |
+
plan = text_config.pop("layer_types", None)
|
| 25 |
+
interval = text_config.pop("global_attention_interval", None)
|
| 26 |
+
if interval is None:
|
| 27 |
+
interval = text_config.pop("full_attention_interval", None)
|
| 28 |
+
if interval is None and plan:
|
| 29 |
+
interval = next(i + 1 for i, t in enumerate(plan) if t == AGNES_GLOBAL)
|
| 30 |
+
text_config["full_attention_interval"] = int(interval or 4)
|
| 31 |
+
main = int(text_config["intermediate_size"])
|
| 32 |
+
# the parallel branch is folded into the main MLP at load time
|
| 33 |
+
text_config["intermediate_size"] = main + width
|
| 34 |
+
text_config["agnes_main_intermediate_size"] = main
|
| 35 |
+
text_config["agnes_parallel_ffn_intermediate_size"] = width
|
| 36 |
+
if isinstance(vision_config, dict):
|
| 37 |
+
vision_config = dict(vision_config)
|
| 38 |
+
vision_config["model_type"] = "qwen3_5"
|
| 39 |
+
kwargs["architectures"] = ["Qwen3_5ForConditionalGeneration"]
|
| 40 |
+
super().__init__(text_config=text_config, vision_config=vision_config, **kwargs)
|
| 41 |
+
# every downstream check sees the built-in hybrid architecture
|
| 42 |
+
self.model_type = "qwen3_5"
|
| 43 |
+
|
| 44 |
+
@classmethod
|
| 45 |
+
def from_pretrained(cls, pretrained_model_name_or_path, *args, **kwargs):
|
| 46 |
+
# The weight loader needs the checkpoint directory to pick up the parallel
|
| 47 |
+
# branch tensors. The path is written into the config *dict* before the
|
| 48 |
+
# object is built: the config reaches the worker processes through a
|
| 49 |
+
# to_dict round trip, which keeps fields that came in through __init__
|
| 50 |
+
# and drops attributes set afterwards (from_pretrained's own kwargs only
|
| 51 |
+
# override known fields, so they cannot carry it either).
|
| 52 |
+
path = str(pretrained_model_name_or_path)
|
| 53 |
+
config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
|
| 54 |
+
config_dict["agnes_model_path"] = path
|
| 55 |
+
if isinstance(config_dict.get("text_config"), dict):
|
| 56 |
+
config_dict["text_config"]["agnes_model_path"] = path
|
| 57 |
+
return cls.from_dict(config_dict, **kwargs)
|
sglang_patch/apply_patch.py
ADDED
|
@@ -0,0 +1,130 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
#!/usr/bin/env python3
|
| 2 |
+
"""Apply the Agnes patch to an sglang python package directory (the one that
|
| 3 |
+
contains `srt/`). Idempotent.
|
| 4 |
+
|
| 5 |
+
1. srt/configs/agnes.py new file
|
| 6 |
+
2. srt/utils/hf_transformers/common.py register AgnesConfig in _CONFIG_REGISTRY
|
| 7 |
+
3. srt/models/qwen3_5.py load_weights: translate the Agnes checkpoint
|
| 8 |
+
(tensor prefixes, parallel-FFN fold)
|
| 9 |
+
|
| 10 |
+
Usage: apply_patch.py <path/to/sglang> e.g. .../site-packages/sglang
|
| 11 |
+
"""
|
| 12 |
+
|
| 13 |
+
import os
|
| 14 |
+
import shutil
|
| 15 |
+
import sys
|
| 16 |
+
|
| 17 |
+
HERE = os.path.dirname(os.path.abspath(__file__))
|
| 18 |
+
MARK = "# === agnes ==="
|
| 19 |
+
|
| 20 |
+
TRANSLATE = '''
|
| 21 |
+
# === agnes ===
|
| 22 |
+
# Agnes 3.0 Flash checkpoints (config model_type "agnes") use their own tensor
|
| 23 |
+
# prefixes and carry a parallel FFN branch per layer. This generator sits at
|
| 24 |
+
# the top of the weight stream and turns it into what the implementation below
|
| 25 |
+
# expects: delta_attn -> linear_attn, global_attn -> self_attn, and the branch
|
| 26 |
+
# concatenated onto the main gate / up (dim 0) and down (dim 1) projections,
|
| 27 |
+
# matching the widened intermediate_size set by sglang.srt.configs.agnes.
|
| 28 |
+
import json as _agnes_json
|
| 29 |
+
import os as _agnes_os
|
| 30 |
+
import re as _agnes_re
|
| 31 |
+
|
| 32 |
+
_AGNES_MLP_RE = _agnes_re.compile(r"^(.*\\.layers\\.\\d+\\.mlp\\.)(gate_proj|up_proj|down_proj)\\.weight$")
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
class _AgnesBranchReader:
|
| 36 |
+
def __init__(self, model_path):
|
| 37 |
+
from safetensors import safe_open
|
| 38 |
+
|
| 39 |
+
self._open = safe_open
|
| 40 |
+
self.path = model_path
|
| 41 |
+
index = _agnes_os.path.join(model_path, "model.safetensors.index.json")
|
| 42 |
+
self.weight_map = _agnes_json.load(open(index))["weight_map"]
|
| 43 |
+
self.handles = {}
|
| 44 |
+
|
| 45 |
+
def get(self, key):
|
| 46 |
+
fn = self.weight_map[key]
|
| 47 |
+
if fn not in self.handles:
|
| 48 |
+
self.handles[fn] = self._open(_agnes_os.path.join(self.path, fn), framework="pt", device="cpu")
|
| 49 |
+
return self.handles[fn].get_tensor(key)
|
| 50 |
+
|
| 51 |
+
|
| 52 |
+
def _agnes_translate_weights(model, weights):
|
| 53 |
+
cfg = getattr(model.config, "text_config", None) or model.config
|
| 54 |
+
width = int(getattr(cfg, "agnes_parallel_ffn_intermediate_size", 0) or 0)
|
| 55 |
+
if width <= 0:
|
| 56 |
+
yield from weights
|
| 57 |
+
return
|
| 58 |
+
model_path = (
|
| 59 |
+
getattr(cfg, "agnes_model_path", None)
|
| 60 |
+
or getattr(model.config, "agnes_model_path", None)
|
| 61 |
+
or _agnes_os.environ.get("AGNES_MODEL_PATH")
|
| 62 |
+
or getattr(model.config, "_name_or_path", None)
|
| 63 |
+
)
|
| 64 |
+
if not model_path or not _agnes_os.path.isdir(model_path):
|
| 65 |
+
raise RuntimeError(
|
| 66 |
+
f"agnes: cannot locate the checkpoint directory (got {model_path!r}); "
|
| 67 |
+
"set AGNES_MODEL_PATH to the model directory"
|
| 68 |
+
)
|
| 69 |
+
reader = _AgnesBranchReader(model_path)
|
| 70 |
+
for name, w in weights:
|
| 71 |
+
if ".mlp.parallel_ffn." in name:
|
| 72 |
+
continue
|
| 73 |
+
m = _AGNES_MLP_RE.match(name)
|
| 74 |
+
if m and "visual" not in name and not name.startswith("mtp"):
|
| 75 |
+
extra = reader.get(f"{m.group(1)}parallel_ffn.{m.group(2)}.weight")
|
| 76 |
+
dim = 1 if m.group(2) == "down_proj" else 0
|
| 77 |
+
w = torch.cat([w, extra.to(device=w.device, dtype=w.dtype)], dim=dim)
|
| 78 |
+
name = name.replace(".delta_attn.", ".linear_attn.").replace(".global_attn.", ".self_attn.")
|
| 79 |
+
yield name, w
|
| 80 |
+
# === /agnes ===
|
| 81 |
+
|
| 82 |
+
'''
|
| 83 |
+
|
| 84 |
+
REGISTER = '''
|
| 85 |
+
# === agnes ===
|
| 86 |
+
from sglang.srt.configs.agnes import AgnesConfig as _AgnesConfig
|
| 87 |
+
|
| 88 |
+
_CONFIG_REGISTRY[_AgnesConfig.model_type] = _AgnesConfig
|
| 89 |
+
'''
|
| 90 |
+
|
| 91 |
+
|
| 92 |
+
def patch_file(path, edit):
|
| 93 |
+
src = open(path, encoding="utf-8").read()
|
| 94 |
+
if MARK in src:
|
| 95 |
+
return "already patched"
|
| 96 |
+
out = edit(src)
|
| 97 |
+
if out is None:
|
| 98 |
+
raise SystemExit(f"anchor not found in {path}")
|
| 99 |
+
open(path, "w", encoding="utf-8").write(out)
|
| 100 |
+
return "patched"
|
| 101 |
+
|
| 102 |
+
|
| 103 |
+
def edit_model_file(src):
|
| 104 |
+
anchor = "QWEN3_5_KV_SCALE_MAPPER = WeightsMapper("
|
| 105 |
+
hook = " weights = QWEN3_5_KV_SCALE_MAPPER.apply(weights)\n"
|
| 106 |
+
if anchor not in src or src.count(hook) < 1:
|
| 107 |
+
return None
|
| 108 |
+
src = src.replace(anchor, TRANSLATE + anchor, 1)
|
| 109 |
+
src = src.replace(hook, " weights = _agnes_translate_weights(self, weights)\n" + hook)
|
| 110 |
+
return src
|
| 111 |
+
|
| 112 |
+
|
| 113 |
+
def main():
|
| 114 |
+
if len(sys.argv) != 2:
|
| 115 |
+
sys.exit(__doc__)
|
| 116 |
+
pkg = os.path.abspath(sys.argv[1])
|
| 117 |
+
srt = os.path.join(pkg, "srt")
|
| 118 |
+
if not os.path.isdir(srt):
|
| 119 |
+
sys.exit(f"{pkg} does not contain srt/")
|
| 120 |
+
dst = os.path.join(srt, "configs", "agnes.py")
|
| 121 |
+
shutil.copy2(os.path.join(HERE, "agnes_sglang_config.py"), dst)
|
| 122 |
+
print(f"configs/agnes.py: installed")
|
| 123 |
+
print("utils/hf_transformers/common.py:", patch_file(
|
| 124 |
+
os.path.join(srt, "utils", "hf_transformers", "common.py"), lambda s: s.rstrip("\n") + "\n" + REGISTER))
|
| 125 |
+
print("models/qwen3_5.py:", patch_file(os.path.join(srt, "models", "qwen3_5.py"), edit_model_file))
|
| 126 |
+
print("APPLY_PATCH_OK")
|
| 127 |
+
|
| 128 |
+
|
| 129 |
+
if __name__ == "__main__":
|
| 130 |
+
main()
|
sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/configs/agnes.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Agnes 3.0 Flash configuration for sglang.
|
| 2 |
+
#
|
| 3 |
+
# The HF checkpoint says model_type "agnes", names its layer types
|
| 4 |
+
# agnes_delta_attention / agnes_global_attention and carries a parallel FFN
|
| 5 |
+
# branch per layer. The server runs it on its built-in hybrid
|
| 6 |
+
# (delta-rule + global attention) implementation, so this class maps those
|
| 7 |
+
# onto the fields that implementation reads; the checkpoint's tensor names
|
| 8 |
+
# are translated while loading (see the model file patched by apply_patch.py).
|
| 9 |
+
from sglang.srt.configs.qwen3_5 import Qwen3_5Config
|
| 10 |
+
|
| 11 |
+
AGNES_DELTA = "agnes_delta_attention"
|
| 12 |
+
AGNES_GLOBAL = "agnes_global_attention"
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class AgnesConfig(Qwen3_5Config):
|
| 16 |
+
model_type = "agnes"
|
| 17 |
+
|
| 18 |
+
def __init__(self, text_config=None, vision_config=None, **kwargs):
|
| 19 |
+
kwargs.pop("auto_map", None) # the transformers remote code is not used in the server
|
| 20 |
+
if isinstance(text_config, dict):
|
| 21 |
+
text_config = dict(text_config)
|
| 22 |
+
text_config["model_type"] = "qwen3_5_text"
|
| 23 |
+
width = int(text_config.pop("parallel_ffn_intermediate_size", 0) or 0)
|
| 24 |
+
plan = text_config.pop("layer_types", None)
|
| 25 |
+
interval = text_config.pop("global_attention_interval", None)
|
| 26 |
+
if interval is None:
|
| 27 |
+
interval = text_config.pop("full_attention_interval", None)
|
| 28 |
+
if interval is None and plan:
|
| 29 |
+
interval = next(i + 1 for i, t in enumerate(plan) if t == AGNES_GLOBAL)
|
| 30 |
+
text_config["full_attention_interval"] = int(interval or 4)
|
| 31 |
+
main = int(text_config["intermediate_size"])
|
| 32 |
+
# the parallel branch is folded into the main MLP at load time
|
| 33 |
+
text_config["intermediate_size"] = main + width
|
| 34 |
+
text_config["agnes_main_intermediate_size"] = main
|
| 35 |
+
text_config["agnes_parallel_ffn_intermediate_size"] = width
|
| 36 |
+
if isinstance(vision_config, dict):
|
| 37 |
+
vision_config = dict(vision_config)
|
| 38 |
+
vision_config["model_type"] = "qwen3_5"
|
| 39 |
+
kwargs["architectures"] = ["Qwen3_5ForConditionalGeneration"]
|
| 40 |
+
super().__init__(text_config=text_config, vision_config=vision_config, **kwargs)
|
| 41 |
+
# every downstream check sees the built-in hybrid architecture
|
| 42 |
+
self.model_type = "qwen3_5"
|
| 43 |
+
|
| 44 |
+
@classmethod
|
| 45 |
+
def from_pretrained(cls, pretrained_model_name_or_path, *args, **kwargs):
|
| 46 |
+
# The weight loader needs the checkpoint directory to pick up the parallel
|
| 47 |
+
# branch tensors. The path is written into the config *dict* before the
|
| 48 |
+
# object is built: the config reaches the worker processes through a
|
| 49 |
+
# to_dict round trip, which keeps fields that came in through __init__
|
| 50 |
+
# and drops attributes set afterwards (from_pretrained's own kwargs only
|
| 51 |
+
# override known fields, so they cannot carry it either).
|
| 52 |
+
path = str(pretrained_model_name_or_path)
|
| 53 |
+
config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
|
| 54 |
+
config_dict["agnes_model_path"] = path
|
| 55 |
+
if isinstance(config_dict.get("text_config"), dict):
|
| 56 |
+
config_dict["text_config"]["agnes_model_path"] = path
|
| 57 |
+
return cls.from_dict(config_dict, **kwargs)
|
sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/models/qwen3_5.py
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
sglang_patch/nightly-dev-20260908-20ca564b/sglang/srt/utils/hf_transformers/common.py
ADDED
|
@@ -0,0 +1,729 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2023-2024 SGLang Team
|
| 2 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 3 |
+
# you may not use this file except in compliance with the License.
|
| 4 |
+
# You may obtain a copy of the License at
|
| 5 |
+
#
|
| 6 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 7 |
+
#
|
| 8 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 9 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 10 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 11 |
+
# See the License for the specific language governing permissions and
|
| 12 |
+
# limitations under the License.
|
| 13 |
+
# ==============================================================================
|
| 14 |
+
"""Shared helpers used by config, tokenizer, and processor modules."""
|
| 15 |
+
|
| 16 |
+
import json
|
| 17 |
+
import os
|
| 18 |
+
from pathlib import Path
|
| 19 |
+
from typing import Any, Dict, Optional, Type, Union
|
| 20 |
+
|
| 21 |
+
import torch
|
| 22 |
+
from huggingface_hub import snapshot_download
|
| 23 |
+
|
| 24 |
+
from sglang.srt.configs import (
|
| 25 |
+
AfmoeConfig,
|
| 26 |
+
BailingHybridConfig,
|
| 27 |
+
ChatGLMConfig,
|
| 28 |
+
Cosmos3Config,
|
| 29 |
+
Cosmos3EdgeConfig,
|
| 30 |
+
Cosmos3EdgeProjectorConfig,
|
| 31 |
+
Cosmos3EdgeTextConfig,
|
| 32 |
+
Cosmos3EdgeVisionConfig,
|
| 33 |
+
DbrxConfig,
|
| 34 |
+
DeepseekVL2Config,
|
| 35 |
+
Dots3Config,
|
| 36 |
+
DotsOCRConfig,
|
| 37 |
+
DotsVLMConfig,
|
| 38 |
+
ExaoneConfig,
|
| 39 |
+
FalconH1Config,
|
| 40 |
+
Glm5NextConfig,
|
| 41 |
+
Glm5NextTextConfig,
|
| 42 |
+
GraniteMoeHybridConfig,
|
| 43 |
+
HYV4Config,
|
| 44 |
+
InklingAudioConfig,
|
| 45 |
+
InklingMMConfig,
|
| 46 |
+
InklingModelConfig,
|
| 47 |
+
InklingVisionConfig,
|
| 48 |
+
InternS2MobiusConfig,
|
| 49 |
+
InternS2MobiusTextConfig,
|
| 50 |
+
InternS2PreviewConfig,
|
| 51 |
+
JetNemotronConfig,
|
| 52 |
+
JetVLMConfig,
|
| 53 |
+
K2HorizonConfig,
|
| 54 |
+
KimiK3Config,
|
| 55 |
+
KimiK25Config,
|
| 56 |
+
KimiLinearConfig,
|
| 57 |
+
KimiVLConfig,
|
| 58 |
+
LagunaConfig,
|
| 59 |
+
LocateAnythingConfig,
|
| 60 |
+
LongcatFlashConfig,
|
| 61 |
+
MiniCPMHybridConfig,
|
| 62 |
+
MiniCPMV4_6Config,
|
| 63 |
+
MiniCPMV4_6VisionConfig,
|
| 64 |
+
MiniMaxM3VLConfig,
|
| 65 |
+
MultiModalityConfig,
|
| 66 |
+
MuseGlimmerAssistantConfig,
|
| 67 |
+
MuseGlimmerConfig,
|
| 68 |
+
NanbeigeConfig,
|
| 69 |
+
NemotronH_Nano_Omni_Reasoning_V3_Config,
|
| 70 |
+
NemotronH_Nano_VL_V2_Config,
|
| 71 |
+
NemotronHConfig,
|
| 72 |
+
NemotronHPuzzleConfig,
|
| 73 |
+
Olmo3Config,
|
| 74 |
+
Qwen3_5Config,
|
| 75 |
+
Qwen3_5MoeConfig,
|
| 76 |
+
Qwen3_5MoeTextConfig,
|
| 77 |
+
Qwen3_5TextConfig,
|
| 78 |
+
Qwen3NextConfig,
|
| 79 |
+
Spark2_5Config,
|
| 80 |
+
Step3p5Config,
|
| 81 |
+
Step3p7Config,
|
| 82 |
+
Step3VLConfig,
|
| 83 |
+
XllmConfig,
|
| 84 |
+
)
|
| 85 |
+
from sglang.srt.configs.deepseek_ocr import DeepseekVLV2Config
|
| 86 |
+
from sglang.srt.configs.internvl import InternVLChatConfig
|
| 87 |
+
from sglang.srt.utils import get_bool_env_var, logger, lru_cache_frozenset
|
| 88 |
+
from sglang.srt.utils.runai_utils import ObjectStorageModel, is_runai_obj_uri
|
| 89 |
+
|
| 90 |
+
from ..hf_transformers_patches import normalize_rope_scaling_compat
|
| 91 |
+
|
| 92 |
+
if get_bool_env_var("SGLANG_USE_MODELSCOPE"):
|
| 93 |
+
from modelscope import AutoConfig, GenerationConfig
|
| 94 |
+
else:
|
| 95 |
+
from transformers import AutoConfig, GenerationConfig
|
| 96 |
+
|
| 97 |
+
from transformers import PretrainedConfig
|
| 98 |
+
|
| 99 |
+
# ---------------------------------------------------------------------------
|
| 100 |
+
# Config registry
|
| 101 |
+
# ---------------------------------------------------------------------------
|
| 102 |
+
|
| 103 |
+
_CONFIG_REGISTRY: Dict[str, Type[PretrainedConfig]] = {
|
| 104 |
+
cls.model_type: cls
|
| 105 |
+
for cls in [
|
| 106 |
+
AfmoeConfig,
|
| 107 |
+
BailingHybridConfig,
|
| 108 |
+
ChatGLMConfig,
|
| 109 |
+
DbrxConfig,
|
| 110 |
+
ExaoneConfig,
|
| 111 |
+
DeepseekVL2Config,
|
| 112 |
+
MultiModalityConfig,
|
| 113 |
+
KimiVLConfig,
|
| 114 |
+
K2HorizonConfig,
|
| 115 |
+
LocateAnythingConfig,
|
| 116 |
+
InternVLChatConfig,
|
| 117 |
+
LagunaConfig,
|
| 118 |
+
Spark2_5Config,
|
| 119 |
+
Step3VLConfig,
|
| 120 |
+
LongcatFlashConfig,
|
| 121 |
+
Olmo3Config,
|
| 122 |
+
MuseGlimmerConfig,
|
| 123 |
+
MuseGlimmerAssistantConfig,
|
| 124 |
+
KimiK3Config,
|
| 125 |
+
Glm5NextConfig,
|
| 126 |
+
Glm5NextTextConfig,
|
| 127 |
+
KimiLinearConfig,
|
| 128 |
+
Qwen3NextConfig,
|
| 129 |
+
FalconH1Config,
|
| 130 |
+
GraniteMoeHybridConfig,
|
| 131 |
+
HYV4Config,
|
| 132 |
+
DotsVLMConfig,
|
| 133 |
+
DotsOCRConfig,
|
| 134 |
+
Dots3Config,
|
| 135 |
+
NemotronH_Nano_VL_V2_Config,
|
| 136 |
+
NemotronH_Nano_Omni_Reasoning_V3_Config,
|
| 137 |
+
NemotronHConfig,
|
| 138 |
+
NemotronHPuzzleConfig,
|
| 139 |
+
NanbeigeConfig,
|
| 140 |
+
DeepseekVLV2Config,
|
| 141 |
+
Qwen3_5Config,
|
| 142 |
+
Qwen3_5MoeConfig,
|
| 143 |
+
Qwen3_5TextConfig,
|
| 144 |
+
Qwen3_5MoeTextConfig,
|
| 145 |
+
InternS2PreviewConfig,
|
| 146 |
+
InternS2MobiusConfig,
|
| 147 |
+
InternS2MobiusTextConfig,
|
| 148 |
+
JetNemotronConfig,
|
| 149 |
+
JetVLMConfig,
|
| 150 |
+
KimiK25Config,
|
| 151 |
+
Step3p5Config,
|
| 152 |
+
Step3p7Config,
|
| 153 |
+
MiniCPMHybridConfig,
|
| 154 |
+
MiniCPMV4_6Config,
|
| 155 |
+
MiniCPMV4_6VisionConfig,
|
| 156 |
+
InklingModelConfig,
|
| 157 |
+
InklingAudioConfig,
|
| 158 |
+
InklingVisionConfig,
|
| 159 |
+
InklingMMConfig,
|
| 160 |
+
MiniMaxM3VLConfig,
|
| 161 |
+
XllmConfig,
|
| 162 |
+
]
|
| 163 |
+
}
|
| 164 |
+
|
| 165 |
+
|
| 166 |
+
# DeepSeek V3.2 / V4 reuse the V3 config schema. Subclass the upstream
|
| 167 |
+
# transformers class with each model_type so AutoConfig.register passes its
|
| 168 |
+
# consistency check (which requires class.model_type == registered key).
|
| 169 |
+
# Default-value divergences (e.g. V4's topk_group) are handled in
|
| 170 |
+
# model_config.py post-load.
|
| 171 |
+
try:
|
| 172 |
+
from transformers import DeepseekV3Config as _HFDeepseekV3Config
|
| 173 |
+
|
| 174 |
+
class _DeepseekV32ConfigAlias(_HFDeepseekV3Config):
|
| 175 |
+
model_type = "deepseek_v32"
|
| 176 |
+
|
| 177 |
+
class _DeepseekV4ConfigAlias(_HFDeepseekV3Config):
|
| 178 |
+
model_type = "deepseek_v4"
|
| 179 |
+
|
| 180 |
+
_CONFIG_REGISTRY["deepseek_v32"] = _DeepseekV32ConfigAlias
|
| 181 |
+
_CONFIG_REGISTRY["deepseek_v4"] = _DeepseekV4ConfigAlias
|
| 182 |
+
|
| 183 |
+
# For kimi_k25_eagle3
|
| 184 |
+
class _KimiK2ConfigAlias(_HFDeepseekV3Config):
|
| 185 |
+
model_type = "kimi_k2"
|
| 186 |
+
|
| 187 |
+
_CONFIG_REGISTRY["kimi_k2"] = _KimiK2ConfigAlias
|
| 188 |
+
except ImportError:
|
| 189 |
+
pass
|
| 190 |
+
|
| 191 |
+
# Newer transformers versions (>=5.10.2) expose MellumConfig directly,
|
| 192 |
+
# but fallback to Qwen3MoeConfig for older versions.
|
| 193 |
+
try:
|
| 194 |
+
import transformers as _hf_transformers
|
| 195 |
+
|
| 196 |
+
_HFMellumConfig = getattr(_hf_transformers, "MellumConfig", None)
|
| 197 |
+
|
| 198 |
+
if _HFMellumConfig is not None:
|
| 199 |
+
_CONFIG_REGISTRY["mellum"] = _HFMellumConfig
|
| 200 |
+
else:
|
| 201 |
+
from transformers import Qwen3MoeConfig as _HFQwen3MoeConfig
|
| 202 |
+
|
| 203 |
+
class _MellumConfigAlias(_HFQwen3MoeConfig):
|
| 204 |
+
model_type = "mellum"
|
| 205 |
+
|
| 206 |
+
def __post_init__(self, **kwargs):
|
| 207 |
+
# Qwen3MoeConfig.__post_init__ wipes sliding_window unless
|
| 208 |
+
# use_sliding_window=True. Mellum gates sliding attention
|
| 209 |
+
# per-layer via layer_types, so preserve sliding_window
|
| 210 |
+
# regardless of the legacy use_sliding_window flag.
|
| 211 |
+
sliding_window = getattr(self, "sliding_window", None)
|
| 212 |
+
super().__post_init__(**kwargs)
|
| 213 |
+
self.sliding_window = sliding_window
|
| 214 |
+
|
| 215 |
+
_CONFIG_REGISTRY["mellum"] = _MellumConfigAlias
|
| 216 |
+
|
| 217 |
+
except ImportError:
|
| 218 |
+
pass
|
| 219 |
+
|
| 220 |
+
|
| 221 |
+
try:
|
| 222 |
+
from transformers import Gemma4Config as _HFGemma4Config
|
| 223 |
+
|
| 224 |
+
class _Gemma4UnifiedConfigAlias(_HFGemma4Config):
|
| 225 |
+
model_type = "gemma4_unified"
|
| 226 |
+
|
| 227 |
+
_CONFIG_REGISTRY["gemma4_unified"] = _Gemma4UnifiedConfigAlias
|
| 228 |
+
except ImportError:
|
| 229 |
+
pass
|
| 230 |
+
|
| 231 |
+
for name, cls in _CONFIG_REGISTRY.items():
|
| 232 |
+
try:
|
| 233 |
+
AutoConfig.register(name, cls)
|
| 234 |
+
except ValueError as e:
|
| 235 |
+
err = str(e).lower()
|
| 236 |
+
if "already registered" not in err and "already used" not in err:
|
| 237 |
+
logger.warning("Failed to register config %s: %s", name, e)
|
| 238 |
+
|
| 239 |
+
# Cosmos3 (understanding tower) reuses the Qwen3-VL config schema. Register it
|
| 240 |
+
# with AutoConfig only (not `_CONFIG_REGISTRY`), so the nested `text_config` is
|
| 241 |
+
# flattened onto the top-level config in `get_config` — the same path the base
|
| 242 |
+
# Qwen3-VL config relies on. Adding it to `_CONFIG_REGISTRY` would trigger a
|
| 243 |
+
# `from_pretrained` reload that drops that flattening.
|
| 244 |
+
try:
|
| 245 |
+
AutoConfig.register(Cosmos3Config.model_type, Cosmos3Config)
|
| 246 |
+
except ValueError as e:
|
| 247 |
+
err = str(e).lower()
|
| 248 |
+
if "already registered" not in err and "already used" not in err:
|
| 249 |
+
logger.warning("Failed to register config %s: %s", Cosmos3Config.model_type, e)
|
| 250 |
+
|
| 251 |
+
# Cosmos3-Edge native text support starts from the checkpoint root config, then
|
| 252 |
+
# consumes ``text_config`` in ``sglang.srt.models.cosmos3_edge``. Keep it out of
|
| 253 |
+
# `_CONFIG_REGISTRY` so the generic parser can flatten text attributes onto the
|
| 254 |
+
# root config after `AutoConfig.from_pretrained`, matching other multimodal
|
| 255 |
+
# configs that use a text sub-config.
|
| 256 |
+
for _cosmos3_edge_config_cls in (
|
| 257 |
+
Cosmos3EdgeTextConfig,
|
| 258 |
+
Cosmos3EdgeVisionConfig,
|
| 259 |
+
Cosmos3EdgeProjectorConfig,
|
| 260 |
+
Cosmos3EdgeConfig,
|
| 261 |
+
):
|
| 262 |
+
try:
|
| 263 |
+
AutoConfig.register(
|
| 264 |
+
_cosmos3_edge_config_cls.model_type, _cosmos3_edge_config_cls
|
| 265 |
+
)
|
| 266 |
+
except ValueError as e:
|
| 267 |
+
err = str(e).lower()
|
| 268 |
+
if "already registered" not in err and "already used" not in err:
|
| 269 |
+
logger.warning(
|
| 270 |
+
"Failed to register config %s: %s",
|
| 271 |
+
_cosmos3_edge_config_cls.model_type,
|
| 272 |
+
e,
|
| 273 |
+
)
|
| 274 |
+
|
| 275 |
+
|
| 276 |
+
# ---------------------------------------------------------------------------
|
| 277 |
+
# Download / path helpers
|
| 278 |
+
# ---------------------------------------------------------------------------
|
| 279 |
+
|
| 280 |
+
|
| 281 |
+
def download_from_hf(
|
| 282 |
+
model_path: str,
|
| 283 |
+
allow_patterns: Optional[Union[str, list]] = None,
|
| 284 |
+
):
|
| 285 |
+
if os.path.exists(model_path):
|
| 286 |
+
return model_path
|
| 287 |
+
|
| 288 |
+
if not allow_patterns:
|
| 289 |
+
allow_patterns = ["*.json", "*.bin", "*.model"]
|
| 290 |
+
|
| 291 |
+
return snapshot_download(model_path, allow_patterns=allow_patterns)
|
| 292 |
+
|
| 293 |
+
|
| 294 |
+
def resolve_runai_obj_uri(model_name_or_path: str) -> str:
|
| 295 |
+
if is_runai_obj_uri(model_name_or_path):
|
| 296 |
+
return ObjectStorageModel.get_path(model_name_or_path)
|
| 297 |
+
return model_name_or_path
|
| 298 |
+
|
| 299 |
+
|
| 300 |
+
def _resolve_local_or_cached_file(model_name_or_path, filename, revision=None):
|
| 301 |
+
"""Resolve a file from a local directory or HF hub cache (no network)."""
|
| 302 |
+
local_path = Path(model_name_or_path) / filename
|
| 303 |
+
if local_path.is_file():
|
| 304 |
+
return str(local_path)
|
| 305 |
+
from huggingface_hub import hf_hub_download
|
| 306 |
+
|
| 307 |
+
return hf_hub_download(
|
| 308 |
+
model_name_or_path, filename, revision=revision, local_files_only=True
|
| 309 |
+
)
|
| 310 |
+
|
| 311 |
+
|
| 312 |
+
def _cached_file_exists(model_name_or_path, filename, revision=None) -> bool:
|
| 313 |
+
"""Whether *filename* is available locally or in the HF cache (no network)."""
|
| 314 |
+
try:
|
| 315 |
+
_resolve_local_or_cached_file(model_name_or_path, filename, revision)
|
| 316 |
+
return True
|
| 317 |
+
except Exception:
|
| 318 |
+
return False
|
| 319 |
+
|
| 320 |
+
|
| 321 |
+
def _remote_file_exists(repo_id, filename, revision=None) -> bool:
|
| 322 |
+
"""Whether *filename* exists on the HF hub (HEAD request only, no download).
|
| 323 |
+
|
| 324 |
+
Returns False on any error (offline, gated, network, invalid id) so callers
|
| 325 |
+
fall back to their default path instead of crashing.
|
| 326 |
+
"""
|
| 327 |
+
from huggingface_hub.constants import HF_HUB_OFFLINE
|
| 328 |
+
|
| 329 |
+
if HF_HUB_OFFLINE:
|
| 330 |
+
return False
|
| 331 |
+
try:
|
| 332 |
+
from huggingface_hub import HfApi
|
| 333 |
+
|
| 334 |
+
return HfApi().file_exists(repo_id, filename, revision=revision)
|
| 335 |
+
except Exception:
|
| 336 |
+
return False
|
| 337 |
+
|
| 338 |
+
|
| 339 |
+
def check_gguf_file(model: Union[str, os.PathLike]) -> bool:
|
| 340 |
+
model = Path(model)
|
| 341 |
+
if not model.is_file():
|
| 342 |
+
return False
|
| 343 |
+
elif model.suffix == ".gguf":
|
| 344 |
+
return True
|
| 345 |
+
|
| 346 |
+
with open(model, "rb") as f:
|
| 347 |
+
header = f.read(4)
|
| 348 |
+
return header == b"GGUF"
|
| 349 |
+
|
| 350 |
+
|
| 351 |
+
def resolve_hf_gguf_reference(
|
| 352 |
+
model: str, revision: Optional[str] = None
|
| 353 |
+
) -> Optional[str]:
|
| 354 |
+
"""Download a .gguf named by Hub reference and return its local path.
|
| 355 |
+
|
| 356 |
+
owner/repo/path/inside/repo.gguf -> exactly that file
|
| 357 |
+
owner/repo:QUANT_TYPE -> the only matching quantization
|
| 358 |
+
owner/repo -> the only .gguf in the repo
|
| 359 |
+
"""
|
| 360 |
+
from sglang.srt.utils import is_remote_url
|
| 361 |
+
|
| 362 |
+
if not model or os.path.exists(model) or is_remote_url(model):
|
| 363 |
+
return None
|
| 364 |
+
|
| 365 |
+
from huggingface_hub import hf_hub_download
|
| 366 |
+
|
| 367 |
+
if ":" in model:
|
| 368 |
+
repo_id, _, quant_type = model.rpartition(":")
|
| 369 |
+
if repo_id.count("/") != 1 or not quant_type:
|
| 370 |
+
return None
|
| 371 |
+
|
| 372 |
+
from huggingface_hub import HfApi
|
| 373 |
+
|
| 374 |
+
files = [
|
| 375 |
+
sibling.rfilename
|
| 376 |
+
for sibling in HfApi().repo_info(repo_id, revision=revision).siblings
|
| 377 |
+
]
|
| 378 |
+
suffix = f"-{quant_type}.gguf"
|
| 379 |
+
candidates = [filename for filename in files if filename.endswith(suffix)]
|
| 380 |
+
if not candidates:
|
| 381 |
+
available = sorted(
|
| 382 |
+
filename for filename in files if filename.endswith(".gguf")
|
| 383 |
+
)
|
| 384 |
+
raise ValueError(
|
| 385 |
+
f"No file matching quant type {quant_type!r} in {repo_id}. "
|
| 386 |
+
f"Available GGUF files: {available}"
|
| 387 |
+
)
|
| 388 |
+
if len(candidates) > 1:
|
| 389 |
+
raise ValueError(
|
| 390 |
+
f"Quant type {quant_type!r} is ambiguous in {repo_id}: "
|
| 391 |
+
f"{sorted(candidates)}. Pass the full owner/repo/path/file.gguf "
|
| 392 |
+
"reference instead."
|
| 393 |
+
)
|
| 394 |
+
return hf_hub_download(repo_id, candidates[0], revision=revision)
|
| 395 |
+
|
| 396 |
+
parts = model.strip("/").split("/")
|
| 397 |
+
if len(parts) < 2:
|
| 398 |
+
return None
|
| 399 |
+
|
| 400 |
+
if len(parts) > 2 and model.endswith(".gguf"):
|
| 401 |
+
repo_id = "/".join(parts[:2])
|
| 402 |
+
filename = "/".join(parts[2:])
|
| 403 |
+
return hf_hub_download(repo_id, filename, revision=revision)
|
| 404 |
+
|
| 405 |
+
if len(parts) != 2:
|
| 406 |
+
return None
|
| 407 |
+
|
| 408 |
+
from huggingface_hub import HfApi
|
| 409 |
+
|
| 410 |
+
try:
|
| 411 |
+
files = [
|
| 412 |
+
s.rfilename for s in HfApi().repo_info(model, revision=revision).siblings
|
| 413 |
+
]
|
| 414 |
+
except Exception:
|
| 415 |
+
return None
|
| 416 |
+
if any(f == "config.json" for f in files):
|
| 417 |
+
return None
|
| 418 |
+
|
| 419 |
+
candidates = [f for f in files if f.endswith(".gguf")]
|
| 420 |
+
if not candidates:
|
| 421 |
+
return None
|
| 422 |
+
if len(candidates) > 1:
|
| 423 |
+
listing = "\n ".join(f"{model}/{f}" for f in sorted(candidates))
|
| 424 |
+
raise ValueError(
|
| 425 |
+
f"{model} contains {len(candidates)} .gguf files; name the one to "
|
| 426 |
+
f"serve:\n {listing}"
|
| 427 |
+
)
|
| 428 |
+
return hf_hub_download(model, candidates[0], revision=revision)
|
| 429 |
+
|
| 430 |
+
|
| 431 |
+
def gguf_sidecar_dir(
|
| 432 |
+
gguf_path: Union[str, os.PathLike], sentinel: str
|
| 433 |
+
) -> Optional[Path]:
|
| 434 |
+
"""Directory containing *sentinel* next to a .gguf file, if there is one."""
|
| 435 |
+
directory = Path(gguf_path).parent
|
| 436 |
+
return directory if (directory / sentinel).is_file() else None
|
| 437 |
+
|
| 438 |
+
|
| 439 |
+
# ---------------------------------------------------------------------------
|
| 440 |
+
# Rope / text config helpers
|
| 441 |
+
# ---------------------------------------------------------------------------
|
| 442 |
+
|
| 443 |
+
|
| 444 |
+
def get_rope_config(config):
|
| 445 |
+
"""Get (rope_theta, rope_params) from config, supporting both v4 and v5.
|
| 446 |
+
|
| 447 |
+
Trust-remote-code configs or parent configs passed to sub-models may not
|
| 448 |
+
have the v5 ``rope_parameters`` property, so we fall back to the v4-style
|
| 449 |
+
``config.rope_theta`` / ``config.rope_scaling`` attributes.
|
| 450 |
+
|
| 451 |
+
Returns:
|
| 452 |
+
(rope_theta, rope_params): In v5, rope_params is the full
|
| 453 |
+
rope_parameters dict (which subsumes rope_scaling and includes
|
| 454 |
+
rope_theta). In v4, rope_params is the rope_scaling dict or None.
|
| 455 |
+
"""
|
| 456 |
+
rope_params = getattr(config, "rope_parameters", None)
|
| 457 |
+
if rope_params is not None:
|
| 458 |
+
rope_theta = rope_params.get("rope_theta", getattr(config, "rope_theta", 10000))
|
| 459 |
+
return rope_theta, rope_params
|
| 460 |
+
return getattr(config, "rope_theta", 10000), getattr(config, "rope_scaling", None)
|
| 461 |
+
|
| 462 |
+
|
| 463 |
+
def _patch_text_config(parent_config: PretrainedConfig, text_config):
|
| 464 |
+
"""Synchronize standard attributes between parent config and text sub-config.
|
| 465 |
+
|
| 466 |
+
In transformers v5, the "untangle config" refactor removed automatic
|
| 467 |
+
inheritance of top-level PretrainedConfig attributes (pad_token_id,
|
| 468 |
+
tie_word_embeddings, etc.) from sub-configs. Downstream code expects
|
| 469 |
+
these attributes to be present on both configs (some models pass the
|
| 470 |
+
parent directly to the language model, others pass the text sub-config),
|
| 471 |
+
so we propagate in both directions when an attribute is missing.
|
| 472 |
+
(See https://github.com/huggingface/transformers/pull/41541)
|
| 473 |
+
"""
|
| 474 |
+
_ATTRS_TO_PROPAGATE = [
|
| 475 |
+
"pad_token_id",
|
| 476 |
+
"bos_token_id",
|
| 477 |
+
"eos_token_id",
|
| 478 |
+
"tie_word_embeddings",
|
| 479 |
+
]
|
| 480 |
+
for attr in _ATTRS_TO_PROPAGATE:
|
| 481 |
+
parent_has = hasattr(parent_config, attr)
|
| 482 |
+
text_has = hasattr(text_config, attr)
|
| 483 |
+
if parent_has and not text_has:
|
| 484 |
+
setattr(text_config, attr, getattr(parent_config, attr))
|
| 485 |
+
elif text_has and not parent_has:
|
| 486 |
+
setattr(parent_config, attr, getattr(text_config, attr))
|
| 487 |
+
return text_config
|
| 488 |
+
|
| 489 |
+
|
| 490 |
+
def get_hf_text_config(config: PretrainedConfig):
|
| 491 |
+
"""Get the "sub" config relevant to llm for multi modal models.
|
| 492 |
+
No op for pure text models.
|
| 493 |
+
"""
|
| 494 |
+
if config.architectures is not None:
|
| 495 |
+
class_name = config.architectures[0]
|
| 496 |
+
if class_name.startswith("Llava") and class_name.endswith("ForCausalLM"):
|
| 497 |
+
# We support non-hf version of llava models, so we do not want to
|
| 498 |
+
# read the wrong values from the unused default text_config.
|
| 499 |
+
# NOTE(HandH1998): We set `torch_dtype` of config to `torch.float16` for the weights, as
|
| 500 |
+
# `torch.float16` is default used for image features in `python/sglang/srt/models/llava.py`.
|
| 501 |
+
setattr(config, "dtype", torch.float16)
|
| 502 |
+
return config
|
| 503 |
+
|
| 504 |
+
text_config = None
|
| 505 |
+
|
| 506 |
+
# Some models (e.g. DeepSeek-OCR) store sub-configs as plain dicts.
|
| 507 |
+
# Convert to PretrainedConfig early so hasattr() checks and asserts work.
|
| 508 |
+
parent_dtype = getattr(config, "dtype", None)
|
| 509 |
+
for _attr in ("text_config", "llm_config", "language_config", "thinker_config"):
|
| 510 |
+
_sub = getattr(config, _attr, None)
|
| 511 |
+
if isinstance(_sub, dict):
|
| 512 |
+
_converted = PretrainedConfig(**_sub)
|
| 513 |
+
if getattr(_converted, "dtype", None) is None and parent_dtype is not None:
|
| 514 |
+
_converted.dtype = parent_dtype
|
| 515 |
+
setattr(config, _attr, _converted)
|
| 516 |
+
elif _sub is not None and parent_dtype is not None:
|
| 517 |
+
# transformers v5 multimodal configs (e.g. Mistral3Config) carry
|
| 518 |
+
# `dtype` only on the top-level config, leaving the sub-configs at
|
| 519 |
+
# None. Without this, _get_and_verify_dtype falls back to float32
|
| 520 |
+
# and then "auto" downcasts to float16, which overflows the Pixtral
|
| 521 |
+
# vision tower on real images and produces NaN features.
|
| 522 |
+
if getattr(_sub, "dtype", None) is None:
|
| 523 |
+
_sub.dtype = parent_dtype
|
| 524 |
+
|
| 525 |
+
# Priority: thinker_config > llm_config > language_config > text_config
|
| 526 |
+
if hasattr(config, "thinker_config"):
|
| 527 |
+
# qwen2.5 omni
|
| 528 |
+
thinker_config = config.thinker_config
|
| 529 |
+
if hasattr(thinker_config, "text_config"):
|
| 530 |
+
setattr(
|
| 531 |
+
thinker_config.text_config,
|
| 532 |
+
"dtype",
|
| 533 |
+
getattr(thinker_config, "dtype", None),
|
| 534 |
+
)
|
| 535 |
+
text_config = thinker_config.text_config
|
| 536 |
+
else:
|
| 537 |
+
text_config = thinker_config
|
| 538 |
+
elif hasattr(config, "llm_config"):
|
| 539 |
+
# PointsV1.5 Chat Model
|
| 540 |
+
assert hasattr(config.llm_config, "num_attention_heads")
|
| 541 |
+
text_config = config.llm_config
|
| 542 |
+
elif hasattr(config, "language_config"):
|
| 543 |
+
text_config = config.language_config
|
| 544 |
+
elif hasattr(config, "text_config"):
|
| 545 |
+
# The code operates under the assumption that text_config should have
|
| 546 |
+
# `num_attention_heads` (among others). Assert here to fail early
|
| 547 |
+
# if transformers config doesn't align with this assumption.
|
| 548 |
+
assert hasattr(config.text_config, "num_attention_heads")
|
| 549 |
+
text_config = config.text_config
|
| 550 |
+
|
| 551 |
+
# Ensure rope_scaling dicts have "type" for remote-code compat (v5).
|
| 552 |
+
normalize_rope_scaling_compat(config)
|
| 553 |
+
|
| 554 |
+
if text_config is not None:
|
| 555 |
+
return _patch_text_config(config, text_config)
|
| 556 |
+
return config
|
| 557 |
+
|
| 558 |
+
|
| 559 |
+
# ---------------------------------------------------------------------------
|
| 560 |
+
# Model-specific helpers
|
| 561 |
+
# ---------------------------------------------------------------------------
|
| 562 |
+
|
| 563 |
+
|
| 564 |
+
def _ensure_sub_configs(config: PretrainedConfig, *attr_names: str) -> None:
|
| 565 |
+
"""Convert dict-valued sub-configs to proper AutoConfig objects in-place."""
|
| 566 |
+
for attr in attr_names:
|
| 567 |
+
sub = getattr(config, attr, None)
|
| 568 |
+
if sub is not None and isinstance(sub, dict):
|
| 569 |
+
setattr(config, attr, AutoConfig.for_model(**sub))
|
| 570 |
+
|
| 571 |
+
|
| 572 |
+
def _is_deepseek_ocr_model(config: PretrainedConfig) -> bool:
|
| 573 |
+
# TODO: Remove this workaround once AutoConfig correctly identifies deepseek-ocr.
|
| 574 |
+
# Hugging Face's AutoConfig currently misidentifies it as deepseekvl2.
|
| 575 |
+
auto_map = getattr(config, "auto_map", None) or {}
|
| 576 |
+
return auto_map.get("AutoModel") == "modeling_deepseekocr.DeepseekOCRForCausalLM"
|
| 577 |
+
|
| 578 |
+
|
| 579 |
+
def _is_deepseek_ocr2_model(config: PretrainedConfig) -> bool:
|
| 580 |
+
auto_map = getattr(config, "auto_map", None) or {}
|
| 581 |
+
return auto_map.get("AutoModel") == "modeling_deepseekocr2.DeepseekOCR2ForCausalLM"
|
| 582 |
+
|
| 583 |
+
|
| 584 |
+
def _override_v_head_dim_if_zero(config: PretrainedConfig, patch: int = 128) -> None:
|
| 585 |
+
patched = False
|
| 586 |
+
for attr in ("text_config", "language_config"):
|
| 587 |
+
sub = getattr(config, attr, None)
|
| 588 |
+
if sub is None:
|
| 589 |
+
continue
|
| 590 |
+
if isinstance(sub, dict):
|
| 591 |
+
if sub.get("v_head_dim") == 0:
|
| 592 |
+
sub["v_head_dim"] = patch
|
| 593 |
+
patched = True
|
| 594 |
+
elif getattr(sub, "v_head_dim", None) == 0:
|
| 595 |
+
sub.v_head_dim = patch
|
| 596 |
+
patched = True
|
| 597 |
+
if patched:
|
| 598 |
+
logger.warning(
|
| 599 |
+
f"Overriding v_head_dim from 0 to {patch} to avoid potential issues."
|
| 600 |
+
)
|
| 601 |
+
|
| 602 |
+
|
| 603 |
+
# ---------------------------------------------------------------------------
|
| 604 |
+
# Context length / generation config / sparse attention
|
| 605 |
+
# ---------------------------------------------------------------------------
|
| 606 |
+
|
| 607 |
+
# Models don't use the same configuration key for determining the maximum
|
| 608 |
+
# context length. Store them here so we can sanely check them.
|
| 609 |
+
# NOTE: The ordering here is important. Some models have two of these and we
|
| 610 |
+
# have a preference for which value gets used.
|
| 611 |
+
CONTEXT_LENGTH_KEYS = [
|
| 612 |
+
"max_sequence_length",
|
| 613 |
+
"seq_length",
|
| 614 |
+
"max_seq_len",
|
| 615 |
+
"model_max_length",
|
| 616 |
+
"max_position_embeddings",
|
| 617 |
+
]
|
| 618 |
+
|
| 619 |
+
|
| 620 |
+
def get_context_length(config):
|
| 621 |
+
"""Get the context length of a model from a huggingface model configs."""
|
| 622 |
+
text_config = config
|
| 623 |
+
rope_scaling = getattr(text_config, "rope_scaling", None)
|
| 624 |
+
if rope_scaling:
|
| 625 |
+
rope_scaling_factor = rope_scaling.get("factor", 1)
|
| 626 |
+
if "original_max_position_embeddings" in rope_scaling:
|
| 627 |
+
rope_scaling_factor = 1
|
| 628 |
+
if rope_scaling.get("rope_type", None) == "llama3":
|
| 629 |
+
rope_scaling_factor = 1
|
| 630 |
+
else:
|
| 631 |
+
rope_scaling_factor = 1
|
| 632 |
+
|
| 633 |
+
for key in CONTEXT_LENGTH_KEYS:
|
| 634 |
+
val = getattr(text_config, key, None)
|
| 635 |
+
if val is not None:
|
| 636 |
+
return int(rope_scaling_factor * val)
|
| 637 |
+
return 2048
|
| 638 |
+
|
| 639 |
+
|
| 640 |
+
@lru_cache_frozenset(maxsize=32)
|
| 641 |
+
def get_generation_config(
|
| 642 |
+
model: str,
|
| 643 |
+
trust_remote_code: bool,
|
| 644 |
+
revision: Optional[str] = None,
|
| 645 |
+
**kwargs,
|
| 646 |
+
):
|
| 647 |
+
if check_gguf_file(model):
|
| 648 |
+
sidecar = gguf_sidecar_dir(model, "generation_config.json")
|
| 649 |
+
if sidecar is not None:
|
| 650 |
+
model = str(sidecar)
|
| 651 |
+
else:
|
| 652 |
+
from .gguf_native import (
|
| 653 |
+
build_gguf_generation_config,
|
| 654 |
+
has_native_gguf_support,
|
| 655 |
+
)
|
| 656 |
+
|
| 657 |
+
if has_native_gguf_support(model):
|
| 658 |
+
return build_gguf_generation_config(model)
|
| 659 |
+
|
| 660 |
+
try:
|
| 661 |
+
return GenerationConfig.from_pretrained(
|
| 662 |
+
model, trust_remote_code=trust_remote_code, revision=revision, **kwargs
|
| 663 |
+
)
|
| 664 |
+
except (FileNotFoundError, OSError) as e:
|
| 665 |
+
# A missing generation_config.json is normal for many checkpoints and
|
| 666 |
+
# is surfaced by HF as a generic OSError (not FileNotFoundError). Treat
|
| 667 |
+
# it as benign — proceed without a generation config, at DEBUG level so
|
| 668 |
+
# normal startup logs stay quiet.
|
| 669 |
+
logger.debug(
|
| 670 |
+
"No generation config for %s: %s. Proceeding without it.",
|
| 671 |
+
model,
|
| 672 |
+
e,
|
| 673 |
+
)
|
| 674 |
+
return None
|
| 675 |
+
|
| 676 |
+
|
| 677 |
+
# Qwen-1M related
|
| 678 |
+
def get_sparse_attention_config(
|
| 679 |
+
model: str,
|
| 680 |
+
sparse_attention_config_filename: str = "sparse_attention_config.json",
|
| 681 |
+
) -> Dict[str, Any]:
|
| 682 |
+
is_local = os.path.isdir(model)
|
| 683 |
+
if not is_local:
|
| 684 |
+
model = download_from_hf(model, allow_patterns=["*.json"])
|
| 685 |
+
|
| 686 |
+
config_file = os.path.join(model, sparse_attention_config_filename)
|
| 687 |
+
if not os.path.exists(config_file):
|
| 688 |
+
return {}
|
| 689 |
+
|
| 690 |
+
with open(config_file) as f:
|
| 691 |
+
config = json.load(f)
|
| 692 |
+
return config
|
| 693 |
+
|
| 694 |
+
|
| 695 |
+
# ---------------------------------------------------------------------------
|
| 696 |
+
# Tokenizer / processor helpers
|
| 697 |
+
# ---------------------------------------------------------------------------
|
| 698 |
+
|
| 699 |
+
|
| 700 |
+
# Some models don't have an available processor, e.g.: InternVL
|
| 701 |
+
def get_tokenizer_from_processor(processor):
|
| 702 |
+
from transformers import PreTrainedTokenizerBase
|
| 703 |
+
|
| 704 |
+
if isinstance(processor, PreTrainedTokenizerBase):
|
| 705 |
+
return processor
|
| 706 |
+
return processor.tokenizer
|
| 707 |
+
|
| 708 |
+
|
| 709 |
+
# Turn-final markers that some checkpoints ship without EOS metadata:
|
| 710 |
+
# <|eom_id|> (Llama-3 tool use), <|content_model_end_sampling|> (Inkling,
|
| 711 |
+
# whose bundled tokenizer config leaves eos_token unset), and
|
| 712 |
+
# <|ifm|im_end|> (some K2 Horizon checkpoints, notably 0.9B, name only
|
| 713 |
+
# <|endoftext|> as EOS).
|
| 714 |
+
_ADDITIONAL_STOP_TOKEN_TEXTS = (
|
| 715 |
+
"<|eom_id|>",
|
| 716 |
+
"<|content_model_end_sampling|>",
|
| 717 |
+
"<|ifm|im_end|>",
|
| 718 |
+
)
|
| 719 |
+
|
| 720 |
+
|
| 721 |
+
def attach_additional_stop_token_ids(tokenizer):
|
| 722 |
+
added = tokenizer.get_added_vocab()
|
| 723 |
+
stop_ids = {added[text] for text in _ADDITIONAL_STOP_TOKEN_TEXTS if text in added}
|
| 724 |
+
tokenizer.additional_stop_token_ids = stop_ids or None
|
| 725 |
+
|
| 726 |
+
# === agnes ===
|
| 727 |
+
from sglang.srt.configs.agnes import AgnesConfig as _AgnesConfig
|
| 728 |
+
|
| 729 |
+
_CONFIG_REGISTRY[_AgnesConfig.model_type] = _AgnesConfig
|
sglang_patch/v0.5.19/sglang/srt/configs/agnes.py
ADDED
|
@@ -0,0 +1,57 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Agnes 3.0 Flash configuration for sglang.
|
| 2 |
+
#
|
| 3 |
+
# The HF checkpoint says model_type "agnes", names its layer types
|
| 4 |
+
# agnes_delta_attention / agnes_global_attention and carries a parallel FFN
|
| 5 |
+
# branch per layer. The server runs it on its built-in hybrid
|
| 6 |
+
# (delta-rule + global attention) implementation, so this class maps those
|
| 7 |
+
# onto the fields that implementation reads; the checkpoint's tensor names
|
| 8 |
+
# are translated while loading (see the model file patched by apply_patch.py).
|
| 9 |
+
from sglang.srt.configs.qwen3_5 import Qwen3_5Config
|
| 10 |
+
|
| 11 |
+
AGNES_DELTA = "agnes_delta_attention"
|
| 12 |
+
AGNES_GLOBAL = "agnes_global_attention"
|
| 13 |
+
|
| 14 |
+
|
| 15 |
+
class AgnesConfig(Qwen3_5Config):
|
| 16 |
+
model_type = "agnes"
|
| 17 |
+
|
| 18 |
+
def __init__(self, text_config=None, vision_config=None, **kwargs):
|
| 19 |
+
kwargs.pop("auto_map", None) # the transformers remote code is not used in the server
|
| 20 |
+
if isinstance(text_config, dict):
|
| 21 |
+
text_config = dict(text_config)
|
| 22 |
+
text_config["model_type"] = "qwen3_5_text"
|
| 23 |
+
width = int(text_config.pop("parallel_ffn_intermediate_size", 0) or 0)
|
| 24 |
+
plan = text_config.pop("layer_types", None)
|
| 25 |
+
interval = text_config.pop("global_attention_interval", None)
|
| 26 |
+
if interval is None:
|
| 27 |
+
interval = text_config.pop("full_attention_interval", None)
|
| 28 |
+
if interval is None and plan:
|
| 29 |
+
interval = next(i + 1 for i, t in enumerate(plan) if t == AGNES_GLOBAL)
|
| 30 |
+
text_config["full_attention_interval"] = int(interval or 4)
|
| 31 |
+
main = int(text_config["intermediate_size"])
|
| 32 |
+
# the parallel branch is folded into the main MLP at load time
|
| 33 |
+
text_config["intermediate_size"] = main + width
|
| 34 |
+
text_config["agnes_main_intermediate_size"] = main
|
| 35 |
+
text_config["agnes_parallel_ffn_intermediate_size"] = width
|
| 36 |
+
if isinstance(vision_config, dict):
|
| 37 |
+
vision_config = dict(vision_config)
|
| 38 |
+
vision_config["model_type"] = "qwen3_5"
|
| 39 |
+
kwargs["architectures"] = ["Qwen3_5ForConditionalGeneration"]
|
| 40 |
+
super().__init__(text_config=text_config, vision_config=vision_config, **kwargs)
|
| 41 |
+
# every downstream check sees the built-in hybrid architecture
|
| 42 |
+
self.model_type = "qwen3_5"
|
| 43 |
+
|
| 44 |
+
@classmethod
|
| 45 |
+
def from_pretrained(cls, pretrained_model_name_or_path, *args, **kwargs):
|
| 46 |
+
# The weight loader needs the checkpoint directory to pick up the parallel
|
| 47 |
+
# branch tensors. The path is written into the config *dict* before the
|
| 48 |
+
# object is built: the config reaches the worker processes through a
|
| 49 |
+
# to_dict round trip, which keeps fields that came in through __init__
|
| 50 |
+
# and drops attributes set afterwards (from_pretrained's own kwargs only
|
| 51 |
+
# override known fields, so they cannot carry it either).
|
| 52 |
+
path = str(pretrained_model_name_or_path)
|
| 53 |
+
config_dict, kwargs = cls.get_config_dict(pretrained_model_name_or_path, **kwargs)
|
| 54 |
+
config_dict["agnes_model_path"] = path
|
| 55 |
+
if isinstance(config_dict.get("text_config"), dict):
|
| 56 |
+
config_dict["text_config"]["agnes_model_path"] = path
|
| 57 |
+
return cls.from_dict(config_dict, **kwargs)
|
sglang_patch/v0.5.19/sglang/srt/models/qwen3_5.py
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
sglang_patch/v0.5.19/sglang/srt/utils/hf_transformers/common.py
ADDED
|
@@ -0,0 +1,670 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2023-2024 SGLang Team
|
| 2 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 3 |
+
# you may not use this file except in compliance with the License.
|
| 4 |
+
# You may obtain a copy of the License at
|
| 5 |
+
#
|
| 6 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 7 |
+
#
|
| 8 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 9 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 10 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 11 |
+
# See the License for the specific language governing permissions and
|
| 12 |
+
# limitations under the License.
|
| 13 |
+
# ==============================================================================
|
| 14 |
+
"""Shared helpers used by config, tokenizer, and processor modules."""
|
| 15 |
+
|
| 16 |
+
import json
|
| 17 |
+
import os
|
| 18 |
+
from pathlib import Path
|
| 19 |
+
from typing import Any, Dict, Optional, Type, Union
|
| 20 |
+
|
| 21 |
+
import torch
|
| 22 |
+
from huggingface_hub import snapshot_download
|
| 23 |
+
|
| 24 |
+
from sglang.srt.configs import (
|
| 25 |
+
AfmoeConfig,
|
| 26 |
+
BailingHybridConfig,
|
| 27 |
+
ChatGLMConfig,
|
| 28 |
+
DbrxConfig,
|
| 29 |
+
DeepseekVL2Config,
|
| 30 |
+
Dots3Config,
|
| 31 |
+
DotsOCRConfig,
|
| 32 |
+
DotsVLMConfig,
|
| 33 |
+
ExaoneConfig,
|
| 34 |
+
FalconH1Config,
|
| 35 |
+
GraniteMoeHybridConfig,
|
| 36 |
+
InklingAudioConfig,
|
| 37 |
+
InklingMMConfig,
|
| 38 |
+
InklingModelConfig,
|
| 39 |
+
InklingVisionConfig,
|
| 40 |
+
InternS2MobiusConfig,
|
| 41 |
+
InternS2MobiusTextConfig,
|
| 42 |
+
InternS2PreviewConfig,
|
| 43 |
+
JetNemotronConfig,
|
| 44 |
+
JetVLMConfig,
|
| 45 |
+
KimiK3Config,
|
| 46 |
+
KimiK25Config,
|
| 47 |
+
KimiLinearConfig,
|
| 48 |
+
KimiVLConfig,
|
| 49 |
+
LagunaConfig,
|
| 50 |
+
LocateAnythingConfig,
|
| 51 |
+
LongcatFlashConfig,
|
| 52 |
+
MiniCPMHybridConfig,
|
| 53 |
+
MiniCPMV4_6Config,
|
| 54 |
+
MiniCPMV4_6VisionConfig,
|
| 55 |
+
MiniMaxM3VLConfig,
|
| 56 |
+
MultiModalityConfig,
|
| 57 |
+
MuseGlimmerAssistantConfig,
|
| 58 |
+
MuseGlimmerConfig,
|
| 59 |
+
NemotronH_Nano_Omni_Reasoning_V3_Config,
|
| 60 |
+
NemotronH_Nano_VL_V2_Config,
|
| 61 |
+
NemotronHConfig,
|
| 62 |
+
NemotronHPuzzleConfig,
|
| 63 |
+
Olmo3Config,
|
| 64 |
+
Qwen3_5Config,
|
| 65 |
+
Qwen3_5MoeConfig,
|
| 66 |
+
Qwen3_5MoeTextConfig,
|
| 67 |
+
Qwen3_5TextConfig,
|
| 68 |
+
Qwen3NextConfig,
|
| 69 |
+
Spark2_5Config,
|
| 70 |
+
Step3p5Config,
|
| 71 |
+
Step3p7Config,
|
| 72 |
+
Step3VLConfig,
|
| 73 |
+
)
|
| 74 |
+
from sglang.srt.configs.deepseek_ocr import DeepseekVLV2Config
|
| 75 |
+
from sglang.srt.configs.internvl import InternVLChatConfig
|
| 76 |
+
from sglang.srt.utils import get_bool_env_var, logger, lru_cache_frozenset
|
| 77 |
+
from sglang.srt.utils.runai_utils import ObjectStorageModel, is_runai_obj_uri
|
| 78 |
+
|
| 79 |
+
from ..hf_transformers_patches import normalize_rope_scaling_compat
|
| 80 |
+
|
| 81 |
+
if get_bool_env_var("SGLANG_USE_MODELSCOPE"):
|
| 82 |
+
from modelscope import AutoConfig, GenerationConfig
|
| 83 |
+
else:
|
| 84 |
+
from transformers import AutoConfig, GenerationConfig
|
| 85 |
+
|
| 86 |
+
from transformers import PretrainedConfig
|
| 87 |
+
|
| 88 |
+
# ---------------------------------------------------------------------------
|
| 89 |
+
# Config registry
|
| 90 |
+
# ---------------------------------------------------------------------------
|
| 91 |
+
|
| 92 |
+
_CONFIG_REGISTRY: Dict[str, Type[PretrainedConfig]] = {
|
| 93 |
+
cls.model_type: cls
|
| 94 |
+
for cls in [
|
| 95 |
+
AfmoeConfig,
|
| 96 |
+
BailingHybridConfig,
|
| 97 |
+
ChatGLMConfig,
|
| 98 |
+
DbrxConfig,
|
| 99 |
+
ExaoneConfig,
|
| 100 |
+
DeepseekVL2Config,
|
| 101 |
+
MultiModalityConfig,
|
| 102 |
+
KimiVLConfig,
|
| 103 |
+
LocateAnythingConfig,
|
| 104 |
+
InternVLChatConfig,
|
| 105 |
+
LagunaConfig,
|
| 106 |
+
Spark2_5Config,
|
| 107 |
+
Step3VLConfig,
|
| 108 |
+
LongcatFlashConfig,
|
| 109 |
+
Olmo3Config,
|
| 110 |
+
MuseGlimmerConfig,
|
| 111 |
+
MuseGlimmerAssistantConfig,
|
| 112 |
+
KimiK3Config,
|
| 113 |
+
KimiLinearConfig,
|
| 114 |
+
Qwen3NextConfig,
|
| 115 |
+
FalconH1Config,
|
| 116 |
+
GraniteMoeHybridConfig,
|
| 117 |
+
DotsVLMConfig,
|
| 118 |
+
DotsOCRConfig,
|
| 119 |
+
Dots3Config,
|
| 120 |
+
NemotronH_Nano_VL_V2_Config,
|
| 121 |
+
NemotronH_Nano_Omni_Reasoning_V3_Config,
|
| 122 |
+
NemotronHConfig,
|
| 123 |
+
NemotronHPuzzleConfig,
|
| 124 |
+
DeepseekVLV2Config,
|
| 125 |
+
Qwen3_5Config,
|
| 126 |
+
Qwen3_5MoeConfig,
|
| 127 |
+
Qwen3_5TextConfig,
|
| 128 |
+
Qwen3_5MoeTextConfig,
|
| 129 |
+
InternS2PreviewConfig,
|
| 130 |
+
InternS2MobiusConfig,
|
| 131 |
+
InternS2MobiusTextConfig,
|
| 132 |
+
JetNemotronConfig,
|
| 133 |
+
JetVLMConfig,
|
| 134 |
+
KimiK25Config,
|
| 135 |
+
Step3p5Config,
|
| 136 |
+
Step3p7Config,
|
| 137 |
+
MiniCPMHybridConfig,
|
| 138 |
+
MiniCPMV4_6Config,
|
| 139 |
+
MiniCPMV4_6VisionConfig,
|
| 140 |
+
InklingModelConfig,
|
| 141 |
+
InklingAudioConfig,
|
| 142 |
+
InklingVisionConfig,
|
| 143 |
+
InklingMMConfig,
|
| 144 |
+
MiniMaxM3VLConfig,
|
| 145 |
+
]
|
| 146 |
+
}
|
| 147 |
+
|
| 148 |
+
|
| 149 |
+
# DeepSeek V3.2 / V4 reuse the V3 config schema. Subclass the upstream
|
| 150 |
+
# transformers class with each model_type so AutoConfig.register passes its
|
| 151 |
+
# consistency check (which requires class.model_type == registered key).
|
| 152 |
+
# Default-value divergences (e.g. V4's topk_group) are handled in
|
| 153 |
+
# model_config.py post-load.
|
| 154 |
+
try:
|
| 155 |
+
from transformers import DeepseekV3Config as _HFDeepseekV3Config
|
| 156 |
+
|
| 157 |
+
class _DeepseekV32ConfigAlias(_HFDeepseekV3Config):
|
| 158 |
+
model_type = "deepseek_v32"
|
| 159 |
+
|
| 160 |
+
class _DeepseekV4ConfigAlias(_HFDeepseekV3Config):
|
| 161 |
+
model_type = "deepseek_v4"
|
| 162 |
+
|
| 163 |
+
_CONFIG_REGISTRY["deepseek_v32"] = _DeepseekV32ConfigAlias
|
| 164 |
+
_CONFIG_REGISTRY["deepseek_v4"] = _DeepseekV4ConfigAlias
|
| 165 |
+
|
| 166 |
+
# For kimi_k25_eagle3
|
| 167 |
+
class _KimiK2ConfigAlias(_HFDeepseekV3Config):
|
| 168 |
+
model_type = "kimi_k2"
|
| 169 |
+
|
| 170 |
+
_CONFIG_REGISTRY["kimi_k2"] = _KimiK2ConfigAlias
|
| 171 |
+
except ImportError:
|
| 172 |
+
pass
|
| 173 |
+
|
| 174 |
+
# Newer transformers versions (>=5.10.2) expose MellumConfig directly,
|
| 175 |
+
# but fallback to Qwen3MoeConfig for older versions.
|
| 176 |
+
try:
|
| 177 |
+
import transformers as _hf_transformers
|
| 178 |
+
|
| 179 |
+
_HFMellumConfig = getattr(_hf_transformers, "MellumConfig", None)
|
| 180 |
+
|
| 181 |
+
if _HFMellumConfig is not None:
|
| 182 |
+
_CONFIG_REGISTRY["mellum"] = _HFMellumConfig
|
| 183 |
+
else:
|
| 184 |
+
from transformers import Qwen3MoeConfig as _HFQwen3MoeConfig
|
| 185 |
+
|
| 186 |
+
class _MellumConfigAlias(_HFQwen3MoeConfig):
|
| 187 |
+
model_type = "mellum"
|
| 188 |
+
|
| 189 |
+
def __post_init__(self, **kwargs):
|
| 190 |
+
# Qwen3MoeConfig.__post_init__ wipes sliding_window unless
|
| 191 |
+
# use_sliding_window=True. Mellum gates sliding attention
|
| 192 |
+
# per-layer via layer_types, so preserve sliding_window
|
| 193 |
+
# regardless of the legacy use_sliding_window flag.
|
| 194 |
+
sliding_window = getattr(self, "sliding_window", None)
|
| 195 |
+
super().__post_init__(**kwargs)
|
| 196 |
+
self.sliding_window = sliding_window
|
| 197 |
+
|
| 198 |
+
_CONFIG_REGISTRY["mellum"] = _MellumConfigAlias
|
| 199 |
+
|
| 200 |
+
except ImportError:
|
| 201 |
+
pass
|
| 202 |
+
|
| 203 |
+
|
| 204 |
+
try:
|
| 205 |
+
from transformers import Gemma4Config as _HFGemma4Config
|
| 206 |
+
|
| 207 |
+
class _Gemma4UnifiedConfigAlias(_HFGemma4Config):
|
| 208 |
+
model_type = "gemma4_unified"
|
| 209 |
+
|
| 210 |
+
_CONFIG_REGISTRY["gemma4_unified"] = _Gemma4UnifiedConfigAlias
|
| 211 |
+
except ImportError:
|
| 212 |
+
pass
|
| 213 |
+
|
| 214 |
+
for name, cls in _CONFIG_REGISTRY.items():
|
| 215 |
+
try:
|
| 216 |
+
AutoConfig.register(name, cls)
|
| 217 |
+
except ValueError as e:
|
| 218 |
+
err = str(e).lower()
|
| 219 |
+
if "already registered" not in err and "already used" not in err:
|
| 220 |
+
logger.warning("Failed to register config %s: %s", name, e)
|
| 221 |
+
|
| 222 |
+
|
| 223 |
+
# ---------------------------------------------------------------------------
|
| 224 |
+
# Download / path helpers
|
| 225 |
+
# ---------------------------------------------------------------------------
|
| 226 |
+
|
| 227 |
+
|
| 228 |
+
def download_from_hf(
|
| 229 |
+
model_path: str,
|
| 230 |
+
allow_patterns: Optional[Union[str, list]] = None,
|
| 231 |
+
):
|
| 232 |
+
if os.path.exists(model_path):
|
| 233 |
+
return model_path
|
| 234 |
+
|
| 235 |
+
if not allow_patterns:
|
| 236 |
+
allow_patterns = ["*.json", "*.bin", "*.model"]
|
| 237 |
+
|
| 238 |
+
return snapshot_download(model_path, allow_patterns=allow_patterns)
|
| 239 |
+
|
| 240 |
+
|
| 241 |
+
def resolve_runai_obj_uri(model_name_or_path: str) -> str:
|
| 242 |
+
if is_runai_obj_uri(model_name_or_path):
|
| 243 |
+
return ObjectStorageModel.get_path(model_name_or_path)
|
| 244 |
+
return model_name_or_path
|
| 245 |
+
|
| 246 |
+
|
| 247 |
+
def _resolve_local_or_cached_file(model_name_or_path, filename, revision=None):
|
| 248 |
+
"""Resolve a file from a local directory or HF hub cache (no network)."""
|
| 249 |
+
local_path = Path(model_name_or_path) / filename
|
| 250 |
+
if local_path.is_file():
|
| 251 |
+
return str(local_path)
|
| 252 |
+
from huggingface_hub import hf_hub_download
|
| 253 |
+
|
| 254 |
+
return hf_hub_download(
|
| 255 |
+
model_name_or_path, filename, revision=revision, local_files_only=True
|
| 256 |
+
)
|
| 257 |
+
|
| 258 |
+
|
| 259 |
+
def _cached_file_exists(model_name_or_path, filename, revision=None) -> bool:
|
| 260 |
+
"""Whether *filename* is available locally or in the HF cache (no network)."""
|
| 261 |
+
try:
|
| 262 |
+
_resolve_local_or_cached_file(model_name_or_path, filename, revision)
|
| 263 |
+
return True
|
| 264 |
+
except Exception:
|
| 265 |
+
return False
|
| 266 |
+
|
| 267 |
+
|
| 268 |
+
def _remote_file_exists(repo_id, filename, revision=None) -> bool:
|
| 269 |
+
"""Whether *filename* exists on the HF hub (HEAD request only, no download).
|
| 270 |
+
|
| 271 |
+
Returns False on any error (offline, gated, network, invalid id) so callers
|
| 272 |
+
fall back to their default path instead of crashing.
|
| 273 |
+
"""
|
| 274 |
+
from huggingface_hub.constants import HF_HUB_OFFLINE
|
| 275 |
+
|
| 276 |
+
if HF_HUB_OFFLINE:
|
| 277 |
+
return False
|
| 278 |
+
try:
|
| 279 |
+
from huggingface_hub import HfApi
|
| 280 |
+
|
| 281 |
+
return HfApi().file_exists(repo_id, filename, revision=revision)
|
| 282 |
+
except Exception:
|
| 283 |
+
return False
|
| 284 |
+
|
| 285 |
+
|
| 286 |
+
def check_gguf_file(model: Union[str, os.PathLike]) -> bool:
|
| 287 |
+
model = Path(model)
|
| 288 |
+
if not model.is_file():
|
| 289 |
+
return False
|
| 290 |
+
elif model.suffix == ".gguf":
|
| 291 |
+
return True
|
| 292 |
+
|
| 293 |
+
with open(model, "rb") as f:
|
| 294 |
+
header = f.read(4)
|
| 295 |
+
return header == b"GGUF"
|
| 296 |
+
|
| 297 |
+
|
| 298 |
+
def resolve_hf_gguf_reference(
|
| 299 |
+
model: str, revision: Optional[str] = None
|
| 300 |
+
) -> Optional[str]:
|
| 301 |
+
"""Download a .gguf named by Hub reference and return its local path.
|
| 302 |
+
|
| 303 |
+
owner/repo/path/inside/repo.gguf -> exactly that file
|
| 304 |
+
owner/repo:QUANT_TYPE -> the only matching quantization
|
| 305 |
+
owner/repo -> the only .gguf in the repo
|
| 306 |
+
"""
|
| 307 |
+
from sglang.srt.utils import is_remote_url
|
| 308 |
+
|
| 309 |
+
if not model or os.path.exists(model) or is_remote_url(model):
|
| 310 |
+
return None
|
| 311 |
+
|
| 312 |
+
from huggingface_hub import hf_hub_download
|
| 313 |
+
|
| 314 |
+
if ":" in model:
|
| 315 |
+
repo_id, _, quant_type = model.rpartition(":")
|
| 316 |
+
if repo_id.count("/") != 1 or not quant_type:
|
| 317 |
+
return None
|
| 318 |
+
|
| 319 |
+
from huggingface_hub import HfApi
|
| 320 |
+
|
| 321 |
+
files = [
|
| 322 |
+
sibling.rfilename
|
| 323 |
+
for sibling in HfApi().repo_info(repo_id, revision=revision).siblings
|
| 324 |
+
]
|
| 325 |
+
suffix = f"-{quant_type}.gguf"
|
| 326 |
+
candidates = [filename for filename in files if filename.endswith(suffix)]
|
| 327 |
+
if not candidates:
|
| 328 |
+
available = sorted(
|
| 329 |
+
filename for filename in files if filename.endswith(".gguf")
|
| 330 |
+
)
|
| 331 |
+
raise ValueError(
|
| 332 |
+
f"No file matching quant type {quant_type!r} in {repo_id}. "
|
| 333 |
+
f"Available GGUF files: {available}"
|
| 334 |
+
)
|
| 335 |
+
if len(candidates) > 1:
|
| 336 |
+
raise ValueError(
|
| 337 |
+
f"Quant type {quant_type!r} is ambiguous in {repo_id}: "
|
| 338 |
+
f"{sorted(candidates)}. Pass the full owner/repo/path/file.gguf "
|
| 339 |
+
"reference instead."
|
| 340 |
+
)
|
| 341 |
+
return hf_hub_download(repo_id, candidates[0], revision=revision)
|
| 342 |
+
|
| 343 |
+
parts = model.strip("/").split("/")
|
| 344 |
+
if len(parts) < 2:
|
| 345 |
+
return None
|
| 346 |
+
|
| 347 |
+
if len(parts) > 2 and model.endswith(".gguf"):
|
| 348 |
+
repo_id = "/".join(parts[:2])
|
| 349 |
+
filename = "/".join(parts[2:])
|
| 350 |
+
return hf_hub_download(repo_id, filename, revision=revision)
|
| 351 |
+
|
| 352 |
+
if len(parts) != 2:
|
| 353 |
+
return None
|
| 354 |
+
|
| 355 |
+
from huggingface_hub import HfApi
|
| 356 |
+
|
| 357 |
+
try:
|
| 358 |
+
files = [
|
| 359 |
+
s.rfilename for s in HfApi().repo_info(model, revision=revision).siblings
|
| 360 |
+
]
|
| 361 |
+
except Exception:
|
| 362 |
+
return None
|
| 363 |
+
if any(f == "config.json" for f in files):
|
| 364 |
+
return None
|
| 365 |
+
|
| 366 |
+
candidates = [f for f in files if f.endswith(".gguf")]
|
| 367 |
+
if not candidates:
|
| 368 |
+
return None
|
| 369 |
+
if len(candidates) > 1:
|
| 370 |
+
listing = "\n ".join(f"{model}/{f}" for f in sorted(candidates))
|
| 371 |
+
raise ValueError(
|
| 372 |
+
f"{model} contains {len(candidates)} .gguf files; name the one to "
|
| 373 |
+
f"serve:\n {listing}"
|
| 374 |
+
)
|
| 375 |
+
return hf_hub_download(model, candidates[0], revision=revision)
|
| 376 |
+
|
| 377 |
+
|
| 378 |
+
def gguf_sidecar_dir(
|
| 379 |
+
gguf_path: Union[str, os.PathLike], sentinel: str
|
| 380 |
+
) -> Optional[Path]:
|
| 381 |
+
"""Directory containing *sentinel* next to a .gguf file, if there is one."""
|
| 382 |
+
directory = Path(gguf_path).parent
|
| 383 |
+
return directory if (directory / sentinel).is_file() else None
|
| 384 |
+
|
| 385 |
+
|
| 386 |
+
# ---------------------------------------------------------------------------
|
| 387 |
+
# Rope / text config helpers
|
| 388 |
+
# ---------------------------------------------------------------------------
|
| 389 |
+
|
| 390 |
+
|
| 391 |
+
def get_rope_config(config):
|
| 392 |
+
"""Get (rope_theta, rope_params) from config, supporting both v4 and v5.
|
| 393 |
+
|
| 394 |
+
Trust-remote-code configs or parent configs passed to sub-models may not
|
| 395 |
+
have the v5 ``rope_parameters`` property, so we fall back to the v4-style
|
| 396 |
+
``config.rope_theta`` / ``config.rope_scaling`` attributes.
|
| 397 |
+
|
| 398 |
+
Returns:
|
| 399 |
+
(rope_theta, rope_params): In v5, rope_params is the full
|
| 400 |
+
rope_parameters dict (which subsumes rope_scaling and includes
|
| 401 |
+
rope_theta). In v4, rope_params is the rope_scaling dict or None.
|
| 402 |
+
"""
|
| 403 |
+
rope_params = getattr(config, "rope_parameters", None)
|
| 404 |
+
if rope_params is not None:
|
| 405 |
+
rope_theta = rope_params.get("rope_theta", getattr(config, "rope_theta", 10000))
|
| 406 |
+
return rope_theta, rope_params
|
| 407 |
+
return getattr(config, "rope_theta", 10000), getattr(config, "rope_scaling", None)
|
| 408 |
+
|
| 409 |
+
|
| 410 |
+
def _patch_text_config(parent_config: PretrainedConfig, text_config):
|
| 411 |
+
"""Synchronize standard attributes between parent config and text sub-config.
|
| 412 |
+
|
| 413 |
+
In transformers v5, the "untangle config" refactor removed automatic
|
| 414 |
+
inheritance of top-level PretrainedConfig attributes (pad_token_id,
|
| 415 |
+
tie_word_embeddings, etc.) from sub-configs. Downstream code expects
|
| 416 |
+
these attributes to be present on both configs (some models pass the
|
| 417 |
+
parent directly to the language model, others pass the text sub-config),
|
| 418 |
+
so we propagate in both directions when an attribute is missing.
|
| 419 |
+
(See https://github.com/huggingface/transformers/pull/41541)
|
| 420 |
+
"""
|
| 421 |
+
_ATTRS_TO_PROPAGATE = [
|
| 422 |
+
"pad_token_id",
|
| 423 |
+
"bos_token_id",
|
| 424 |
+
"eos_token_id",
|
| 425 |
+
"tie_word_embeddings",
|
| 426 |
+
]
|
| 427 |
+
for attr in _ATTRS_TO_PROPAGATE:
|
| 428 |
+
parent_has = hasattr(parent_config, attr)
|
| 429 |
+
text_has = hasattr(text_config, attr)
|
| 430 |
+
if parent_has and not text_has:
|
| 431 |
+
setattr(text_config, attr, getattr(parent_config, attr))
|
| 432 |
+
elif text_has and not parent_has:
|
| 433 |
+
setattr(parent_config, attr, getattr(text_config, attr))
|
| 434 |
+
return text_config
|
| 435 |
+
|
| 436 |
+
|
| 437 |
+
def get_hf_text_config(config: PretrainedConfig):
|
| 438 |
+
"""Get the "sub" config relevant to llm for multi modal models.
|
| 439 |
+
No op for pure text models.
|
| 440 |
+
"""
|
| 441 |
+
if config.architectures is not None:
|
| 442 |
+
class_name = config.architectures[0]
|
| 443 |
+
if class_name.startswith("Llava") and class_name.endswith("ForCausalLM"):
|
| 444 |
+
# We support non-hf version of llava models, so we do not want to
|
| 445 |
+
# read the wrong values from the unused default text_config.
|
| 446 |
+
# NOTE(HandH1998): We set `torch_dtype` of config to `torch.float16` for the weights, as
|
| 447 |
+
# `torch.float16` is default used for image features in `python/sglang/srt/models/llava.py`.
|
| 448 |
+
setattr(config, "dtype", torch.float16)
|
| 449 |
+
return config
|
| 450 |
+
|
| 451 |
+
text_config = None
|
| 452 |
+
|
| 453 |
+
# Some models (e.g. DeepSeek-OCR) store sub-configs as plain dicts.
|
| 454 |
+
# Convert to PretrainedConfig early so hasattr() checks and asserts work.
|
| 455 |
+
parent_dtype = getattr(config, "dtype", None)
|
| 456 |
+
for _attr in ("text_config", "llm_config", "language_config", "thinker_config"):
|
| 457 |
+
_sub = getattr(config, _attr, None)
|
| 458 |
+
if isinstance(_sub, dict):
|
| 459 |
+
_converted = PretrainedConfig(**_sub)
|
| 460 |
+
if getattr(_converted, "dtype", None) is None and parent_dtype is not None:
|
| 461 |
+
_converted.dtype = parent_dtype
|
| 462 |
+
setattr(config, _attr, _converted)
|
| 463 |
+
elif _sub is not None and parent_dtype is not None:
|
| 464 |
+
# transformers v5 multimodal configs (e.g. Mistral3Config) carry
|
| 465 |
+
# `dtype` only on the top-level config, leaving the sub-configs at
|
| 466 |
+
# None. Without this, _get_and_verify_dtype falls back to float32
|
| 467 |
+
# and then "auto" downcasts to float16, which overflows the Pixtral
|
| 468 |
+
# vision tower on real images and produces NaN features.
|
| 469 |
+
if getattr(_sub, "dtype", None) is None:
|
| 470 |
+
_sub.dtype = parent_dtype
|
| 471 |
+
|
| 472 |
+
# Priority: thinker_config > llm_config > language_config > text_config
|
| 473 |
+
if hasattr(config, "thinker_config"):
|
| 474 |
+
# qwen2.5 omni
|
| 475 |
+
thinker_config = config.thinker_config
|
| 476 |
+
if hasattr(thinker_config, "text_config"):
|
| 477 |
+
setattr(
|
| 478 |
+
thinker_config.text_config,
|
| 479 |
+
"dtype",
|
| 480 |
+
getattr(thinker_config, "dtype", None),
|
| 481 |
+
)
|
| 482 |
+
text_config = thinker_config.text_config
|
| 483 |
+
else:
|
| 484 |
+
text_config = thinker_config
|
| 485 |
+
elif hasattr(config, "llm_config"):
|
| 486 |
+
# PointsV1.5 Chat Model
|
| 487 |
+
assert hasattr(config.llm_config, "num_attention_heads")
|
| 488 |
+
text_config = config.llm_config
|
| 489 |
+
elif hasattr(config, "language_config"):
|
| 490 |
+
text_config = config.language_config
|
| 491 |
+
elif hasattr(config, "text_config"):
|
| 492 |
+
# The code operates under the assumption that text_config should have
|
| 493 |
+
# `num_attention_heads` (among others). Assert here to fail early
|
| 494 |
+
# if transformers config doesn't align with this assumption.
|
| 495 |
+
assert hasattr(config.text_config, "num_attention_heads")
|
| 496 |
+
text_config = config.text_config
|
| 497 |
+
|
| 498 |
+
# Ensure rope_scaling dicts have "type" for remote-code compat (v5).
|
| 499 |
+
normalize_rope_scaling_compat(config)
|
| 500 |
+
|
| 501 |
+
if text_config is not None:
|
| 502 |
+
return _patch_text_config(config, text_config)
|
| 503 |
+
return config
|
| 504 |
+
|
| 505 |
+
|
| 506 |
+
# ---------------------------------------------------------------------------
|
| 507 |
+
# Model-specific helpers
|
| 508 |
+
# ---------------------------------------------------------------------------
|
| 509 |
+
|
| 510 |
+
|
| 511 |
+
def _ensure_sub_configs(config: PretrainedConfig, *attr_names: str) -> None:
|
| 512 |
+
"""Convert dict-valued sub-configs to proper AutoConfig objects in-place."""
|
| 513 |
+
for attr in attr_names:
|
| 514 |
+
sub = getattr(config, attr, None)
|
| 515 |
+
if sub is not None and isinstance(sub, dict):
|
| 516 |
+
setattr(config, attr, AutoConfig.for_model(**sub))
|
| 517 |
+
|
| 518 |
+
|
| 519 |
+
def _is_deepseek_ocr_model(config: PretrainedConfig) -> bool:
|
| 520 |
+
# TODO: Remove this workaround once AutoConfig correctly identifies deepseek-ocr.
|
| 521 |
+
# Hugging Face's AutoConfig currently misidentifies it as deepseekvl2.
|
| 522 |
+
auto_map = getattr(config, "auto_map", None) or {}
|
| 523 |
+
return auto_map.get("AutoModel") == "modeling_deepseekocr.DeepseekOCRForCausalLM"
|
| 524 |
+
|
| 525 |
+
|
| 526 |
+
def _is_deepseek_ocr2_model(config: PretrainedConfig) -> bool:
|
| 527 |
+
auto_map = getattr(config, "auto_map", None) or {}
|
| 528 |
+
return auto_map.get("AutoModel") == "modeling_deepseekocr2.DeepseekOCR2ForCausalLM"
|
| 529 |
+
|
| 530 |
+
|
| 531 |
+
def _override_v_head_dim_if_zero(config: PretrainedConfig, patch: int = 128) -> None:
|
| 532 |
+
patched = False
|
| 533 |
+
for attr in ("text_config", "language_config"):
|
| 534 |
+
sub = getattr(config, attr, None)
|
| 535 |
+
if sub is None:
|
| 536 |
+
continue
|
| 537 |
+
if isinstance(sub, dict):
|
| 538 |
+
if sub.get("v_head_dim") == 0:
|
| 539 |
+
sub["v_head_dim"] = patch
|
| 540 |
+
patched = True
|
| 541 |
+
elif getattr(sub, "v_head_dim", None) == 0:
|
| 542 |
+
sub.v_head_dim = patch
|
| 543 |
+
patched = True
|
| 544 |
+
if patched:
|
| 545 |
+
logger.warning(
|
| 546 |
+
f"Overriding v_head_dim from 0 to {patch} to avoid potential issues."
|
| 547 |
+
)
|
| 548 |
+
|
| 549 |
+
|
| 550 |
+
# ---------------------------------------------------------------------------
|
| 551 |
+
# Context length / generation config / sparse attention
|
| 552 |
+
# ---------------------------------------------------------------------------
|
| 553 |
+
|
| 554 |
+
# Models don't use the same configuration key for determining the maximum
|
| 555 |
+
# context length. Store them here so we can sanely check them.
|
| 556 |
+
# NOTE: The ordering here is important. Some models have two of these and we
|
| 557 |
+
# have a preference for which value gets used.
|
| 558 |
+
CONTEXT_LENGTH_KEYS = [
|
| 559 |
+
"max_sequence_length",
|
| 560 |
+
"seq_length",
|
| 561 |
+
"max_seq_len",
|
| 562 |
+
"model_max_length",
|
| 563 |
+
"max_position_embeddings",
|
| 564 |
+
]
|
| 565 |
+
|
| 566 |
+
|
| 567 |
+
def get_context_length(config):
|
| 568 |
+
"""Get the context length of a model from a huggingface model configs."""
|
| 569 |
+
text_config = config
|
| 570 |
+
rope_scaling = getattr(text_config, "rope_scaling", None)
|
| 571 |
+
if rope_scaling:
|
| 572 |
+
rope_scaling_factor = rope_scaling.get("factor", 1)
|
| 573 |
+
if "original_max_position_embeddings" in rope_scaling:
|
| 574 |
+
rope_scaling_factor = 1
|
| 575 |
+
if rope_scaling.get("rope_type", None) == "llama3":
|
| 576 |
+
rope_scaling_factor = 1
|
| 577 |
+
else:
|
| 578 |
+
rope_scaling_factor = 1
|
| 579 |
+
|
| 580 |
+
for key in CONTEXT_LENGTH_KEYS:
|
| 581 |
+
val = getattr(text_config, key, None)
|
| 582 |
+
if val is not None:
|
| 583 |
+
return int(rope_scaling_factor * val)
|
| 584 |
+
return 2048
|
| 585 |
+
|
| 586 |
+
|
| 587 |
+
@lru_cache_frozenset(maxsize=32)
|
| 588 |
+
def get_generation_config(
|
| 589 |
+
model: str,
|
| 590 |
+
trust_remote_code: bool,
|
| 591 |
+
revision: Optional[str] = None,
|
| 592 |
+
**kwargs,
|
| 593 |
+
):
|
| 594 |
+
if check_gguf_file(model):
|
| 595 |
+
sidecar = gguf_sidecar_dir(model, "generation_config.json")
|
| 596 |
+
if sidecar is not None:
|
| 597 |
+
model = str(sidecar)
|
| 598 |
+
else:
|
| 599 |
+
from .gguf_native import (
|
| 600 |
+
build_gguf_generation_config,
|
| 601 |
+
has_native_gguf_support,
|
| 602 |
+
)
|
| 603 |
+
|
| 604 |
+
if has_native_gguf_support(model):
|
| 605 |
+
return build_gguf_generation_config(model)
|
| 606 |
+
|
| 607 |
+
try:
|
| 608 |
+
return GenerationConfig.from_pretrained(
|
| 609 |
+
model, trust_remote_code=trust_remote_code, revision=revision, **kwargs
|
| 610 |
+
)
|
| 611 |
+
except (FileNotFoundError, OSError) as e:
|
| 612 |
+
# A missing generation_config.json is normal for many checkpoints and
|
| 613 |
+
# is surfaced by HF as a generic OSError (not FileNotFoundError). Treat
|
| 614 |
+
# it as benign — proceed without a generation config, at DEBUG level so
|
| 615 |
+
# normal startup logs stay quiet.
|
| 616 |
+
logger.debug(
|
| 617 |
+
"No generation config for %s: %s. Proceeding without it.",
|
| 618 |
+
model,
|
| 619 |
+
e,
|
| 620 |
+
)
|
| 621 |
+
return None
|
| 622 |
+
|
| 623 |
+
|
| 624 |
+
# Qwen-1M related
|
| 625 |
+
def get_sparse_attention_config(
|
| 626 |
+
model: str,
|
| 627 |
+
sparse_attention_config_filename: str = "sparse_attention_config.json",
|
| 628 |
+
) -> Dict[str, Any]:
|
| 629 |
+
is_local = os.path.isdir(model)
|
| 630 |
+
if not is_local:
|
| 631 |
+
model = download_from_hf(model, allow_patterns=["*.json"])
|
| 632 |
+
|
| 633 |
+
config_file = os.path.join(model, sparse_attention_config_filename)
|
| 634 |
+
if not os.path.exists(config_file):
|
| 635 |
+
return {}
|
| 636 |
+
|
| 637 |
+
with open(config_file) as f:
|
| 638 |
+
config = json.load(f)
|
| 639 |
+
return config
|
| 640 |
+
|
| 641 |
+
|
| 642 |
+
# ---------------------------------------------------------------------------
|
| 643 |
+
# Tokenizer / processor helpers
|
| 644 |
+
# ---------------------------------------------------------------------------
|
| 645 |
+
|
| 646 |
+
|
| 647 |
+
# Some models don't have an available processor, e.g.: InternVL
|
| 648 |
+
def get_tokenizer_from_processor(processor):
|
| 649 |
+
from transformers import PreTrainedTokenizerBase
|
| 650 |
+
|
| 651 |
+
if isinstance(processor, PreTrainedTokenizerBase):
|
| 652 |
+
return processor
|
| 653 |
+
return processor.tokenizer
|
| 654 |
+
|
| 655 |
+
|
| 656 |
+
# Turn-final markers that some checkpoints ship without EOS metadata:
|
| 657 |
+
# <|eom_id|> (Llama-3 tool use) and <|content_model_end_sampling|> (Inkling,
|
| 658 |
+
# whose bundled tokenizer config leaves eos_token unset).
|
| 659 |
+
_ADDITIONAL_STOP_TOKEN_TEXTS = ("<|eom_id|>", "<|content_model_end_sampling|>")
|
| 660 |
+
|
| 661 |
+
|
| 662 |
+
def attach_additional_stop_token_ids(tokenizer):
|
| 663 |
+
added = tokenizer.get_added_vocab()
|
| 664 |
+
stop_ids = {added[text] for text in _ADDITIONAL_STOP_TOKEN_TEXTS if text in added}
|
| 665 |
+
tokenizer.additional_stop_token_ids = stop_ids or None
|
| 666 |
+
|
| 667 |
+
# === agnes ===
|
| 668 |
+
from sglang.srt.configs.agnes import AgnesConfig as _AgnesConfig
|
| 669 |
+
|
| 670 |
+
_CONFIG_REGISTRY[_AgnesConfig.model_type] = _AgnesConfig
|
tokenizer.json
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c7eae8d08432055ad259acfe49c8b17b3e68e4a92f43170f8c85505ae7a586f3
|
| 3 |
+
size 19991784
|
tokenizer_config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"audio_bos_token": "<|audio_start|>",
|
| 4 |
+
"audio_eos_token": "<|audio_end|>",
|
| 5 |
+
"audio_token": "<|audio_pad|>",
|
| 6 |
+
"backend": "tokenizers",
|
| 7 |
+
"bos_token": null,
|
| 8 |
+
"clean_up_tokenization_spaces": false,
|
| 9 |
+
"eos_token": "<|im_end|>",
|
| 10 |
+
"errors": "replace",
|
| 11 |
+
"fix_mistral_regex": true,
|
| 12 |
+
"image_token": "<|image_pad|>",
|
| 13 |
+
"is_local": true,
|
| 14 |
+
"local_files_only": false,
|
| 15 |
+
"model_max_length": 262144,
|
| 16 |
+
"model_specific_special_tokens": {
|
| 17 |
+
"audio_bos_token": "<|audio_start|>",
|
| 18 |
+
"audio_eos_token": "<|audio_end|>",
|
| 19 |
+
"audio_token": "<|audio_pad|>",
|
| 20 |
+
"image_token": "<|image_pad|>",
|
| 21 |
+
"video_token": "<|video_pad|>",
|
| 22 |
+
"vision_bos_token": "<|vision_start|>",
|
| 23 |
+
"vision_eos_token": "<|vision_end|>"
|
| 24 |
+
},
|
| 25 |
+
"pad_token": "<|endoftext|>",
|
| 26 |
+
"pretokenize_regex": "(?i:'s|'t|'re|'ve|'m|'ll|'d)|[^\\r\\n\\p{L}\\p{N}]?[\\p{L}\\p{M}]+|\\p{N}| ?[^\\s\\p{L}\\p{M}\\p{N}]+[\\r\\n]*|\\s*[\\r\\n]+|\\s+(?!\\S)|\\s+",
|
| 27 |
+
"split_special_tokens": false,
|
| 28 |
+
"tokenizer_class": "TokenizersBackend",
|
| 29 |
+
"unk_token": null,
|
| 30 |
+
"video_token": "<|video_pad|>",
|
| 31 |
+
"vision_bos_token": "<|vision_start|>",
|
| 32 |
+
"vision_eos_token": "<|vision_end|>"
|
| 33 |
+
}
|
video_preprocessor_config.json
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"video_processor_type": "AgnesVideoProcessor",
|
| 3 |
+
"auto_map": {
|
| 4 |
+
"AutoVideoProcessor": "video_processing_agnes.AgnesVideoProcessor"
|
| 5 |
+
},
|
| 6 |
+
"size": {
|
| 7 |
+
"shortest_edge": 131072,
|
| 8 |
+
"longest_edge": 786432
|
| 9 |
+
},
|
| 10 |
+
"patch_size": 16,
|
| 11 |
+
"temporal_patch_size": 2,
|
| 12 |
+
"merge_size": 2,
|
| 13 |
+
"image_mean": [
|
| 14 |
+
0.5,
|
| 15 |
+
0.5,
|
| 16 |
+
0.5
|
| 17 |
+
],
|
| 18 |
+
"image_std": [
|
| 19 |
+
0.5,
|
| 20 |
+
0.5,
|
| 21 |
+
0.5
|
| 22 |
+
],
|
| 23 |
+
"fps": 2,
|
| 24 |
+
"min_frames": 4,
|
| 25 |
+
"max_frames": 768,
|
| 26 |
+
"do_sample_frames": true
|
| 27 |
+
}
|
video_processing_agnes.py
ADDED
|
@@ -0,0 +1,197 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
# Copyright 2026 Agnes AI. All rights reserved.
|
| 2 |
+
#
|
| 3 |
+
# Licensed under the Apache License, Version 2.0 (the "License");
|
| 4 |
+
# you may not use this file except in compliance with the License.
|
| 5 |
+
# You may obtain a copy of the License at
|
| 6 |
+
#
|
| 7 |
+
# http://www.apache.org/licenses/LICENSE-2.0
|
| 8 |
+
#
|
| 9 |
+
# Unless required by applicable law or agreed to in writing, software
|
| 10 |
+
# distributed under the License is distributed on an "AS IS" BASIS,
|
| 11 |
+
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
| 12 |
+
# See the License for the specific language governing permissions and
|
| 13 |
+
# limitations under the License.
|
| 14 |
+
"""Video processor for Agnes 3.0 Flash: frame sampling and dynamic-resolution patching."""
|
| 15 |
+
|
| 16 |
+
import math
|
| 17 |
+
|
| 18 |
+
import numpy as np
|
| 19 |
+
import torch
|
| 20 |
+
|
| 21 |
+
from transformers.feature_extraction_utils import BatchFeature
|
| 22 |
+
from transformers.image_utils import ChannelDimension, PILImageResampling, SizeDict, get_image_size
|
| 23 |
+
from transformers.processing_utils import Unpack, VideosKwargs
|
| 24 |
+
from transformers.utils import TensorType, add_start_docstrings, is_torchvision_available, logging
|
| 25 |
+
from transformers.video_processing_utils import BASE_VIDEO_PROCESSOR_DOCSTRING, BaseVideoProcessor
|
| 26 |
+
from transformers.video_utils import VideoMetadata, group_videos_by_shape, reorder_videos
|
| 27 |
+
|
| 28 |
+
|
| 29 |
+
if is_torchvision_available():
|
| 30 |
+
from torchvision.transforms.v2 import functional as tvF
|
| 31 |
+
|
| 32 |
+
logger = logging.get_logger(__name__)
|
| 33 |
+
|
| 34 |
+
|
| 35 |
+
def fit_video_to_grid(
|
| 36 |
+
num_frames: int,
|
| 37 |
+
height: int,
|
| 38 |
+
width: int,
|
| 39 |
+
temporal_factor: int = 2,
|
| 40 |
+
factor: int = 32,
|
| 41 |
+
min_pixels: int = 128 * 128,
|
| 42 |
+
max_pixels: int = 16 * 16 * 2 * 2 * 2 * 6144,
|
| 43 |
+
):
|
| 44 |
+
"""Spatial size for a clip: multiples of `factor`, with the frame count
|
| 45 |
+
rounded up to `temporal_factor` and the total voxel count kept inside
|
| 46 |
+
[min_pixels, max_pixels]."""
|
| 47 |
+
if height < factor or width < factor:
|
| 48 |
+
raise ValueError(f"height:{height} or width:{width} must be larger than factor:{factor}")
|
| 49 |
+
elif max(height, width) / min(height, width) > 200:
|
| 50 |
+
raise ValueError(f"absolute aspect ratio must be smaller than 200, got {max(height, width) / min(height, width)}")
|
| 51 |
+
h = round(height / factor) * factor
|
| 52 |
+
w = round(width / factor) * factor
|
| 53 |
+
t = math.ceil(num_frames / temporal_factor) * temporal_factor
|
| 54 |
+
if t * h * w > max_pixels:
|
| 55 |
+
scale = math.sqrt((num_frames * height * width) / max_pixels)
|
| 56 |
+
h = max(factor, math.floor(height / scale / factor) * factor)
|
| 57 |
+
w = max(factor, math.floor(width / scale / factor) * factor)
|
| 58 |
+
elif t * h * w < min_pixels:
|
| 59 |
+
scale = math.sqrt(min_pixels / (num_frames * height * width))
|
| 60 |
+
h = math.ceil(height * scale / factor) * factor
|
| 61 |
+
w = math.ceil(width * scale / factor) * factor
|
| 62 |
+
return h, w
|
| 63 |
+
|
| 64 |
+
|
| 65 |
+
class AgnesVideoProcessorInitKwargs(VideosKwargs, total=False):
|
| 66 |
+
patch_size: int
|
| 67 |
+
temporal_patch_size: int
|
| 68 |
+
merge_size: int
|
| 69 |
+
min_frames: int
|
| 70 |
+
max_frames: int
|
| 71 |
+
|
| 72 |
+
|
| 73 |
+
@add_start_docstrings(
|
| 74 |
+
"Video processor for Agnes 3.0 Flash; resizes each clip to a patch grid that fits its own resolution.",
|
| 75 |
+
BASE_VIDEO_PROCESSOR_DOCSTRING,
|
| 76 |
+
"""
|
| 77 |
+
patch_size (`int`, *optional*, defaults to 16):
|
| 78 |
+
Spatial patch size of the vision tower.
|
| 79 |
+
temporal_patch_size (`int`, *optional*, defaults to 2):
|
| 80 |
+
Temporal patch size of the vision tower.
|
| 81 |
+
merge_size (`int`, *optional*, defaults to 2):
|
| 82 |
+
Side of the patch square merged into one language-model token.
|
| 83 |
+
""",
|
| 84 |
+
)
|
| 85 |
+
class AgnesVideoProcessor(BaseVideoProcessor):
|
| 86 |
+
resample = PILImageResampling.BICUBIC
|
| 87 |
+
size = {"shortest_edge": 128 * 32 * 32, "longest_edge": 32 * 32 * 768}
|
| 88 |
+
image_mean = [0.5, 0.5, 0.5]
|
| 89 |
+
image_std = [0.5, 0.5, 0.5]
|
| 90 |
+
do_resize = True
|
| 91 |
+
do_rescale = True
|
| 92 |
+
do_normalize = True
|
| 93 |
+
do_convert_rgb = True
|
| 94 |
+
patch_size = 16
|
| 95 |
+
temporal_patch_size = 2
|
| 96 |
+
merge_size = 2
|
| 97 |
+
fps = 2
|
| 98 |
+
min_frames = 4
|
| 99 |
+
max_frames = 768
|
| 100 |
+
do_sample_frames = True
|
| 101 |
+
valid_kwargs = AgnesVideoProcessorInitKwargs
|
| 102 |
+
model_input_names = ["pixel_values_videos", "video_grid_thw"]
|
| 103 |
+
|
| 104 |
+
def __init__(self, **kwargs: Unpack[AgnesVideoProcessorInitKwargs]):
|
| 105 |
+
super().__init__(**kwargs)
|
| 106 |
+
|
| 107 |
+
def _standardize_kwargs(self, **kwargs) -> dict:
|
| 108 |
+
kwargs = super()._standardize_kwargs(**kwargs)
|
| 109 |
+
size = kwargs.get("size", self.size)
|
| 110 |
+
if not size.shortest_edge or not size.longest_edge:
|
| 111 |
+
raise ValueError("size must contain 'shortest_edge' and 'longest_edge' keys.")
|
| 112 |
+
return kwargs
|
| 113 |
+
|
| 114 |
+
def sample_frames(self, metadata: VideoMetadata, num_frames: int | None = None, fps: int | float | None = None, **kwargs):
|
| 115 |
+
"""Frame indices to keep: `fps` frames per second of source video when
|
| 116 |
+
metadata is available, clamped to [min_frames, max_frames]; `num_frames`
|
| 117 |
+
overrides that. Indices are spread uniformly over the clip."""
|
| 118 |
+
if fps is not None and num_frames is not None:
|
| 119 |
+
raise ValueError("`num_frames` and `fps` are mutually exclusive arguments, please use only one!")
|
| 120 |
+
total = metadata.total_num_frames
|
| 121 |
+
fps = fps if fps is not None else self.fps
|
| 122 |
+
if num_frames is None and fps is not None:
|
| 123 |
+
if metadata.fps is None:
|
| 124 |
+
metadata.fps = 24
|
| 125 |
+
logger.warning_once(
|
| 126 |
+
"Asked to sample `fps` frames per second but no video metadata was provided which is required when sampling with `fps`. "
|
| 127 |
+
"Defaulting to `fps=24`. Please provide `video_metadata` for more accurate results."
|
| 128 |
+
)
|
| 129 |
+
num_frames = int(total / metadata.fps * fps)
|
| 130 |
+
num_frames = min(max(num_frames, self.min_frames), self.max_frames, total)
|
| 131 |
+
if num_frames is None:
|
| 132 |
+
num_frames = min(max(total, self.min_frames), self.max_frames)
|
| 133 |
+
return np.linspace(0, total - 1, num_frames).round().astype(int)
|
| 134 |
+
|
| 135 |
+
def _preprocess(
|
| 136 |
+
self,
|
| 137 |
+
videos: list[torch.Tensor],
|
| 138 |
+
do_convert_rgb: bool = True,
|
| 139 |
+
do_resize: bool = True,
|
| 140 |
+
size: SizeDict | None = None,
|
| 141 |
+
resample: "PILImageResampling | tvF.InterpolationMode | int | None" = PILImageResampling.BICUBIC,
|
| 142 |
+
do_rescale: bool = True,
|
| 143 |
+
rescale_factor: float = 1 / 255.0,
|
| 144 |
+
do_normalize: bool = True,
|
| 145 |
+
image_mean: float | list[float] | None = None,
|
| 146 |
+
image_std: float | list[float] | None = None,
|
| 147 |
+
patch_size: int | None = None,
|
| 148 |
+
temporal_patch_size: int | None = None,
|
| 149 |
+
merge_size: int | None = None,
|
| 150 |
+
return_tensors: str | TensorType | None = None,
|
| 151 |
+
**kwargs,
|
| 152 |
+
):
|
| 153 |
+
# 1. resize, batched per input shape
|
| 154 |
+
by_shape, order = group_videos_by_shape(videos)
|
| 155 |
+
resized = {}
|
| 156 |
+
for shape, batch in by_shape.items():
|
| 157 |
+
if do_convert_rgb:
|
| 158 |
+
batch = self.convert_to_rgb(batch)
|
| 159 |
+
n, t, c, h, w = batch.shape
|
| 160 |
+
if do_resize:
|
| 161 |
+
new_h, new_w = fit_video_to_grid(
|
| 162 |
+
num_frames=t, height=h, width=w, temporal_factor=temporal_patch_size,
|
| 163 |
+
factor=patch_size * merge_size, min_pixels=size.shortest_edge, max_pixels=size.longest_edge,
|
| 164 |
+
)
|
| 165 |
+
batch = self.resize(batch.view(n * t, c, h, w), size=SizeDict(height=new_h, width=new_w), resample=resample)
|
| 166 |
+
batch = batch.view(n, t, c, new_h, new_w)
|
| 167 |
+
resized[shape] = batch
|
| 168 |
+
videos = reorder_videos(resized, order)
|
| 169 |
+
|
| 170 |
+
# 2. normalise, pad the frame count to the temporal patch, cut into patches
|
| 171 |
+
by_shape, order = group_videos_by_shape(videos)
|
| 172 |
+
flat = {}
|
| 173 |
+
grids = {}
|
| 174 |
+
for shape, batch in by_shape.items():
|
| 175 |
+
new_h, new_w = get_image_size(batch[0], channel_dim=ChannelDimension.FIRST)
|
| 176 |
+
px = self.rescale_and_normalize(batch, do_rescale, rescale_factor, do_normalize, image_mean, image_std)
|
| 177 |
+
t = px.shape[1]
|
| 178 |
+
if pad := -t % temporal_patch_size:
|
| 179 |
+
px = torch.cat((px, px[:, -1:].expand(-1, pad, -1, -1, -1)), dim=1)
|
| 180 |
+
n, gt, c = px.shape[:3]
|
| 181 |
+
gt = gt // temporal_patch_size
|
| 182 |
+
gh, gw = new_h // patch_size, new_w // patch_size
|
| 183 |
+
px = px.view(
|
| 184 |
+
n, gt, temporal_patch_size, c, gh // merge_size, merge_size, patch_size, gw // merge_size, merge_size, patch_size
|
| 185 |
+
)
|
| 186 |
+
px = px.permute(0, 1, 4, 7, 5, 8, 3, 2, 6, 9)
|
| 187 |
+
flat[shape] = px.reshape(n, gt * gh * gw, c * temporal_patch_size * patch_size * patch_size)
|
| 188 |
+
grids[shape] = [[gt, gh, gw]] * n
|
| 189 |
+
|
| 190 |
+
pixel_values_videos = torch.cat(reorder_videos(flat, order), dim=0)
|
| 191 |
+
video_grid_thw = torch.tensor(reorder_videos(grids, order))
|
| 192 |
+
return BatchFeature(
|
| 193 |
+
data={"pixel_values_videos": pixel_values_videos, "video_grid_thw": video_grid_thw}, tensor_type=return_tensors
|
| 194 |
+
)
|
| 195 |
+
|
| 196 |
+
|
| 197 |
+
__all__ = ["AgnesVideoProcessor"]
|