diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000000000000000000000000000000000000..91ff68db37e482e4e072616df262e782108563d1 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +benchmarks.png filter=lfs diff=lfs merge=lfs -text diff --git a/CONTRIBUTING.md b/CONTRIBUTING.md new file mode 100644 index 0000000000000000000000000000000000000000..bddbc48a30e7476ef8e9c0dadea103c1a0ea900d --- /dev/null +++ b/CONTRIBUTING.md @@ -0,0 +1,31 @@ +## Notice to external contributors +### General info +Hello! In order for us (YANDEX LLC) to accept patches and other contributions from you, you will have to adopt our Contributor License Agreement (the “CLA”). The current version of the CLA you may find here: + +* https://yandex.ru/legal/cla/en/ (in English) and +* https://yandex.ru/legal/cla/ru/ (in Russian). + +By adopting the CLA, you state the following: + +* You obviously wish and are willingly licensing your contributions to us for our open source projects under the terms of the CLA, +* You have read the terms and conditions of the CLA and agree with them in full, +* You are legally able to provide and license your contributions as stated, +* We may use your contributions for our open source projects and for any other our project too, +* We rely on your assurances concerning the rights of third parties in relation to your contributions. + +If you agree with these principles, please read and adopt our CLA. By providing us your contributions, you hereby declare that you have read and adopted our CLA, and we may freely merge your contributions with our corresponding open source project and use it in further in accordance with terms and conditions of the CLA. + +### Provide contributions + +If you have adopted terms and conditions of the CLA, you are able to provide your contributions. When you submit your pull request, please add the following information into it: + +I hereby agree to the terms of the CLA available at: [link]. + +Replace the bracketed text as follows: + +* [link] is the link at the current version of the CLA (you may add here a link https://yandex.ru/legal/cla/?lang=en (in English) or a link https://yandex.ru/legal/cla/?lang=ru (in Russian). + +It is enough to provide us with such notification once. + +### Other questions +If you have any questions, please write us at opensource-support@yandex-team.ru. \ No newline at end of file diff --git a/LICENSE b/LICENSE new file mode 100644 index 0000000000000000000000000000000000000000..c64e81782a5b2b64f19d26147691f92403025234 --- /dev/null +++ b/LICENSE @@ -0,0 +1,13 @@ +Copyright 2026 YANDEX LLC + + Licensed under the Apache License, Version 2.0 (the "License"); + you may not use this file except in compliance with the License. + You may obtain a copy of the License at + + http://www.apache.org/licenses/LICENSE-2.0 + + Unless required by applicable law or agreed to in writing, software + distributed under the License is distributed on an "AS IS" BASIS, + WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + See the License for the specific language governing permissions and + limitations under the License. \ No newline at end of file diff --git a/NOTICES b/NOTICES new file mode 100644 index 0000000000000000000000000000000000000000..458b731d2b68b798aa4fa507606d66c9d5359949 --- /dev/null +++ b/NOTICES @@ -0,0 +1,8 @@ +------------------------------------------------------------------------------- +Export control notice +------------------------------------------------------------------------------- + +It is necessary to comply with the applicable export control laws and +regulations. We declare that we comply with the applicable export control laws +and regulations for the published software, and we expect the users of the +software and the contributors to it to be compliant with them as well. \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000000000000000000000000000000000000..138bcf9333a2082719b30eb7fa63a867e3ea8efe --- /dev/null +++ b/README.md @@ -0,0 +1,294 @@ +--- +license: apache-2.0 +language: + - ru + - en +library_name: transformers +pipeline_tag: text-generation +tags: + - custom_code + - mixture-of-experts + - vllm +--- + +# AliceAI-Foundation-80B-A3B-Base + +[English version](./README_en.md) + +AliceAI-Foundation-80B-A3B-Base – базовая языковая модель с гибридной +архитектурой и MoE-слоями. Модель содержит 80 млрд параметров, из которых для каждого токена активируются 3 млрд, и поддерживает контекст длиной до 262 144 токенов. Модель была обучена полностью с нуля. + +При создании модели мы заново собрали обучающий корпус, выбрали архитектуру и +гиперпараметры, а также подготовили данные для сложных рассуждений и +взаимодействия с инструментами. Ключевые решения проверялись в серии отдельных +обучений с нуля объёмом по 2 трлн токенов каждое. + +В математике, программировании и других задачах на рассуждения модель показывает результаты на уровне более крупных опенсорс-моделей, а особенно сильна в задачах на фактические знания на русском языке. Вместе с весами мы публикуем фактологические бенчмарки +[WikiWebFacts](https://huggingface.co/datasets/yandex/WikiWebFacts) и +[HardMultiQA](https://huggingface.co/datasets/yandex/HardMultiQA), ориентированные на +русскоязычный контекст, и протоколы их оценки. + +Сравнение по бенчмаркам + +## Обзор модели + +- Тип: авторегрессионная языковая модель +- Этап обучения: предобучение +- Языковая модель + - Количетсво параметров: 80B всего, 3B активных + - Рзамер скрытого состояния: 2048 + - Размер словаря: 129024 + - Количество слоев: 48 + - Схема слоёв: 12 × (3 × (KDA → MoE) → 1 × (Gated Attention → MoE)) + - KDA: + - Количество query-голов: 32 + - Количество KV-голов: 32 + - Размерность query-головы: 128 + - Размерность KV-головы: 128 + - Рамер ядра свертки: 4 + - Gated Attention: + - Количество query-голов: 16 + - Количество KV-голов: 2 + - Размерность query-головы: 256 + - MoE: + - Количество экспертов: 512 + - Top-K: 10 + 1 общий эксперт + - Промежуточная размерность эксперта: 512 + - MTP: 1 слой +- Длина контекста: 262144 + +## Бенчмарки + +
+

Названия русскоязычных бенчмарков выделены зелёным, англоязычных — синим.

+

Все замеры в этом разделе проведены во внутренней инфраструктуре замеров, инференс в фреймворке vllm с t=0 для всех моделей. Жирным выделен победитель в каждой строчке.

+ ++++ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
БенчмаркAliceAI-Foundation-80B-A3B-BaseQwen3.5-35B-A3B-BaseGLM-4.5-Air-Base (106B-A12B)Nemotron-3-Super-120B-A12B-BaseDeepSeek-V4-Flash-Base (284B-A13B)
Факты
WikiWebFacts
5-shot, бенчмарк на знание фактов на русском языке.
86.562.470.272.883.2
HardMultiQA
5-shot, бенчмарк на знание фактов на русском языке.
67.947.248.654.565.4
CultCat
4-shot, бенчмарк на знание культурных фактов. Подробнее — в статье на Хабре.
86.559.259.166.380.7
TriviaQA
5-shot, открытый бенчмарк на знание фактов на английском языке; вместо метрики Exact Match используется LLM-as-a-judge.
79.071.483.589.889.4
Образовательные бенчмарки
EduBench Russian
5-shot, бенчмарк образования, собранный на основе запросов к Алисе.
74.242.939.044.067.7
EduBench Literature
5-shot, бенчмарк образования, собранный на основе запросов к Алисе.
73.851.851.455.869.1
EduBench History
5-shot, бенчмарк образования, собранный на основе запросов к Алисе.
82.065.962.870.276.9
EduBench English
5-shot, бенчмарк образования, собранный на основе запросов к Алисе.
76.171.767.271.382.9
Экспертные знания
ExpertFactsQA Medicine
5-shot, бенчмарк на фактические знания, составленный профильными экспертами.
63.659.050.642.360.7
ExpertFactsQA Law
Сложный 5-shot, бенчмарк на фактические знания, составленный профильными экспертами.
49.627.922.524.340.5
Экзамены
EGE CoT
5-shot, бенчмарк из заданий части А ЕГЭ по различным предметам.
90.584.777.884.390.3
MMLU-Pro CoT
5-shot, открытый бенчмарк на знания и рассуждения по широкому набору предметов на английском языке.
66.863.258.469.966.5
SuperGPQA CoT
5-shot, открытый бенчмарк с вопросами, составленными экспертами из разных научных областей.
44.343.635.446.646.1
Математика
MATH-500
5-shot, бенчмарк с математическими задачами; используется LLM-as-a-judge и более длинные рассуждения в few-shot-примерах.
91.181.960.284.880.7
EduBench Math
5-shot, бенчмарк образования, собранный на основе запросов к Алисе
79.380.056.969.776.3
EduBench Math University
5-shot, бенчмарк образования, собранный на основе запросов к Алисе.
70.169.951.467.468.6
Код
BigCodeBench 1-shot pass@1
1-shot, наша реализация BigCodeBench с улучшенными тестами.
48.343.544.548.849.1
LiveCodeBench v5-6 CoT 1-shot pass@1
1-shot, открытый бенчмарк со сложными задачами по программированию, в которых требуется найти алгоритм и реализовать его в коде.
50.550.422.650.438.1
Длинный контекст
FinQA 128k
5-shot, длинная модификация опенсорсного FinQA, задачи на аналитику по финансовым отчётам.
74.173.535.571.774.1
LongMemEval 128k
5-shot, открытый бенчмарк с задачами на поиск и использование информации из длинной истории диалога.
64.655.650.664.868.0
+
+ +
+

Все замеры в этом разделе проведены во внутренней инфраструктуре замеров, инференс в фреймворке vllm с t=1 и штрафами за повторы (repetition_penalty=1, presence_penalty=1.5) для всех моделей. Жирным выделен победитель в каждой строчке.

+ ++++ + + + + + + + + + + + + + + +
БенчмаркAliceAI-Foundation-80B-A3B-BaseQwen3.5-35B-A3B-BaseNemotron-3-Super-120B-A12B-Base
Сложные рассуждения
AIME 2026 pass@32
0-shot, задачи American Invitational Mathematics Examination.
96.796.790.0
HMMT 2026 Feb pass@32
0-shot, задачи февральского математического турнира Harvard–MIT.
96.987.966.7
IMO Answerbench pass@8
0-shot, задачи Международной математической олимпиады.
88.784.564.5
CodeForces CPP pass@8
0-shot, соревновательные задачи Codeforces на C++.
68.973.756.6
LiveCodeBench v5-6 pass@1
0-shot, открытый бенчмарк с задачами на поиск и использование информации из длинной истории диалога.
60.451.934.7
LiveCodeBench v5-6 pass@8
0-shot, открытый бенчмарк с задачами на поиск и использование информации из длинной истории диалога.
82.982.159.8
+
+ + +## Как использовать + +### Transformers + +Модель можно запустить через Transformers. Референсная версия Transformers — 5.16.1. Для выполнения KDA-слоёв на GPU требуется `flash-linear-attention` с поддержкой KDA: + +```bash +pip install \ + transformers==5.16.1 \ + accelerate==1.14.0 \ + flash-linear-attention==0.5.0 +``` + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "yandex/AliceAI-Foundation-80B-A3B-Base" + +tokenizer = AutoTokenizer.from_pretrained(model_id, trust_remote_code=True) +model = AutoModelForCausalLM.from_pretrained( + model_id, + trust_remote_code=True, + dtype=torch.bfloat16, + device_map="auto", +) + +prompt = "Есть 256 монет с разным весом, за какое минимальное количество попарных взвешиваний можно найти вторую по весу монету?" +inputs = tokenizer(prompt, return_tensors="pt").to(model.device) +output_ids = model.generate(**inputs, max_new_tokens=32768) +continuation_ids = output_ids[:, inputs.input_ids.shape[1] :] +print(tokenizer.decode(continuation_ids[0], skip_special_tokens=True)) +``` + +### vLLM + +Также модель можно запустить через vLLM. Для запуска требуются Docker и NVIDIA Container Toolkit. + +```bash +docker run --name alice-vllm --pull=always --gpus '"device=0,1,2,3"' --ipc=host \ + -p 8001:8000 \ + yamlbrand/alice-ai-vllm:latest \ + yandex/AliceAI-Foundation-80B-A3B-Base \ + --tensor-parallel-size 4 \ + --max-model-len auto \ + --attention-backend FLASH_ATTN \ + --attention-config.flash_attn_version=2 \ + --speculative-config '{"method":"mtp","num_speculative_tokens":1}' +``` + +Повторный запуск после остановки, с сохранённым кешем: + +```bash +docker start -a alice-vllm +``` + +Чтобы использовать все GPU, замените `--gpus '"device=0,1,2,3"'` на `--gpus all` +и укажите соответствующее значение размера тензорного параллелизма. + +После запуска сервера отправьте запрос: + +```bash +curl http://127.0.0.1:8001/v1/completions \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "yandex/AliceAI-Foundation-80B-A3B-Base", + "prompt": "Есть 256 монет с разным весом, за какое минимальное количество попарных взвешиваний можно найти вторую по весу монету?", + "max_tokens": 32768, + "temperature": 0 + }' +``` + +### Токенизатор + +Токенизатор загружается как `LlamaTokenizer` из файла `tokenizer.model`. Модель +токенизации использует SentencePiece BPE. Маркеры `[COT_ENABLE]`, `[COT_START]`, `[COT_END]`, а также маркеры инструментов являются обычными атомарными токенами словаря, а не специальными токенами Hugging Face. + +В `tokenizer_config.json` для параметра `legacy` явно установлено значение +`false`, чтобы сохранить ожидаемую обработку пробелов. Не переопределяйте его +значением `true`. + +### Как дообучить под свои задачи + +#### Формат данных + +Для подготовки агентских данных, использованных при обучении модели, мы +использовали стандартный формат OpenAI Messages. В нём траектория задаётся +последовательностью сообщений с ролями `system`, `user`, `assistant`, `tool` и +`meta`, а определения доступных инструментов передаются отдельно в поле +`tools`. + +Перед токенизацией каждая такая траектория рендерилась с помощью +[`chat_template.jinja`](finetune/chat_template.jinja). Шаблон задаёт префиксы +ролей, представление reasoning-трейсов, описаний инструментов, вызовов функций +и результатов их выполнения. Именно в таком текстовом представлении эти данные +использовались при обучении модели. + +Для sft и RL рекомендуем хранить данные в формате OpenAI +Messages и рендерить их с помощью этого шаблона. Так формат новых данных будет +совпадать с форматом, который модель видела во время претрейна. + +Мы намеренно не указываем этот шаблон как `chat_template` в +`tokenizer_config.json`: Alice-AI-Foundation-80B-A3B-Base — базовая модель, поэтому у неё нет +единственного формата диалога, который должен автоматически применяться при +инференсе. Приведённый шаблон предназначен именно для подготовки данных к +дообучению. + +#### Пример LoRA-дообучения + +В репозитории есть минимальный пример PEFT-дообучения +[`finetune_lora.py`](finetune/finetune_lora.py). Он загружает зафиксированную ревизию +датасета `tatsu-lab/alpaca`, обучается только на ответах и сохраняет только +LoRA-адаптер. Для модели такого размера необходим FSDP2, пример ниже рассчитан +на четыре GPU с 80 GB памяти. + +```bash +pip install \ + transformers==5.16.1 \ + accelerate==1.14.0 \ + peft==0.20.0 \ + datasets==5.0.1 \ + flash-linear-attention==0.5.0 + +pip install flash-attn==2.8.1 --no-build-isolation + +CUDA_VISIBLE_DEVICES=0,1,2,3 \ +PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ +accelerate launch \ + --use_fsdp \ + --num_processes 4 \ + --num_machines 1 \ + --dynamo_backend no \ + --mixed_precision no \ + --fsdp_version 2 \ + --fsdp_reshard_after_forward true \ + --fsdp_auto_wrap_policy TRANSFORMER_BASED_WRAP \ + --fsdp_transformer_layer_cls_to_wrap AliceAIDecoderLayer \ + --fsdp_cpu_ram_efficient_loading true \ + --fsdp_sync_module_states true \ + --fsdp_state_dict_type SHARDED_STATE_DICT \ + finetune/finetune_lora.py \ + --model yandex/AliceAI-Foundation-80B-A3B-Base \ + --steps 100 \ + --sequence-length 512 \ + --output-dir alice-lora +``` + +`--mixed_precision no` здесь не означает FP32-модель: базовые веса загружаются +в BF16, а LoRA-параметры PEFT хранит в FP32. + +При RAM-efficient загрузке полные веса чекпоинта загружает только rank 0; +остальные процессы создают модель на meta device и получают свои шарды через +FSDP2. + +--- +Данная модель является предварительно обученной (pretrained) моделью и предоставляется в исходном виде, без дополнительного этапа post-training / alignment. + +Модель предназначена прежде всего для исследований, экспериментов и дальнейшей доработки. Она не является готовым решением для непосредственного использования в пользовательских продуктах и сервисах. + +Перед использованием модели в production-среде рекомендуется провести собственное тестирование и, исходя из сценария применения, реализовать необходимые этапы дообучения, настройки и контроля поведения модели. + +Пользователь самостоятельно определяет применимость модели для конкретного сценария и несет ответственность за ее интеграцию и использование. diff --git a/README_en.md b/README_en.md new file mode 100644 index 0000000000000000000000000000000000000000..8c3e26eca25bc6d7abc32d85bcf562cdea1af6e6 --- /dev/null +++ b/README_en.md @@ -0,0 +1,292 @@ +--- +license: apache-2.0 +language: + - ru + - en +library_name: transformers +pipeline_tag: text-generation +tags: + - custom_code + - mixture-of-experts + - vllm +--- + +# AliceAI-Foundation-80B-A3B-Base + +[Русская версия](./README.md) + +AliceAI-Foundation-80B-A3B-Base is a base language model with a hybrid +architecture and MoE layers. The model has 80 billion parameters, of which +3 billion are activated for each token, and supports a context length of up to +262,144 tokens. The model was trained entirely from scratch. + +To build the model, we assembled a new training corpus, selected the architecture +and hyperparameters, and prepared data for complex reasoning and tool use. We +validated key design decisions through a series of separate training runs from +scratch, each using 2 trillion tokens. + +On mathematics, coding, and other reasoning tasks, the model performs on par +with larger open-source models and is particularly strong on Russian factual +knowledge. Alongside the model weights, we release the factual benchmarks +[WikiWebFacts](https://huggingface.co/datasets/yandex/WikiWebFacts) and +[HardMultiQA](https://huggingface.co/datasets/yandex/HardMultiQA), which focus on +Russian-language contexts, together with their evaluation protocols. + +Benchmark comparison + +## Model Overview + +- Type: autoregressive language model +- Training stage: pre-training +- Language model + - Number of parameters: 80B total, 3B activated + - Hidden size: 2048 + - Vocabulary size: 129024 + - Number of layers: 48 + - Layer layout: 12 × (3 × (KDA → MoE) → 1 × (Gated Attention → MoE)) + - KDA: + - Number of query heads: 32 + - Number of KV heads: 32 + - Query head dimension: 128 + - KV head dimension: 128 + - Convolution kernel size: 4 + - Gated Attention: + - Number of query heads: 16 + - Number of KV heads: 2 + - Query head dimension: 256 + - MoE: + - Number of experts: 512 + - Top-K: 10 routed + 1 shared expert + - Expert intermediate size: 512 + - MTP: 1 layer +- Context length: 262144 + +## Benchmarks + +
+

Russian-language benchmark names are shown in green; English-language benchmark names are shown in blue.

+

All results in this section were obtained using our internal evaluation infrastructure, with inference performed in vLLM at t=0 for every model. The best result in each row is shown in bold.

+ ++++ + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
BenchmarkAliceAI-Foundation-80B-A3B-BaseQwen3.5-35B-A3B-BaseGLM-4.5-Air-Base (106B-A12B)Nemotron-3-Super-120B-A12B-BaseDeepSeek-V4-Flash-Base (284B-A13B)
Facts
WikiWebFacts
5-shot benchmark of factual knowledge in Russian.
86.562.470.272.883.2
HardMultiQA
5-shot benchmark of factual knowledge in Russian.
67.947.248.654.565.4
CultCat
4-shot benchmark of cultural knowledge. Read more in our article on Habr.
86.559.259.166.380.7
TriviaQA
5-shot open benchmark of factual knowledge in English; LLM-as-a-judge is used instead of Exact Match.
79.071.483.589.889.4
Educational benchmarks
EduBench Russian
5-shot education benchmark built from queries submitted to Alice.
74.242.939.044.067.7
EduBench Literature
5-shot education benchmark built from queries submitted to Alice.
73.851.851.455.869.1
EduBench History
5-shot education benchmark built from queries submitted to Alice.
82.065.962.870.276.9
EduBench English
5-shot education benchmark built from queries submitted to Alice.
76.171.767.271.382.9
Expert knowledge
ExpertFactsQA Medicine
5-shot factual-knowledge benchmark created by domain experts.
63.659.050.642.360.7
ExpertFactsQA Law
Challenging 5-shot factual-knowledge benchmark created by domain experts.
49.627.922.524.340.5
Exams
EGE CoT
5-shot benchmark based on multiple-choice Unified State Exam tasks across various subjects.
90.584.777.884.390.3
MMLU-Pro CoT
5-shot open benchmark of knowledge and reasoning across a broad range of subjects in English.
66.863.258.469.966.5
SuperGPQA CoT
5-shot open benchmark containing questions written by experts from different scientific fields.
44.343.635.446.646.1
Mathematics
MATH-500
5-shot benchmark of mathematical problems; it uses LLM-as-a-judge and longer reasoning traces in the few-shot examples.
91.181.960.284.880.7
EduBench Math
5-shot education benchmark built from queries submitted to Alice
79.380.056.969.776.3
EduBench Math University
5-shot education benchmark built from queries submitted to Alice.
70.169.951.467.468.6
Coding
BigCodeBench 1-shot pass@1
1-shot, our implementation of BigCodeBench with improved tests.
48.343.544.548.849.1
LiveCodeBench v5-6 CoT 1-shot pass@1
1-shot open benchmark of challenging programming problems that require finding an algorithm and implementing it in code.
50.550.422.650.438.1
Long context
FinQA 128k
5-shot long-context adaptation of the open-source FinQA benchmark, featuring financial-report analysis tasks.
74.173.535.571.774.1
LongMemEval 128k
5-shot open benchmark of finding and using information from long dialogue histories.
64.655.650.664.868.0
+
+ +
+

All results in this section were obtained using our internal evaluation infrastructure, with inference performed in vLLM at t=1 and repetition penalties (repetition_penalty=1, presence_penalty=1.5) for every model. The best result in each row is shown in bold.

+ ++++ + + + + + + + + + + + + + + +
BenchmarkAliceAI-Foundation-80B-A3B-BaseQwen3.5-35B-A3B-BaseNemotron-3-Super-120B-A12B-Base
Complex reasoning
AIME 2026 pass@32
0-shot problems from the American Invitational Mathematics Examination.
96.796.790.0
HMMT 2026 Feb pass@32
0-shot problems from the February Harvard–MIT Mathematics Tournament.
96.987.966.7
IMO Answerbench pass@8
0-shot problems from the International Mathematical Olympiad.
88.784.564.5
CodeForces CPP pass@8
0-shot competitive-programming problems from Codeforces in C++.
68.973.756.6
LiveCodeBench v5-6 pass@1
0-shot open benchmark of challenging programming problems.
60.451.934.7
LiveCodeBench v5-6 pass@8
0-shot open benchmark of challenging programming problems.
82.982.159.8
+
+ + +## Usage + +### Transformers + +The model can be run with Transformers. The reference Transformers version is +5.16.1. Running the KDA layers on GPU requires `flash-linear-attention` with +KDA support: + +```bash +pip install \ + transformers==5.16.1 \ + accelerate==1.14.0 \ + flash-linear-attention==0.5.0 +``` + +```python +import torch +from transformers import AutoModelForCausalLM, AutoTokenizer + +model_id = "yandex/AliceAI-Foundation-80B-A3B-Base" + +tokenizer = AutoTokenizer.from_pretrained(model_id, trust_remote_code=True) +model = AutoModelForCausalLM.from_pretrained( + model_id, + trust_remote_code=True, + dtype=torch.bfloat16, + device_map="auto", +) + +prompt = "There are 256 coins of different weights. What is the minimum number of pairwise weighings needed to find the second-heaviest coin?" +inputs = tokenizer(prompt, return_tensors="pt").to(model.device) +output_ids = model.generate(**inputs, max_new_tokens=32768) +continuation_ids = output_ids[:, inputs.input_ids.shape[1] :] +print(tokenizer.decode(continuation_ids[0], skip_special_tokens=True)) +``` + +### vLLM + +The model can also be run with vLLM. Docker and NVIDIA Container Toolkit are +required. + +```bash +docker run --name alice-vllm --pull=always --gpus '"device=0,1,2,3"' --ipc=host \ + -p 8001:8000 \ + yamlbrand/alice-ai-vllm:latest \ + yandex/AliceAI-Foundation-80B-A3B-Base \ + --tensor-parallel-size 4 \ + --max-model-len auto \ + --attention-backend FLASH_ATTN \ + --attention-config.flash_attn_version=2 \ + --speculative-config '{"method":"mtp","num_speculative_tokens":1}' +``` + +To restart the stopped container while preserving its cache: + +```bash +docker start -a alice-vllm +``` + +To use all available GPUs, replace `--gpus '"device=0,1,2,3"'` with +`--gpus all` and set the tensor-parallel size accordingly. + +Once the server is running, send a request: + +```bash +curl http://127.0.0.1:8001/v1/completions \ + -H 'Content-Type: application/json' \ + -d '{ + "model": "yandex/AliceAI-Foundation-80B-A3B-Base", + "prompt": "There are 256 coins of different weights. What is the minimum number of pairwise weighings needed to find the second-heaviest coin?", + "max_tokens": 32768, + "temperature": 0 + }' +``` + +### Tokenizer + +The tokenizer is loaded as `LlamaTokenizer` from `tokenizer.model` and uses +SentencePiece BPE. The `[COT_ENABLE]`, `[COT_START]`, and `[COT_END]` markers, +as well as the tool-use markers, are ordinary atomic vocabulary tokens rather +than Hugging Face special tokens. + +In `tokenizer_config.json`, `legacy` is explicitly set to `false` to preserve +the expected whitespace handling. Do not override it with `true`. + +### Fine-tuning for your tasks + +#### Data format + +To prepare the agentic data used during model training, we used the standard +OpenAI Messages format. A trajectory is represented as a sequence of messages +with the `system`, `user`, `assistant`, `tool`, and `meta` roles, while the +definitions of the available tools are passed separately in the `tools` field. + +Before tokenization, each trajectory was rendered with +[`chat_template.jinja`](finetune/chat_template.jinja). The template defines the +role prefixes and the representation of reasoning traces, tool descriptions, +function calls, and tool results. This is the textual representation in which +the model encountered these data during training. + +For sft and RL, we recommend storing data in the OpenAI +Messages format and rendering it with this template. This keeps the new data +consistent with the format seen by the model during pretraining. + +We intentionally do not set this template as `chat_template` in +`tokenizer_config.json`: Alice-AI-Foundation-80B-A3B-Base is a base model and therefore does +not have a single conversational format that should be applied automatically +during inference. The provided template is intended specifically for preparing +fine-tuning data. + +#### LoRA fine-tuning example + +The repository includes a minimal PEFT fine-tuning example, +[`finetune_lora.py`](finetune/finetune_lora.py). It loads a pinned revision of +the `tatsu-lab/alpaca` dataset, computes the training loss only on responses, +and saves only the LoRA adapter. A model of this size requires FSDP2; the +example below is designed for four GPUs with 80 GB of memory each. + +```bash +pip install \ + transformers==5.16.1 \ + accelerate==1.14.0 \ + peft==0.20.0 \ + datasets==5.0.1 \ + flash-linear-attention==0.5.0 + +pip install flash-attn==2.8.1 --no-build-isolation + +CUDA_VISIBLE_DEVICES=0,1,2,3 \ +PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \ +accelerate launch \ + --use_fsdp \ + --num_processes 4 \ + --num_machines 1 \ + --dynamo_backend no \ + --mixed_precision no \ + --fsdp_version 2 \ + --fsdp_reshard_after_forward true \ + --fsdp_auto_wrap_policy TRANSFORMER_BASED_WRAP \ + --fsdp_transformer_layer_cls_to_wrap AliceAIDecoderLayer \ + --fsdp_cpu_ram_efficient_loading true \ + --fsdp_sync_module_states true \ + --fsdp_state_dict_type SHARDED_STATE_DICT \ + finetune/finetune_lora.py \ + --model yandex/AliceAI-Foundation-80B-A3B-Base \ + --steps 100 \ + --sequence-length 512 \ + --output-dir alice-lora +``` + +Here, `--mixed_precision no` does not mean that the model uses FP32: the base +weights are loaded in BF16, while PEFT stores the LoRA parameters in FP32. + +With RAM-efficient loading, only rank 0 loads the full checkpoint weights. The +other processes construct the model on the meta device and receive their shards +through FSDP2. diff --git a/assets/benchmarks.png b/assets/benchmarks.png new file mode 100644 index 0000000000000000000000000000000000000000..242d95eda79fc61a209d07cd1e40170825d2faf5 --- /dev/null +++ b/assets/benchmarks.png @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ccf544ac6032cdbc1e15b15c8cc1f6dfd086690d356b37f916e9019022974e27 +size 125003 diff --git a/config.json b/config.json new file mode 100644 index 0000000000000000000000000000000000000000..a303954001398ff2ce4ab2ce99988f37ac45a3af --- /dev/null +++ b/config.json @@ -0,0 +1,58 @@ +{ + "architectures": [ + "AliceAIForCausalLM" + ], + "attention_dropout": 0.0, + "bos_token_id": 1, + "auto_map": { + "AutoConfig": "configuration_alice_ai.AliceAIConfig", + "AutoModelForCausalLM": "modeling_alice_ai.AliceAIForCausalLM" + }, + "block_attn_res_block_size": 4, + "head_dim": 256, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "kda_allow_negative_eigenvalues": false, + "layer_types": [ + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention", + "linear_attention", "linear_attention", "linear_attention", "full_attention" + ], + "eos_token_id": 2, + "linear_conv_kernel_dim": 4, + "linear_key_head_dim": 128, + "linear_num_key_heads": 32, + "linear_num_value_heads": 32, + "linear_value_head_dim": 128, + "max_position_embeddings": 262144, + "model_type": "alice_ai", + "moe_intermediate_size": 512, + "mtp_num_hidden_layers": 1, + "num_attention_heads": 16, + "number_of_conv_states": 3, + "num_experts": 512, + "num_experts_per_tok": 10, + "num_hidden_layers": 48, + "num_key_value_heads": 2, + "output_router_logits": false, + "partial_rotary_factor": 0.25, + "rms_norm_eps": 1e-6, + "rope_theta": 1000000.0, + "router_bias_correction": true, + "router_score_function": "sigmoid", + "shared_expert_intermediate_size": 512, + "tie_word_embeddings": false, + "transformers_version": "5.16.1", + "use_cache": true, + "vocab_size": 129024 +} diff --git a/configuration_alice_ai.py b/configuration_alice_ai.py new file mode 100644 index 0000000000000000000000000000000000000000..1abe631201c41950ce5ae7462db59ad6d567db96 --- /dev/null +++ b/configuration_alice_ai.py @@ -0,0 +1,109 @@ +from transformers.configuration_utils import PretrainedConfig + + +class AliceAIConfig(PretrainedConfig): + model_type = "alice_ai" + + def __init__( + self, + vocab_size: int = 129024, + hidden_size: int = 2048, + num_hidden_layers: int = 48, + num_attention_heads: int = 16, + num_key_value_heads: int = 2, + head_dim: int = 256, + linear_num_key_heads: int = 32, + linear_num_value_heads: int = 32, + linear_key_head_dim: int = 128, + linear_value_head_dim: int = 128, + linear_conv_kernel_dim: int = 4, + num_experts: int = 512, + num_experts_per_tok: int = 10, + moe_intermediate_size: int = 512, + shared_expert_intermediate_size: int = 512, + block_attn_res_block_size: int = 4, + router_score_function: str = "sigmoid", + router_bias_correction: bool = True, + kda_allow_negative_eigenvalues: bool = False, + max_position_embeddings: int = 262144, + rope_theta: float = 1_000_000.0, + partial_rotary_factor: float = 0.25, + rms_norm_eps: float = 1e-6, + hidden_act: str = "silu", + initializer_range: float = 0.02, + attention_dropout: float = 0.0, + use_cache: bool = True, + output_router_logits: bool = False, + layer_types: list[str] | None = None, + tie_word_embeddings: bool = False, + pad_token_id: int | None = None, + bos_token_id: int | None = None, + eos_token_id: int | list[int] | None = None, + **kwargs, + ) -> None: + if layer_types is None: + layer_types = [ + "full_attention" if (layer_idx + 1) % 4 == 0 else "linear_attention" + for layer_idx in range(num_hidden_layers) + ] + super().__init__( + pad_token_id=pad_token_id, + bos_token_id=bos_token_id, + eos_token_id=eos_token_id, + tie_word_embeddings=tie_word_embeddings, + **kwargs, + ) + self.vocab_size = vocab_size + self.hidden_size = hidden_size + self.num_hidden_layers = num_hidden_layers + self.num_attention_heads = num_attention_heads + self.num_key_value_heads = num_key_value_heads + self.head_dim = head_dim + self.linear_num_key_heads = linear_num_key_heads + self.linear_num_value_heads = linear_num_value_heads + self.linear_key_head_dim = linear_key_head_dim + self.linear_value_head_dim = linear_value_head_dim + self.linear_conv_kernel_dim = linear_conv_kernel_dim + self.num_experts = num_experts + self.num_experts_per_tok = num_experts_per_tok + self.moe_intermediate_size = moe_intermediate_size + self.shared_expert_intermediate_size = shared_expert_intermediate_size + self.block_attn_res_block_size = block_attn_res_block_size + self.router_score_function = router_score_function + self.router_bias_correction = router_bias_correction + self.kda_allow_negative_eigenvalues = kda_allow_negative_eigenvalues + self.max_position_embeddings = max_position_embeddings + self.rope_theta = rope_theta + self.partial_rotary_factor = partial_rotary_factor + self.rms_norm_eps = rms_norm_eps + self.hidden_act = hidden_act + self.initializer_range = initializer_range + self.attention_dropout = attention_dropout + self.use_cache = use_cache + self.output_router_logits = output_router_logits + self.layer_types = layer_types + self.number_of_conv_states = 3 + self._validate_fields() + + def _validate_fields(self) -> None: + if self.block_attn_res_block_size <= 0: + raise ValueError("block_attn_res_block_size must be positive") + if self.linear_conv_kernel_dim < 2: + raise ValueError("linear_conv_kernel_dim must be at least 2") + if self.router_score_function != "sigmoid": + raise ValueError("This architecture requires sigmoid routing") + if not 0 < self.num_experts_per_tok <= self.num_experts: + raise ValueError("num_experts_per_tok must be between 1 and num_experts") + if self.num_attention_heads % self.num_key_value_heads != 0: + raise ValueError( + "num_attention_heads must be divisible by num_key_value_heads" + ) + if self.linear_num_value_heads % self.linear_num_key_heads != 0: + raise ValueError( + "linear_num_value_heads must be divisible by linear_num_key_heads" + ) + if len(self.layer_types) != self.num_hidden_layers: + raise ValueError("layer_types must contain one entry per hidden layer") + unknown = set(self.layer_types) - {"linear_attention", "full_attention"} + if unknown: + raise ValueError(f"Unsupported layer types: {sorted(unknown)}") diff --git a/finetune/chat_template.jinja b/finetune/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..6efa912272419b8a32320797976a9dc1ad282259 --- /dev/null +++ b/finetune/chat_template.jinja @@ -0,0 +1,94 @@ +{%- set roles = {"assistant": "assistant:", "user": "user:", "system": "system:", "meta": "meta:"} -%} +{%- set TOOL_CALL_START = "[TOOL_CALL_START]" -%} +{%- set COT_START = "[COT_START]" -%} +{%- set COT_END = "[COT_END]" -%} +{%- set tools_prefix = "Тебе доступны следующие функции:" -%} +{%- set json_seps = (",", ":") -%} + +{%- macro js(x) -%} + {{- x | tojson(separators=json_seps) | replace("&", "\\u0026") | replace("<", "\\u003c") | replace(">", "\\u003e") | replace("'", "\\u0027") | replace("/", "\\/") -}} +{%- endmacro -%} + +{%- macro tool_definition(tool) -%} + {%- if tool.function is defined -%} + {%- set tool = tool.function -%} + {%- endif -%} + {{- 'function {"name":"' ~ tool.name ~ '",' -}} + {%- if tool.description is defined and tool.description -%} + {{- '"description":"' ~ tool.description ~ '",' -}} + {%- endif -%} + {{- '"parameters":' ~ js(tool.parameters) -}} + {%- if tool.return_parameters is defined and tool.return_parameters -%} + {{- ',"return_parameters":' ~ js(tool.return_parameters) -}} + {%- endif -%} + {%- if tool.few_shot_examples is defined and tool.few_shot_examples -%} + {{- ',"few_shot_examples":' ~ js(tool.few_shot_examples) -}} + {%- endif -%} + {{- "}" -}} +{%- endmacro -%} + +{%- macro render_tools(tools) -%} + {{- tools_prefix -}} + {%- for tool in tools -%} + {{- "\n" ~ tool_definition(tool) -}} + {%- endfor -%} +{%- endmacro -%} + +{%- macro render_tool_calls(calls) -%} + {%- for call in calls -%} + {%- set fn = call.function if call.function is defined else call -%} + {{- "\n" ~ TOOL_CALL_START ~ fn.name ~ "\n" -}} + {%- if fn.arguments is mapping -%} + {{- js(fn.arguments) -}} + {%- else -%} + {{- fn.arguments -}} + {%- endif -%} + {%- endfor -%} +{%- endmacro -%} + +{%- macro render_assistant(message) -%} + {%- if message.tool_calls is defined and message.tool_calls -%} + {%- set content = render_tool_calls(message.tool_calls) -%} + {%- elif message.content is defined and message.content is not none -%} + {%- set content = message.content -%} + {%- else -%} + {%- set content = "" -%} + {%- endif -%} + {{- roles.assistant -}} + {%- if message.reasoning_content is defined and message.reasoning_content -%} + {{- COT_START ~ message.reasoning_content ~ COT_END -}} + {%- endif -%} + {{- " " ~ content -}} +{%- endmacro -%} + +{%- macro render_message(message) -%} + {%- if message.role == "system" -%} + {{- roles.system ~ " " ~ message.content -}} + {%- elif message.role == "user" -%} + {{- roles.user ~ " " ~ message.content -}} + {%- elif message.role == "assistant" -%} + {{- render_assistant(message) -}} + {%- elif message.role == "tool" -%} + {{- "Tool " ~ message.name ~ ": " ~ (message.content | default("", true)) -}} + {%- elif message.role == "meta" -%} + {{- roles.meta ~ message.content -}} + {%- endif -%} +{%- endmacro -%} + +{{- bos_token | default("", true) -}} +{%- if tools is defined and tools -%} + {{- render_tools(tools) -}} +{%- endif -%} +{%- for message in messages -%} + {%- set sep = "\n\n " if (tools is defined and tools) or not loop.first else "" -%} + {%- set block = render_message(message) -%} + {%- if loop.last and not (add_generation_prompt | default(false)) -%} + {%- set block = block | trim -%} + {%- endif -%} + {%- if block -%} + {{- sep ~ block -}} + {%- endif -%} +{%- endfor -%} +{%- if add_generation_prompt | default(false) -%} + {{- "\n\n " ~ roles.assistant -}} +{%- endif -%} diff --git a/finetune/finetune_lora.py b/finetune/finetune_lora.py new file mode 100644 index 0000000000000000000000000000000000000000..080a41cea1eca4440716271166daa7dbb39eff38 --- /dev/null +++ b/finetune/finetune_lora.py @@ -0,0 +1,274 @@ +from __future__ import annotations + +import argparse +from pathlib import Path + +import torch +from accelerate import Accelerator, DistributedType +from datasets import load_dataset +from torch.distributed.tensor import DTensor +from torch.utils.data import DataLoader +from transformers import AutoConfig, AutoModelForCausalLM, AutoTokenizer + +from peft import LoraConfig, TaskType, get_peft_model, get_peft_model_state_dict + +DEFAULT_MODEL = Path(__file__).resolve().parent.parent +DATASET_NAME = "tatsu-lab/alpaca" +DATASET_REVISION = "dce01c9b08f87459cf36a430d809084718273017" +FSDP_VERSION = 2 +LORA_TARGET_MODULES = ["q_proj", "k_proj", "v_proj", "o_proj"] +MAX_RESPONSE_TOKENS = 128 +ROUTER_BUFFER_NAME = "e_score_correction_bias" + + +def parse_args() -> argparse.Namespace: + parser = argparse.ArgumentParser(description="LoRA fine-tuning for Alice AI") + parser.add_argument("--model", default=str(DEFAULT_MODEL)) + parser.add_argument("--output-dir", type=Path) + parser.add_argument("--steps", type=int, default=1) + parser.add_argument("--sequence-length", type=int, default=32) + parser.add_argument("--learning-rate", type=float, default=2e-4) + parser.add_argument("--lora-rank", type=int, default=8) + parser.add_argument("--seed", type=int, default=42) + return parser.parse_args() + + +def encode_alpaca_row(row, tokenizer, sequence_length: int) -> dict[str, list[int]]: + prompt = f"Instruction:\n{row['instruction']}" + if row["input"]: + prompt += f"\n\nInput:\n{row['input']}" + prompt += "\n\nResponse:\n" + + prompt_ids = tokenizer(prompt, add_special_tokens=False).input_ids + response_ids = tokenizer(row["output"], add_special_tokens=False).input_ids + response_budget = min(MAX_RESPONSE_TOKENS, max(1, sequence_length // 4)) + response_ids = response_ids[:response_budget] + prompt_ids = prompt_ids[: sequence_length - len(response_ids) - 2] + + input_ids = [ + tokenizer.bos_token_id, + *prompt_ids, + *response_ids, + tokenizer.eos_token_id, + ] + labels = [-100] * (len(prompt_ids) + 1) + [*response_ids, tokenizer.eos_token_id] + attention_mask = [1] * len(input_ids) + padding_length = sequence_length - len(input_ids) + input_ids.extend([tokenizer.pad_token_id] * padding_length) + labels.extend([-100] * padding_length) + attention_mask.extend([0] * padding_length) + return { + "input_ids": input_ids, + "attention_mask": attention_mask, + "labels": labels, + } + + +def build_dataloader(tokenizer, sequence_length: int) -> DataLoader: + if tokenizer.pad_token_id is None: + tokenizer.pad_token = tokenizer.eos_token + dataset = load_dataset( + DATASET_NAME, + revision=DATASET_REVISION, + split="train", + ) + dataset = dataset.map( + encode_alpaca_row, + fn_kwargs={"tokenizer": tokenizer, "sequence_length": sequence_length}, + remove_columns=dataset.column_names, + ) + return DataLoader(dataset.with_format("torch"), batch_size=1, shuffle=True) + + +def save_adapter(accelerator: Accelerator, model, tokenizer, output_dir: Path) -> None: + unwrapped_model = accelerator.unwrap_model(model) + adapter_state = get_peft_model_state_dict(unwrapped_model) + adapter_state = { + name: value.full_tensor().cpu() if isinstance(value, DTensor) else value.cpu() + for name, value in adapter_state.items() + } + if accelerator.is_main_process: + unwrapped_model.save_pretrained(output_dir, state_dict=adapter_state) + tokenizer.save_pretrained(output_dir) + accelerator.wait_for_everyone() + + +def restore_rotary_buffer( + accelerator: Accelerator, + model, + device: torch.device | None = None, +) -> None: + unwrapped_model = accelerator.unwrap_model(model) + config = unwrapped_model.config + base_model = ( + unwrapped_model.get_base_model() + if hasattr(unwrapped_model, "get_base_model") + else unwrapped_model + ) + rotary_embedding = base_model.model.rotary_emb + rotary_dim = int(config.head_dim * config.partial_rotary_factor) + inv_freq = 1.0 / ( + config.rope_theta + ** ( + torch.arange( + 0, + rotary_dim, + 2, + dtype=torch.float32, + device=device or rotary_embedding.inv_freq.device, + ) + / rotary_dim + ) + ) + rotary_embedding.inv_freq = inv_freq + + +def load_model(model_path: str, accelerator: Accelerator): + load_kwargs = { + "trust_remote_code": True, + "dtype": torch.bfloat16, + "attn_implementation": "flash_attention_2", + } + if ( + not accelerator.state.fsdp_plugin.cpu_ram_efficient_loading + or accelerator.is_main_process + ): + return AutoModelForCausalLM.from_pretrained(model_path, **load_kwargs) + + config = AutoConfig.from_pretrained(model_path, trust_remote_code=True) + # Transformers validates FlashAttention against a real device even though + # the attention backend does not affect the module layout. + config._attn_implementation = "eager" + previous_dtype = torch.get_default_dtype() + torch.set_default_dtype(load_kwargs["dtype"]) + try: + with torch.device("meta"): + model = AutoModelForCausalLM.from_config(config, trust_remote_code=True) + finally: + torch.set_default_dtype(previous_dtype) + model.config._attn_implementation = load_kwargs["attn_implementation"] + return model + + +def remove_router_buffers(model, is_main_process: bool) -> dict[str, torch.Tensor]: + router_buffers = {} + for module_name, module in model.named_modules(): + buffer = module._buffers.pop(ROUTER_BUFFER_NAME, None) + if buffer is None: + continue + router_buffers[module_name] = ( + buffer.detach().cpu() + if is_main_process + else torch.empty(buffer.shape, dtype=buffer.dtype) + ) + return router_buffers + + +def restore_router_buffers( + accelerator: Accelerator, + model, + router_buffers: dict[str, torch.Tensor], +) -> None: + unwrapped_model = accelerator.unwrap_model(model) + for module_name, buffer in router_buffers.items(): + device_buffer = buffer.to(accelerator.device) + torch.distributed.broadcast(device_buffer, src=0) + unwrapped_model.get_submodule(module_name).register_buffer( + ROUTER_BUFFER_NAME, + device_buffer, + ) + + +def materialize_trainable_parameters(model) -> None: + # Accelerate 1.14 remaps FSDP2 optimizer parameters by data_ptr(). All meta + # tensors use pointer 0, so give the small trainable LoRA tensors real storage. + for module in model.modules(): + for parameter_name, parameter in tuple(module.named_parameters(recurse=False)): + if parameter.requires_grad and parameter.is_meta: + module._parameters[parameter_name] = torch.nn.Parameter( + torch.empty_like(parameter, device="cpu"), + ) + + +def main() -> None: + args = parse_args() + if args.steps < 1: + raise ValueError("--steps must be positive") + if args.sequence_length <= 1: + raise ValueError("--sequence-length must be at least 2") + + accelerator = Accelerator() + if accelerator.distributed_type != DistributedType.FSDP: + raise RuntimeError("Launch this script with Accelerate FSDP") + if accelerator.state.fsdp_plugin.fsdp_version != FSDP_VERSION: + raise RuntimeError("This example requires FSDP2") + + torch.manual_seed(args.seed) + tokenizer = AutoTokenizer.from_pretrained(args.model, trust_remote_code=True) + with accelerator.main_process_first(): + dataloader = build_dataloader(tokenizer, args.sequence_length) + + lora_config = LoraConfig( + task_type=TaskType.CAUSAL_LM, + r=args.lora_rank, + lora_alpha=2 * args.lora_rank, + lora_dropout=0.0, + target_modules=LORA_TARGET_MODULES, + bias="none", + ) + + model = load_model(args.model, accelerator) + if accelerator.state.fsdp_plugin.cpu_ram_efficient_loading: + restore_rotary_buffer(accelerator, model, device=torch.device("cpu")) + model.config.use_cache = False + model = get_peft_model(model, lora_config) + if accelerator.state.fsdp_plugin.cpu_ram_efficient_loading: + materialize_trainable_parameters(model) + router_buffers = ( + remove_router_buffers(model, accelerator.is_main_process) + if accelerator.state.fsdp_plugin.cpu_ram_efficient_loading + else {} + ) + + model.gradient_checkpointing_enable( + gradient_checkpointing_kwargs={"use_reentrant": False} + ) + trainable = sum( + parameter.numel() for parameter in model.parameters() if parameter.requires_grad + ) + total = sum(parameter.numel() for parameter in model.parameters()) + accelerator.print(f"Trainable parameters: {trainable:,} / {total:,}") + + optimizer = torch.optim.AdamW( + (parameter for parameter in model.parameters() if parameter.requires_grad), + lr=args.learning_rate, + ) + model, optimizer, dataloader = accelerator.prepare(model, optimizer, dataloader) + restore_router_buffers(accelerator, model, router_buffers) + restore_rotary_buffer(accelerator, model) + + model.train() + data_iterator = iter(dataloader) + for step in range(args.steps): + try: + batch = next(data_iterator) + except StopIteration: + data_iterator = iter(dataloader) + batch = next(data_iterator) + + optimizer.zero_grad(set_to_none=True) + outputs = model(**batch, use_cache=False) + accelerator.backward(outputs.loss) + optimizer.step() + accelerator.print( + f"step={step + 1} loss={outputs.loss.detach().float().item():.6f}" + ) + + if args.output_dir is not None: + save_adapter(accelerator, model, tokenizer, args.output_dir) + accelerator.print("LoRA fine-tuning smoke test completed") + accelerator.end_training() + + +if __name__ == "__main__": + main() diff --git a/model-00001-of-00049.safetensors b/model-00001-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..65939f8bd7c4ad44929c78c52169005a7fda6dc3 --- /dev/null +++ b/model-00001-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc8752372a7af35f5c9385b32a24b15783307f1e56e11902bbca3944e57d68a0 +size 3907510776 diff --git a/model-00002-of-00049.safetensors b/model-00002-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7a14231b96c4f251fc52128aad07e32c9efde350 --- /dev/null +++ b/model-00002-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2cd4cf968c6ec6d8d8b6c03d6dcc6e93b181bf93b8934b7f9b9cf69fc1310ad1 +size 3300139664 diff --git a/model-00003-of-00049.safetensors b/model-00003-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9e43d8dca7553659d5444936fcf3a7f15e681907 --- /dev/null +++ b/model-00003-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa2bdd505ea6c76a1ba921bc457133f21e7e9520277dd61a9c3415c08d048205 +size 3284173088 diff --git a/model-00004-of-00049.safetensors b/model-00004-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4dd53d99a466f7ca4591c6026c04479dd3928256 --- /dev/null +++ b/model-00004-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ddd0ff34293459b64fdee66d7fea280ef373c1f0af45560194b254a64dc923e6 +size 3300139664 diff --git a/model-00005-of-00049.safetensors b/model-00005-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4b9b1a66443a9850ed5c0b7fda0013a68679fac7 --- /dev/null +++ b/model-00005-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:78e8cccae89bf63a7123f5a44c575edbcdbc47c049906b17887a01a3b5b85703 +size 3300139664 diff --git a/model-00006-of-00049.safetensors b/model-00006-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..587cf22afc29842ff7b9d0a78cb8564b38920c19 --- /dev/null +++ b/model-00006-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ab3cdf27eb28f20be5a1854b6ccd297e4c2089f819ec380a2386c2715cbf35da +size 3300139664 diff --git a/model-00007-of-00049.safetensors b/model-00007-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4bc1f8550bdecd47ef5e845aba29e103aca85bf7 --- /dev/null +++ b/model-00007-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ee49728c98e1008fb8302573a55cb42c1af8acd104a9deb6f817f1a223f6cf1 +size 3284173088 diff --git a/model-00008-of-00049.safetensors b/model-00008-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9dda0c481570273d7e31b21f6b3b04c5fc146576 --- /dev/null +++ b/model-00008-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9d039f8b7ab5870d74965ecfa6555f730dd03b6feda8cb4dc07f486a24e8638 +size 3300139664 diff --git a/model-00009-of-00049.safetensors b/model-00009-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a20158edbd027e0d970ebae56a40fe40fa96f736 --- /dev/null +++ b/model-00009-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8acb2ccc31d519bf5814e54d4731ea077f9b4993e86f111dddeabf2d2c633f3 +size 3300139664 diff --git a/model-00010-of-00049.safetensors b/model-00010-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..12c6ff37506e2c5c1adfc9ad4891a0c159aee9f3 --- /dev/null +++ b/model-00010-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ba4a9bab910a68d0ea6e43035bd7c9041baaf5003671df066cad950ae905738 +size 3300139576 diff --git a/model-00011-of-00049.safetensors b/model-00011-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9a3179151553b649dab5405de93a52a5e6b22623 --- /dev/null +++ b/model-00011-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53b2e775863ade93ee3da92005b09f53bc1b06d6348df76424178ce3ab42cf30 +size 3284173104 diff --git a/model-00012-of-00049.safetensors b/model-00012-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..912293046c969182c8b60328cbc075a35e1c8b03 --- /dev/null +++ b/model-00012-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc8d2867f1ca4c018caa0d0520e54b8b73ca1071ad2c883b82ad7c5be307731f +size 3300139696 diff --git a/model-00013-of-00049.safetensors b/model-00013-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..46c7153673c0484dbe63f74eeff56678b50bae83 --- /dev/null +++ b/model-00013-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6717e4cc976388adfb12dfc2e367c370b2c2d10bb779e7b9527ce8e8a4aaf161 +size 3300139696 diff --git a/model-00014-of-00049.safetensors b/model-00014-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6e0441ac8f00e00b6af772573e9a3cc4a7a40348 --- /dev/null +++ b/model-00014-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ab428fefdf329131479a8b43fb51878c74cd73f4732c2cd774773f506663afe +size 3300139696 diff --git a/model-00015-of-00049.safetensors b/model-00015-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9749d06d393b85afb269a42ae424051ad901fe30 --- /dev/null +++ b/model-00015-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7946bda09928d412a9dfecac35f85613306c43f677e3004eea5c322ec171bf97 +size 3284173104 diff --git a/model-00016-of-00049.safetensors b/model-00016-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4de23538b0d614dabff198274517c4b518ce42f3 --- /dev/null +++ b/model-00016-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4844a9f4d8d5ff7ae063d78e1f6b86ed219b0fe9542c3b6473fe853ad5ac1329 +size 3300139696 diff --git a/model-00017-of-00049.safetensors b/model-00017-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2f0aa092426b25bd8309f5c1fbb2c70ba4e371ee --- /dev/null +++ b/model-00017-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:acc60ddc1e922706d75b7583d003f235c2ff43cf656f31d3de0bfa86c5892cee +size 3300139696 diff --git a/model-00018-of-00049.safetensors b/model-00018-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..cfa79a4d4bd2f59de1f7608ab39bf77ceebf7745 --- /dev/null +++ b/model-00018-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d45cbc29dd70710361ca57a7f89ef6087e04f6fed1b25646bb7d7874ceebdf25 +size 3300139696 diff --git a/model-00019-of-00049.safetensors b/model-00019-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..24a91efa927a632480159996dcecae504f050123 --- /dev/null +++ b/model-00019-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ba1eda354b8b364b38b18460254d94304602693e4ad8cc634c48bff187a2ba39 +size 3284173104 diff --git a/model-00020-of-00049.safetensors b/model-00020-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..14cf082a54bef45dfe43ded2e659a0e1f36d1302 --- /dev/null +++ b/model-00020-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f133b92c9de031f997bfe04f8cf01ed49694ebf32dbb76cc633cf8186c9d172 +size 3300139696 diff --git a/model-00021-of-00049.safetensors b/model-00021-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..19e38de740824a452e312d81865f7ccad900e788 --- /dev/null +++ b/model-00021-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fae72d324834b897fe9973d5fdf94c9a81f2a594fb39b10c41c912e444ec2a69 +size 3300139696 diff --git a/model-00022-of-00049.safetensors b/model-00022-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..78b221c1ade3832212bac2412f1a8231dad0959f --- /dev/null +++ b/model-00022-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ba0f129fe6694303eb26385a70d628b0ee88be625ee34d04acecb130c8716e3 +size 3300139696 diff --git a/model-00023-of-00049.safetensors b/model-00023-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a1acd1f011ab75aa206cd8654d095ab231b7686d --- /dev/null +++ b/model-00023-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d9632bd7065a4f8c4ba884e28492e90aabbce5b9f1c3638ec5d80fd2ce71a35 +size 3284173104 diff --git a/model-00024-of-00049.safetensors b/model-00024-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ef455b8ccc3c84e7c0cb76f33eb0556afe1180a4 --- /dev/null +++ b/model-00024-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c937a38a72f9b1e87af22c5a80115b832cf1630485019a56c38bb60d42bcd85 +size 3300139696 diff --git a/model-00025-of-00049.safetensors b/model-00025-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4f187edddf4b7d67699c6f00dcb489793c19a1f6 --- /dev/null +++ b/model-00025-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8f61093583ce87a5b236a3beb5384ee41a363db3dcbbe3811aa4786bdf5cb072 +size 3300139696 diff --git a/model-00026-of-00049.safetensors b/model-00026-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9ade294ca7571a17b997ba2b74ce4e31aea9938e --- /dev/null +++ b/model-00026-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4225886e8dbf1e07a75228a2e455d00932c6f537de6f5541d12e7377df0d7872 +size 3300139696 diff --git a/model-00027-of-00049.safetensors b/model-00027-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..c6e402b63a40d3ed3c6baafec7e11a0c80b3f4e2 --- /dev/null +++ b/model-00027-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad7a806a91bcc936c35e9190bac39de33871ac11c65471cdc3082fd52e609819 +size 3284173104 diff --git a/model-00028-of-00049.safetensors b/model-00028-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..fdee564013affd2099e2a8b48cbf8a72b9ee4320 --- /dev/null +++ b/model-00028-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad364d878d318cc67b3e5d5a06405ee5f2065cf66ba908d823483295c35a62b7 +size 3300139696 diff --git a/model-00029-of-00049.safetensors b/model-00029-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..108ce497341a6360bdc7816a14949f73ea674df3 --- /dev/null +++ b/model-00029-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:24336b293f97861b7442bcdb4c2e9fa68b02554ae7afddccd5c48f9ec79c0a11 +size 3300139696 diff --git a/model-00030-of-00049.safetensors b/model-00030-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3157e42955e6aeef302e724f8d49143fa8fd4a08 --- /dev/null +++ b/model-00030-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d7e63e8ffba289606d19eee08c6ddbcc85fb0ed1c8b28f2aa90b3d278bfe49c7 +size 3300139696 diff --git a/model-00031-of-00049.safetensors b/model-00031-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..8774836d5fbb38a68f76524d1f66aac4ff37481e --- /dev/null +++ b/model-00031-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:34849f8d875cbc6f360c3ca2e657762963bd15fbefec8dea11b817bb646215bc +size 3284173104 diff --git a/model-00032-of-00049.safetensors b/model-00032-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..457e5f00a51ba54335eb37b3f89ae0e23793bd18 --- /dev/null +++ b/model-00032-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2388667584c9c5bc9c5ea52453ede54818b87e8036911f3ad68154b14a1fb7e6 +size 3300139696 diff --git a/model-00033-of-00049.safetensors b/model-00033-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7175b6f44f8e427f8eb6442884d1aacb162df032 --- /dev/null +++ b/model-00033-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15b8af443b13f3e3b9f7509885b262af0a12031b1e1372ac4516f759d4a9e919 +size 3300139696 diff --git a/model-00034-of-00049.safetensors b/model-00034-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..aaf0ef247d2044cf490bd932d52c877c17871dc0 --- /dev/null +++ b/model-00034-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:785f5e1c7395bd155f8bfaffa114a3a3da6ad3480bc2530a8d901c2016be202a +size 3300139696 diff --git a/model-00035-of-00049.safetensors b/model-00035-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3c001471e9ccc3f276903915d32a2c4bc391c58f --- /dev/null +++ b/model-00035-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f440c22516776bda8622abb8e62818413b8cbc7c30e43032c6cd858f04282bea +size 3284173104 diff --git a/model-00036-of-00049.safetensors b/model-00036-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..450189ef0e3161f1ef4dd1a0c3b61907af5421cc --- /dev/null +++ b/model-00036-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2c0b96f186169421ab809742b34cc2d6b36d90ab8a724a68293e5408be3f14a7 +size 3300139696 diff --git a/model-00037-of-00049.safetensors b/model-00037-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..dc1b4d142b4857d85188b6b80c20ca2a7259159b --- /dev/null +++ b/model-00037-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3554c046833ba5156a2ffe2806789929fe5926f2d3ada3edb4ad605239f72252 +size 3300139696 diff --git a/model-00038-of-00049.safetensors b/model-00038-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5f33be4fae670a1e16e2044b942606588f219a86 --- /dev/null +++ b/model-00038-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27d841a333d25b7f81e342b7d73f058f36fa88aafec9dd61cb5b4e4440942740 +size 3300139696 diff --git a/model-00039-of-00049.safetensors b/model-00039-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1f5ddd0f71309482b1da1394e64cbb2e28bca9ec --- /dev/null +++ b/model-00039-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d6a4d680c385d9366642b15346a2bf5fe41fc5e02b097b5d46953843e1b75be4 +size 3284173104 diff --git a/model-00040-of-00049.safetensors b/model-00040-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..92048982a74a3ce1fa2509096e025f5582df4668 --- /dev/null +++ b/model-00040-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71208df7f7472c5ed0f8aa1c3d33e11162cb1163d36840117bbbbd20e63bee1a +size 3300139696 diff --git a/model-00041-of-00049.safetensors b/model-00041-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ad0507319114096f50328a7076010b4de2a09e97 --- /dev/null +++ b/model-00041-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1a29b760d53e9fd1d4966a86b63f022687afcf97b22da088ab5c572a5a3591d4 +size 3300139696 diff --git a/model-00042-of-00049.safetensors b/model-00042-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..15e573c05cdf3c0f1571dfa1a72dbaae9a740bda --- /dev/null +++ b/model-00042-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f48cfd81e48c5aed3165bb5dacf75cdaaf23da7a9c90da1cd6f932761ff434ca +size 3300139696 diff --git a/model-00043-of-00049.safetensors b/model-00043-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..72144843e0d04bbb658b4c78c1f5223d1d69a60a --- /dev/null +++ b/model-00043-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9756a4fa5c60470d306c20a847f80bcff09c103f7de1267f107acb9edc543ef6 +size 3284173104 diff --git a/model-00044-of-00049.safetensors b/model-00044-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7aab19d39fe64847b88adbda4d3748c518f24061 --- /dev/null +++ b/model-00044-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:939c3a3a9e3c6d98576a07bfac7c4e7c3bc1eacd11974f12ead3f5b1623e5846 +size 3300139696 diff --git a/model-00045-of-00049.safetensors b/model-00045-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4810936bc860d4476c526d525b5e2eba66f73487 --- /dev/null +++ b/model-00045-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47cbf75ec6400d37cb0d9d17549ca2e719dd942b0cb1ff5913e6a298000a3299 +size 3300139696 diff --git a/model-00046-of-00049.safetensors b/model-00046-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..536bddcfde8777ee3026784607c65401045642f1 --- /dev/null +++ b/model-00046-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a486589f62e415e0f85e35fc089b578f5e27438ec6a66195f52acc0d7480fafd +size 3300139696 diff --git a/model-00047-of-00049.safetensors b/model-00047-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..8725f0c6d628ca94ab3316698a0888d982084e6e --- /dev/null +++ b/model-00047-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c8144d5737d35181ef709ea8a1a0f3156f6e21c22b01fb1bfd68f64ca8785496 +size 3284173104 diff --git a/model-00048-of-00049.safetensors b/model-00048-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d8499926abcd4a878719f04101d868bcdd433905 --- /dev/null +++ b/model-00048-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93adf046951009afe3610356e152fa1cd80da48f295cf2c24a838a4d54e18e1e +size 3812668080 diff --git a/model-00049-of-00049.safetensors b/model-00049-of-00049.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ba8a6bf4fd1b6ece1613481e10b6f72eeae9204b --- /dev/null +++ b/model-00049-of-00049.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:831494ed9b55ed01f1e4b69d8ad86d882ac71049cfa4d6f3c3b16abc5d913f6f +size 3238015632 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..74341ec7fd30f10466a950e354e17723716b5383 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,1314 @@ +{ + "metadata": { + "total_size": 162572866816 + }, + "weight_map": { + "model.embed_tokens.weight": "model-00001-of-00049.safetensors", + "model.layers.0.input_layernorm.weight": "model-00001-of-00049.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.q_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.k_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.v_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.q_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.k_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.v_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.f_a_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.f_b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.g_a_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.g_b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.o_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.a_log_bias": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.dt_bias": "model-00001-of-00049.safetensors", + "model.layers.0.linear_attn.o_norm.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.gate.e_score_correction_bias": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.gate.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.shared_expert_gate.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.shared_expert.up_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.shared_expert.gate_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.shared_expert.down_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.experts.gate_up_proj": "model-00001-of-00049.safetensors", + "model.layers.0.mlp.experts.down_proj": "model-00001-of-00049.safetensors", + "model.layers.0.mlp_res_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.0.mlp_res_norm_weight": "model-00001-of-00049.safetensors", + "model.layers.1.input_layernorm.weight": "model-00001-of-00049.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.q_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.k_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.v_conv1d.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.q_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.k_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.v_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.f_a_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.f_b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.g_a_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.g_b_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.o_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.a_log_bias": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.dt_bias": "model-00001-of-00049.safetensors", + "model.layers.1.linear_attn.o_norm.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.gate.e_score_correction_bias": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.gate.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.shared_expert_gate.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.shared_expert.up_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.shared_expert.gate_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.shared_expert.down_proj.weight": "model-00001-of-00049.safetensors", + "model.layers.1.mlp.experts.gate_up_proj": "model-00002-of-00049.safetensors", + "model.layers.1.mlp.experts.down_proj": "model-00002-of-00049.safetensors", + "model.layers.1.attn_res_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.1.attn_res_norm_weight": "model-00002-of-00049.safetensors", + "model.layers.1.mlp_res_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.1.mlp_res_norm_weight": "model-00002-of-00049.safetensors", + "model.layers.2.input_layernorm.weight": "model-00002-of-00049.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.q_conv1d.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.k_conv1d.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.v_conv1d.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.q_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.k_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.v_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.f_a_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.f_b_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.b_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.g_a_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.g_b_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.o_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.a_log_bias": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.dt_bias": "model-00002-of-00049.safetensors", + "model.layers.2.linear_attn.o_norm.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.gate.e_score_correction_bias": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.gate.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.shared_expert_gate.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.shared_expert.up_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.shared_expert.gate_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.shared_expert.down_proj.weight": "model-00002-of-00049.safetensors", + "model.layers.2.mlp.experts.gate_up_proj": "model-00003-of-00049.safetensors", + "model.layers.2.mlp.experts.down_proj": "model-00003-of-00049.safetensors", + "model.layers.2.attn_res_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.2.attn_res_norm_weight": "model-00003-of-00049.safetensors", + "model.layers.2.mlp_res_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.2.mlp_res_norm_weight": "model-00003-of-00049.safetensors", + "model.layers.3.input_layernorm.weight": "model-00003-of-00049.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.k_norm.weight": "model-00003-of-00049.safetensors", + "model.layers.3.self_attn.q_norm.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.gate.e_score_correction_bias": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.gate.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.shared_expert_gate.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.shared_expert.up_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.shared_expert.gate_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.shared_expert.down_proj.weight": "model-00003-of-00049.safetensors", + "model.layers.3.mlp.experts.gate_up_proj": "model-00004-of-00049.safetensors", + "model.layers.3.mlp.experts.down_proj": "model-00004-of-00049.safetensors", + "model.layers.3.attn_res_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.3.attn_res_norm_weight": "model-00004-of-00049.safetensors", + "model.layers.3.mlp_res_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.3.mlp_res_norm_weight": "model-00004-of-00049.safetensors", + "model.layers.4.input_layernorm.weight": "model-00004-of-00049.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.q_conv1d.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.k_conv1d.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.v_conv1d.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.q_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.k_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.v_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.f_a_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.f_b_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.b_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.g_a_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.g_b_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.o_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.a_log_bias": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.dt_bias": "model-00004-of-00049.safetensors", + "model.layers.4.linear_attn.o_norm.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.gate.e_score_correction_bias": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.gate.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.shared_expert_gate.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.shared_expert.up_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.shared_expert.gate_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.shared_expert.down_proj.weight": "model-00004-of-00049.safetensors", + "model.layers.4.mlp.experts.gate_up_proj": "model-00005-of-00049.safetensors", + "model.layers.4.mlp.experts.down_proj": "model-00005-of-00049.safetensors", + "model.layers.4.attn_res_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.4.attn_res_norm_weight": "model-00005-of-00049.safetensors", + "model.layers.4.mlp_res_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.4.mlp_res_norm_weight": "model-00005-of-00049.safetensors", + "model.layers.5.input_layernorm.weight": "model-00005-of-00049.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.q_conv1d.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.k_conv1d.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.v_conv1d.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.q_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.k_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.v_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.f_a_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.f_b_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.b_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.g_a_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.g_b_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.o_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.a_log_bias": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.dt_bias": "model-00005-of-00049.safetensors", + "model.layers.5.linear_attn.o_norm.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.gate.e_score_correction_bias": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.gate.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.shared_expert_gate.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.shared_expert.up_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.shared_expert.gate_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.shared_expert.down_proj.weight": "model-00005-of-00049.safetensors", + "model.layers.5.mlp.experts.gate_up_proj": "model-00006-of-00049.safetensors", + "model.layers.5.mlp.experts.down_proj": "model-00006-of-00049.safetensors", + "model.layers.5.attn_res_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.5.attn_res_norm_weight": "model-00006-of-00049.safetensors", + "model.layers.5.mlp_res_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.5.mlp_res_norm_weight": "model-00006-of-00049.safetensors", + "model.layers.6.input_layernorm.weight": "model-00006-of-00049.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.q_conv1d.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.k_conv1d.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.v_conv1d.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.q_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.k_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.v_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.f_a_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.f_b_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.b_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.g_a_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.g_b_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.o_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.a_log_bias": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.dt_bias": "model-00006-of-00049.safetensors", + "model.layers.6.linear_attn.o_norm.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.gate.e_score_correction_bias": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.gate.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.shared_expert_gate.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.shared_expert.up_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.shared_expert.gate_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.shared_expert.down_proj.weight": "model-00006-of-00049.safetensors", + "model.layers.6.mlp.experts.gate_up_proj": "model-00007-of-00049.safetensors", + "model.layers.6.mlp.experts.down_proj": "model-00007-of-00049.safetensors", + "model.layers.6.attn_res_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.6.attn_res_norm_weight": "model-00007-of-00049.safetensors", + "model.layers.6.mlp_res_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.6.mlp_res_norm_weight": "model-00007-of-00049.safetensors", + "model.layers.7.input_layernorm.weight": "model-00007-of-00049.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.k_norm.weight": "model-00007-of-00049.safetensors", + "model.layers.7.self_attn.q_norm.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.gate.e_score_correction_bias": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.gate.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.shared_expert_gate.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.shared_expert.up_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.shared_expert.gate_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.shared_expert.down_proj.weight": "model-00007-of-00049.safetensors", + "model.layers.7.mlp.experts.gate_up_proj": "model-00008-of-00049.safetensors", + "model.layers.7.mlp.experts.down_proj": "model-00008-of-00049.safetensors", + "model.layers.7.attn_res_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.7.attn_res_norm_weight": "model-00008-of-00049.safetensors", + "model.layers.7.mlp_res_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.7.mlp_res_norm_weight": "model-00008-of-00049.safetensors", + "model.layers.8.input_layernorm.weight": "model-00008-of-00049.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.q_conv1d.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.k_conv1d.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.v_conv1d.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.q_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.k_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.v_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.f_a_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.f_b_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.b_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.g_a_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.g_b_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.o_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.a_log_bias": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.dt_bias": "model-00008-of-00049.safetensors", + "model.layers.8.linear_attn.o_norm.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.gate.e_score_correction_bias": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.gate.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.shared_expert_gate.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.shared_expert.up_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.shared_expert.gate_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.shared_expert.down_proj.weight": "model-00008-of-00049.safetensors", + "model.layers.8.mlp.experts.gate_up_proj": "model-00009-of-00049.safetensors", + "model.layers.8.mlp.experts.down_proj": "model-00009-of-00049.safetensors", + "model.layers.8.attn_res_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.8.attn_res_norm_weight": "model-00009-of-00049.safetensors", + "model.layers.8.mlp_res_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.8.mlp_res_norm_weight": "model-00009-of-00049.safetensors", + "model.layers.9.input_layernorm.weight": "model-00009-of-00049.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.q_conv1d.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.k_conv1d.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.v_conv1d.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.q_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.k_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.v_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.f_a_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.f_b_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.b_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.g_a_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.g_b_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.o_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.a_log_bias": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.dt_bias": "model-00009-of-00049.safetensors", + "model.layers.9.linear_attn.o_norm.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.gate.e_score_correction_bias": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.gate.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.shared_expert_gate.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.shared_expert.up_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.shared_expert.gate_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.shared_expert.down_proj.weight": "model-00009-of-00049.safetensors", + "model.layers.9.mlp.experts.gate_up_proj": "model-00010-of-00049.safetensors", + "model.layers.9.mlp.experts.down_proj": "model-00010-of-00049.safetensors", + "model.layers.9.attn_res_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.9.attn_res_norm_weight": "model-00010-of-00049.safetensors", + "model.layers.9.mlp_res_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.9.mlp_res_norm_weight": "model-00010-of-00049.safetensors", + "model.layers.10.input_layernorm.weight": "model-00010-of-00049.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.q_conv1d.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.k_conv1d.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.v_conv1d.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.q_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.k_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.v_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.f_a_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.f_b_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.b_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.g_a_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.g_b_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.o_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.a_log_bias": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.dt_bias": "model-00010-of-00049.safetensors", + "model.layers.10.linear_attn.o_norm.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.gate.e_score_correction_bias": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.gate.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.shared_expert_gate.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.shared_expert.up_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.shared_expert.gate_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.shared_expert.down_proj.weight": "model-00010-of-00049.safetensors", + "model.layers.10.mlp.experts.gate_up_proj": "model-00011-of-00049.safetensors", + "model.layers.10.mlp.experts.down_proj": "model-00011-of-00049.safetensors", + "model.layers.10.attn_res_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.10.attn_res_norm_weight": "model-00011-of-00049.safetensors", + "model.layers.10.mlp_res_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.10.mlp_res_norm_weight": "model-00011-of-00049.safetensors", + "model.layers.11.input_layernorm.weight": "model-00011-of-00049.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.k_norm.weight": "model-00011-of-00049.safetensors", + "model.layers.11.self_attn.q_norm.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.gate.e_score_correction_bias": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.gate.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.shared_expert_gate.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.shared_expert.up_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.shared_expert.gate_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.shared_expert.down_proj.weight": "model-00011-of-00049.safetensors", + "model.layers.11.mlp.experts.gate_up_proj": "model-00012-of-00049.safetensors", + "model.layers.11.mlp.experts.down_proj": "model-00012-of-00049.safetensors", + "model.layers.11.attn_res_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.11.attn_res_norm_weight": "model-00012-of-00049.safetensors", + "model.layers.11.mlp_res_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.11.mlp_res_norm_weight": "model-00012-of-00049.safetensors", + "model.layers.12.input_layernorm.weight": "model-00012-of-00049.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.q_conv1d.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.k_conv1d.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.v_conv1d.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.q_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.k_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.v_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.f_a_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.f_b_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.b_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.g_a_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.g_b_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.o_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.a_log_bias": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.dt_bias": "model-00012-of-00049.safetensors", + "model.layers.12.linear_attn.o_norm.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.gate.e_score_correction_bias": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.gate.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.shared_expert_gate.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.shared_expert.up_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.shared_expert.gate_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.shared_expert.down_proj.weight": "model-00012-of-00049.safetensors", + "model.layers.12.mlp.experts.gate_up_proj": "model-00013-of-00049.safetensors", + "model.layers.12.mlp.experts.down_proj": "model-00013-of-00049.safetensors", + "model.layers.12.attn_res_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.12.attn_res_norm_weight": "model-00013-of-00049.safetensors", + "model.layers.12.mlp_res_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.12.mlp_res_norm_weight": "model-00013-of-00049.safetensors", + "model.layers.13.input_layernorm.weight": "model-00013-of-00049.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.q_conv1d.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.k_conv1d.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.v_conv1d.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.q_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.k_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.v_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.f_a_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.f_b_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.b_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.g_a_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.g_b_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.o_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.a_log_bias": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.dt_bias": "model-00013-of-00049.safetensors", + "model.layers.13.linear_attn.o_norm.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.gate.e_score_correction_bias": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.gate.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.shared_expert_gate.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.shared_expert.up_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.shared_expert.gate_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.shared_expert.down_proj.weight": "model-00013-of-00049.safetensors", + "model.layers.13.mlp.experts.gate_up_proj": "model-00014-of-00049.safetensors", + "model.layers.13.mlp.experts.down_proj": "model-00014-of-00049.safetensors", + "model.layers.13.attn_res_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.13.attn_res_norm_weight": "model-00014-of-00049.safetensors", + "model.layers.13.mlp_res_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.13.mlp_res_norm_weight": "model-00014-of-00049.safetensors", + "model.layers.14.input_layernorm.weight": "model-00014-of-00049.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.q_conv1d.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.k_conv1d.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.v_conv1d.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.q_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.k_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.v_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.f_a_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.f_b_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.b_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.g_a_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.g_b_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.o_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.a_log_bias": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.dt_bias": "model-00014-of-00049.safetensors", + "model.layers.14.linear_attn.o_norm.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.gate.e_score_correction_bias": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.gate.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.shared_expert_gate.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.shared_expert.up_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.shared_expert.gate_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.shared_expert.down_proj.weight": "model-00014-of-00049.safetensors", + "model.layers.14.mlp.experts.gate_up_proj": "model-00015-of-00049.safetensors", + "model.layers.14.mlp.experts.down_proj": "model-00015-of-00049.safetensors", + "model.layers.14.attn_res_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.14.attn_res_norm_weight": "model-00015-of-00049.safetensors", + "model.layers.14.mlp_res_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.14.mlp_res_norm_weight": "model-00015-of-00049.safetensors", + "model.layers.15.input_layernorm.weight": "model-00015-of-00049.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.k_norm.weight": "model-00015-of-00049.safetensors", + "model.layers.15.self_attn.q_norm.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.gate.e_score_correction_bias": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.gate.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.shared_expert_gate.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.shared_expert.up_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.shared_expert.gate_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.shared_expert.down_proj.weight": "model-00015-of-00049.safetensors", + "model.layers.15.mlp.experts.gate_up_proj": "model-00016-of-00049.safetensors", + "model.layers.15.mlp.experts.down_proj": "model-00016-of-00049.safetensors", + "model.layers.15.attn_res_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.15.attn_res_norm_weight": "model-00016-of-00049.safetensors", + "model.layers.15.mlp_res_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.15.mlp_res_norm_weight": "model-00016-of-00049.safetensors", + "model.layers.16.input_layernorm.weight": "model-00016-of-00049.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.q_conv1d.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.k_conv1d.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.v_conv1d.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.q_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.k_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.v_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.f_a_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.f_b_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.b_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.g_a_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.g_b_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.o_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.a_log_bias": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.dt_bias": "model-00016-of-00049.safetensors", + "model.layers.16.linear_attn.o_norm.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.gate.e_score_correction_bias": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.gate.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.shared_expert_gate.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.shared_expert.up_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.shared_expert.gate_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.shared_expert.down_proj.weight": "model-00016-of-00049.safetensors", + "model.layers.16.mlp.experts.gate_up_proj": "model-00017-of-00049.safetensors", + "model.layers.16.mlp.experts.down_proj": "model-00017-of-00049.safetensors", + "model.layers.16.attn_res_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.16.attn_res_norm_weight": "model-00017-of-00049.safetensors", + "model.layers.16.mlp_res_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.16.mlp_res_norm_weight": "model-00017-of-00049.safetensors", + "model.layers.17.input_layernorm.weight": "model-00017-of-00049.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.q_conv1d.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.k_conv1d.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.v_conv1d.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.q_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.k_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.v_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.f_a_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.f_b_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.b_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.g_a_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.g_b_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.o_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.a_log_bias": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.dt_bias": "model-00017-of-00049.safetensors", + "model.layers.17.linear_attn.o_norm.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.gate.e_score_correction_bias": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.gate.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.shared_expert_gate.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.shared_expert.up_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.shared_expert.gate_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.shared_expert.down_proj.weight": "model-00017-of-00049.safetensors", + "model.layers.17.mlp.experts.gate_up_proj": "model-00018-of-00049.safetensors", + "model.layers.17.mlp.experts.down_proj": "model-00018-of-00049.safetensors", + "model.layers.17.attn_res_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.17.attn_res_norm_weight": "model-00018-of-00049.safetensors", + "model.layers.17.mlp_res_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.17.mlp_res_norm_weight": "model-00018-of-00049.safetensors", + "model.layers.18.input_layernorm.weight": "model-00018-of-00049.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.q_conv1d.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.k_conv1d.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.v_conv1d.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.q_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.k_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.v_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.f_a_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.f_b_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.b_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.g_a_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.g_b_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.o_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.a_log_bias": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.dt_bias": "model-00018-of-00049.safetensors", + "model.layers.18.linear_attn.o_norm.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.gate.e_score_correction_bias": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.gate.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.shared_expert_gate.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.shared_expert.up_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.shared_expert.gate_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.shared_expert.down_proj.weight": "model-00018-of-00049.safetensors", + "model.layers.18.mlp.experts.gate_up_proj": "model-00019-of-00049.safetensors", + "model.layers.18.mlp.experts.down_proj": "model-00019-of-00049.safetensors", + "model.layers.18.attn_res_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.18.attn_res_norm_weight": "model-00019-of-00049.safetensors", + "model.layers.18.mlp_res_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.18.mlp_res_norm_weight": "model-00019-of-00049.safetensors", + "model.layers.19.input_layernorm.weight": "model-00019-of-00049.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.k_norm.weight": "model-00019-of-00049.safetensors", + "model.layers.19.self_attn.q_norm.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.gate.e_score_correction_bias": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.gate.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.shared_expert_gate.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.shared_expert.up_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.shared_expert.gate_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.shared_expert.down_proj.weight": "model-00019-of-00049.safetensors", + "model.layers.19.mlp.experts.gate_up_proj": "model-00020-of-00049.safetensors", + "model.layers.19.mlp.experts.down_proj": "model-00020-of-00049.safetensors", + "model.layers.19.attn_res_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.19.attn_res_norm_weight": "model-00020-of-00049.safetensors", + "model.layers.19.mlp_res_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.19.mlp_res_norm_weight": "model-00020-of-00049.safetensors", + "model.layers.20.input_layernorm.weight": "model-00020-of-00049.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.q_conv1d.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.k_conv1d.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.v_conv1d.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.q_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.k_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.v_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.f_a_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.f_b_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.b_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.g_a_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.g_b_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.o_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.a_log_bias": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.dt_bias": "model-00020-of-00049.safetensors", + "model.layers.20.linear_attn.o_norm.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.gate.e_score_correction_bias": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.gate.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.shared_expert_gate.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.shared_expert.up_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.shared_expert.gate_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.shared_expert.down_proj.weight": "model-00020-of-00049.safetensors", + "model.layers.20.mlp.experts.gate_up_proj": "model-00021-of-00049.safetensors", + "model.layers.20.mlp.experts.down_proj": "model-00021-of-00049.safetensors", + "model.layers.20.attn_res_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.20.attn_res_norm_weight": "model-00021-of-00049.safetensors", + "model.layers.20.mlp_res_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.20.mlp_res_norm_weight": "model-00021-of-00049.safetensors", + "model.layers.21.input_layernorm.weight": "model-00021-of-00049.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.q_conv1d.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.k_conv1d.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.v_conv1d.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.q_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.k_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.v_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.f_a_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.f_b_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.b_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.g_a_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.g_b_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.o_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.a_log_bias": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.dt_bias": "model-00021-of-00049.safetensors", + "model.layers.21.linear_attn.o_norm.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.gate.e_score_correction_bias": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.gate.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.shared_expert_gate.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.shared_expert.up_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.shared_expert.gate_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.shared_expert.down_proj.weight": "model-00021-of-00049.safetensors", + "model.layers.21.mlp.experts.gate_up_proj": "model-00022-of-00049.safetensors", + "model.layers.21.mlp.experts.down_proj": "model-00022-of-00049.safetensors", + "model.layers.21.attn_res_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.21.attn_res_norm_weight": "model-00022-of-00049.safetensors", + "model.layers.21.mlp_res_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.21.mlp_res_norm_weight": "model-00022-of-00049.safetensors", + "model.layers.22.input_layernorm.weight": "model-00022-of-00049.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.q_conv1d.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.k_conv1d.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.v_conv1d.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.q_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.k_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.v_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.f_a_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.f_b_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.b_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.g_a_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.g_b_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.o_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.a_log_bias": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.dt_bias": "model-00022-of-00049.safetensors", + "model.layers.22.linear_attn.o_norm.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.gate.e_score_correction_bias": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.gate.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.shared_expert_gate.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.shared_expert.up_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.shared_expert.gate_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.shared_expert.down_proj.weight": "model-00022-of-00049.safetensors", + "model.layers.22.mlp.experts.gate_up_proj": "model-00023-of-00049.safetensors", + "model.layers.22.mlp.experts.down_proj": "model-00023-of-00049.safetensors", + "model.layers.22.attn_res_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.22.attn_res_norm_weight": "model-00023-of-00049.safetensors", + "model.layers.22.mlp_res_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.22.mlp_res_norm_weight": "model-00023-of-00049.safetensors", + "model.layers.23.input_layernorm.weight": "model-00023-of-00049.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.k_norm.weight": "model-00023-of-00049.safetensors", + "model.layers.23.self_attn.q_norm.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.gate.e_score_correction_bias": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.gate.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.shared_expert_gate.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.shared_expert.up_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.shared_expert.gate_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.shared_expert.down_proj.weight": "model-00023-of-00049.safetensors", + "model.layers.23.mlp.experts.gate_up_proj": "model-00024-of-00049.safetensors", + "model.layers.23.mlp.experts.down_proj": "model-00024-of-00049.safetensors", + "model.layers.23.attn_res_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.23.attn_res_norm_weight": "model-00024-of-00049.safetensors", + "model.layers.23.mlp_res_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.23.mlp_res_norm_weight": "model-00024-of-00049.safetensors", + "model.layers.24.input_layernorm.weight": "model-00024-of-00049.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.q_conv1d.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.k_conv1d.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.v_conv1d.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.q_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.k_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.v_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.f_a_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.f_b_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.b_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.g_a_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.g_b_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.o_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.a_log_bias": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.dt_bias": "model-00024-of-00049.safetensors", + "model.layers.24.linear_attn.o_norm.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.gate.e_score_correction_bias": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.gate.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.shared_expert_gate.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.shared_expert.up_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.shared_expert.gate_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.shared_expert.down_proj.weight": "model-00024-of-00049.safetensors", + "model.layers.24.mlp.experts.gate_up_proj": "model-00025-of-00049.safetensors", + "model.layers.24.mlp.experts.down_proj": "model-00025-of-00049.safetensors", + "model.layers.24.attn_res_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.24.attn_res_norm_weight": "model-00025-of-00049.safetensors", + "model.layers.24.mlp_res_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.24.mlp_res_norm_weight": "model-00025-of-00049.safetensors", + "model.layers.25.input_layernorm.weight": "model-00025-of-00049.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.q_conv1d.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.k_conv1d.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.v_conv1d.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.q_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.k_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.v_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.f_a_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.f_b_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.b_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.g_a_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.g_b_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.o_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.a_log_bias": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.dt_bias": "model-00025-of-00049.safetensors", + "model.layers.25.linear_attn.o_norm.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.gate.e_score_correction_bias": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.gate.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.shared_expert_gate.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.shared_expert.up_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.shared_expert.gate_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.shared_expert.down_proj.weight": "model-00025-of-00049.safetensors", + "model.layers.25.mlp.experts.gate_up_proj": "model-00026-of-00049.safetensors", + "model.layers.25.mlp.experts.down_proj": "model-00026-of-00049.safetensors", + "model.layers.25.attn_res_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.25.attn_res_norm_weight": "model-00026-of-00049.safetensors", + "model.layers.25.mlp_res_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.25.mlp_res_norm_weight": "model-00026-of-00049.safetensors", + "model.layers.26.input_layernorm.weight": "model-00026-of-00049.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.q_conv1d.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.k_conv1d.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.v_conv1d.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.q_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.k_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.v_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.f_a_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.f_b_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.b_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.g_a_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.g_b_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.o_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.a_log_bias": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.dt_bias": "model-00026-of-00049.safetensors", + "model.layers.26.linear_attn.o_norm.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.gate.e_score_correction_bias": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.gate.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.shared_expert_gate.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.shared_expert.up_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.shared_expert.gate_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.shared_expert.down_proj.weight": "model-00026-of-00049.safetensors", + "model.layers.26.mlp.experts.gate_up_proj": "model-00027-of-00049.safetensors", + "model.layers.26.mlp.experts.down_proj": "model-00027-of-00049.safetensors", + "model.layers.26.attn_res_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.26.attn_res_norm_weight": "model-00027-of-00049.safetensors", + "model.layers.26.mlp_res_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.26.mlp_res_norm_weight": "model-00027-of-00049.safetensors", + "model.layers.27.input_layernorm.weight": "model-00027-of-00049.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.k_norm.weight": "model-00027-of-00049.safetensors", + "model.layers.27.self_attn.q_norm.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.gate.e_score_correction_bias": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.gate.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.shared_expert_gate.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.shared_expert.up_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.shared_expert.gate_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.shared_expert.down_proj.weight": "model-00027-of-00049.safetensors", + "model.layers.27.mlp.experts.gate_up_proj": "model-00028-of-00049.safetensors", + "model.layers.27.mlp.experts.down_proj": "model-00028-of-00049.safetensors", + "model.layers.27.attn_res_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.27.attn_res_norm_weight": "model-00028-of-00049.safetensors", + "model.layers.27.mlp_res_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.27.mlp_res_norm_weight": "model-00028-of-00049.safetensors", + "model.layers.28.input_layernorm.weight": "model-00028-of-00049.safetensors", + "model.layers.28.post_attention_layernorm.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.q_conv1d.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.k_conv1d.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.v_conv1d.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.q_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.k_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.v_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.f_a_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.f_b_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.b_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.g_a_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.g_b_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.o_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.a_log_bias": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.dt_bias": "model-00028-of-00049.safetensors", + "model.layers.28.linear_attn.o_norm.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.gate.e_score_correction_bias": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.gate.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.shared_expert_gate.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.shared_expert.up_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.shared_expert.gate_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.shared_expert.down_proj.weight": "model-00028-of-00049.safetensors", + "model.layers.28.mlp.experts.gate_up_proj": "model-00029-of-00049.safetensors", + "model.layers.28.mlp.experts.down_proj": "model-00029-of-00049.safetensors", + "model.layers.28.attn_res_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.28.attn_res_norm_weight": "model-00029-of-00049.safetensors", + "model.layers.28.mlp_res_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.28.mlp_res_norm_weight": "model-00029-of-00049.safetensors", + "model.layers.29.input_layernorm.weight": "model-00029-of-00049.safetensors", + "model.layers.29.post_attention_layernorm.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.q_conv1d.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.k_conv1d.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.v_conv1d.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.q_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.k_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.v_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.f_a_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.f_b_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.b_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.g_a_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.g_b_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.o_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.a_log_bias": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.dt_bias": "model-00029-of-00049.safetensors", + "model.layers.29.linear_attn.o_norm.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.gate.e_score_correction_bias": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.gate.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.shared_expert_gate.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.shared_expert.up_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.shared_expert.gate_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.shared_expert.down_proj.weight": "model-00029-of-00049.safetensors", + "model.layers.29.mlp.experts.gate_up_proj": "model-00030-of-00049.safetensors", + "model.layers.29.mlp.experts.down_proj": "model-00030-of-00049.safetensors", + "model.layers.29.attn_res_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.29.attn_res_norm_weight": "model-00030-of-00049.safetensors", + "model.layers.29.mlp_res_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.29.mlp_res_norm_weight": "model-00030-of-00049.safetensors", + "model.layers.30.input_layernorm.weight": "model-00030-of-00049.safetensors", + "model.layers.30.post_attention_layernorm.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.q_conv1d.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.k_conv1d.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.v_conv1d.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.q_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.k_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.v_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.f_a_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.f_b_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.b_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.g_a_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.g_b_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.o_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.a_log_bias": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.dt_bias": "model-00030-of-00049.safetensors", + "model.layers.30.linear_attn.o_norm.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.gate.e_score_correction_bias": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.gate.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.shared_expert_gate.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.shared_expert.up_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.shared_expert.gate_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.shared_expert.down_proj.weight": "model-00030-of-00049.safetensors", + "model.layers.30.mlp.experts.gate_up_proj": "model-00031-of-00049.safetensors", + "model.layers.30.mlp.experts.down_proj": "model-00031-of-00049.safetensors", + "model.layers.30.attn_res_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.30.attn_res_norm_weight": "model-00031-of-00049.safetensors", + "model.layers.30.mlp_res_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.30.mlp_res_norm_weight": "model-00031-of-00049.safetensors", + "model.layers.31.input_layernorm.weight": "model-00031-of-00049.safetensors", + "model.layers.31.post_attention_layernorm.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.k_norm.weight": "model-00031-of-00049.safetensors", + "model.layers.31.self_attn.q_norm.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.gate.e_score_correction_bias": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.gate.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.shared_expert_gate.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.shared_expert.up_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.shared_expert.gate_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.shared_expert.down_proj.weight": "model-00031-of-00049.safetensors", + "model.layers.31.mlp.experts.gate_up_proj": "model-00032-of-00049.safetensors", + "model.layers.31.mlp.experts.down_proj": "model-00032-of-00049.safetensors", + "model.layers.31.attn_res_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.31.attn_res_norm_weight": "model-00032-of-00049.safetensors", + "model.layers.31.mlp_res_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.31.mlp_res_norm_weight": "model-00032-of-00049.safetensors", + "model.layers.32.input_layernorm.weight": "model-00032-of-00049.safetensors", + "model.layers.32.post_attention_layernorm.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.q_conv1d.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.k_conv1d.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.v_conv1d.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.q_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.k_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.v_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.f_a_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.f_b_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.b_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.g_a_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.g_b_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.o_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.a_log_bias": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.dt_bias": "model-00032-of-00049.safetensors", + "model.layers.32.linear_attn.o_norm.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.gate.e_score_correction_bias": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.gate.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.shared_expert_gate.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.shared_expert.up_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.shared_expert.gate_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.shared_expert.down_proj.weight": "model-00032-of-00049.safetensors", + "model.layers.32.mlp.experts.gate_up_proj": "model-00033-of-00049.safetensors", + "model.layers.32.mlp.experts.down_proj": "model-00033-of-00049.safetensors", + "model.layers.32.attn_res_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.32.attn_res_norm_weight": "model-00033-of-00049.safetensors", + "model.layers.32.mlp_res_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.32.mlp_res_norm_weight": "model-00033-of-00049.safetensors", + "model.layers.33.input_layernorm.weight": "model-00033-of-00049.safetensors", + "model.layers.33.post_attention_layernorm.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.q_conv1d.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.k_conv1d.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.v_conv1d.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.q_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.k_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.v_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.f_a_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.f_b_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.b_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.g_a_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.g_b_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.o_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.a_log_bias": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.dt_bias": "model-00033-of-00049.safetensors", + "model.layers.33.linear_attn.o_norm.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.gate.e_score_correction_bias": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.gate.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.shared_expert_gate.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.shared_expert.up_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.shared_expert.gate_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.shared_expert.down_proj.weight": "model-00033-of-00049.safetensors", + "model.layers.33.mlp.experts.gate_up_proj": "model-00034-of-00049.safetensors", + "model.layers.33.mlp.experts.down_proj": "model-00034-of-00049.safetensors", + "model.layers.33.attn_res_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.33.attn_res_norm_weight": "model-00034-of-00049.safetensors", + "model.layers.33.mlp_res_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.33.mlp_res_norm_weight": "model-00034-of-00049.safetensors", + "model.layers.34.input_layernorm.weight": "model-00034-of-00049.safetensors", + "model.layers.34.post_attention_layernorm.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.q_conv1d.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.k_conv1d.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.v_conv1d.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.q_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.k_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.v_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.f_a_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.f_b_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.b_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.g_a_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.g_b_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.o_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.a_log_bias": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.dt_bias": "model-00034-of-00049.safetensors", + "model.layers.34.linear_attn.o_norm.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.gate.e_score_correction_bias": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.gate.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.shared_expert_gate.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.shared_expert.up_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.shared_expert.gate_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.shared_expert.down_proj.weight": "model-00034-of-00049.safetensors", + "model.layers.34.mlp.experts.gate_up_proj": "model-00035-of-00049.safetensors", + "model.layers.34.mlp.experts.down_proj": "model-00035-of-00049.safetensors", + "model.layers.34.attn_res_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.34.attn_res_norm_weight": "model-00035-of-00049.safetensors", + "model.layers.34.mlp_res_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.34.mlp_res_norm_weight": "model-00035-of-00049.safetensors", + "model.layers.35.input_layernorm.weight": "model-00035-of-00049.safetensors", + "model.layers.35.post_attention_layernorm.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.k_norm.weight": "model-00035-of-00049.safetensors", + "model.layers.35.self_attn.q_norm.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.gate.e_score_correction_bias": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.gate.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.shared_expert_gate.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.shared_expert.up_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.shared_expert.gate_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.shared_expert.down_proj.weight": "model-00035-of-00049.safetensors", + "model.layers.35.mlp.experts.gate_up_proj": "model-00036-of-00049.safetensors", + "model.layers.35.mlp.experts.down_proj": "model-00036-of-00049.safetensors", + "model.layers.35.attn_res_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.35.attn_res_norm_weight": "model-00036-of-00049.safetensors", + "model.layers.35.mlp_res_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.35.mlp_res_norm_weight": "model-00036-of-00049.safetensors", + "model.layers.36.input_layernorm.weight": "model-00036-of-00049.safetensors", + "model.layers.36.post_attention_layernorm.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.q_conv1d.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.k_conv1d.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.v_conv1d.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.q_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.k_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.v_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.f_a_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.f_b_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.b_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.g_a_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.g_b_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.o_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.a_log_bias": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.dt_bias": "model-00036-of-00049.safetensors", + "model.layers.36.linear_attn.o_norm.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.gate.e_score_correction_bias": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.gate.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.shared_expert_gate.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.shared_expert.up_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.shared_expert.gate_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.shared_expert.down_proj.weight": "model-00036-of-00049.safetensors", + "model.layers.36.mlp.experts.gate_up_proj": "model-00037-of-00049.safetensors", + "model.layers.36.mlp.experts.down_proj": "model-00037-of-00049.safetensors", + "model.layers.36.attn_res_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.36.attn_res_norm_weight": "model-00037-of-00049.safetensors", + "model.layers.36.mlp_res_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.36.mlp_res_norm_weight": "model-00037-of-00049.safetensors", + "model.layers.37.input_layernorm.weight": "model-00037-of-00049.safetensors", + "model.layers.37.post_attention_layernorm.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.q_conv1d.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.k_conv1d.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.v_conv1d.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.q_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.k_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.v_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.f_a_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.f_b_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.b_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.g_a_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.g_b_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.o_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.a_log_bias": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.dt_bias": "model-00037-of-00049.safetensors", + "model.layers.37.linear_attn.o_norm.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.gate.e_score_correction_bias": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.gate.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.shared_expert_gate.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.shared_expert.up_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.shared_expert.gate_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.shared_expert.down_proj.weight": "model-00037-of-00049.safetensors", + "model.layers.37.mlp.experts.gate_up_proj": "model-00038-of-00049.safetensors", + "model.layers.37.mlp.experts.down_proj": "model-00038-of-00049.safetensors", + "model.layers.37.attn_res_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.37.attn_res_norm_weight": "model-00038-of-00049.safetensors", + "model.layers.37.mlp_res_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.37.mlp_res_norm_weight": "model-00038-of-00049.safetensors", + "model.layers.38.input_layernorm.weight": "model-00038-of-00049.safetensors", + "model.layers.38.post_attention_layernorm.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.q_conv1d.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.k_conv1d.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.v_conv1d.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.q_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.k_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.v_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.f_a_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.f_b_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.b_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.g_a_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.g_b_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.o_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.a_log_bias": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.dt_bias": "model-00038-of-00049.safetensors", + "model.layers.38.linear_attn.o_norm.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.gate.e_score_correction_bias": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.gate.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.shared_expert_gate.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.shared_expert.up_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.shared_expert.gate_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.shared_expert.down_proj.weight": "model-00038-of-00049.safetensors", + "model.layers.38.mlp.experts.gate_up_proj": "model-00039-of-00049.safetensors", + "model.layers.38.mlp.experts.down_proj": "model-00039-of-00049.safetensors", + "model.layers.38.attn_res_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.38.attn_res_norm_weight": "model-00039-of-00049.safetensors", + "model.layers.38.mlp_res_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.38.mlp_res_norm_weight": "model-00039-of-00049.safetensors", + "model.layers.39.input_layernorm.weight": "model-00039-of-00049.safetensors", + "model.layers.39.post_attention_layernorm.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.q_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.k_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.v_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.o_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.k_norm.weight": "model-00039-of-00049.safetensors", + "model.layers.39.self_attn.q_norm.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.gate.e_score_correction_bias": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.gate.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.shared_expert_gate.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.shared_expert.up_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.shared_expert.gate_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.shared_expert.down_proj.weight": "model-00039-of-00049.safetensors", + "model.layers.39.mlp.experts.gate_up_proj": "model-00040-of-00049.safetensors", + "model.layers.39.mlp.experts.down_proj": "model-00040-of-00049.safetensors", + "model.layers.39.attn_res_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.39.attn_res_norm_weight": "model-00040-of-00049.safetensors", + "model.layers.39.mlp_res_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.39.mlp_res_norm_weight": "model-00040-of-00049.safetensors", + "model.layers.40.input_layernorm.weight": "model-00040-of-00049.safetensors", + "model.layers.40.post_attention_layernorm.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.q_conv1d.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.k_conv1d.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.v_conv1d.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.q_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.k_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.v_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.f_a_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.f_b_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.b_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.g_a_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.g_b_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.o_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.a_log_bias": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.dt_bias": "model-00040-of-00049.safetensors", + "model.layers.40.linear_attn.o_norm.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.gate.e_score_correction_bias": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.gate.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.shared_expert_gate.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.shared_expert.up_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.shared_expert.gate_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.shared_expert.down_proj.weight": "model-00040-of-00049.safetensors", + "model.layers.40.mlp.experts.gate_up_proj": "model-00041-of-00049.safetensors", + "model.layers.40.mlp.experts.down_proj": "model-00041-of-00049.safetensors", + "model.layers.40.attn_res_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.40.attn_res_norm_weight": "model-00041-of-00049.safetensors", + "model.layers.40.mlp_res_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.40.mlp_res_norm_weight": "model-00041-of-00049.safetensors", + "model.layers.41.input_layernorm.weight": "model-00041-of-00049.safetensors", + "model.layers.41.post_attention_layernorm.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.q_conv1d.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.k_conv1d.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.v_conv1d.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.q_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.k_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.v_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.f_a_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.f_b_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.b_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.g_a_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.g_b_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.o_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.a_log_bias": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.dt_bias": "model-00041-of-00049.safetensors", + "model.layers.41.linear_attn.o_norm.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.gate.e_score_correction_bias": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.gate.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.shared_expert_gate.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.shared_expert.up_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.shared_expert.gate_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.shared_expert.down_proj.weight": "model-00041-of-00049.safetensors", + "model.layers.41.mlp.experts.gate_up_proj": "model-00042-of-00049.safetensors", + "model.layers.41.mlp.experts.down_proj": "model-00042-of-00049.safetensors", + "model.layers.41.attn_res_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.41.attn_res_norm_weight": "model-00042-of-00049.safetensors", + "model.layers.41.mlp_res_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.41.mlp_res_norm_weight": "model-00042-of-00049.safetensors", + "model.layers.42.input_layernorm.weight": "model-00042-of-00049.safetensors", + "model.layers.42.post_attention_layernorm.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.q_conv1d.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.k_conv1d.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.v_conv1d.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.q_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.k_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.v_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.f_a_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.f_b_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.b_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.g_a_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.g_b_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.o_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.a_log_bias": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.dt_bias": "model-00042-of-00049.safetensors", + "model.layers.42.linear_attn.o_norm.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.gate.e_score_correction_bias": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.gate.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.shared_expert_gate.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.shared_expert.up_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.shared_expert.gate_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.shared_expert.down_proj.weight": "model-00042-of-00049.safetensors", + "model.layers.42.mlp.experts.gate_up_proj": "model-00043-of-00049.safetensors", + "model.layers.42.mlp.experts.down_proj": "model-00043-of-00049.safetensors", + "model.layers.42.attn_res_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.42.attn_res_norm_weight": "model-00043-of-00049.safetensors", + "model.layers.42.mlp_res_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.42.mlp_res_norm_weight": "model-00043-of-00049.safetensors", + "model.layers.43.input_layernorm.weight": "model-00043-of-00049.safetensors", + "model.layers.43.post_attention_layernorm.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.q_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.k_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.v_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.o_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.k_norm.weight": "model-00043-of-00049.safetensors", + "model.layers.43.self_attn.q_norm.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.gate.e_score_correction_bias": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.gate.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.shared_expert_gate.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.shared_expert.up_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.shared_expert.gate_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.shared_expert.down_proj.weight": "model-00043-of-00049.safetensors", + "model.layers.43.mlp.experts.gate_up_proj": "model-00044-of-00049.safetensors", + "model.layers.43.mlp.experts.down_proj": "model-00044-of-00049.safetensors", + "model.layers.43.attn_res_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.43.attn_res_norm_weight": "model-00044-of-00049.safetensors", + "model.layers.43.mlp_res_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.43.mlp_res_norm_weight": "model-00044-of-00049.safetensors", + "model.layers.44.input_layernorm.weight": "model-00044-of-00049.safetensors", + "model.layers.44.post_attention_layernorm.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.q_conv1d.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.k_conv1d.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.v_conv1d.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.q_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.k_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.v_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.f_a_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.f_b_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.b_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.g_a_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.g_b_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.o_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.a_log_bias": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.dt_bias": "model-00044-of-00049.safetensors", + "model.layers.44.linear_attn.o_norm.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.gate.e_score_correction_bias": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.gate.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.shared_expert_gate.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.shared_expert.up_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.shared_expert.gate_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.shared_expert.down_proj.weight": "model-00044-of-00049.safetensors", + "model.layers.44.mlp.experts.gate_up_proj": "model-00045-of-00049.safetensors", + "model.layers.44.mlp.experts.down_proj": "model-00045-of-00049.safetensors", + "model.layers.44.attn_res_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.44.attn_res_norm_weight": "model-00045-of-00049.safetensors", + "model.layers.44.mlp_res_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.44.mlp_res_norm_weight": "model-00045-of-00049.safetensors", + "model.layers.45.input_layernorm.weight": "model-00045-of-00049.safetensors", + "model.layers.45.post_attention_layernorm.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.q_conv1d.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.k_conv1d.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.v_conv1d.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.q_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.k_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.v_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.f_a_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.f_b_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.b_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.g_a_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.g_b_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.o_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.a_log_bias": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.dt_bias": "model-00045-of-00049.safetensors", + "model.layers.45.linear_attn.o_norm.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.gate.e_score_correction_bias": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.gate.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.shared_expert_gate.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.shared_expert.up_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.shared_expert.gate_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.shared_expert.down_proj.weight": "model-00045-of-00049.safetensors", + "model.layers.45.mlp.experts.gate_up_proj": "model-00046-of-00049.safetensors", + "model.layers.45.mlp.experts.down_proj": "model-00046-of-00049.safetensors", + "model.layers.45.attn_res_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.45.attn_res_norm_weight": "model-00046-of-00049.safetensors", + "model.layers.45.mlp_res_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.45.mlp_res_norm_weight": "model-00046-of-00049.safetensors", + "model.layers.46.input_layernorm.weight": "model-00046-of-00049.safetensors", + "model.layers.46.post_attention_layernorm.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.q_conv1d.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.k_conv1d.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.v_conv1d.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.q_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.k_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.v_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.f_a_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.f_b_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.b_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.g_a_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.g_b_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.o_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.a_log_bias": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.dt_bias": "model-00046-of-00049.safetensors", + "model.layers.46.linear_attn.o_norm.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.gate.e_score_correction_bias": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.gate.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.shared_expert_gate.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.shared_expert.up_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.shared_expert.gate_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.shared_expert.down_proj.weight": "model-00046-of-00049.safetensors", + "model.layers.46.mlp.experts.gate_up_proj": "model-00047-of-00049.safetensors", + "model.layers.46.mlp.experts.down_proj": "model-00047-of-00049.safetensors", + "model.layers.46.attn_res_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.46.attn_res_norm_weight": "model-00047-of-00049.safetensors", + "model.layers.46.mlp_res_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.46.mlp_res_norm_weight": "model-00047-of-00049.safetensors", + "model.layers.47.input_layernorm.weight": "model-00047-of-00049.safetensors", + "model.layers.47.post_attention_layernorm.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.q_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.k_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.v_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.o_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.k_norm.weight": "model-00047-of-00049.safetensors", + "model.layers.47.self_attn.q_norm.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.gate.e_score_correction_bias": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.gate.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.shared_expert_gate.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.shared_expert.up_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.shared_expert.gate_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.shared_expert.down_proj.weight": "model-00047-of-00049.safetensors", + "model.layers.47.mlp.experts.gate_up_proj": "model-00048-of-00049.safetensors", + "model.layers.47.mlp.experts.down_proj": "model-00048-of-00049.safetensors", + "model.layers.47.attn_res_proj.weight": "model-00048-of-00049.safetensors", + "model.layers.47.attn_res_norm_weight": "model-00048-of-00049.safetensors", + "model.layers.47.mlp_res_proj.weight": "model-00048-of-00049.safetensors", + "model.layers.47.mlp_res_norm_weight": "model-00048-of-00049.safetensors", + "model.norm.weight": "model-00048-of-00049.safetensors", + "model.attnres_final.res_proj.weight": "model-00048-of-00049.safetensors", + "model.attnres_final.res_norm_weight": "model-00048-of-00049.safetensors", + "lm_head.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.input_layernorm.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.post_attention_layernorm.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.q_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.k_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.v_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.o_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.k_norm.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.self_attn.q_norm.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.gate.e_score_correction_bias": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.gate.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.shared_expert_gate.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.shared_expert.up_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.shared_expert.gate_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.shared_expert.down_proj.weight": "model-00048-of-00049.safetensors", + "mtp.layers.0.mlp.experts.gate_up_proj": "model-00049-of-00049.safetensors", + "mtp.layers.0.mlp.experts.down_proj": "model-00049-of-00049.safetensors", + "mtp.fc.weight": "model-00049-of-00049.safetensors", + "mtp.pre_fc_norm_embedding.weight": "model-00049-of-00049.safetensors", + "mtp.pre_fc_norm_hidden.weight": "model-00049-of-00049.safetensors", + "mtp.norm.weight": "model-00049-of-00049.safetensors" + } +} \ No newline at end of file diff --git a/modeling_alice_ai.py b/modeling_alice_ai.py new file mode 100644 index 0000000000000000000000000000000000000000..0e2fa7749a959da08ffda337caeb0d2399ad97b2 --- /dev/null +++ b/modeling_alice_ai.py @@ -0,0 +1,876 @@ +from __future__ import annotations + +import math +from typing import Any, ClassVar + +import torch +import torch.nn.functional as functional +from torch import nn +from transformers.activations import ACT2FN +from transformers.cache_utils import Cache, DynamicCache +from transformers.generation import GenerationMixin +from transformers.modeling_layers import GradientCheckpointingLayer +from transformers.modeling_outputs import ( + MoeCausalLMOutputWithPast, + MoeModelOutputWithPast, +) +from transformers.modeling_utils import ALL_ATTENTION_FUNCTIONS, PreTrainedModel +from transformers.utils import can_return_tuple +from transformers.utils.generic import merge_with_config_defaults +from transformers.utils.output_capturing import capture_outputs + +from .configuration_alice_ai import AliceAIConfig + + +def _rms_norm( + hidden_states: torch.Tensor, + weight: torch.Tensor, + eps: float, + *, + zero_centered: bool = False, +) -> torch.Tensor: + dtype = hidden_states.dtype + normalized = hidden_states.float() * torch.rsqrt( + hidden_states.float().square().mean(dim=-1, keepdim=True) + eps + ) + scale = 1.0 + weight.float() if zero_centered else weight.float() + return (normalized * scale).to(dtype) + + +class AliceAIRMSNorm(nn.Module): + def __init__(self, hidden_size: int, eps: float, *, zero_centered: bool) -> None: + super().__init__() + initial_value = 0.0 if zero_centered else 1.0 + self.weight = nn.Parameter(torch.full((hidden_size,), initial_value)) + self.eps = eps + self.zero_centered = zero_centered + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + return _rms_norm( + hidden_states, + self.weight, + self.eps, + zero_centered=self.zero_centered, + ) + + +def _depth_softmax_mix( + sources: list[torch.Tensor], + query_weight: torch.Tensor, + norm_weight: torch.Tensor, + eps: float, +) -> torch.Tensor: + stacked = torch.stack(sources, dim=0) + keys = _rms_norm(stacked, norm_weight, eps) + scores = functional.linear(keys.float(), query_weight.float()).squeeze(-1) + weights = scores.softmax(dim=0).to(stacked.dtype).unsqueeze(-1) + return (stacked * weights).sum(dim=0) + + +class AliceAIFinalBlockAttnRes(nn.Module): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__() + self.res_proj = nn.Linear(config.hidden_size, 1, bias=False) + self.res_norm_weight = nn.Parameter(torch.ones(config.hidden_size)) + self.eps = config.rms_norm_eps + + def forward( + self, completed: list[torch.Tensor], partial: torch.Tensor | None = None + ) -> torch.Tensor: + sources = completed if partial is None else [*completed, partial] + return _depth_softmax_mix( + sources, self.res_proj.weight, self.res_norm_weight, self.eps + ) + + +class AliceAIRotaryEmbedding(nn.Module): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__() + rotary_dim = int(config.head_dim * config.partial_rotary_factor) + if rotary_dim % 2: + raise ValueError("The rotary dimension must be even") + inv_freq = 1.0 / ( + config.rope_theta + ** (torch.arange(0, rotary_dim, 2, dtype=torch.float32) / rotary_dim) + ) + self.register_buffer("inv_freq", inv_freq, persistent=False) + + def forward( + self, position_ids: torch.LongTensor, dtype: torch.dtype + ) -> tuple[torch.Tensor, torch.Tensor]: + frequencies = torch.einsum( + "bi,j->bij", position_ids.float(), self.inv_freq.float() + ) + embeddings = torch.cat((frequencies, frequencies), dim=-1) + return embeddings.cos().to(dtype), embeddings.sin().to(dtype) + + +def _rotate_half(hidden_states: torch.Tensor) -> torch.Tensor: + first, second = hidden_states.chunk(2, dim=-1) + return torch.cat((-second, first), dim=-1) + + +def _apply_rotary( + hidden_states: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor +) -> torch.Tensor: + rotary_dim = cos.shape[-1] + rotary, remainder = hidden_states[..., :rotary_dim], hidden_states[..., rotary_dim:] + cos = cos.unsqueeze(1) + sin = sin.unsqueeze(1) + rotary = rotary * cos + _rotate_half(rotary) * sin + return torch.cat((rotary, remainder), dim=-1) + + +def eager_attention_forward( + module: nn.Module, + query: torch.Tensor, + key: torch.Tensor, + value: torch.Tensor, + attention_mask: torch.Tensor | None, + dropout: float = 0.0, + scaling: float | None = None, + **_: Any, +) -> tuple[torch.Tensor, torch.Tensor]: + key = key.repeat_interleave(module.num_key_value_groups, dim=1) + value = value.repeat_interleave(module.num_key_value_groups, dim=1) + scores = torch.matmul(query.float(), key.float().transpose(-1, -2)) + scores = scores * (module.scaling if scaling is None else scaling) + if attention_mask is not None: + if attention_mask.dtype == torch.bool: + scores = scores.masked_fill(~attention_mask, torch.finfo(scores.dtype).min) + else: + scores = scores + attention_mask + probabilities = scores.softmax(dim=-1).to(query.dtype) + probabilities = functional.dropout( + probabilities, p=dropout, training=module.training + ) + output = torch.matmul(probabilities, value).transpose(1, 2).contiguous() + return output, probabilities + + +class AliceAIAttention(nn.Module): + def __init__(self, config: AliceAIConfig, layer_idx: int) -> None: + super().__init__() + self.config = config + self.layer_idx = layer_idx + self.num_heads = config.num_attention_heads + self.num_kv_heads = config.num_key_value_heads + self.num_key_value_groups = self.num_heads // self.num_kv_heads + self.head_dim = config.head_dim + self.scaling = self.head_dim**-0.5 + self.attention_dropout = config.attention_dropout + self.is_causal = True + self.q_proj = nn.Linear( + config.hidden_size, 2 * self.num_heads * self.head_dim, bias=False + ) + self.k_proj = nn.Linear( + config.hidden_size, self.num_kv_heads * self.head_dim, bias=False + ) + self.v_proj = nn.Linear( + config.hidden_size, self.num_kv_heads * self.head_dim, bias=False + ) + self.o_proj = nn.Linear( + self.num_heads * self.head_dim, config.hidden_size, bias=False + ) + self.q_norm = AliceAIRMSNorm( + self.head_dim, config.rms_norm_eps, zero_centered=True + ) + self.k_norm = AliceAIRMSNorm( + self.head_dim, config.rms_norm_eps, zero_centered=True + ) + + def _backend_mask( + self, + attention_mask: torch.Tensor | None, + cache_position: torch.LongTensor, + key_length: int, + implementation: str, + ) -> torch.Tensor | None: + if attention_mask is not None and attention_mask.ndim == 4: + return attention_mask[..., :key_length] + padding_mask = ( + attention_mask[..., :key_length].to(torch.bool) + if attention_mask is not None + else None + ) + if padding_mask is not None and bool(padding_mask.all()): + padding_mask = None + if implementation.startswith("flash_attention"): + return padding_mask + + query_length = cache_position.shape[0] + is_plain_prefill = query_length == key_length and cache_position[0] == 0 + is_single_token_decode = query_length == 1 + if ( + implementation == "sdpa" + and padding_mask is None + and (is_plain_prefill or is_single_token_decode) + ): + return None + + key_positions = torch.arange(key_length, device=cache_position.device) + causal_mask = key_positions.view(1, 1, 1, -1) <= cache_position.view( + 1, 1, -1, 1 + ) + if padding_mask is not None: + causal_mask = causal_mask & padding_mask[:, None, None, :] + return causal_mask + + def forward( + self, + hidden_states: torch.Tensor, + position_embeddings: tuple[torch.Tensor, torch.Tensor], + attention_mask: torch.Tensor | None, + cache: Cache | None, + cache_position: torch.LongTensor, + ) -> tuple[torch.Tensor, torch.Tensor | None]: + batch_size, sequence_length, _ = hidden_states.shape + query, output_gate = ( + self.q_proj(hidden_states) + .view(batch_size, sequence_length, self.num_heads, 2 * self.head_dim) + .chunk(2, dim=-1) + ) + output_gate = output_gate.reshape( + batch_size, sequence_length, self.num_heads * self.head_dim + ) + key = self.k_proj(hidden_states).view( + batch_size, sequence_length, self.num_kv_heads, self.head_dim + ) + value = self.v_proj(hidden_states).view( + batch_size, sequence_length, self.num_kv_heads, self.head_dim + ) + query = self.q_norm(query).transpose(1, 2) + key = self.k_norm(key).transpose(1, 2) + value = value.transpose(1, 2) + cos, sin = position_embeddings + query = _apply_rotary(query, cos, sin) + key = _apply_rotary(key, cos, sin) + if cache is not None: + key, value = cache.update(key, value, self.layer_idx) + + implementation = self.config._attn_implementation + backend_mask = self._backend_mask( + attention_mask, cache_position, key.shape[-2], implementation + ) + attention_interface = ALL_ATTENTION_FUNCTIONS.get_interface( + implementation, eager_attention_forward + ) + output, attention_weights = attention_interface( + self, + query, + key, + value, + backend_mask, + dropout=0.0 if not self.training else self.attention_dropout, + scaling=self.scaling, + is_causal=True, + ) + output = output.reshape( + batch_size, sequence_length, self.num_heads * self.head_dim + ) + output = output * torch.sigmoid(output_gate) + return self.o_proj(output), attention_weights + + +class AliceAIKDA(nn.Module): + def __init__(self, config: AliceAIConfig, layer_idx: int) -> None: + super().__init__() + self.config = config + self.layer_idx = layer_idx + self.num_k_heads = config.linear_num_key_heads + self.num_v_heads = config.linear_num_value_heads + self.head_k_dim = config.linear_key_head_dim + self.head_v_dim = config.linear_value_head_dim + self.key_dim = self.num_k_heads * self.head_k_dim + self.value_dim = self.num_v_heads * self.head_v_dim + self.conv_kernel_size = config.linear_conv_kernel_dim + self.act_fn = ACT2FN[config.hidden_act] + conv_kwargs = { + "kernel_size": self.conv_kernel_size, + "padding": self.conv_kernel_size - 1, + "bias": False, + } + self.q_conv1d = nn.Conv1d( + self.key_dim, self.key_dim, groups=self.key_dim, **conv_kwargs + ) + self.k_conv1d = nn.Conv1d( + self.key_dim, self.key_dim, groups=self.key_dim, **conv_kwargs + ) + self.v_conv1d = nn.Conv1d( + self.value_dim, self.value_dim, groups=self.value_dim, **conv_kwargs + ) + self.q_proj = nn.Linear(config.hidden_size, self.key_dim, bias=False) + self.k_proj = nn.Linear(config.hidden_size, self.key_dim, bias=False) + self.v_proj = nn.Linear(config.hidden_size, self.value_dim, bias=False) + self.f_a_proj = nn.Linear(config.hidden_size, self.head_v_dim, bias=False) + self.f_b_proj = nn.Linear(self.head_v_dim, self.key_dim, bias=False) + self.b_proj = nn.Linear(config.hidden_size, self.num_k_heads, bias=False) + self.g_a_proj = nn.Linear(config.hidden_size, self.head_v_dim, bias=False) + self.g_b_proj = nn.Linear(self.head_v_dim, self.value_dim, bias=False) + self.o_norm = AliceAIRMSNorm( + self.head_v_dim, config.rms_norm_eps, zero_centered=False + ) + self.o_proj = nn.Linear(self.value_dim, config.hidden_size, bias=False) + self.a_log_bias = nn.Parameter(torch.empty(self.num_k_heads)) + self.dt_bias = nn.Parameter(torch.empty(self.key_dim)) + + def _causal_conv( + self, + hidden_states: torch.Tensor, + convolution: nn.Conv1d, + cache: Cache | None, + state_idx: int, + ) -> torch.Tensor: + channels_first = hidden_states.transpose(1, 2) + sequence_length = channels_first.shape[-1] + if cache is not None: + channels_first = cache.update_conv_state( + channels_first, + self.layer_idx, + state_idx, + conv_kernel_size=self.conv_kernel_size, + ) + output = self.act_fn( + convolution(channels_first)[..., : channels_first.shape[-1]] + ) + return output[..., -sequence_length:].transpose(1, 2) + + def _torch_kda( + self, + query: torch.Tensor, + key: torch.Tensor, + value: torch.Tensor, + alpha: torch.Tensor, + beta: torch.Tensor, + initial_state: torch.Tensor | None, + ) -> tuple[torch.Tensor, torch.Tensor]: + query = functional.normalize(query.float(), dim=-1) * self.head_k_dim**-0.5 + key = functional.normalize(key.float(), dim=-1) + gate = -self.a_log_bias.float().exp().view( + 1, 1, self.num_k_heads, 1 + ) * functional.softplus( + alpha.float() + + self.dt_bias.float().view(1, 1, self.num_k_heads, self.head_k_dim) + ) + beta = beta.float().sigmoid() + if self.config.kda_allow_negative_eigenvalues: + beta = beta * 2.0 + state = ( + initial_state.float() + if initial_state is not None + else query.new_zeros( + query.shape[0], + self.num_v_heads, + self.head_k_dim, + self.head_v_dim, + ) + ) + repeat = self.num_v_heads // self.num_k_heads + if repeat > 1: + query = query.repeat_interleave(repeat, dim=2) + key = key.repeat_interleave(repeat, dim=2) + gate = gate.repeat_interleave(repeat, dim=2) + beta = beta.repeat_interleave(repeat, dim=2) + outputs = [] + for token_idx in range(query.shape[1]): + query_i = query[:, token_idx] + key_i = key[:, token_idx] + value_i = value[:, token_idx].float() + state = state * gate[:, token_idx].exp().unsqueeze(-1) + prediction = torch.einsum("bhk,bhkv->bhv", key_i, state) + delta = (value_i - prediction) * beta[:, token_idx].unsqueeze(-1) + state = state + torch.einsum("bhk,bhv->bhkv", key_i, delta) + outputs.append(torch.einsum("bhk,bhkv->bhv", query_i, state)) + return torch.stack(outputs, dim=1).to(value.dtype), state + + def _kda( + self, + query: torch.Tensor, + key: torch.Tensor, + value: torch.Tensor, + alpha: torch.Tensor, + beta: torch.Tensor, + initial_state: torch.Tensor | None, + use_cache: bool, + ) -> tuple[torch.Tensor, torch.Tensor | None]: + if query.is_cuda: + try: + from fla.ops.kda import chunk_kda, fused_recurrent_kda # noqa: PLC0415 + + kernel = ( + fused_recurrent_kda + if query.shape[1] == 1 and initial_state is not None + else chunk_kda + ) + kernel_beta = beta.float().sigmoid() + if self.config.kda_allow_negative_eigenvalues: + kernel_beta = kernel_beta * 2.0 + return kernel( + q=query, + k=key, + v=value, + g=alpha, + beta=kernel_beta, + A_log=self.a_log_bias.float(), + dt_bias=self.dt_bias.float(), + initial_state=( + initial_state.float() if initial_state is not None else None + ), + output_final_state=use_cache, + use_qk_l2norm_in_kernel=True, + use_gate_in_kernel=True, + ) + except ImportError as error: + raise ImportError( + "CUDA execution requires flash-linear-attention with KDA support; " + "install flash-linear-attention>=0.5.0" + ) from error + output, state = self._torch_kda(query, key, value, alpha, beta, initial_state) + return output, state if use_cache else None + + def forward( + self, + hidden_states: torch.Tensor, + cache: Cache | None, + attention_mask: torch.Tensor | None, + ) -> torch.Tensor: + if attention_mask is not None: + hidden_states = ( + hidden_states * attention_mask[..., -hidden_states.shape[1] :, None] + ) + query = self.q_proj(hidden_states) + key = self.k_proj(hidden_states) + value = self.v_proj(hidden_states) + output_gate = self.g_b_proj(self.g_a_proj(hidden_states)) + alpha = self.f_b_proj(self.f_a_proj(hidden_states)) + beta = self.b_proj(hidden_states) + has_previous_state = cache is not None and cache.has_previous_state( + self.layer_idx + ) + query = self._causal_conv(query, self.q_conv1d, cache, 0) + key = self._causal_conv(key, self.k_conv1d, cache, 1) + value = self._causal_conv(value, self.v_conv1d, cache, 2) + query = query.view(*query.shape[:2], self.num_k_heads, self.head_k_dim) + key = key.view(*key.shape[:2], self.num_k_heads, self.head_k_dim) + value = value.view(*value.shape[:2], self.num_v_heads, self.head_v_dim) + alpha = alpha.view(*alpha.shape[:2], self.num_k_heads, self.head_k_dim) + initial_state = ( + cache.layers[self.layer_idx].recurrent_states[0] + if has_previous_state + else None + ) + output, final_state = self._kda( + query, key, value, alpha, beta, initial_state, cache is not None + ) + if cache is not None: + cache.update_recurrent_state(final_state, self.layer_idx) + output = self.o_norm(output) + output = output * torch.sigmoid(output_gate.view_as(output)) + return self.o_proj(output.flatten(-2)) + + +class AliceAIMLP(nn.Module): + def __init__( + self, hidden_size: int, intermediate_size: int, hidden_act: str + ) -> None: + super().__init__() + self.gate_proj = nn.Linear(hidden_size, intermediate_size, bias=False) + self.up_proj = nn.Linear(hidden_size, intermediate_size, bias=False) + self.down_proj = nn.Linear(intermediate_size, hidden_size, bias=False) + self.act_fn = ACT2FN[hidden_act] + + def forward(self, hidden_states: torch.Tensor) -> torch.Tensor: + return self.down_proj( + self.act_fn(self.gate_proj(hidden_states)) * self.up_proj(hidden_states) + ) + + +class AliceAIExperts(nn.Module): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__() + self.num_experts = config.num_experts + self.act_fn = ACT2FN[config.hidden_act] + self.gate_up_proj = nn.Parameter( + torch.empty( + config.num_experts, + 2 * config.moe_intermediate_size, + config.hidden_size, + ) + ) + self.down_proj = nn.Parameter( + torch.empty( + config.num_experts, + config.hidden_size, + config.moe_intermediate_size, + ) + ) + + def forward( + self, + hidden_states: torch.Tensor, + expert_indices: torch.LongTensor, + expert_weights: torch.Tensor, + ) -> torch.Tensor: + output = torch.zeros_like(hidden_states) + for expert_idx in range(self.num_experts): + token_indices, slots = torch.where(expert_indices == expert_idx) + if token_indices.numel() == 0: + continue + expert_input = hidden_states[token_indices] + gate_up = functional.linear(expert_input, self.gate_up_proj[expert_idx]) + gate, up = gate_up.chunk(2, dim=-1) + expert_output = functional.linear( + self.act_fn(gate) * up, self.down_proj[expert_idx] + ) + expert_output = expert_output * expert_weights[token_indices, slots, None] + output.index_add_(0, token_indices, expert_output.to(output.dtype)) + return output + + +class AliceAISigmoidTopKRouter(nn.Module): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__() + self.weight = nn.Parameter(torch.empty(config.num_experts, config.hidden_size)) + self.top_k = config.num_experts_per_tok + if config.router_bias_correction: + self.register_buffer( + "e_score_correction_bias", torch.zeros(config.num_experts) + ) + else: + self.e_score_correction_bias = None + + def forward( + self, hidden_states: torch.Tensor + ) -> tuple[torch.Tensor, torch.Tensor, torch.LongTensor]: + router_logits = functional.linear(hidden_states.float(), self.weight.float()) + scores = router_logits.sigmoid() + selection_scores = ( + scores + if self.e_score_correction_bias is None + else scores + self.e_score_correction_bias + ) + indices = selection_scores.topk(self.top_k, dim=-1).indices + weights = scores.gather(-1, indices) + weights = weights / weights.sum(dim=-1, keepdim=True) + return router_logits, weights.to(hidden_states.dtype), indices + + +class AliceAISparseMoEBlock(nn.Module): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__() + self.gate = AliceAISigmoidTopKRouter(config) + self.experts = AliceAIExperts(config) + self.shared_expert = AliceAIMLP( + config.hidden_size, + config.shared_expert_intermediate_size, + config.hidden_act, + ) + self.shared_expert_gate = nn.Linear(config.hidden_size, 1, bias=False) + + def forward(self, hidden_states: torch.Tensor) -> tuple[torch.Tensor, torch.Tensor]: + original_shape = hidden_states.shape + flattened = hidden_states.reshape(-1, original_shape[-1]) + router_logits, weights, indices = self.gate(flattened) + routed = self.experts(flattened, indices, weights) + shared = self.shared_expert(flattened) * torch.sigmoid( + self.shared_expert_gate(flattened) + ) + return (routed + shared).reshape(original_shape), router_logits + + +class AliceAIDecoderLayer(GradientCheckpointingLayer): + def __init__(self, config: AliceAIConfig, layer_idx: int) -> None: + super().__init__() + self.layer_idx = layer_idx + self.layer_type = config.layer_types[layer_idx] + if self.layer_type == "linear_attention": + self.linear_attn = AliceAIKDA(config, layer_idx) + else: + self.self_attn = AliceAIAttention(config, layer_idx) + self.input_layernorm = AliceAIRMSNorm( + config.hidden_size, config.rms_norm_eps, zero_centered=True + ) + self.post_attention_layernorm = AliceAIRMSNorm( + config.hidden_size, config.rms_norm_eps, zero_centered=True + ) + self.mlp = AliceAISparseMoEBlock(config) + if layer_idx != 0: + self.attn_res_proj = nn.Linear(config.hidden_size, 1, bias=False) + self.attn_res_norm_weight = nn.Parameter(torch.ones(config.hidden_size)) + self.mlp_res_proj = nn.Linear(config.hidden_size, 1, bias=False) + self.mlp_res_norm_weight = nn.Parameter(torch.ones(config.hidden_size)) + self.eps = config.rms_norm_eps + + def _mix( + self, + completed: list[torch.Tensor], + partial: torch.Tensor | None, + projection: nn.Linear, + norm_weight: torch.Tensor, + ) -> torch.Tensor: + sources = completed if partial is None else [*completed, partial] + return _depth_softmax_mix(sources, projection.weight, norm_weight, self.eps) + + def forward( + self, + embeddings: torch.Tensor, + completed_blocks: tuple[torch.Tensor, ...], + partial: torch.Tensor | None, + position_embeddings: tuple[torch.Tensor, torch.Tensor], + attention_mask: torch.Tensor | None, + past_key_values: Cache | None, + cache_position: torch.LongTensor, + ) -> tuple[torch.Tensor, torch.Tensor]: + if self.layer_idx == 0: + mixed = embeddings + else: + mixed = self._mix( + completed_blocks, partial, self.attn_res_proj, self.attn_res_norm_weight + ) + mixed = self.input_layernorm(mixed) + if self.layer_type == "linear_attention": + attention_output = self.linear_attn( + mixed, past_key_values, attention_mask + ) + else: + attention_output, _ = self.self_attn( + mixed, + position_embeddings, + attention_mask, + past_key_values, + cache_position, + ) + partial = attention_output if partial is None else partial + attention_output + mixed = self._mix( + completed_blocks, partial, self.mlp_res_proj, self.mlp_res_norm_weight + ) + moe_output, router_logits = self.mlp(self.post_attention_layernorm(mixed)) + return partial + moe_output, router_logits + + +class AliceAIPreTrainedModel(PreTrainedModel): + config_class = AliceAIConfig + base_model_prefix = "model" + supports_gradient_checkpointing = True + _is_stateful = True + _no_split_modules: ClassVar[list[str]] = ["AliceAIDecoderLayer"] + _keys_to_ignore_on_load_unexpected: ClassVar[list[str]] = [r"^mtp\."] + _supports_sdpa = True + _supports_flash_attn = True + _can_record_outputs: ClassVar[dict[str, type[nn.Module]]] = { + "hidden_states": AliceAIDecoderLayer, + "attentions": AliceAIAttention, + } + + @staticmethod + def _needs_initialization(parameter: torch.Tensor) -> bool: + return not getattr(parameter, "_is_hf_initialized", False) + + @torch.no_grad() + def _init_weights(self, module: nn.Module) -> None: # noqa: PLR0912 + if isinstance(module, (nn.Linear, nn.Conv1d)): + module.weight.normal_(mean=0.0, std=self.config.initializer_range) + if module.bias is not None: + module.bias.zero_() + elif isinstance(module, nn.Embedding): + module.weight.normal_(mean=0.0, std=self.config.initializer_range) + if module.padding_idx is not None: + module.weight[module.padding_idx].zero_() + elif isinstance(module, AliceAIExperts): + if self._needs_initialization(module.gate_up_proj): + module.gate_up_proj.normal_(mean=0.0, std=self.config.initializer_range) + if self._needs_initialization(module.down_proj): + module.down_proj.normal_(mean=0.0, std=self.config.initializer_range) + elif isinstance(module, AliceAISigmoidTopKRouter): + if self._needs_initialization(module.weight): + module.weight.normal_(mean=0.0, std=self.config.initializer_range) + elif isinstance(module, AliceAIKDA): + if self._needs_initialization(module.a_log_bias): + module.a_log_bias.uniform_(0, 16).log_() + if self._needs_initialization(module.dt_bias): + dt = torch.exp( + torch.rand_like(module.dt_bias) * (math.log(0.1) - math.log(0.001)) + + math.log(0.001) + ).clamp_min(1e-4) + module.dt_bias.copy_(dt + torch.log(-torch.expm1(-dt))) + elif isinstance(module, AliceAIFinalBlockAttnRes): + if self._needs_initialization(module.res_proj.weight): + module.res_proj.weight.zero_() + elif isinstance(module, AliceAIDecoderLayer): + if hasattr(module, "attn_res_proj"): + if self._needs_initialization(module.attn_res_proj.weight): + module.attn_res_proj.weight.zero_() + if self._needs_initialization(module.mlp_res_proj.weight): + module.mlp_res_proj.weight.zero_() + + +class AliceAIModel(AliceAIPreTrainedModel): + def __init__(self, config: AliceAIConfig) -> None: + super().__init__(config) + self.embed_tokens = nn.Embedding( + config.vocab_size, config.hidden_size, config.pad_token_id + ) + self.layers = nn.ModuleList( + [ + AliceAIDecoderLayer(config, index) + for index in range(config.num_hidden_layers) + ] + ) + self.attnres_final = AliceAIFinalBlockAttnRes(config) + self.norm = AliceAIRMSNorm( + config.hidden_size, config.rms_norm_eps, zero_centered=True + ) + self.rotary_emb = AliceAIRotaryEmbedding(config) + self.gradient_checkpointing = False + self.post_init() + + @merge_with_config_defaults + @capture_outputs + def forward( + self, + input_ids: torch.LongTensor | None = None, + attention_mask: torch.Tensor | None = None, + position_ids: torch.LongTensor | None = None, + past_key_values: Cache | None = None, + inputs_embeds: torch.FloatTensor | None = None, + use_cache: bool | None = None, + output_router_logits: bool | None = None, + **_: Any, + ) -> MoeModelOutputWithPast: + if (input_ids is None) == (inputs_embeds is None): + raise ValueError("Specify exactly one of input_ids or inputs_embeds") + if inputs_embeds is None: + inputs_embeds = self.embed_tokens(input_ids) + use_cache = self.config.use_cache if use_cache is None else use_cache + output_router_logits = ( + self.config.output_router_logits + if output_router_logits is None + else output_router_logits + ) + if use_cache and past_key_values is None: + past_key_values = DynamicCache(config=self.config) + past_length = ( + past_key_values.get_seq_length() if past_key_values is not None else 0 + ) + cache_position = torch.arange( + past_length, + past_length + inputs_embeds.shape[1], + device=inputs_embeds.device, + ) + if position_ids is None: + position_ids = cache_position.unsqueeze(0) + position_embeddings = self.rotary_emb(position_ids, inputs_embeds.dtype) + if isinstance(attention_mask, dict): + linear_mask = attention_mask["linear_attention"] + full_attention_mask = attention_mask["full_attention"] + else: + linear_mask = attention_mask + full_attention_mask = attention_mask + if cache_position[0] > 0 or ( + attention_mask is not None and torch.all(attention_mask == 1) + ): + linear_mask = None + + completed_blocks = [inputs_embeds] + partial = None + router_logits = [] + for layer_idx, layer in enumerate(self.layers): + if layer_idx > 0 and layer_idx % self.config.block_attn_res_block_size == 0: + completed_blocks.append(partial) + partial = None + layer_mask = ( + linear_mask + if layer.layer_type == "linear_attention" + else full_attention_mask + ) + partial, layer_router_logits = layer( + inputs_embeds, + completed_blocks=tuple(completed_blocks), + partial=partial, + position_embeddings=position_embeddings, + attention_mask=layer_mask, + past_key_values=past_key_values, + cache_position=cache_position, + ) + if output_router_logits: + router_logits.append(layer_router_logits) + hidden_states = self.norm(self.attnres_final(completed_blocks, partial)) + return MoeModelOutputWithPast( + last_hidden_state=hidden_states, + past_key_values=past_key_values, + router_logits=tuple(router_logits) if output_router_logits else None, + ) + + +class AliceAIForCausalLM(AliceAIPreTrainedModel, GenerationMixin): + _tied_weights_keys: ClassVar[dict[str, str]] = { + "lm_head.weight": "model.embed_tokens.weight" + } + + def __init__(self, config: AliceAIConfig) -> None: + super().__init__(config) + self.model = AliceAIModel(config) + self.lm_head = nn.Linear(config.hidden_size, config.vocab_size, bias=False) + self.vocab_size = config.vocab_size + self.post_init() + + def get_input_embeddings(self) -> nn.Module: + return self.model.embed_tokens + + def set_input_embeddings(self, value: nn.Module) -> None: + self.model.embed_tokens = value + + def get_output_embeddings(self) -> nn.Module: + return self.lm_head + + def set_output_embeddings(self, value: nn.Module) -> None: + self.lm_head = value + + @can_return_tuple + def forward( + self, + input_ids: torch.LongTensor | None = None, + attention_mask: torch.Tensor | None = None, + position_ids: torch.LongTensor | None = None, + past_key_values: Cache | None = None, + inputs_embeds: torch.FloatTensor | None = None, + labels: torch.LongTensor | None = None, + use_cache: bool | None = None, + output_router_logits: bool | None = None, + logits_to_keep: int | torch.Tensor = 0, + **kwargs: Any, + ) -> MoeCausalLMOutputWithPast: + outputs = self.model( + input_ids=input_ids, + attention_mask=attention_mask, + position_ids=position_ids, + past_key_values=past_key_values, + inputs_embeds=inputs_embeds, + use_cache=use_cache, + output_router_logits=output_router_logits, + **kwargs, + ) + indices = ( + slice(-logits_to_keep, None) + if isinstance(logits_to_keep, int) + else logits_to_keep + ) + logits = self.lm_head(outputs.last_hidden_state[:, indices, :]) + loss = ( + self.loss_function(logits, labels, self.vocab_size, **kwargs) + if labels is not None + else None + ) + return MoeCausalLMOutputWithPast( + loss=loss, + logits=logits, + past_key_values=outputs.past_key_values, + hidden_states=outputs.hidden_states, + attentions=outputs.attentions, + router_logits=outputs.router_logits, + ) + + +AliceAIConfig.register_for_auto_class() +AliceAIModel.register_for_auto_class("AutoModel") +AliceAIForCausalLM.register_for_auto_class("AutoModelForCausalLM") diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000000000000000000000000000000000000..a52c50a199269393cd1548c7e6a77a654bd2001b --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,5 @@ +{ + "bos_token": "", + "eos_token": "", + "unk_token": "" +} diff --git a/tokenizer.model b/tokenizer.model new file mode 100644 index 0000000000000000000000000000000000000000..d589fcd5c44f6600eb3052349405f202ce667c49 --- /dev/null +++ b/tokenizer.model @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aed6fe7cbde4daaf8da3f2d121932e1445e831fd294273cec1aeeddf0920dcea +size 2573190 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..22b173676c1b96c9beecf417b38ce211f0ad5114 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,10 @@ +{ + "add_bos_token": true, + "add_eos_token": false, + "bos_token": "", + "eos_token": "", + "legacy": false, + "model_max_length": 1000000000000000019884624838656, + "tokenizer_class": "LlamaTokenizer", + "unk_token": "" +}