初始化项目,由ModelHub XC社区提供模型
Model: upstage/solar-pro-preview-instruct Source: Original Platform
This commit is contained in:
59
.gitattributes
vendored
Normal file
59
.gitattributes
vendored
Normal file
@@ -0,0 +1,59 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zstandard filter=lfs diff=lfs merge=lfs -text
|
||||
*.tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
*.db* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ark* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.gguf* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ggml filter=lfs diff=lfs merge=lfs -text
|
||||
*.llamafile* filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
model-00004-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00006-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00009-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00003-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00007-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00001-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00008-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00005-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00002-of-00009.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.model filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
21
LICENSE
Normal file
21
LICENSE
Normal file
@@ -0,0 +1,21 @@
|
||||
MIT License
|
||||
|
||||
Copyright (c) Upstage Corporation.
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in all
|
||||
copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
||||
SOFTWARE.
|
||||
129
README.md
Normal file
129
README.md
Normal file
@@ -0,0 +1,129 @@
|
||||
---
|
||||
license: mit
|
||||
license_link: https://huggingface.co/upstage/solar-pro-preview-instruct/blob/main/LICENSE
|
||||
language:
|
||||
- en
|
||||
pipeline_tag: text-generation
|
||||
tags:
|
||||
- nlp
|
||||
library_name: transformers
|
||||
---
|
||||
|
||||
<p align="left">
|
||||
<a href="https://go.upstage.ai/3Xk9J6X">
|
||||
<img src="https://cdn-uploads.huggingface.co/production/uploads/5fd90c758fe27b1a6b077abb/jwMkqV88Hj8sJu7NjTedm.png" width="100%"/>
|
||||
</a>
|
||||
<p>
|
||||
|
||||
# **Solar Pro Preview: The most intelligent LLM on a single GPU**
|
||||
|
||||
# **Summary**
|
||||
|
||||
We introduce **Solar Pro Preview**, an advanced large language model (LLM) with 22 billion parameters designed to [fit into a single GPU](https://www.upstage.ai/products/solar-pro-preview?utm_source=%08platform&utm_medium=huggingface&utm_campaign=solarpro-preview-launch). Solar Pro Preview shows superior performance compared to LLMs with less than 30 billion parameters and delivers performance comparable to models over three times its size, such as Llama 3.1 with 70 billion parameters.
|
||||
|
||||
Solar Pro Preview is developed using an enhanced version of our previous depth up-scaling method, which scales a Phi-3-medium model with 14 billion parameters to 22 billion parameters, intended to run on a GPU with 80GB of VRAM. Our carefully curated training strategy and dataset have significantly enhanced performance from Phi-3-medium, particularly on the MMLU-Pro and IFEval benchmarks, both respected for evaluating a model’s knowledge and instruction-following abilities.
|
||||
|
||||
Solar Pro Preview is a pre-release version of the official Solar Pro, with limitations on language coverage and a maximum context length of 4K. However, we believe Solar Pro Preview not only stands out as a highly efficient and capable model, but has the potential to be further extended to cover more languages and capabilities. The official version of Solar Pro will be released this November 2024 with expanded language support beyond English and longer context windows. To stay informed about the latest updates, please sign up for [our mailing list](https://www.upstage.ai/get-upstage-updates). If you have any feedback or questions about the model, please visit our [model discussion board](https://huggingface.co/upstage/solar-pro-preview-instruct/discussions) and connect with us directly.
|
||||
|
||||
# **Usage**
|
||||
|
||||
Solar Pro Preview is an instruction-tuned language model. This model is specifically designed to follow instructions and engage in conversational tasks.
|
||||
|
||||
### Chat Template
|
||||
|
||||
As an instruction-tuned model, Solar Pro Preview uses the ChatML template for optimal performance in conversational and instruction-following tasks. This approach aligns with the model's training data and is likely to yield more accurate and relevant responses. For instance, a question formatted in the ChatML template looks like the following, where the model generates the answer after <|im_start|>assistant. Note that system prompts are not currently supported in Solar Pro Preview. This feature will be available in the official release.
|
||||
|
||||
```
|
||||
<|im_start|>user
|
||||
Please, introduce yourself.<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
### Text Generation
|
||||
|
||||
Below is an example inference code that details loading the model, applying the chat template, and generating the model answer.
|
||||
|
||||
```python
|
||||
# Install requirements
|
||||
# !pip install transformers==4.44.2 torch==2.3.1 flash_attn==2.5.8 accelerate==0.31.0
|
||||
|
||||
# Load model
|
||||
import torch
|
||||
from transformers import AutoModelForCausalLM, AutoTokenizer
|
||||
tokenizer = AutoTokenizer.from_pretrained("upstage/solar-pro-preview-instruct")
|
||||
model = AutoModelForCausalLM.from_pretrained(
|
||||
"upstage/solar-pro-preview-instruct",
|
||||
device_map="cuda",
|
||||
torch_dtype="auto",
|
||||
trust_remote_code=True,
|
||||
)
|
||||
# Apply chat template
|
||||
messages = [
|
||||
{"role": "user", "content": "Please, introduce yourself."},
|
||||
]
|
||||
prompt = tokenizer.apply_chat_template(messages, return_tensors="pt", add_generation_prompt=True).to(model.device)
|
||||
# Generate text
|
||||
outputs = model.generate(prompt, max_new_tokens=512)
|
||||
print(tokenizer.decode(outputs[0]))
|
||||
```
|
||||
|
||||
Solar Pro Preview is also available as an API in [Upstage Console](https://go.upstage.ai/3Xl0Hqv) and we provide other easy-to-use methods as well. If you'd like to explore these options, please visit our [blog page](https://www.upstage.ai/products/solar-pro-preview?utm_source=%08platform&utm_medium=huggingface&utm_campaign=solarpro-preview-launch).
|
||||
|
||||
|
||||
# **Evaluation**
|
||||
|
||||
Solar Pro Preview is evaluated over a variety of benchmarks.
|
||||
|
||||
| | Solar-pro-preview | Phi-3-medium-4K-instruct | Phi-3.5-MoE-instruct | Gemma 2 27B IT | Llama-3.1-8B-instruct | Llama-3.1-70B-instruct |
|
||||
| ------------- | :---------------: | :----------------------: | :------------------: | :----------------------------------------: | :-------------------------------------------------------------------------------: | :-------------------------------------------------------------------------------: |
|
||||
| *Release Date* | 2024.09.08 | 2024.05.02 | 2024.08.20 | 2024.06.25 | 2024.06.18 | 2024.06.16 |
|
||||
| *Model size* | 22B | 14B | 41.9B (6.6B) | 27B | 8B | 70B |
|
||||
| *License* | MIT | MIT | MIT | [gemma](https://ai.google.dev/gemma/terms) | [llama3.1](https://huggingface.co/meta-llama/Meta-Llama-3.1-8B/blob/main/LICENSE) | [llama3.1](https://huggingface.co/meta-llama/Meta-Llama-3.1-8B/blob/main/LICENSE) |
|
||||
| **MMLU** | 79.14 | 78.02 | 78.66 | 76.13 | 68.25 | 82.09 |
|
||||
| **MMLU Pro** | 52.11 | 47.51 | 46.99 | 45.68 | 37.88 | 53.01 |
|
||||
| **IFEval** | 84.37 | 64.37 | 69.15 | 75.36 | 77.40 | 84.13 |
|
||||
| **ARC-C** | 68.86 | 66.55 | 68.34 | 74.06 | 60.24 | 70.39 |
|
||||
| **GPQA** | 36.38 | 35.78 | 34.38 | 36.38 | 35.26 | 41.06 |
|
||||
| **HellaSwag** | 86.36 | 85.68 | 85.97 | 86.02 | 80.08 | 86.42 |
|
||||
| **EQBench** | 77.91 | 76.78 | 77.22 | 80.32 | 65.80 | 82.52 |
|
||||
| **BigBench Hard** | 67.31 | 63.09 | 62.58 | 64.88 | 51.06 | 69.54 |
|
||||
| **MUSR** | 45.85 | 42.28 | 46.79 | 45.67 | 29.68 | 47.22 |
|
||||
| **GSM8K** | 89.69 | 84.76 | 82.26 | 62.85 | 75.97 | 92.12 |
|
||||
| **MBPP** | 61.59 | 60.27 | N/A (\*) | 63.08 | 52.20 | 65.51 |
|
||||
|
||||
(*) Since the model tends to generate a chat template, the score can't be accurately determined.
|
||||
|
||||
### Evaluation Protocol
|
||||
|
||||
For easy reproduction of our evaluation results, we list the evaluation tools and settings used below. All evaluations are conducted with NVIDIA DGX H100.
|
||||
|
||||
| | Evaluation setting | Metric | Evaluation tool |
|
||||
| ------------- | :-------------------- | :------------------------------------------------------------- | :------------------------------------------------------------------------------------------------------------------------------------------------- |
|
||||
| MMLU | 5-shot | macro_avg / acc | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| MMLU Pro | 5-shot | macro_avg / acc | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| IFEval | 0-shot, chat_template | mean of prompt_level_strict_acc and instruction_level_strict_acc | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| ARC-C | 25-shot | acc_norm | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| GPQA | 0-shot | acc_norm | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| HellaSwag | 10-shot | acc_norm | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| EQBench | 0-shot, chat_template | eqbench score | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| BigBench Hard | 3-shot | macro_avg / acc_norm | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| MUSR | 0-shot | macro_avg / acc_norm | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| GSM8K | 8-shot, CoT | acc, exact_match & strict_extract | [lm-eval-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/928e8bb6f50d1e93ef5d0bcaa81f8c5fd9a6f4d8) #928e8bb |
|
||||
| MBPP | 0-shot | pass@1 | [bigcode-evaluation-harness](https://github.com/bigcode-project/bigcode-evaluation-harness/tree/0f3e95f0806e78a4f432056cdb1be93604a51d69) #0f3e95f |
|
||||
|
||||
The results may vary slightly for different batch sizes and experimental environment such as GPU type.
|
||||
|
||||
# **Contact us**
|
||||
|
||||
For any questions and suggestions regarding the model, please visit the [discussion board](https://huggingface.co/upstage/solar-pro-preview-instruct/discussions).
|
||||
|
||||
Learn more:
|
||||
|
||||
- [Chat with Solar Pro Preview](https://chat.upstage.ai)
|
||||
- [Solar Pro Preview blog](https://www.upstage.ai/products/solar-pro-preview)
|
||||
- [Solar Pro Preview developer documents](https://developers.upstage.ai/docs/apis/chat?utm_campaign=solarpro-preview-launch)
|
||||
|
||||
Also try out:
|
||||
|
||||
- [Document Parse](http://developers.upstage.ai/docs/apis/document-parse?utm_campaign=solarpro-preview-launch): An industry-leading model for converting complex document files to LLM-compatible HTML formats.
|
||||
- [Solar DocVision Preview](http://developers.upstage.ai/docs/apis/document-qa?utm_campaign=solarpro-preview-launch): A vision LLM specialized on documents.
|
||||
130
added_tokens.json
Normal file
130
added_tokens.json
Normal file
@@ -0,0 +1,130 @@
|
||||
{
|
||||
"<|assistant|>": 32001,
|
||||
"<|end|>": 32000,
|
||||
"<|im_end|>": 32007,
|
||||
"<|im_start|>": 32010,
|
||||
"<|placeholder100|>": 32104,
|
||||
"<|placeholder101|>": 32105,
|
||||
"<|placeholder102|>": 32106,
|
||||
"<|placeholder103|>": 32107,
|
||||
"<|placeholder104|>": 32108,
|
||||
"<|placeholder105|>": 32109,
|
||||
"<|placeholder106|>": 32110,
|
||||
"<|placeholder107|>": 32111,
|
||||
"<|placeholder108|>": 32112,
|
||||
"<|placeholder109|>": 32113,
|
||||
"<|placeholder10|>": 32014,
|
||||
"<|placeholder110|>": 32114,
|
||||
"<|placeholder111|>": 32115,
|
||||
"<|placeholder112|>": 32116,
|
||||
"<|placeholder113|>": 32117,
|
||||
"<|placeholder114|>": 32118,
|
||||
"<|placeholder115|>": 32119,
|
||||
"<|placeholder116|>": 32120,
|
||||
"<|placeholder117|>": 32121,
|
||||
"<|placeholder118|>": 32122,
|
||||
"<|placeholder119|>": 32123,
|
||||
"<|placeholder11|>": 32015,
|
||||
"<|placeholder120|>": 32124,
|
||||
"<|placeholder121|>": 32125,
|
||||
"<|placeholder122|>": 32126,
|
||||
"<|placeholder123|>": 32127,
|
||||
"<|placeholder12|>": 32016,
|
||||
"<|placeholder13|>": 32017,
|
||||
"<|placeholder14|>": 32018,
|
||||
"<|placeholder15|>": 32019,
|
||||
"<|placeholder16|>": 32020,
|
||||
"<|placeholder17|>": 32021,
|
||||
"<|placeholder18|>": 32022,
|
||||
"<|placeholder19|>": 32023,
|
||||
"<|placeholder1|>": 32002,
|
||||
"<|placeholder20|>": 32024,
|
||||
"<|placeholder21|>": 32025,
|
||||
"<|placeholder22|>": 32026,
|
||||
"<|placeholder23|>": 32027,
|
||||
"<|placeholder24|>": 32028,
|
||||
"<|placeholder25|>": 32029,
|
||||
"<|placeholder26|>": 32030,
|
||||
"<|placeholder27|>": 32031,
|
||||
"<|placeholder28|>": 32032,
|
||||
"<|placeholder29|>": 32033,
|
||||
"<|placeholder2|>": 32003,
|
||||
"<|placeholder30|>": 32034,
|
||||
"<|placeholder31|>": 32035,
|
||||
"<|placeholder32|>": 32036,
|
||||
"<|placeholder33|>": 32037,
|
||||
"<|placeholder34|>": 32038,
|
||||
"<|placeholder35|>": 32039,
|
||||
"<|placeholder36|>": 32040,
|
||||
"<|placeholder37|>": 32041,
|
||||
"<|placeholder38|>": 32042,
|
||||
"<|placeholder39|>": 32043,
|
||||
"<|placeholder3|>": 32004,
|
||||
"<|placeholder40|>": 32044,
|
||||
"<|placeholder41|>": 32045,
|
||||
"<|placeholder42|>": 32046,
|
||||
"<|placeholder43|>": 32047,
|
||||
"<|placeholder44|>": 32048,
|
||||
"<|placeholder45|>": 32049,
|
||||
"<|placeholder46|>": 32050,
|
||||
"<|placeholder47|>": 32051,
|
||||
"<|placeholder48|>": 32052,
|
||||
"<|placeholder49|>": 32053,
|
||||
"<|placeholder4|>": 32005,
|
||||
"<|placeholder50|>": 32054,
|
||||
"<|placeholder51|>": 32055,
|
||||
"<|placeholder52|>": 32056,
|
||||
"<|placeholder53|>": 32057,
|
||||
"<|placeholder54|>": 32058,
|
||||
"<|placeholder55|>": 32059,
|
||||
"<|placeholder56|>": 32060,
|
||||
"<|placeholder57|>": 32061,
|
||||
"<|placeholder58|>": 32062,
|
||||
"<|placeholder59|>": 32063,
|
||||
"<|placeholder5|>": 32008,
|
||||
"<|placeholder60|>": 32064,
|
||||
"<|placeholder61|>": 32065,
|
||||
"<|placeholder62|>": 32066,
|
||||
"<|placeholder63|>": 32067,
|
||||
"<|placeholder64|>": 32068,
|
||||
"<|placeholder65|>": 32069,
|
||||
"<|placeholder66|>": 32070,
|
||||
"<|placeholder67|>": 32071,
|
||||
"<|placeholder68|>": 32072,
|
||||
"<|placeholder69|>": 32073,
|
||||
"<|placeholder6|>": 32009,
|
||||
"<|placeholder70|>": 32074,
|
||||
"<|placeholder71|>": 32075,
|
||||
"<|placeholder72|>": 32076,
|
||||
"<|placeholder73|>": 32077,
|
||||
"<|placeholder74|>": 32078,
|
||||
"<|placeholder75|>": 32079,
|
||||
"<|placeholder76|>": 32080,
|
||||
"<|placeholder77|>": 32081,
|
||||
"<|placeholder78|>": 32082,
|
||||
"<|placeholder79|>": 32083,
|
||||
"<|placeholder7|>": 32011,
|
||||
"<|placeholder80|>": 32084,
|
||||
"<|placeholder81|>": 32085,
|
||||
"<|placeholder82|>": 32086,
|
||||
"<|placeholder83|>": 32087,
|
||||
"<|placeholder84|>": 32088,
|
||||
"<|placeholder85|>": 32089,
|
||||
"<|placeholder86|>": 32090,
|
||||
"<|placeholder87|>": 32091,
|
||||
"<|placeholder88|>": 32092,
|
||||
"<|placeholder89|>": 32093,
|
||||
"<|placeholder8|>": 32012,
|
||||
"<|placeholder90|>": 32094,
|
||||
"<|placeholder91|>": 32095,
|
||||
"<|placeholder92|>": 32096,
|
||||
"<|placeholder93|>": 32097,
|
||||
"<|placeholder94|>": 32098,
|
||||
"<|placeholder95|>": 32099,
|
||||
"<|placeholder96|>": 32100,
|
||||
"<|placeholder97|>": 32101,
|
||||
"<|placeholder98|>": 32102,
|
||||
"<|placeholder99|>": 32103,
|
||||
"<|placeholder9|>": 32013,
|
||||
"<|system|>": 32006
|
||||
}
|
||||
57
config.json
Normal file
57
config.json
Normal file
@@ -0,0 +1,57 @@
|
||||
{
|
||||
"architectures": [
|
||||
"SolarForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"auto_map": {
|
||||
"AutoConfig": "configuration_solar.SolarConfig",
|
||||
"AutoModelForCausalLM": "modeling_solar.SolarForCausalLM"
|
||||
},
|
||||
"bos_token_id": 1,
|
||||
"bskcn_1": [
|
||||
12,
|
||||
20,
|
||||
32,
|
||||
44
|
||||
],
|
||||
"bskcn_2": [
|
||||
20,
|
||||
32
|
||||
],
|
||||
"bskcn_3": [
|
||||
16,
|
||||
24,
|
||||
36,
|
||||
48
|
||||
],
|
||||
"bskcn_4": [
|
||||
28,
|
||||
40
|
||||
],
|
||||
"bskcn_tv": [
|
||||
0.9,
|
||||
0.8
|
||||
],
|
||||
"eos_token_id": 32007,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 5120,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 17920,
|
||||
"max_position_embeddings": 4096,
|
||||
"mlp_bias": false,
|
||||
"model_type": "solar",
|
||||
"num_attention_heads": 40,
|
||||
"num_hidden_layers": 64,
|
||||
"num_key_value_heads": 10,
|
||||
"pretraining_tp": 1,
|
||||
"rms_norm_eps": 1e-05,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 10000.0,
|
||||
"sliding_window": 2047,
|
||||
"tie_word_embeddings": false,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.44.2",
|
||||
"use_cache": true,
|
||||
"vocab_size": 32128
|
||||
}
|
||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
||||
{"framework": "pytorch", "task": "text-generation", "allow_remote": true}
|
||||
206
configuration_solar.py
Normal file
206
configuration_solar.py
Normal file
@@ -0,0 +1,206 @@
|
||||
# coding=utf-8
|
||||
# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.
|
||||
#
|
||||
# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
|
||||
# and OPT implementations in this library. It has been modified from its
|
||||
# original forms to accommodate minor architectural differences compared
|
||||
# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""Solar model configuration"""
|
||||
|
||||
from transformers.configuration_utils import PretrainedConfig
|
||||
from transformers.utils import logging
|
||||
|
||||
|
||||
logger = logging.get_logger(__name__)
|
||||
|
||||
|
||||
class SolarConfig(PretrainedConfig):
|
||||
r"""
|
||||
This is the configuration class to store the configuration of a [`SolarModel`]. It is used to instantiate an LLaMA
|
||||
model according to the specified arguments, defining the model architecture. Instantiating a configuration with the
|
||||
defaults will yield a similar configuration to that of the LLaMA-7B.
|
||||
|
||||
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
|
||||
documentation from [`PretrainedConfig`] for more information.
|
||||
|
||||
|
||||
Args:
|
||||
vocab_size (`int`, *optional*, defaults to 32000):
|
||||
Vocabulary size of the LLaMA model. Defines the number of different tokens that can be represented by the
|
||||
`inputs_ids` passed when calling [`SolarModel`]
|
||||
hidden_size (`int`, *optional*, defaults to 4096):
|
||||
Dimension of the hidden representations.
|
||||
intermediate_size (`int`, *optional*, defaults to 11008):
|
||||
Dimension of the MLP representations.
|
||||
num_hidden_layers (`int`, *optional*, defaults to 32):
|
||||
Number of hidden layers in the Transformer decoder.
|
||||
num_attention_heads (`int`, *optional*, defaults to 32):
|
||||
Number of attention heads for each attention layer in the Transformer decoder.
|
||||
num_key_value_heads (`int`, *optional*):
|
||||
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
|
||||
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
|
||||
`num_key_value_heads=1` the model will use Multi Query Attention (MQA) otherwise GQA is used. When
|
||||
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
|
||||
by meanpooling all the original heads within that group. For more details checkout [this
|
||||
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to
|
||||
`num_attention_heads`.
|
||||
hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
|
||||
The non-linear activation function (function or string) in the decoder.
|
||||
max_position_embeddings (`int`, *optional*, defaults to 2048):
|
||||
The maximum sequence length that this model might ever be used with. Solar 1 supports up to 2048 tokens,
|
||||
Solar 2 up to 4096, CodeSolar up to 16384.
|
||||
initializer_range (`float`, *optional*, defaults to 0.02):
|
||||
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
|
||||
rms_norm_eps (`float`, *optional*, defaults to 1e-06):
|
||||
The epsilon used by the rms normalization layers.
|
||||
use_cache (`bool`, *optional*, defaults to `True`):
|
||||
Whether or not the model should return the last key/values attentions (not used by all models). Only
|
||||
relevant if `config.is_decoder=True`.
|
||||
pad_token_id (`int`, *optional*):
|
||||
Padding token id.
|
||||
bos_token_id (`int`, *optional*, defaults to 1):
|
||||
Beginning of stream token id.
|
||||
eos_token_id (`int`, *optional*, defaults to 2):
|
||||
End of stream token id.
|
||||
pretraining_tp (`int`, *optional*, defaults to 1):
|
||||
Experimental feature. Tensor parallelism rank used during pretraining. Please refer to [this
|
||||
document](https://huggingface.co/docs/transformers/main/perf_train_gpu_many#tensor-parallelism) to understand more about it. This value is
|
||||
necessary to ensure exact reproducibility of the pretraining results. Please refer to [this
|
||||
issue](https://github.com/pytorch/pytorch/issues/76232).
|
||||
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
|
||||
Whether to tie weight embeddings
|
||||
rope_theta (`float`, *optional*, defaults to 10000.0):
|
||||
The base period of the RoPE embeddings.
|
||||
rope_scaling (`Dict`, *optional*):
|
||||
Dictionary containing the scaling configuration for the RoPE embeddings. Currently supports two scaling
|
||||
strategies: linear and dynamic. Their scaling factor must be a float greater than 1. The expected format is
|
||||
`{"type": strategy name, "factor": scaling factor}`. When using this flag, don't update
|
||||
`max_position_embeddings` to the expected new maximum. See the following thread for more information on how
|
||||
these scaling strategies behave:
|
||||
https://www.reddit.com/r/LocalLLaMA/comments/14mrgpr/dynamically_scaled_rope_further_increases/. This is an
|
||||
experimental feature, subject to breaking API changes in future versions.
|
||||
attention_bias (`bool`, *optional*, defaults to `False`):
|
||||
Whether to use a bias in the query, key, value and output projection layers during self-attention.
|
||||
attention_dropout (`float`, *optional*, defaults to 0.0):
|
||||
The dropout ratio for the attention probabilities.
|
||||
mlp_bias (`bool`, *optional*, defaults to `False`):
|
||||
Whether to use a bias in up_proj, down_proj and gate_proj layers in the MLP layers.
|
||||
sliding_window (`int`, *optional*, defaults to 2047):
|
||||
Sliding window attention window size. If not specified, will default to `2047`.
|
||||
|
||||
```python
|
||||
>>> from transformers import SolarModel, SolarConfig
|
||||
|
||||
>>> # Initializing a Solar-pro style configuration
|
||||
>>> configuration = SolarConfig()
|
||||
|
||||
>>> # Initializing a model from the Solar-pro style configuration
|
||||
>>> model = SolarModel(configuration)
|
||||
|
||||
>>> # Accessing the model configuration
|
||||
>>> configuration = model.config
|
||||
```"""
|
||||
|
||||
model_type = "solar"
|
||||
keys_to_ignore_at_inference = ["past_key_values"]
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
vocab_size=32000,
|
||||
hidden_size=4096,
|
||||
intermediate_size=11008,
|
||||
num_hidden_layers=32,
|
||||
num_attention_heads=32,
|
||||
num_key_value_heads=None,
|
||||
hidden_act="silu",
|
||||
max_position_embeddings=2048,
|
||||
initializer_range=0.02,
|
||||
rms_norm_eps=1e-6,
|
||||
use_cache=True,
|
||||
pad_token_id=None,
|
||||
bos_token_id=1,
|
||||
eos_token_id=2,
|
||||
pretraining_tp=1,
|
||||
tie_word_embeddings=False,
|
||||
rope_theta=10000.0,
|
||||
rope_scaling=None,
|
||||
attention_bias=False,
|
||||
attention_dropout=0.0,
|
||||
mlp_bias=False,
|
||||
sliding_window=2047,
|
||||
bskcn_1=[12, 20, 32, 44],
|
||||
bskcn_2=[20, 32],
|
||||
bskcn_3=[16, 24, 36, 48],
|
||||
bskcn_4=[28, 40],
|
||||
bskcn_tv=[0.9,0.8],
|
||||
**kwargs,
|
||||
):
|
||||
self.vocab_size = vocab_size
|
||||
self.max_position_embeddings = max_position_embeddings
|
||||
self.hidden_size = hidden_size
|
||||
self.intermediate_size = intermediate_size
|
||||
self.num_hidden_layers = num_hidden_layers
|
||||
self.num_attention_heads = num_attention_heads
|
||||
|
||||
# for backward compatibility
|
||||
if num_key_value_heads is None:
|
||||
num_key_value_heads = num_attention_heads
|
||||
|
||||
self.num_key_value_heads = num_key_value_heads
|
||||
self.hidden_act = hidden_act
|
||||
self.initializer_range = initializer_range
|
||||
self.rms_norm_eps = rms_norm_eps
|
||||
self.pretraining_tp = pretraining_tp
|
||||
self.use_cache = use_cache
|
||||
self.rope_theta = rope_theta
|
||||
self.rope_scaling = rope_scaling
|
||||
self._rope_scaling_validation()
|
||||
self.attention_bias = attention_bias
|
||||
self.attention_dropout = attention_dropout
|
||||
self.mlp_bias = mlp_bias
|
||||
self.sliding_window = sliding_window
|
||||
self.bskcn_1 = bskcn_1
|
||||
self.bskcn_2 = bskcn_2
|
||||
self.bskcn_3 = bskcn_3
|
||||
self.bskcn_4 = bskcn_4
|
||||
self.bskcn_tv = bskcn_tv
|
||||
|
||||
super().__init__(
|
||||
pad_token_id=pad_token_id,
|
||||
bos_token_id=bos_token_id,
|
||||
eos_token_id=eos_token_id,
|
||||
tie_word_embeddings=tie_word_embeddings,
|
||||
**kwargs,
|
||||
)
|
||||
|
||||
def _rope_scaling_validation(self):
|
||||
"""
|
||||
Validate the `rope_scaling` configuration.
|
||||
"""
|
||||
if self.rope_scaling is None:
|
||||
return
|
||||
|
||||
if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 2:
|
||||
raise ValueError(
|
||||
"`rope_scaling` must be a dictionary with two fields, `type` and `factor`, " f"got {self.rope_scaling}"
|
||||
)
|
||||
rope_scaling_type = self.rope_scaling.get("type", None)
|
||||
rope_scaling_factor = self.rope_scaling.get("factor", None)
|
||||
if rope_scaling_type is None or rope_scaling_type not in ["linear", "dynamic"]:
|
||||
raise ValueError(
|
||||
f"`rope_scaling`'s type field must be one of ['linear', 'dynamic'], got {rope_scaling_type}"
|
||||
)
|
||||
if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor <= 1.0:
|
||||
raise ValueError(f"`rope_scaling`'s factor field must be a float > 1, got {rope_scaling_factor}")
|
||||
11
generation_config.json
Normal file
11
generation_config.json
Normal file
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"_from_model_config": true,
|
||||
"bos_token_id": 1,
|
||||
"eos_token_id": [
|
||||
2,
|
||||
32000,
|
||||
32007
|
||||
],
|
||||
"use_cache": true,
|
||||
"transformers_version": "4.44.2"
|
||||
}
|
||||
3
model-00001-of-00009.safetensors
Normal file
3
model-00001-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:bce5ef2fe88dbf56e12271e674d40408d8884fbcd2878179904be71f8d311bfa
|
||||
size 4916640672
|
||||
3
model-00002-of-00009.safetensors
Normal file
3
model-00002-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8aeb0f191bfa0ff6a5316fcc99e817dfef70c7c7fcf1bf7423c1c0d0c145d7b3
|
||||
size 4954693120
|
||||
3
model-00003-of-00009.safetensors
Normal file
3
model-00003-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d3565f2a2f9d6820f9457e6890bc0ed480786022fa023d0b5c69d5bcfc9e2ed7
|
||||
size 4902243992
|
||||
3
model-00004-of-00009.safetensors
Normal file
3
model-00004-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cbb538fb3e865651a8243160df5ee1bd0dd7ab9f9e624bac2946a7d63e5b8d8c
|
||||
size 4954672440
|
||||
3
model-00005-of-00009.safetensors
Normal file
3
model-00005-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9a93c34d1340d596ca741a2bf037ca3892b30e6166b745dd2f41d99de5309e08
|
||||
size 4954672432
|
||||
3
model-00006-of-00009.safetensors
Normal file
3
model-00006-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:813a5b5c0623d9fe090972c27e535147be0e76ebaaa3f90847dfaa02bded02b1
|
||||
size 4954693144
|
||||
3
model-00007-of-00009.safetensors
Normal file
3
model-00007-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b4c5b2f4d811d8e1568b745382b532dc911ce905af109bdabd6357de55466b61
|
||||
size 4902243992
|
||||
3
model-00008-of-00009.safetensors
Normal file
3
model-00008-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:e3b715fd390c37b43bf91e280e89921ca056c5b28f028868f5790ef55a504943
|
||||
size 4954672440
|
||||
3
model-00009-of-00009.safetensors
Normal file
3
model-00009-of-00009.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0af544234a1e6f6dcbf25323f848003b639df26e5d9892e87d59a67118f39a2a
|
||||
size 4785599296
|
||||
586
model.safetensors.index.json
Normal file
586
model.safetensors.index.json
Normal file
@@ -0,0 +1,586 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 44280064000
|
||||
},
|
||||
"weight_map": {
|
||||
"lm_head.weight": "model-00009-of-00009.safetensors",
|
||||
"model.embed_tokens.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.0.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.1.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.10.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.10.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.11.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.12.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.13.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.14.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.14.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.15.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.16.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.17.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.18.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.19.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.2.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.2.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.20.input_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.mlp.down_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.mlp.gate_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.mlp.up_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.post_attention_layernorm.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.20.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.21.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.21.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.21.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.21.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.21.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.21.self_attn.k_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.21.self_attn.o_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.21.self_attn.q_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.21.self_attn.v_proj.weight": "model-00003-of-00009.safetensors",
|
||||
"model.layers.22.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.22.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.23.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.24.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.25.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.26.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.input_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.mlp.down_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.mlp.up_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.post_attention_layernorm.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.27.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.28.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.28.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.28.mlp.gate_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.28.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.28.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.28.self_attn.k_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.28.self_attn.o_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.28.self_attn.q_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.28.self_attn.v_proj.weight": "model-00004-of-00009.safetensors",
|
||||
"model.layers.29.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.29.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.3.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.3.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.30.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.30.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.31.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.32.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.33.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.input_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.mlp.down_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.post_attention_layernorm.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.34.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.35.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.35.mlp.gate_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.mlp.up_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.35.self_attn.k_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.self_attn.o_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.self_attn.q_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.35.self_attn.v_proj.weight": "model-00005-of-00009.safetensors",
|
||||
"model.layers.36.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.36.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.37.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.38.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.39.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.4.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.4.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.40.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.40.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.41.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.input_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.mlp.down_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.mlp.gate_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.mlp.up_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.post_attention_layernorm.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.self_attn.k_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.self_attn.o_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.self_attn.q_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.42.self_attn.v_proj.weight": "model-00006-of-00009.safetensors",
|
||||
"model.layers.43.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.43.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.44.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.45.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.46.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.47.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.48.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.input_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.mlp.down_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.mlp.gate_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.mlp.up_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.post_attention_layernorm.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.49.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.5.input_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.mlp.down_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.post_attention_layernorm.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.5.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.50.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.50.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.50.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.50.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.50.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.50.self_attn.k_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.50.self_attn.o_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.50.self_attn.q_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.50.self_attn.v_proj.weight": "model-00007-of-00009.safetensors",
|
||||
"model.layers.51.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.51.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.52.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.53.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.54.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.55.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.input_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.mlp.down_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.mlp.up_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.post_attention_layernorm.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.56.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.57.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.57.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.57.mlp.gate_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.57.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.57.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.57.self_attn.k_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.57.self_attn.o_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.57.self_attn.q_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.57.self_attn.v_proj.weight": "model-00008-of-00009.safetensors",
|
||||
"model.layers.58.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.58.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.59.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.6.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.6.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.6.mlp.gate_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.6.mlp.up_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.6.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.6.self_attn.k_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.6.self_attn.o_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.6.self_attn.q_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.6.self_attn.v_proj.weight": "model-00001-of-00009.safetensors",
|
||||
"model.layers.60.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.60.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.61.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.62.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.input_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.mlp.down_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.mlp.gate_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.mlp.up_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.post_attention_layernorm.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.self_attn.k_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.self_attn.o_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.self_attn.q_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.63.self_attn.v_proj.weight": "model-00009-of-00009.safetensors",
|
||||
"model.layers.7.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.7.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.8.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.input_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.mlp.down_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.mlp.gate_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.mlp.up_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.post_attention_layernorm.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.self_attn.k_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.self_attn.o_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.self_attn.q_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.layers.9.self_attn.v_proj.weight": "model-00002-of-00009.safetensors",
|
||||
"model.norm.weight": "model-00009-of-00009.safetensors"
|
||||
}
|
||||
}
|
||||
1745
modeling_solar.py
Normal file
1745
modeling_solar.py
Normal file
File diff suppressed because it is too large
Load Diff
BIN
solar-pro-banner.png
Normal file
BIN
solar-pro-banner.png
Normal file
Binary file not shown.
|
After Width: | Height: | Size: 376 KiB |
30
special_tokens_map.json
Normal file
30
special_tokens_map.json
Normal file
@@ -0,0 +1,30 @@
|
||||
{
|
||||
"bos_token": {
|
||||
"content": "<|startoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"eos_token": {
|
||||
"content": "<|im_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"pad_token": {
|
||||
"content": "<|im_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"unk_token": {
|
||||
"content": "<unk>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
}
|
||||
}
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:592a70a0f7aae41e1e2711383a233def76a85ef1698e35a0935ec5892d5403bd
|
||||
size 1866587
|
||||
3
tokenizer.model
Normal file
3
tokenizer.model
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:e79e5412a9810832e8183fb69a7fef269d8558215223b5e1bd07480e711119b7
|
||||
size 499744
|
||||
1067
tokenizer_config.json
Normal file
1067
tokenizer_config.json
Normal file
File diff suppressed because it is too large
Load Diff
552
vllm_solar.py
Normal file
552
vllm_solar.py
Normal file
@@ -0,0 +1,552 @@
|
||||
# coding=utf-8
|
||||
# Adapted from
|
||||
# https://github.com/huggingface/transformers/blob/v4.28.0/src/transformers/models/llama/modeling_llama.py
|
||||
# Copyright 2023 The vLLM team.
|
||||
# Copyright 2022 EleutherAI and the HuggingFace Inc. team. All rights reserved.
|
||||
#
|
||||
# This code is based on EleutherAI's GPT-NeoX library and the GPT-NeoX
|
||||
# and OPT implementations in this library. It has been modified from its
|
||||
# original forms to accommodate minor architectural differences compared
|
||||
# to GPT-NeoX and OPT used by the Meta AI team that trained the model.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
"""Inference-only Solar model compatible with HuggingFace weights."""
|
||||
from typing import Any, Dict, Iterable, List, Optional, Tuple, Union
|
||||
|
||||
import torch
|
||||
from torch import nn
|
||||
|
||||
from vllm.attention import Attention, AttentionMetadata
|
||||
from vllm.config import CacheConfig, LoRAConfig
|
||||
from vllm.distributed import (get_pp_group, get_tensor_model_parallel_rank,
|
||||
get_tensor_model_parallel_world_size)
|
||||
from vllm.model_executor.layers.activation import SiluAndMul
|
||||
from vllm.model_executor.layers.layernorm import RMSNorm
|
||||
from vllm.model_executor.layers.linear import (MergedColumnParallelLinear,
|
||||
QKVParallelLinear,
|
||||
RowParallelLinear)
|
||||
from vllm.model_executor.layers.logits_processor import LogitsProcessor
|
||||
from vllm.model_executor.layers.quantization.base_config import (
|
||||
QuantizationConfig)
|
||||
from vllm.model_executor.layers.quantization.compressed_tensors.utils import (
|
||||
get_compressed_tensors_cache_scale)
|
||||
from vllm.model_executor.layers.rotary_embedding import get_rope
|
||||
from vllm.model_executor.layers.sampler import Sampler
|
||||
from vllm.model_executor.layers.vocab_parallel_embedding import (
|
||||
DEFAULT_VOCAB_PADDING_SIZE, ParallelLMHead, VocabParallelEmbedding)
|
||||
from vllm.model_executor.model_loader.weight_utils import (
|
||||
default_weight_loader, kv_cache_scales_loader, maybe_remap_kv_scale_name)
|
||||
from vllm.model_executor.sampling_metadata import SamplingMetadata
|
||||
from vllm.sequence import IntermediateTensors, SamplerOutput
|
||||
from vllm.utils import is_hip
|
||||
|
||||
from vllm.model_executor.models.interfaces import SupportsLoRA
|
||||
from vllm.model_executor.models.utils import PPMissingLayer, is_pp_missing_parameter, make_layers
|
||||
|
||||
class SolarMLP(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
hidden_size: int,
|
||||
intermediate_size: int,
|
||||
hidden_act: str,
|
||||
quant_config: Optional[QuantizationConfig] = None,
|
||||
bias: bool = False,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.gate_up_proj = MergedColumnParallelLinear(
|
||||
input_size=hidden_size,
|
||||
output_sizes=[intermediate_size] * 2,
|
||||
bias=bias,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.gate_up_proj")
|
||||
self.down_proj = RowParallelLinear(input_size=intermediate_size,
|
||||
output_size=hidden_size,
|
||||
bias=bias,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.down_proj")
|
||||
if hidden_act != "silu":
|
||||
raise ValueError(f"Unsupported activation: {hidden_act}. "
|
||||
"Only silu is supported for now.")
|
||||
self.act_fn = SiluAndMul()
|
||||
|
||||
def forward(self, x):
|
||||
gate_up, _ = self.gate_up_proj(x)
|
||||
x = self.act_fn(gate_up)
|
||||
x, _ = self.down_proj(x)
|
||||
return x
|
||||
|
||||
|
||||
class SolarAttention(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config,
|
||||
hidden_size: int,
|
||||
num_heads: int,
|
||||
num_kv_heads: int,
|
||||
rope_theta: float = 10000,
|
||||
rope_scaling: Optional[Dict[str, Any]] = None,
|
||||
max_position_embeddings: int = 8192,
|
||||
quant_config: Optional[QuantizationConfig] = None,
|
||||
bias: bool = False,
|
||||
cache_config: Optional[CacheConfig] = None,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.hidden_size = hidden_size
|
||||
tp_size = get_tensor_model_parallel_world_size()
|
||||
self.total_num_heads = num_heads
|
||||
assert self.total_num_heads % tp_size == 0
|
||||
self.num_heads = self.total_num_heads // tp_size
|
||||
self.total_num_kv_heads = num_kv_heads
|
||||
if self.total_num_kv_heads >= tp_size:
|
||||
# Number of KV heads is greater than TP size, so we partition
|
||||
# the KV heads across multiple tensor parallel GPUs.
|
||||
assert self.total_num_kv_heads % tp_size == 0
|
||||
else:
|
||||
# Number of KV heads is less than TP size, so we replicate
|
||||
# the KV heads across multiple tensor parallel GPUs.
|
||||
assert tp_size % self.total_num_kv_heads == 0
|
||||
self.num_kv_heads = max(1, self.total_num_kv_heads // tp_size)
|
||||
# MistralConfig has an optional head_dim introduced by Mistral-Nemo
|
||||
self.head_dim = getattr(config, "head_dim",
|
||||
self.hidden_size // self.total_num_heads)
|
||||
self.q_size = self.num_heads * self.head_dim
|
||||
self.kv_size = self.num_kv_heads * self.head_dim
|
||||
self.scaling = self.head_dim**-0.5
|
||||
self.rope_theta = rope_theta
|
||||
self.max_position_embeddings = max_position_embeddings
|
||||
|
||||
self.qkv_proj = QKVParallelLinear(
|
||||
hidden_size=hidden_size,
|
||||
head_size=self.head_dim,
|
||||
total_num_heads=self.total_num_heads,
|
||||
total_num_kv_heads=self.total_num_kv_heads,
|
||||
bias=bias,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.qkv_proj",
|
||||
)
|
||||
self.o_proj = RowParallelLinear(
|
||||
input_size=self.total_num_heads * self.head_dim,
|
||||
output_size=hidden_size,
|
||||
bias=bias,
|
||||
quant_config=quant_config,
|
||||
prefix=f"{prefix}.o_proj",
|
||||
)
|
||||
|
||||
self.rotary_emb = get_rope(
|
||||
self.head_dim,
|
||||
rotary_dim=self.head_dim,
|
||||
max_position=max_position_embeddings,
|
||||
base=rope_theta,
|
||||
rope_scaling=rope_scaling,
|
||||
)
|
||||
self.attn = Attention(self.num_heads,
|
||||
self.head_dim,
|
||||
self.scaling,
|
||||
num_kv_heads=self.num_kv_heads,
|
||||
cache_config=cache_config,
|
||||
quant_config=quant_config)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
positions: torch.Tensor,
|
||||
hidden_states: torch.Tensor,
|
||||
kv_cache: torch.Tensor,
|
||||
attn_metadata: AttentionMetadata,
|
||||
) -> torch.Tensor:
|
||||
qkv, _ = self.qkv_proj(hidden_states)
|
||||
q, k, v = qkv.split([self.q_size, self.kv_size, self.kv_size], dim=-1)
|
||||
q, k = self.rotary_emb(positions, q, k)
|
||||
attn_output = self.attn(q, k, v, kv_cache, attn_metadata)
|
||||
output, _ = self.o_proj(attn_output)
|
||||
return output
|
||||
|
||||
|
||||
class SolarDecoderLayer(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config,
|
||||
cache_config: Optional[CacheConfig] = None,
|
||||
quant_config: Optional[QuantizationConfig] = None,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.hidden_size = config.hidden_size
|
||||
rope_theta = getattr(config, "rope_theta", 10000)
|
||||
rope_scaling = getattr(config, "rope_scaling", None)
|
||||
if rope_scaling is not None and getattr(
|
||||
config, "original_max_position_embeddings", None):
|
||||
rope_scaling["original_max_position_embeddings"] = (
|
||||
config.original_max_position_embeddings)
|
||||
max_position_embeddings = getattr(config, "max_position_embeddings",
|
||||
8192)
|
||||
# Support abacusai/Smaug-72B-v0.1 with attention_bias
|
||||
# Support internlm/internlm-7b with bias
|
||||
attention_bias = getattr(config, "attention_bias", False) or getattr(
|
||||
config, "bias", False)
|
||||
self.self_attn = SolarAttention(
|
||||
config=config,
|
||||
hidden_size=self.hidden_size,
|
||||
num_heads=config.num_attention_heads,
|
||||
num_kv_heads=getattr(config, "num_key_value_heads",
|
||||
config.num_attention_heads),
|
||||
rope_theta=rope_theta,
|
||||
rope_scaling=rope_scaling,
|
||||
max_position_embeddings=max_position_embeddings,
|
||||
quant_config=quant_config,
|
||||
bias=attention_bias,
|
||||
cache_config=cache_config,
|
||||
prefix=f"{prefix}.self_attn",
|
||||
)
|
||||
self.mlp = SolarMLP(
|
||||
hidden_size=self.hidden_size,
|
||||
intermediate_size=config.intermediate_size,
|
||||
hidden_act=config.hidden_act,
|
||||
quant_config=quant_config,
|
||||
bias=getattr(config, "mlp_bias", False),
|
||||
prefix=f"{prefix}.mlp",
|
||||
)
|
||||
self.input_layernorm = RMSNorm(config.hidden_size,
|
||||
eps=config.rms_norm_eps)
|
||||
self.post_attention_layernorm = RMSNorm(config.hidden_size,
|
||||
eps=config.rms_norm_eps)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
positions: torch.Tensor,
|
||||
hidden_states: torch.Tensor,
|
||||
kv_cache: torch.Tensor,
|
||||
attn_metadata: AttentionMetadata,
|
||||
residual: Optional[torch.Tensor],
|
||||
) -> Tuple[torch.Tensor, torch.Tensor]:
|
||||
# Self Attention
|
||||
if residual is None:
|
||||
residual = hidden_states
|
||||
hidden_states = self.input_layernorm(hidden_states)
|
||||
else:
|
||||
hidden_states, residual = self.input_layernorm(
|
||||
hidden_states, residual)
|
||||
hidden_states = self.self_attn(
|
||||
positions=positions,
|
||||
hidden_states=hidden_states,
|
||||
kv_cache=kv_cache,
|
||||
attn_metadata=attn_metadata,
|
||||
)
|
||||
|
||||
# Fully Connected
|
||||
hidden_states, residual = self.post_attention_layernorm(
|
||||
hidden_states, residual)
|
||||
hidden_states = self.mlp(hidden_states)
|
||||
return hidden_states, residual
|
||||
|
||||
|
||||
class SolarModel(nn.Module):
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config,
|
||||
cache_config: Optional[CacheConfig] = None,
|
||||
quant_config: Optional[QuantizationConfig] = None,
|
||||
lora_config: Optional[LoRAConfig] = None,
|
||||
prefix: str = "",
|
||||
) -> None:
|
||||
super().__init__()
|
||||
self.config = config
|
||||
self.padding_idx = config.pad_token_id
|
||||
lora_vocab = (lora_config.lora_extra_vocab_size *
|
||||
(lora_config.max_loras or 1)) if lora_config else 0
|
||||
self.vocab_size = config.vocab_size + lora_vocab
|
||||
self.org_vocab_size = config.vocab_size
|
||||
if get_pp_group().is_first_rank or (config.tie_word_embeddings
|
||||
and get_pp_group().is_last_rank):
|
||||
self.embed_tokens = VocabParallelEmbedding(
|
||||
self.vocab_size,
|
||||
config.hidden_size,
|
||||
org_num_embeddings=config.vocab_size,
|
||||
)
|
||||
else:
|
||||
self.embed_tokens = PPMissingLayer()
|
||||
self.start_layer, self.end_layer, self.layers = make_layers(
|
||||
config.num_hidden_layers,
|
||||
lambda prefix: SolarDecoderLayer(config=config,
|
||||
cache_config=cache_config,
|
||||
quant_config=quant_config,
|
||||
prefix=prefix),
|
||||
prefix=f"{prefix}.layers")
|
||||
if get_pp_group().is_last_rank:
|
||||
self.norm = RMSNorm(config.hidden_size, eps=config.rms_norm_eps)
|
||||
else:
|
||||
self.norm = PPMissingLayer()
|
||||
|
||||
def get_input_embeddings(self, input_ids: torch.Tensor) -> torch.Tensor:
|
||||
return self.embed_tokens(input_ids)
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: Optional[torch.Tensor],
|
||||
positions: torch.Tensor,
|
||||
kv_caches: List[torch.Tensor],
|
||||
attn_metadata: AttentionMetadata,
|
||||
intermediate_tensors: Optional[IntermediateTensors],
|
||||
inputs_embeds: Optional[torch.Tensor] = None,
|
||||
) -> Union[torch.Tensor, IntermediateTensors]:
|
||||
if get_pp_group().is_first_rank:
|
||||
if inputs_embeds is not None:
|
||||
hidden_states = inputs_embeds
|
||||
else:
|
||||
hidden_states = self.get_input_embeddings(input_ids)
|
||||
residual = None
|
||||
else:
|
||||
assert intermediate_tensors is not None
|
||||
hidden_states = intermediate_tensors["hidden_states"]
|
||||
residual = intermediate_tensors["residual"]
|
||||
|
||||
bskcn_h_1 = None
|
||||
bskcn_h_2 = None
|
||||
bskcn_r_1 = None
|
||||
bskcn_r_2 = None
|
||||
bskcn_tv = self.config.bskcn_tv[0] if self.training else self.config.bskcn_tv[1]
|
||||
|
||||
for i in range(self.start_layer, self.end_layer):
|
||||
if i in self.config.bskcn_1:
|
||||
bskcn_h_1 = hidden_states.clone()
|
||||
bskcn_r_1 = residual.clone()
|
||||
if i in self.config.bskcn_2:
|
||||
bskcn_h_2 = hidden_states.clone()
|
||||
bskcn_r_2 = residual.clone()
|
||||
if i in self.config.bskcn_3:
|
||||
hidden_states = bskcn_h_1*bskcn_tv + hidden_states*(1-bskcn_tv)
|
||||
residual = bskcn_r_1*bskcn_tv + residual*(1-bskcn_tv)
|
||||
if i in self.config.bskcn_4:
|
||||
hidden_states = bskcn_h_2*bskcn_tv + hidden_states*(1-bskcn_tv)
|
||||
residual = bskcn_r_2*bskcn_tv + residual*(1-bskcn_tv)
|
||||
layer = self.layers[i]
|
||||
hidden_states, residual = layer(
|
||||
positions,
|
||||
hidden_states,
|
||||
kv_caches[i - self.start_layer],
|
||||
attn_metadata,
|
||||
residual,
|
||||
)
|
||||
|
||||
if not get_pp_group().is_last_rank:
|
||||
return IntermediateTensors({
|
||||
"hidden_states": hidden_states,
|
||||
"residual": residual
|
||||
})
|
||||
|
||||
hidden_states, _ = self.norm(hidden_states, residual)
|
||||
return hidden_states
|
||||
|
||||
|
||||
class SolarForCausalLM(nn.Module, SupportsLoRA):
|
||||
packed_modules_mapping = {
|
||||
"qkv_proj": [
|
||||
"q_proj",
|
||||
"k_proj",
|
||||
"v_proj",
|
||||
],
|
||||
"gate_up_proj": [
|
||||
"gate_proj",
|
||||
"up_proj",
|
||||
],
|
||||
}
|
||||
|
||||
# LoRA specific attributes
|
||||
supported_lora_modules = [
|
||||
"qkv_proj", "o_proj", "gate_up_proj", "down_proj", "embed_tokens",
|
||||
"lm_head"
|
||||
]
|
||||
embedding_modules = {
|
||||
"embed_tokens": "input_embeddings",
|
||||
"lm_head": "output_embeddings",
|
||||
}
|
||||
embedding_padding_modules = ["lm_head"]
|
||||
bitsandbytes_stacked_params_mapping = {
|
||||
# shard_name, weight_name, index
|
||||
"q_proj": ("qkv_proj", 0),
|
||||
"k_proj": ("qkv_proj", 1),
|
||||
"v_proj": ("qkv_proj", 2),
|
||||
"gate_proj": ("gate_up_proj", 0),
|
||||
"up_proj": ("gate_up_proj", 1),
|
||||
}
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
config,
|
||||
cache_config: Optional[CacheConfig] = None,
|
||||
quant_config: Optional[QuantizationConfig] = None,
|
||||
lora_config: Optional[LoRAConfig] = None,
|
||||
) -> None:
|
||||
super().__init__()
|
||||
|
||||
self.config = config
|
||||
self.lora_config = lora_config
|
||||
|
||||
self.model = SolarModel(config,
|
||||
cache_config,
|
||||
quant_config,
|
||||
lora_config=lora_config,
|
||||
prefix="model")
|
||||
if get_pp_group().is_last_rank:
|
||||
self.unpadded_vocab_size = config.vocab_size
|
||||
if lora_config:
|
||||
self.unpadded_vocab_size += lora_config.lora_extra_vocab_size
|
||||
self.lm_head = ParallelLMHead(
|
||||
self.unpadded_vocab_size,
|
||||
config.hidden_size,
|
||||
org_num_embeddings=config.vocab_size,
|
||||
padding_size=DEFAULT_VOCAB_PADDING_SIZE
|
||||
# We need bigger padding if using lora for kernel
|
||||
# compatibility
|
||||
if not lora_config else lora_config.lora_vocab_padding_size,
|
||||
quant_config=quant_config,
|
||||
)
|
||||
if config.tie_word_embeddings:
|
||||
self.lm_head.weight = self.model.embed_tokens.weight
|
||||
|
||||
logit_scale = getattr(config, "logit_scale", 1.0)
|
||||
self.logits_processor = LogitsProcessor(self.unpadded_vocab_size,
|
||||
config.vocab_size,
|
||||
logit_scale)
|
||||
self.sampler = Sampler()
|
||||
else:
|
||||
self.lm_head = PPMissingLayer()
|
||||
|
||||
def forward(
|
||||
self,
|
||||
input_ids: torch.Tensor,
|
||||
positions: torch.Tensor,
|
||||
kv_caches: List[torch.Tensor],
|
||||
attn_metadata: AttentionMetadata,
|
||||
intermediate_tensors: Optional[IntermediateTensors] = None,
|
||||
) -> Union[torch.Tensor, IntermediateTensors]:
|
||||
model_output = self.model(input_ids, positions, kv_caches,
|
||||
attn_metadata, intermediate_tensors)
|
||||
return model_output
|
||||
|
||||
def compute_logits(self, hidden_states: torch.Tensor,
|
||||
sampling_metadata: SamplingMetadata) -> torch.Tensor:
|
||||
logits = self.logits_processor(self.lm_head, hidden_states,
|
||||
sampling_metadata)
|
||||
return logits
|
||||
|
||||
def sample(
|
||||
self,
|
||||
logits: torch.Tensor,
|
||||
sampling_metadata: SamplingMetadata,
|
||||
) -> Optional[SamplerOutput]:
|
||||
next_tokens = self.sampler(logits, sampling_metadata)
|
||||
return next_tokens
|
||||
|
||||
def make_empty_intermediate_tensors(
|
||||
self, batch_size: int, dtype: torch.dtype,
|
||||
device: torch.device) -> IntermediateTensors:
|
||||
return IntermediateTensors({
|
||||
"hidden_states":
|
||||
torch.zeros((batch_size, self.config.hidden_size),
|
||||
dtype=dtype,
|
||||
device=device),
|
||||
"residual":
|
||||
torch.zeros((batch_size, self.config.hidden_size),
|
||||
dtype=dtype,
|
||||
device=device),
|
||||
})
|
||||
|
||||
def load_weights(self, weights: Iterable[Tuple[str, torch.Tensor]]):
|
||||
stacked_params_mapping = [
|
||||
# (param_name, shard_name, shard_id)
|
||||
(".qkv_proj", ".q_proj", "q"),
|
||||
(".qkv_proj", ".k_proj", "k"),
|
||||
(".qkv_proj", ".v_proj", "v"),
|
||||
(".gate_up_proj", ".gate_proj", 0),
|
||||
(".gate_up_proj", ".up_proj", 1),
|
||||
]
|
||||
params_dict = dict(self.named_parameters())
|
||||
for name, loaded_weight in weights:
|
||||
if "rotary_emb.inv_freq" in name:
|
||||
continue
|
||||
if ("rotary_emb.cos_cached" in name
|
||||
or "rotary_emb.sin_cached" in name):
|
||||
# Models trained using ColossalAI may include these tensors in
|
||||
# the checkpoint. Skip them.
|
||||
continue
|
||||
if scale_name := get_compressed_tensors_cache_scale(name):
|
||||
# Loading kv cache scales for compressed-tensors quantization
|
||||
param = params_dict[scale_name]
|
||||
weight_loader = getattr(param, "weight_loader",
|
||||
default_weight_loader)
|
||||
loaded_weight = loaded_weight[0]
|
||||
weight_loader(param, loaded_weight)
|
||||
continue
|
||||
for (param_name, weight_name, shard_id) in stacked_params_mapping:
|
||||
if weight_name not in name:
|
||||
continue
|
||||
name = name.replace(weight_name, param_name)
|
||||
# Skip loading extra bias for GPTQ models.
|
||||
if name.endswith(".bias") and name not in params_dict:
|
||||
continue
|
||||
|
||||
if is_pp_missing_parameter(name, self):
|
||||
continue
|
||||
|
||||
param = params_dict[name]
|
||||
weight_loader = param.weight_loader
|
||||
weight_loader(param, loaded_weight, shard_id)
|
||||
|
||||
break
|
||||
else:
|
||||
# Skip loading extra bias for GPTQ models.
|
||||
if name.endswith(".bias") and name not in params_dict:
|
||||
continue
|
||||
# Remapping the name of FP8 kv-scale.
|
||||
name = maybe_remap_kv_scale_name(name, params_dict)
|
||||
if name is None:
|
||||
continue
|
||||
|
||||
if is_pp_missing_parameter(name, self):
|
||||
continue
|
||||
|
||||
param = params_dict[name]
|
||||
weight_loader = getattr(param, "weight_loader",
|
||||
default_weight_loader)
|
||||
weight_loader(param, loaded_weight)
|
||||
|
||||
# If this function is called, it should always initialize KV cache scale
|
||||
# factors (or else raise an exception). Thus, handled exceptions should
|
||||
# make sure to leave KV cache scale factors in a known good (dummy) state
|
||||
def load_kv_cache_scales(self, quantization_param_path: str) -> None:
|
||||
tp_size = get_tensor_model_parallel_world_size()
|
||||
tp_rank = get_tensor_model_parallel_rank()
|
||||
for layer_idx, scaling_factor in kv_cache_scales_loader(
|
||||
quantization_param_path, tp_rank, tp_size,
|
||||
self.config.num_hidden_layers,
|
||||
self.config.__class__.model_type):
|
||||
if not isinstance(self.model.layers[layer_idx], nn.Identity):
|
||||
layer_self_attn = self.model.layers[layer_idx].self_attn
|
||||
|
||||
if is_hip():
|
||||
# The scaling factor convention we are assuming is
|
||||
# quantized_value * scaling_factor ~= true_value
|
||||
# which is consistent with the practice of setting
|
||||
# scaling_factor = tensor_amax / FPtype_max
|
||||
scaling_factor *= 2
|
||||
if hasattr(layer_self_attn, "kv_scale"):
|
||||
layer_self_attn.attn._kv_scale = scaling_factor
|
||||
else:
|
||||
raise RuntimeError("Self attention has no KV cache scaling "
|
||||
"factor attribute!")
|
||||
Reference in New Issue
Block a user