初始化项目,由ModelHub XC社区提供模型
Model: Xilabs/calypso-3b-alpha-v2 Source: Original Platform
This commit is contained in:
35
.gitattributes
vendored
Normal file
35
.gitattributes
vendored
Normal file
@@ -0,0 +1,35 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
195
README.md
Normal file
195
README.md
Normal file
@@ -0,0 +1,195 @@
|
||||
---
|
||||
license: cc-by-nc-sa-4.0
|
||||
datasets:
|
||||
- Xilabs/PIPPA-alpaca
|
||||
language:
|
||||
- en
|
||||
pipeline_tag: text-generation
|
||||
---
|
||||
|
||||
# Calypso 3B - Alpha V2 Model Card
|
||||
|
||||
## Model Description
|
||||
|
||||
**Model Name:** Calypso 3B
|
||||
**Version:** Calypso 3B - Alpha V2
|
||||
<img src="https://i.imgur.com/zhLV66U.jpg" alt="Calypso" width="300">
|
||||
|
||||
**Based on:** [openlm-research/open_llama_3b_v2](https://huggingface.co/openlm-research/open_llama_3b_v2)
|
||||
|
||||
Calypso 3B is a language model designed for one-on-one chat interactions with a character or persona. It has been finetuned on the PIPPA-Alpaca dataset and a private dataset of human-generated chats. The model is particularly suited for providing conversational responses in a variety of contexts, making it suitable for role-playing, or one-on-one chatting.
|
||||
|
||||
## Intended Use
|
||||
|
||||
Calypso 3B is intended to facilitate engaging and interactive one-on-one chat experiences.
|
||||
|
||||
## Limitations and Ethical Considerations
|
||||
|
||||
- **Safety Note:** Calypso 3B can produce content that may not be safe for all audiences. It may generate inappropriate, offensive, or sensitive content. User discretion is advised.
|
||||
|
||||
- **Factual Accuracy:** The model's responses may not always be factually accurate. It should not be relied upon to provide accurate information, especially in critical or sensitive contexts.
|
||||
|
||||
- **Bias and Fairness:** As with many language models, Calypso 3B might inadvertently exhibit biases present in the training data. Efforts have been made to mitigate this, but biases may still be present.
|
||||
|
||||
## Example Usage
|
||||
|
||||
```python
|
||||
import gradio as gr
|
||||
from transformers import LlamaTokenizer, LlamaForCausalLM, GenerationConfig
|
||||
import torch
|
||||
from transformers import LlamaForCausalLM, LlamaTokenizer
|
||||
|
||||
|
||||
class Chat:
|
||||
def __init__(self, model, tokenizer, conv_prompt, user_alias='User', character_name='Chatbot', message_history=[], chat_buffer_size=10):
|
||||
self.model = model
|
||||
self.tokenizer = tokenizer
|
||||
self.conv_prompt = conv_prompt
|
||||
self.user_alias = user_alias
|
||||
self.character_name = character_name
|
||||
self.chat_buffer_size = chat_buffer_size
|
||||
self.message_history = message_history
|
||||
self.display_messages = []
|
||||
for message_pairs in message_history:
|
||||
message1, message2 = message_pairs
|
||||
self.display_messages.append([message1['text'], message2['text']])
|
||||
|
||||
def evaluate(self, message, temperature=0.6, top_p=0.75, top_k=50, num_beams=5, max_new_tokens=256, repetition_penalty=1.4, **kwargs):
|
||||
prompt = self.prompt_gen_chat(self.message_history, message)
|
||||
inputs = self.tokenizer(prompt, return_tensors="pt")
|
||||
input_ids = inputs["input_ids"].to(self.model.device)
|
||||
generation_config = GenerationConfig(
|
||||
temperature=temperature,
|
||||
top_p=top_p,
|
||||
top_k=top_k,
|
||||
num_beams=num_beams,
|
||||
early_stopping=True,
|
||||
repetition_penalty=repetition_penalty,
|
||||
**kwargs,
|
||||
)
|
||||
with torch.no_grad():
|
||||
generation_output = self.model.generate(
|
||||
input_ids=input_ids,
|
||||
generation_config=generation_config,
|
||||
return_dict_in_generate=True,
|
||||
output_scores=True,
|
||||
max_new_tokens=max_new_tokens,
|
||||
)
|
||||
s = generation_output.sequences[0]
|
||||
output = self.tokenizer.decode(s, skip_special_tokens=True)
|
||||
split_str = """### Response:\n{self.character_name}:"""
|
||||
output = output.split(split_str)[1].strip()
|
||||
return output
|
||||
|
||||
def gradio_helper(self, message):
|
||||
# make response
|
||||
response = self.evaluate(message)
|
||||
# update message history
|
||||
self.message_history.append(
|
||||
(
|
||||
{"speaker": self.user_alias, "text": message},
|
||||
{"speaker": self.character_name, "text": response},
|
||||
)
|
||||
)
|
||||
if len(self.message_history) > self.chat_buffer_size:
|
||||
self.message_history = self.message_history[-self.chat_buffer_size:]
|
||||
# update display messages
|
||||
self.display_messages.append([message, response])
|
||||
return self.display_messages
|
||||
|
||||
def prompt_gen_chat(self, message_history, message):
|
||||
past_dialogue = []
|
||||
for message_pairs in message_history:
|
||||
message1, message2 = message_pairs
|
||||
past_dialogue.append(f"{message1['speaker']}: {message1['text']}")
|
||||
past_dialogue.append(f"{message2['speaker']}: {message2['text']}")
|
||||
past_dialogue_formatted = "\n".join(past_dialogue)
|
||||
|
||||
prompt = f"""Below is an instruction that describes a task, paired with an input that provides further context. Write a response that appropriately completes the request.
|
||||
|
||||
### Instruction:
|
||||
{self.conv_prompt}
|
||||
|
||||
This is the conversation between {self.user_alias} and {self.character_name} till now:
|
||||
{past_dialogue_formatted}
|
||||
|
||||
Continuing from the previous conversation, write what {self.character_name} says to {self.user_alias}:
|
||||
### Input:
|
||||
{self.user_alias}: {message}
|
||||
### Response:
|
||||
{self.character_name}:"""
|
||||
|
||||
return prompt
|
||||
|
||||
def launch_gradio(self):
|
||||
with gr.Blocks(theme="JohnSmith9982/small_and_pretty") as demo:
|
||||
chatbot = gr.Chatbot(elem_id="chatbot")
|
||||
with gr.Row():
|
||||
txt = gr.Textbox(show_label=False,
|
||||
placeholder="Enter text and press enter")
|
||||
txt.submit(self.gradio_helper, txt, chatbot)
|
||||
txt.submit(lambda: "", None, txt)
|
||||
|
||||
demo.launch(debug=True, share=True)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
model_path = "Xilabs/calypso-3b-alpha-v2"
|
||||
load_in_8bit = False
|
||||
model = LlamaForCausalLM.from_pretrained(
|
||||
model_path, device_map="auto", load_in_8bit=load_in_8bit)
|
||||
tokenizer = LlamaTokenizer.from_pretrained(model_path)
|
||||
conv_prompt = "Two people are texting each other on a messaging platform."
|
||||
message_history = [
|
||||
(
|
||||
{
|
||||
"speaker": "Bob",
|
||||
"text": "Hey, Alice! How are you doing? What's the status on those reports?",
|
||||
},
|
||||
{
|
||||
"speaker": "Alice",
|
||||
"text": "Hey, Bob! I'm doing well. I'm almost done with the reports. I'll send them to you by the end of the day.",
|
||||
},
|
||||
),
|
||||
(
|
||||
{
|
||||
"speaker": "Bob",
|
||||
"text": "That's great! Thanks, Alice. I'll be waiting for them. Btw, I have approved your leave for next week.",
|
||||
},
|
||||
{
|
||||
"speaker": "Alice",
|
||||
"text": "Oh, thanks, Bob! I really appreciate it. I will be sure to send you the reports before I leave. Anything else you need from me?",
|
||||
},
|
||||
)
|
||||
]
|
||||
|
||||
chat_instance = Chat(model, tokenizer, conv_prompt, user_alias='Bob',
|
||||
character_name='Alice', message_history=message_history)
|
||||
chat_instance.launch_gradio()
|
||||
```
|
||||
|
||||
## Future Improvements
|
||||
|
||||
Calypso 3B is an ongoing project, and future iterations will focus on enhancing safety, improving factual accuracy, and reducing biases in its responses. The development team is committed to addressing user feedback and continuously improving the model's performance.
|
||||
|
||||
## Licensing and Commercial Use
|
||||
|
||||
Larger and more permissive versions of Calypso will be released in the future. If you're interested in using Calypso 3B or its future iterations for commercial purposes, obtaining a license, or accessing the model via an API, please reach out to us for more information.
|
||||
|
||||
---
|
||||
|
||||
**Disclaimer:** This model card is provided for informational purposes only. Users are responsible for using the model in accordance with applicable laws and ethical considerations.
|
||||
# [Open LLM Leaderboard Evaluation Results](https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard)
|
||||
Detailed results can be found [here](https://huggingface.co/datasets/open-llm-leaderboard/details_Xilabs__calypso-3b-alpha-v2)
|
||||
|
||||
| Metric | Value |
|
||||
|-----------------------|---------------------------|
|
||||
| Avg. | 37.52 |
|
||||
| ARC (25-shot) | 41.55 |
|
||||
| HellaSwag (10-shot) | 71.48 |
|
||||
| MMLU (5-shot) | 25.82 |
|
||||
| TruthfulQA (0-shot) | 35.73 |
|
||||
| Winogrande (5-shot) | 65.27 |
|
||||
| GSM8K (5-shot) | 0.68 |
|
||||
| DROP (3-shot) | 22.08 |
|
||||
|
||||
26
config.json
Normal file
26
config.json
Normal file
@@ -0,0 +1,26 @@
|
||||
{
|
||||
"_name_or_path": "CobraMamba/mamba-gpt-3b-v3",
|
||||
"architectures": [
|
||||
"LlamaForCausalLM"
|
||||
],
|
||||
"bos_token_id": 1,
|
||||
"eos_token_id": 2,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 3200,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 8640,
|
||||
"max_position_embeddings": 2048,
|
||||
"model_type": "llama",
|
||||
"num_attention_heads": 32,
|
||||
"num_hidden_layers": 26,
|
||||
"num_key_value_heads": 32,
|
||||
"pad_token_id": 0,
|
||||
"pretraining_tp": 1,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"tie_word_embeddings": false,
|
||||
"torch_dtype": "float32",
|
||||
"transformers_version": "4.32.0.dev0",
|
||||
"use_cache": true,
|
||||
"vocab_size": 32000
|
||||
}
|
||||
7
generation_config.json
Normal file
7
generation_config.json
Normal file
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"_from_model_config": true,
|
||||
"bos_token_id": 1,
|
||||
"eos_token_id": 2,
|
||||
"pad_token_id": 0,
|
||||
"transformers_version": "4.32.0.dev0"
|
||||
}
|
||||
3
pytorch_model-00001-of-00005.bin
Normal file
3
pytorch_model-00001-of-00005.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6c2dc6c6a940078bb264710e752a5740501595a6e658937478acbda3d999245d
|
||||
size 2969744687
|
||||
3
pytorch_model-00002-of-00005.bin
Normal file
3
pytorch_model-00002-of-00005.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:de1e920672c64267f9b34019be0f6189aceb6ce961c9321fb3b5839ebc067ac4
|
||||
size 2973868379
|
||||
3
pytorch_model-00003-of-00005.bin
Normal file
3
pytorch_model-00003-of-00005.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:03263a163b1019a56b009f8134cbdd09529616d9587c25fed1820b7372d3ebe9
|
||||
size 2973868443
|
||||
3
pytorch_model-00004-of-00005.bin
Normal file
3
pytorch_model-00004-of-00005.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:38c91a4c9a6a16436b84bf8fe678e1210dfe32eaa37f3c8248da855822119573
|
||||
size 2973868443
|
||||
3
pytorch_model-00005-of-00005.bin
Normal file
3
pytorch_model-00005-of-00005.bin
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:083dca58ed366002a0cebfd3d49118fc47cea8dd0b3d514ff3c9b9a6be2adb15
|
||||
size 1814626933
|
||||
244
pytorch_model.bin.index.json
Normal file
244
pytorch_model.bin.index.json
Normal file
@@ -0,0 +1,244 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 13705894400
|
||||
},
|
||||
"weight_map": {
|
||||
"lm_head.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.embed_tokens.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.input_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.mlp.down_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.mlp.gate_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.mlp.up_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.post_attention_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.self_attn.o_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.0.self_attn.v_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.input_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.mlp.down_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.mlp.gate_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.mlp.up_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.post_attention_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.self_attn.o_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.1.self_attn.v_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.10.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.10.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.11.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.11.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.11.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.11.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.12.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.13.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.14.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.15.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.input_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.mlp.down_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.mlp.gate_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.mlp.up_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.post_attention_layernorm.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.self_attn.o_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.16.self_attn.v_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.17.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.self_attn.k_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.17.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.17.self_attn.q_proj.weight": "pytorch_model-00003-of-00005.bin",
|
||||
"model.layers.17.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.18.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.19.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.2.input_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.mlp.down_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.mlp.gate_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.mlp.up_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.post_attention_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.self_attn.o_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.2.self_attn.v_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.20.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.20.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.21.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.input_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.mlp.down_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.mlp.gate_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.mlp.up_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.post_attention_layernorm.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.self_attn.o_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.22.self_attn.v_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.23.input_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.mlp.down_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.mlp.gate_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.mlp.up_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.post_attention_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.self_attn.k_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.23.self_attn.o_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.23.self_attn.q_proj.weight": "pytorch_model-00004-of-00005.bin",
|
||||
"model.layers.23.self_attn.v_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.input_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.mlp.down_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.mlp.gate_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.mlp.up_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.post_attention_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.self_attn.k_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.self_attn.o_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.self_attn.q_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.24.self_attn.v_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.input_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.mlp.down_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.mlp.gate_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.mlp.up_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.post_attention_layernorm.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.self_attn.k_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.self_attn.o_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.self_attn.q_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.25.self_attn.v_proj.weight": "pytorch_model-00005-of-00005.bin",
|
||||
"model.layers.3.input_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.mlp.down_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.mlp.gate_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.mlp.up_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.post_attention_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.self_attn.o_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.3.self_attn.v_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.input_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.mlp.down_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.mlp.gate_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.mlp.up_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.post_attention_layernorm.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.self_attn.o_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.4.self_attn.v_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.5.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.self_attn.k_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.5.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.5.self_attn.q_proj.weight": "pytorch_model-00001-of-00005.bin",
|
||||
"model.layers.5.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.6.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.7.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.8.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.input_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.mlp.down_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.mlp.gate_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.mlp.up_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.post_attention_layernorm.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.self_attn.k_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.self_attn.o_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.self_attn.q_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.layers.9.self_attn.v_proj.weight": "pytorch_model-00002-of-00005.bin",
|
||||
"model.norm.weight": "pytorch_model-00005-of-00005.bin"
|
||||
}
|
||||
}
|
||||
24
special_tokens_map.json
Normal file
24
special_tokens_map.json
Normal file
@@ -0,0 +1,24 @@
|
||||
{
|
||||
"bos_token": {
|
||||
"content": "<s>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"eos_token": {
|
||||
"content": "</s>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"pad_token": "<unk>",
|
||||
"unk_token": {
|
||||
"content": "<unk>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
}
|
||||
}
|
||||
3
tokenizer.model
Normal file
3
tokenizer.model
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:91b289e85fa20fd375d8b33dc12f77616f18abc6359804471d1fafcb425fecb8
|
||||
size 511574
|
||||
36
tokenizer_config.json
Normal file
36
tokenizer_config.json
Normal file
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"add_bos_token": true,
|
||||
"add_eos_token": false,
|
||||
"bos_token": {
|
||||
"__type": "AddedToken",
|
||||
"content": "<s>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": {
|
||||
"__type": "AddedToken",
|
||||
"content": "</s>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"legacy": true,
|
||||
"model_max_length": 2048,
|
||||
"pad_token": null,
|
||||
"padding_side": "left",
|
||||
"sp_model_kwargs": {},
|
||||
"spaces_between_special_tokens": false,
|
||||
"tokenizer_class": "LlamaTokenizer",
|
||||
"unk_token": {
|
||||
"__type": "AddedToken",
|
||||
"content": "<unk>",
|
||||
"lstrip": false,
|
||||
"normalized": true,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user