初始化项目,由ModelHub XC社区提供模型
Model: IggyLux/MN-VelvetCafe-RP-12B Source: Original Platform
This commit is contained in:
37
.gitattributes
vendored
Normal file
37
.gitattributes
vendored
Normal file
@@ -0,0 +1,37 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
tekken.json filter=lfs diff=lfs merge=lfs -text
|
||||||
|
Velvet_Cafe.png filter=lfs diff=lfs merge=lfs -text
|
||||||
125
Iggy's-RP-Preset.json
Normal file
125
Iggy's-RP-Preset.json
Normal file
@@ -0,0 +1,125 @@
|
|||||||
|
{
|
||||||
|
"temp": 0.8,
|
||||||
|
"temperature_last": true,
|
||||||
|
"top_p": 0.88,
|
||||||
|
"top_k": 150,
|
||||||
|
"top_a": 0,
|
||||||
|
"tfs": 1,
|
||||||
|
"epsilon_cutoff": 0,
|
||||||
|
"eta_cutoff": 0,
|
||||||
|
"typical_p": 1,
|
||||||
|
"min_p": 0.025,
|
||||||
|
"rep_pen": 1.12,
|
||||||
|
"rep_pen_range": 2048,
|
||||||
|
"rep_pen_decay": 0,
|
||||||
|
"rep_pen_slope": 1,
|
||||||
|
"no_repeat_ngram_size": 0,
|
||||||
|
"penalty_alpha": 0,
|
||||||
|
"num_beams": 1,
|
||||||
|
"length_penalty": 0,
|
||||||
|
"min_length": 0,
|
||||||
|
"encoder_rep_pen": 1,
|
||||||
|
"freq_pen": 0.1,
|
||||||
|
"presence_pen": 0,
|
||||||
|
"skew": 0,
|
||||||
|
"do_sample": true,
|
||||||
|
"early_stopping": false,
|
||||||
|
"dynatemp": false,
|
||||||
|
"min_temp": 0,
|
||||||
|
"max_temp": 2,
|
||||||
|
"dynatemp_exponent": 1,
|
||||||
|
"smoothing_factor": 0.25,
|
||||||
|
"smoothing_curve": 1,
|
||||||
|
"dry_allowed_length": 2,
|
||||||
|
"dry_multiplier": 1,
|
||||||
|
"dry_base": 1.75,
|
||||||
|
"dry_sequence_breakers": "[\"\\n\", \":\", \"\\\"\", \"'\",\"*\", \"<\", \">\", \"/s\", \"[\", \"]\", \"INST\", \"/INST\", \"[INST]\", \"[/INST]\", \"s\", \"|\", \"im_start\", \"im_end\", \"im\", \"<|im_start|>\", \"<|im_end|>\", \"user\", \"assistant\", \"USER\", \"ASSISTANT\", ",
|
||||||
|
"dry_penalty_last_n": 0,
|
||||||
|
"add_bos_token": true,
|
||||||
|
"ban_eos_token": false,
|
||||||
|
"skip_special_tokens": true,
|
||||||
|
"mirostat_mode": 0,
|
||||||
|
"mirostat_tau": 5,
|
||||||
|
"mirostat_eta": 0.1,
|
||||||
|
"guidance_scale": 1,
|
||||||
|
"negative_prompt": "",
|
||||||
|
"grammar_string": "",
|
||||||
|
"json_schema": {},
|
||||||
|
"json_schema_allow_empty": false,
|
||||||
|
"banned_tokens": "",
|
||||||
|
"sampler_priority": [
|
||||||
|
"repetition_penalty",
|
||||||
|
"presence_penalty",
|
||||||
|
"frequency_penalty",
|
||||||
|
"dry",
|
||||||
|
"temperature",
|
||||||
|
"dynamic_temperature",
|
||||||
|
"quadratic_sampling",
|
||||||
|
"top_n_sigma",
|
||||||
|
"top_k",
|
||||||
|
"top_p",
|
||||||
|
"typical_p",
|
||||||
|
"epsilon_cutoff",
|
||||||
|
"eta_cutoff",
|
||||||
|
"tfs",
|
||||||
|
"top_a",
|
||||||
|
"min_p",
|
||||||
|
"mirostat",
|
||||||
|
"xtc",
|
||||||
|
"encoder_repetition_penalty",
|
||||||
|
"no_repeat_ngram"
|
||||||
|
],
|
||||||
|
"samplers": [
|
||||||
|
"penalties",
|
||||||
|
"dry",
|
||||||
|
"top_n_sigma",
|
||||||
|
"top_k",
|
||||||
|
"typ_p",
|
||||||
|
"tfs_z",
|
||||||
|
"typical_p",
|
||||||
|
"top_p",
|
||||||
|
"min_p",
|
||||||
|
"adaptive_p",
|
||||||
|
"xtc",
|
||||||
|
"temperature"
|
||||||
|
],
|
||||||
|
"samplers_priorities": [
|
||||||
|
"dry",
|
||||||
|
"penalties",
|
||||||
|
"no_repeat_ngram",
|
||||||
|
"temperature",
|
||||||
|
"top_nsigma",
|
||||||
|
"top_p_top_k",
|
||||||
|
"top_a",
|
||||||
|
"min_p",
|
||||||
|
"tfs",
|
||||||
|
"eta_cutoff",
|
||||||
|
"epsilon_cutoff",
|
||||||
|
"typical_p",
|
||||||
|
"quadratic",
|
||||||
|
"xtc"
|
||||||
|
],
|
||||||
|
"ignore_eos_token": false,
|
||||||
|
"spaces_between_special_tokens": false,
|
||||||
|
"speculative_ngram": false,
|
||||||
|
"sampler_order": [
|
||||||
|
6,
|
||||||
|
0,
|
||||||
|
1,
|
||||||
|
3,
|
||||||
|
4,
|
||||||
|
2,
|
||||||
|
5
|
||||||
|
],
|
||||||
|
"logit_bias": [],
|
||||||
|
"xtc_threshold": 0.1,
|
||||||
|
"xtc_probability": 0,
|
||||||
|
"nsigma": 0,
|
||||||
|
"min_keep": 0,
|
||||||
|
"extensions": {},
|
||||||
|
"adaptive_target": -0.01,
|
||||||
|
"adaptive_decay": 0.9,
|
||||||
|
"rep_pen_size": 0,
|
||||||
|
"genamt": 356,
|
||||||
|
"max_length": 8192
|
||||||
|
}
|
||||||
139
README.md
Normal file
139
README.md
Normal file
@@ -0,0 +1,139 @@
|
|||||||
|
---
|
||||||
|
base_model:
|
||||||
|
- PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
|
||||||
|
- kyx0r/Neona-12B
|
||||||
|
library_name: transformers
|
||||||
|
tags:
|
||||||
|
- mergekit
|
||||||
|
- merge
|
||||||
|
- roleplaying
|
||||||
|
- RP
|
||||||
|
- Writing
|
||||||
|
- creative
|
||||||
|
- story
|
||||||
|
- fiction
|
||||||
|
- mistral
|
||||||
|
- text
|
||||||
|
- adventure
|
||||||
|
- conversational
|
||||||
|
license: cc-by-4.0
|
||||||
|
---
|
||||||
|
#
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
Static Quants:
|
||||||
|
|
||||||
|
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B-Q4_K_M-GGUF
|
||||||
|
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B-Q8_0-GGUF
|
||||||
|
|
||||||
|
This is my 4th Attempt at a merge of finetunes and the only one I've been happy with. I'm always looking for new merges/finetunes of 12b's due to my 8gb VRAM limitations so I decided to merge my own. I focus mainly on Group Chat RP's personally so when I RP it's mostly +2 Characters if not more.
|
||||||
|
|
||||||
|
My take at what I think makes this merged finetune model good:
|
||||||
|
- 🌟 Strong scene/position/clothing tracking for immersive multi-turn RP
|
||||||
|
- ❤️ Balanced emotional responses — no sudden aggression or refusal spikes unless fitting the narrative of RP (sometimes due to relations you might want this type of response)
|
||||||
|
- 📝 Handles author's notes/system prompts reliably
|
||||||
|
|
||||||
|
My Goal was to take Dans PE hoping that it's character/clothes/personality tracking and consistency would shine when combined with Neona. Neona is really good at adapting to writing styles and instruction following from my experience using it as a daily driver. Combining the two resulted in very good visual focused RP.
|
||||||
|
|
||||||
|
I dislike when models forget clothing, positioning and don't reply in responses detailing changes like that. This often leads to models hallucinating/forgetting positions and clothing specifics that breaks immersion for me. This model seems to feel more visually detailed and descriptive and aware of some of the better things Dan does while keeping some of the instruction following and closer to neurtral emotional responses of Neona.
|
||||||
|
|
||||||
|
I encourage you to try both down below as I really love these models. Thank you for making them @kyx0r and @PocketDoc
|
||||||
|
|
||||||
|
- Dans-PersonalityEngine-V1.3.0-12b is one of those local models that just clicks really well for roleplay. The creators tuned it hard on a ton of different datasets, and they made sure roleplay and creative writing were right up there as core strengths, not some side feature tacked on. That means it naturally picks up on writing good dialogue, keeping descriptions flowing, and building scenes that feel alive instead of stiff or robotic. Unfortunately it looks like it was created before tokenizer issues with mistral nemo were fixed. Due to that it might have format/puncuation issues that sometimes can be a bit annoying. It also has a tendency to favor shorter replies.
|
||||||
|
- https://huggingface.co/PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
|
||||||
|
|
||||||
|
- Neona-12B is a personal favorite of mine. It's a model that seems unbiased in roleplay. If you want slice of life and keeping things SFW it really adapts well. If you want NSFW ERP it can adjust and adapt to that as well too. The model doesn't seem to jump you based on subtle contact or act in extremes like some other finetunes do. I feel like it has a stability emotionally most models don't have. It also seems to handle system prompts/authors notes and instructions well which not all models do.
|
||||||
|
- https://huggingface.co/kyx0r/Neona-12B
|
||||||
|
|
||||||
|
My preferred format for Roleplaying in Sillytavern is:
|
||||||
|
- ChatML
|
||||||
|
or
|
||||||
|
- Mistral V3-Tekken
|
||||||
|
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
My sampler settings for Text Completion preset are included as well with the model, though I personally believe you should find what you like best yourself instead of relying on others. But if you need it, feel free to use it as a place to start:
|
||||||
|
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B/blob/main/Iggy's-RP-Preset.json
|
||||||
|
|
||||||
|
It's set to 8192 context. Which is a great starting point for 8gb VRAM users and with 356 response length to conserve context, I tweak it to 512 for more detail and 1024 for scene climaxes in great detail.
|
||||||
|
|
||||||
|
My preset temp is 0.8 if for some reason you want it to be less creative or more grounded you can go as low as 0.4 (play around with it)
|
||||||
|
|
||||||
|
My setup is:
|
||||||
|
- KoboldCpp GUI for the backend GGUF model loading found here: https://github.com/LostRuins/koboldcpp
|
||||||
|
- Sillytavern for the front end chat interface https://github.com/SillyTavern/SillyTavern (current version 1.16.0)
|
||||||
|
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
- 📖 https://github.com/aikohanasaki/SillyTavern-MemoryBooks/ - For keeping context low and saving older responses as Memories in a Lore Book
|
||||||
|
- 👀 https://github.com/leandrojofre/SillyTavern-Presence - For Group Chats: Using Presence lets you select what characters can see the user and char's messages.
|
||||||
|
- 🗣️ https://github.com/mattjaybe/SillyTavern-EchoChamber - New* I recently found this and thought it was pretty cool, you can have a chat comment on your RP.
|
||||||
|
- 🎯 https://github.com/Samueras/GuidedGenerations-Extension - Helps steer stubborn models, use guides to lock in scenes/details/clothing/positions and more.
|
||||||
|
|
||||||
|
* * *
|
||||||
|
Character Cards and Roleplay Usage/Examples:
|
||||||
|
For some reason a lot of people do things differently (usually based on old tutorials) but I refrain from using opening messages on character cards, example dialogue and things that would sway the model to speak for the user. I also make my own characters after using Chub/Venus/Playground seeing how there's a lot of 1500-2000 token character cards with p-lists like this:
|
||||||
|
|
||||||
|

|
||||||
|
If you see a character card like this it might work, but honestly it's not really neccesary to format using P lists and other stuff like that now days. Models can read standard text formatting just fine.
|
||||||
|
|
||||||
|
Another thing I try to avoid doing or downloading is character cards that use example dialogue, especially one's that speak for the user in examples:
|
||||||
|
|
||||||
|

|
||||||
|
As you can see in this example the creator has mostly example dialogue between a "Interviewer" and the Jinn. This kind of example formatting might lead to the introduction of a 3rd character and start speaking for the "interviewer" or maybe even talk for your character in that way as well.
|
||||||
|
|
||||||
|
I was helping someone troubleshoot using the model and decided (after just waking up without my first cup of coffee so ignore some mistakes in text!) and figured I'd share how I structure my roleplays in sillytavern:
|
||||||
|
|
||||||
|

|
||||||
|
I'm officially naming this the Iggy format since I don't see anyone else start RP's in sillytavern this way.
|
||||||
|
|
||||||
|
For example I'm not a big fan of Character Cards starting the scenario with a first message (tends to set bad habits) so I'll either let my character open up with the first Message and a lot of times before that first message I'll send a system level prompt using /sys to set the scenario in some format of narrative summary.
|
||||||
|
|
||||||
|
If for some reason this leaves context due to limits I'll repurpose my opening scenario into the group chat scenario here:
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
Or pause the summary extenstion and paste it into there, another option is tossing part of it into authors notes (condensed version to save context) or if you are using memory books extension suggested above, it will generally be included in it's worldinfo lore books entry it creates.
|
||||||
|
|
||||||
|
I do my best to share what I've learned having technical limitations over the last few years roleplaying, if you have any issues or problems feel free to ask and I'll try to help!
|
||||||
|
|
||||||
|
This is a merge of pre-trained language models created using [mergekit](https://github.com/cg123/mergekit).
|
||||||
|
|
||||||
|
## Merge Details
|
||||||
|
### Merge Method
|
||||||
|
|
||||||
|
This model was merged using the [SLERP](https://en.wikipedia.org/wiki/Slerp) merge method.
|
||||||
|
|
||||||
|
### Models Merged
|
||||||
|
|
||||||
|
The following models were included in the merge:
|
||||||
|
* [PocketDoc/Dans-PersonalityEngine-V1.3.0-12b](https://huggingface.co/PocketDoc/Dans-PersonalityEngine-V1.3.0-12b)
|
||||||
|
* [kyx0r/Neona-12B](https://huggingface.co/kyx0r/Neona-12B)
|
||||||
|
|
||||||
|
### Configuration
|
||||||
|
|
||||||
|
The following YAML configuration was used to produce this model:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
models:
|
||||||
|
- model: kyx0r/Neona-12B
|
||||||
|
- model: PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
|
||||||
|
merge_method: slerp
|
||||||
|
base_model: kyx0r/Neona-12B
|
||||||
|
parameters:
|
||||||
|
t:
|
||||||
|
- value: 0.2
|
||||||
|
- filter: self_attn
|
||||||
|
value: [0, 0.2, 0.4, 0.6, 0.8, 1]
|
||||||
|
- filter: mlp
|
||||||
|
value: [1, 0.8, 0.6, 0.4, 0.2, 0]
|
||||||
|
dtype: bfloat16
|
||||||
|
chat_template: "chatml"
|
||||||
|
tokenizer:
|
||||||
|
source: "base"
|
||||||
|
```
|
||||||
3
Velvet_Cafe.png
Normal file
3
Velvet_Cafe.png
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:5577e9552667232e50a1d604c2251f061f119997250ef7b9d52eeb6f7b9875ec
|
||||||
|
size 13571518
|
||||||
2
chat_template.jinja
Normal file
2
chat_template.jinja
Normal file
@@ -0,0 +1,2 @@
|
|||||||
|
{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}
|
||||||
|
{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}
|
||||||
26
config.json
Normal file
26
config.json
Normal file
@@ -0,0 +1,26 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"MistralForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"dtype": "bfloat16",
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"head_dim": 128,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 5120,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 14336,
|
||||||
|
"max_position_embeddings": 131072,
|
||||||
|
"model_type": "mistral",
|
||||||
|
"num_attention_heads": 32,
|
||||||
|
"num_hidden_layers": 40,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"rms_norm_eps": 1e-05,
|
||||||
|
"rope_theta": 1000000.0,
|
||||||
|
"sliding_window": null,
|
||||||
|
"tie_word_embeddings": false,
|
||||||
|
"transformers_version": "4.57.6",
|
||||||
|
"use_cache": false,
|
||||||
|
"vocab_size": 131072
|
||||||
|
}
|
||||||
16
mergekit_config.yml
Normal file
16
mergekit_config.yml
Normal file
@@ -0,0 +1,16 @@
|
|||||||
|
models:
|
||||||
|
- model: kyx0r/Neona-12B
|
||||||
|
- model: PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
|
||||||
|
merge_method: slerp
|
||||||
|
base_model: kyx0r/Neona-12B
|
||||||
|
parameters:
|
||||||
|
t:
|
||||||
|
- value: 0.2 # Fallback for unfiltered tensors (e.g., embeddings) - adjust as needed
|
||||||
|
- filter: self_attn
|
||||||
|
value: [0, 0.2, 0.4, 0.6, 0.8, 1]
|
||||||
|
- filter: mlp
|
||||||
|
value: [1, 0.8, 0.6, 0.4, 0.2, 0]
|
||||||
|
dtype: bfloat16
|
||||||
|
chat_template: "chatml" # Added back from your original for completeness
|
||||||
|
tokenizer:
|
||||||
|
source: "base"
|
||||||
3
model-00001-of-00014.safetensors
Normal file
3
model-00001-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:b0457392628e2db935bdef2ee27d8675e7f43ea57df4c96d6e745ba399209ff9
|
||||||
|
size 1342177408
|
||||||
3
model-00002-of-00014.safetensors
Normal file
3
model-00002-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:fe4953150c846b5a65223432e1041b50278a2750ca152cde2440006a2f738005
|
||||||
|
size 1887468816
|
||||||
3
model-00003-of-00014.safetensors
Normal file
3
model-00003-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:ec666d5898974d337dc3206bfa5bf57042f7c3fccc6d4246b09267588c57c3c4
|
||||||
|
size 1929444664
|
||||||
3
model-00004-of-00014.safetensors
Normal file
3
model-00004-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:97d6c29afa79cc225b049f07b3251cb3dcce78aef6399ddd6548a9358d4389ff
|
||||||
|
size 1887522696
|
||||||
3
model-00005-of-00014.safetensors
Normal file
3
model-00005-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:6d67604e35eb062ad7d48f77cd1e07f8fd86a90985e1e6caf3632f3f3917f651
|
||||||
|
size 1929444672
|
||||||
3
model-00006-of-00014.safetensors
Normal file
3
model-00006-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:35db7930e084273a68c321fd36f6fce862634c474425ebc16834863cc22965d2
|
||||||
|
size 1887522680
|
||||||
3
model-00007-of-00014.safetensors
Normal file
3
model-00007-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c0a9b9b84e504cd76d028709746556a9f927c888529a9727f707b7c1e190577b
|
||||||
|
size 1929444672
|
||||||
3
model-00008-of-00014.safetensors
Normal file
3
model-00008-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:ee28f99187f54d76b53fe70bd0789a7f249472eff2b34536407170bf03dca1db
|
||||||
|
size 1887522696
|
||||||
3
model-00009-of-00014.safetensors
Normal file
3
model-00009-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:8dd66451be9fe7a6f2da5a40556915ef835ca2eb78e5abcda0ef64943eaa472a
|
||||||
|
size 1929444664
|
||||||
3
model-00010-of-00014.safetensors
Normal file
3
model-00010-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c8f01b58c3759d9081dd1f6ada8bf3c17992908c82b97e9e33fa97f1f58eee06
|
||||||
|
size 1887522696
|
||||||
3
model-00011-of-00014.safetensors
Normal file
3
model-00011-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:f8c448a266f1c46e7d7751889317fd922bbfa651dfcef80421fcb95855493093
|
||||||
|
size 1929444672
|
||||||
3
model-00012-of-00014.safetensors
Normal file
3
model-00012-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:325db5b4a0be87881bf7ccac99ef657cffe52da258a3474352bb1ea7062b3f7d
|
||||||
|
size 1887522672
|
||||||
3
model-00013-of-00014.safetensors
Normal file
3
model-00013-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:85bfd99ae14331939d43e0058782d22a3e65e391bcc19a436e5021328ce27b07
|
||||||
|
size 1929444640
|
||||||
3
model-00014-of-00014.safetensors
Normal file
3
model-00014-of-00014.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:8b9787f4545baf5e3acec88fe93ed273cabb68b1f8b203d6b09b9c3ccbf26bb3
|
||||||
|
size 251679536
|
||||||
371
model.safetensors.index.json
Normal file
371
model.safetensors.index.json
Normal file
@@ -0,0 +1,371 @@
|
|||||||
|
{
|
||||||
|
"metadata": {
|
||||||
|
"total_size": 24495564800,
|
||||||
|
"mergekit_version": "0.1.4"
|
||||||
|
},
|
||||||
|
"weight_map": {
|
||||||
|
"lm_head.weight": "model-00001-of-00014.safetensors",
|
||||||
|
"model.embed_tokens.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.input_layernorm.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.mlp.down_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.mlp.gate_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.mlp.up_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.post_attention_layernorm.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.self_attn.k_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.self_attn.o_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.self_attn.q_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.0.self_attn.v_proj.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.1.input_layernorm.weight": "model-00002-of-00014.safetensors",
|
||||||
|
"model.layers.1.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.1.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.input_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.10.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.input_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.11.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.12.input_layernorm.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.12.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.12.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
|
||||||
|
"model.layers.12.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.12.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.12.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.12.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.12.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.12.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.input_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.13.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.input_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.14.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.input_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.15.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.16.input_layernorm.weight": "model-00004-of-00014.safetensors",
|
||||||
|
"model.layers.16.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.16.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.input_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.17.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.input_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.18.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.19.input_layernorm.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.19.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.19.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
|
||||||
|
"model.layers.19.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.19.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.19.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.19.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.19.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.19.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.input_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.2.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.input_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.20.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.input_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.21.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.22.input_layernorm.weight": "model-00006-of-00014.safetensors",
|
||||||
|
"model.layers.22.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.22.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.input_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.23.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.input_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.24.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.25.input_layernorm.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.25.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.25.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
|
||||||
|
"model.layers.25.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.25.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.25.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.25.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.25.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.25.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.input_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.26.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.input_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.27.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.input_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.28.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.29.input_layernorm.weight": "model-00008-of-00014.safetensors",
|
||||||
|
"model.layers.29.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.29.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.input_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.3.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.input_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.30.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.31.input_layernorm.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.31.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.31.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
|
||||||
|
"model.layers.31.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.31.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.31.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.31.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.31.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.31.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.input_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.32.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.input_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.33.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.input_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.34.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.35.input_layernorm.weight": "model-00010-of-00014.safetensors",
|
||||||
|
"model.layers.35.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.35.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.input_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.36.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.input_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.37.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.38.input_layernorm.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.38.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.38.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
|
||||||
|
"model.layers.38.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.38.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.38.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.38.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.38.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.38.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.input_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.39.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.input_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.4.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.input_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.5.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.6.input_layernorm.weight": "model-00012-of-00014.safetensors",
|
||||||
|
"model.layers.6.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.6.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.input_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.7.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.input_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.8.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.9.input_layernorm.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.9.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.9.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
|
||||||
|
"model.layers.9.mlp.up_proj.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.layers.9.post_attention_layernorm.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.layers.9.self_attn.k_proj.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.layers.9.self_attn.o_proj.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.layers.9.self_attn.q_proj.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.layers.9.self_attn.v_proj.weight": "model-00014-of-00014.safetensors",
|
||||||
|
"model.norm.weight": "model-00014-of-00014.safetensors"
|
||||||
|
}
|
||||||
|
}
|
||||||
3
tekken.json
Normal file
3
tekken.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:eccd1665d2e477697c33cb7f0daa6f6dfefc57a0a6bceb66d4be52952f827516
|
||||||
|
size 14801223
|
||||||
409625
tokenizer.json
Normal file
409625
tokenizer.json
Normal file
File diff suppressed because it is too large
Load Diff
8013
tokenizer_config.json
Normal file
8013
tokenizer_config.json
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user