初始化项目,由ModelHub XC社区提供模型

Model: IggyLux/MN-VelvetCafe-RP-12B
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-20 09:07:18 +08:00
commit 638533527e
25 changed files with 418402 additions and 0 deletions

37
.gitattributes vendored Normal file
View File

@@ -0,0 +1,37 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tekken.json filter=lfs diff=lfs merge=lfs -text
Velvet_Cafe.png filter=lfs diff=lfs merge=lfs -text

125
Iggy's-RP-Preset.json Normal file
View File

@@ -0,0 +1,125 @@
{
"temp": 0.8,
"temperature_last": true,
"top_p": 0.88,
"top_k": 150,
"top_a": 0,
"tfs": 1,
"epsilon_cutoff": 0,
"eta_cutoff": 0,
"typical_p": 1,
"min_p": 0.025,
"rep_pen": 1.12,
"rep_pen_range": 2048,
"rep_pen_decay": 0,
"rep_pen_slope": 1,
"no_repeat_ngram_size": 0,
"penalty_alpha": 0,
"num_beams": 1,
"length_penalty": 0,
"min_length": 0,
"encoder_rep_pen": 1,
"freq_pen": 0.1,
"presence_pen": 0,
"skew": 0,
"do_sample": true,
"early_stopping": false,
"dynatemp": false,
"min_temp": 0,
"max_temp": 2,
"dynatemp_exponent": 1,
"smoothing_factor": 0.25,
"smoothing_curve": 1,
"dry_allowed_length": 2,
"dry_multiplier": 1,
"dry_base": 1.75,
"dry_sequence_breakers": "[\"\\n\", \":\", \"\\\"\", \"'\",\"*\", \"<\", \">\", \"/s\", \"[\", \"]\", \"INST\", \"/INST\", \"[INST]\", \"[/INST]\", \"s\", \"|\", \"im_start\", \"im_end\", \"im\", \"<|im_start|>\", \"<|im_end|>\", \"user\", \"assistant\", \"USER\", \"ASSISTANT\", ",
"dry_penalty_last_n": 0,
"add_bos_token": true,
"ban_eos_token": false,
"skip_special_tokens": true,
"mirostat_mode": 0,
"mirostat_tau": 5,
"mirostat_eta": 0.1,
"guidance_scale": 1,
"negative_prompt": "",
"grammar_string": "",
"json_schema": {},
"json_schema_allow_empty": false,
"banned_tokens": "",
"sampler_priority": [
"repetition_penalty",
"presence_penalty",
"frequency_penalty",
"dry",
"temperature",
"dynamic_temperature",
"quadratic_sampling",
"top_n_sigma",
"top_k",
"top_p",
"typical_p",
"epsilon_cutoff",
"eta_cutoff",
"tfs",
"top_a",
"min_p",
"mirostat",
"xtc",
"encoder_repetition_penalty",
"no_repeat_ngram"
],
"samplers": [
"penalties",
"dry",
"top_n_sigma",
"top_k",
"typ_p",
"tfs_z",
"typical_p",
"top_p",
"min_p",
"adaptive_p",
"xtc",
"temperature"
],
"samplers_priorities": [
"dry",
"penalties",
"no_repeat_ngram",
"temperature",
"top_nsigma",
"top_p_top_k",
"top_a",
"min_p",
"tfs",
"eta_cutoff",
"epsilon_cutoff",
"typical_p",
"quadratic",
"xtc"
],
"ignore_eos_token": false,
"spaces_between_special_tokens": false,
"speculative_ngram": false,
"sampler_order": [
6,
0,
1,
3,
4,
2,
5
],
"logit_bias": [],
"xtc_threshold": 0.1,
"xtc_probability": 0,
"nsigma": 0,
"min_keep": 0,
"extensions": {},
"adaptive_target": -0.01,
"adaptive_decay": 0.9,
"rep_pen_size": 0,
"genamt": 356,
"max_length": 8192
}

139
README.md Normal file
View File

@@ -0,0 +1,139 @@
---
base_model:
- PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
- kyx0r/Neona-12B
library_name: transformers
tags:
- mergekit
- merge
- roleplaying
- RP
- Writing
- creative
- story
- fiction
- mistral
- text
- adventure
- conversational
license: cc-by-4.0
---
#
![LuxBannerVelvetCafe](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/F-SR5UFdMWsHtiGxJRtlL.png)
![Velvet_Cafe](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/yo1cOV4_gg4I1AvBncAya.jpeg)
Static Quants:
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B-Q4_K_M-GGUF
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B-Q8_0-GGUF
This is my 4th Attempt at a merge of finetunes and the only one I've been happy with. I'm always looking for new merges/finetunes of 12b's due to my 8gb VRAM limitations so I decided to merge my own. I focus mainly on Group Chat RP's personally so when I RP it's mostly +2 Characters if not more.
My take at what I think makes this merged finetune model good:
- 🌟 Strong scene/position/clothing tracking for immersive multi-turn RP
- ❤️ Balanced emotional responses — no sudden aggression or refusal spikes unless fitting the narrative of RP (sometimes due to relations you might want this type of response)
- 📝 Handles author's notes/system prompts reliably
My Goal was to take Dans PE hoping that it's character/clothes/personality tracking and consistency would shine when combined with Neona. Neona is really good at adapting to writing styles and instruction following from my experience using it as a daily driver. Combining the two resulted in very good visual focused RP.
I dislike when models forget clothing, positioning and don't reply in responses detailing changes like that. This often leads to models hallucinating/forgetting positions and clothing specifics that breaks immersion for me. This model seems to feel more visually detailed and descriptive and aware of some of the better things Dan does while keeping some of the instruction following and closer to neurtral emotional responses of Neona.
I encourage you to try both down below as I really love these models. Thank you for making them @kyx0r and @PocketDoc
- Dans-PersonalityEngine-V1.3.0-12b is one of those local models that just clicks really well for roleplay. The creators tuned it hard on a ton of different datasets, and they made sure roleplay and creative writing were right up there as core strengths, not some side feature tacked on. That means it naturally picks up on writing good dialogue, keeping descriptions flowing, and building scenes that feel alive instead of stiff or robotic. Unfortunately it looks like it was created before tokenizer issues with mistral nemo were fixed. Due to that it might have format/puncuation issues that sometimes can be a bit annoying. It also has a tendency to favor shorter replies.
- https://huggingface.co/PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
- Neona-12B is a personal favorite of mine. It's a model that seems unbiased in roleplay. If you want slice of life and keeping things SFW it really adapts well. If you want NSFW ERP it can adjust and adapt to that as well too. The model doesn't seem to jump you based on subtle contact or act in extremes like some other finetunes do. I feel like it has a stability emotionally most models don't have. It also seems to handle system prompts/authors notes and instructions well which not all models do.
- https://huggingface.co/kyx0r/Neona-12B
My preferred format for Roleplaying in Sillytavern is:
- ChatML
or
- Mistral V3-Tekken
![SamplerSettingsModelButton200](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/dxljI9mHZMzcFQsjvNl2F.png)
My sampler settings for Text Completion preset are included as well with the model, though I personally believe you should find what you like best yourself instead of relying on others. But if you need it, feel free to use it as a place to start:
https://huggingface.co/IggyLux/MN-VelvetCafe-RP-12B/blob/main/Iggy's-RP-Preset.json
It's set to 8192 context. Which is a great starting point for 8gb VRAM users and with 356 response length to conserve context, I tweak it to 512 for more detail and 1024 for scene climaxes in great detail.
My preset temp is 0.8 if for some reason you want it to be less creative or more grounded you can go as low as 0.4 (play around with it)
My setup is:
- KoboldCpp GUI for the backend GGUF model loading found here: https://github.com/LostRuins/koboldcpp
- Sillytavern for the front end chat interface https://github.com/SillyTavern/SillyTavern (current version 1.16.0)
![ModelButtonIL200](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/H3TxfWVWCpjU1SXZAw6HX.png)
- 📖 https://github.com/aikohanasaki/SillyTavern-MemoryBooks/ - For keeping context low and saving older responses as Memories in a Lore Book
- 👀 https://github.com/leandrojofre/SillyTavern-Presence - For Group Chats: Using Presence lets you select what characters can see the user and char's messages.
- 🗣️ https://github.com/mattjaybe/SillyTavern-EchoChamber - New* I recently found this and thought it was pretty cool, you can have a chat comment on your RP.
- 🎯 https://github.com/Samueras/GuidedGenerations-Extension - Helps steer stubborn models, use guides to lock in scenes/details/clothing/positions and more.
* * *
Character Cards and Roleplay Usage/Examples:
For some reason a lot of people do things differently (usually based on old tutorials) but I refrain from using opening messages on character cards, example dialogue and things that would sway the model to speak for the user. I also make my own characters after using Chub/Venus/Playground seeing how there's a lot of 1500-2000 token character cards with p-lists like this:
![image](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/GkghPuNEcZ_PxusavrxTn.png)
If you see a character card like this it might work, but honestly it's not really neccesary to format using P lists and other stuff like that now days. Models can read standard text formatting just fine.
Another thing I try to avoid doing or downloading is character cards that use example dialogue, especially one's that speak for the user in examples:
![image](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/LasAhD9QTSR8MNE01F_GZ.png)
As you can see in this example the creator has mostly example dialogue between a "Interviewer" and the Jinn. This kind of example formatting might lead to the introduction of a 3rd character and start speaking for the "interviewer" or maybe even talk for your character in that way as well.
I was helping someone troubleshoot using the model and decided (after just waking up without my first cup of coffee so ignore some mistakes in text!) and figured I'd share how I structure my roleplays in sillytavern:
![image](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/gPI9cvR21R1PYrS-6zEns.png)
I'm officially naming this the Iggy format since I don't see anyone else start RP's in sillytavern this way.
For example I'm not a big fan of Character Cards starting the scenario with a first message (tends to set bad habits) so I'll either let my character open up with the first Message and a lot of times before that first message I'll send a system level prompt using /sys to set the scenario in some format of narrative summary.
If for some reason this leaves context due to limits I'll repurpose my opening scenario into the group chat scenario here:
![image](https://cdn-uploads.huggingface.co/production/uploads/65bbcee1320702b1043ef8ae/8SnDZVcXnrT1926666mCQ.png)
Or pause the summary extenstion and paste it into there, another option is tossing part of it into authors notes (condensed version to save context) or if you are using memory books extension suggested above, it will generally be included in it's worldinfo lore books entry it creates.
I do my best to share what I've learned having technical limitations over the last few years roleplaying, if you have any issues or problems feel free to ask and I'll try to help!
This is a merge of pre-trained language models created using [mergekit](https://github.com/cg123/mergekit).
## Merge Details
### Merge Method
This model was merged using the [SLERP](https://en.wikipedia.org/wiki/Slerp) merge method.
### Models Merged
The following models were included in the merge:
* [PocketDoc/Dans-PersonalityEngine-V1.3.0-12b](https://huggingface.co/PocketDoc/Dans-PersonalityEngine-V1.3.0-12b)
* [kyx0r/Neona-12B](https://huggingface.co/kyx0r/Neona-12B)
### Configuration
The following YAML configuration was used to produce this model:
```yaml
models:
- model: kyx0r/Neona-12B
- model: PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
merge_method: slerp
base_model: kyx0r/Neona-12B
parameters:
t:
- value: 0.2
- filter: self_attn
value: [0, 0.2, 0.4, 0.6, 0.8, 1]
- filter: mlp
value: [1, 0.8, 0.6, 0.4, 0.2, 0]
dtype: bfloat16
chat_template: "chatml"
tokenizer:
source: "base"
```

3
Velvet_Cafe.png Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5577e9552667232e50a1d604c2251f061f119997250ef7b9d52eeb6f7b9875ec
size 13571518

2
chat_template.jinja Normal file
View File

@@ -0,0 +1,2 @@
{% for message in messages %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}
{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}

26
config.json Normal file
View File

@@ -0,0 +1,26 @@
{
"architectures": [
"MistralForCausalLM"
],
"attention_dropout": 0.0,
"bos_token_id": 1,
"dtype": "bfloat16",
"eos_token_id": 2,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 5120,
"initializer_range": 0.02,
"intermediate_size": 14336,
"max_position_embeddings": 131072,
"model_type": "mistral",
"num_attention_heads": 32,
"num_hidden_layers": 40,
"num_key_value_heads": 8,
"rms_norm_eps": 1e-05,
"rope_theta": 1000000.0,
"sliding_window": null,
"tie_word_embeddings": false,
"transformers_version": "4.57.6",
"use_cache": false,
"vocab_size": 131072
}

16
mergekit_config.yml Normal file
View File

@@ -0,0 +1,16 @@
models:
- model: kyx0r/Neona-12B
- model: PocketDoc/Dans-PersonalityEngine-V1.3.0-12b
merge_method: slerp
base_model: kyx0r/Neona-12B
parameters:
t:
- value: 0.2 # Fallback for unfiltered tensors (e.g., embeddings) - adjust as needed
- filter: self_attn
value: [0, 0.2, 0.4, 0.6, 0.8, 1]
- filter: mlp
value: [1, 0.8, 0.6, 0.4, 0.2, 0]
dtype: bfloat16
chat_template: "chatml" # Added back from your original for completeness
tokenizer:
source: "base"

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b0457392628e2db935bdef2ee27d8675e7f43ea57df4c96d6e745ba399209ff9
size 1342177408

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fe4953150c846b5a65223432e1041b50278a2750ca152cde2440006a2f738005
size 1887468816

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ec666d5898974d337dc3206bfa5bf57042f7c3fccc6d4246b09267588c57c3c4
size 1929444664

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:97d6c29afa79cc225b049f07b3251cb3dcce78aef6399ddd6548a9358d4389ff
size 1887522696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6d67604e35eb062ad7d48f77cd1e07f8fd86a90985e1e6caf3632f3f3917f651
size 1929444672

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:35db7930e084273a68c321fd36f6fce862634c474425ebc16834863cc22965d2
size 1887522680

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c0a9b9b84e504cd76d028709746556a9f927c888529a9727f707b7c1e190577b
size 1929444672

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ee28f99187f54d76b53fe70bd0789a7f249472eff2b34536407170bf03dca1db
size 1887522696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8dd66451be9fe7a6f2da5a40556915ef835ca2eb78e5abcda0ef64943eaa472a
size 1929444664

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c8f01b58c3759d9081dd1f6ada8bf3c17992908c82b97e9e33fa97f1f58eee06
size 1887522696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f8c448a266f1c46e7d7751889317fd922bbfa651dfcef80421fcb95855493093
size 1929444672

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:325db5b4a0be87881bf7ccac99ef657cffe52da258a3474352bb1ea7062b3f7d
size 1887522672

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:85bfd99ae14331939d43e0058782d22a3e65e391bcc19a436e5021328ce27b07
size 1929444640

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8b9787f4545baf5e3acec88fe93ed273cabb68b1f8b203d6b09b9c3ccbf26bb3
size 251679536

View File

@@ -0,0 +1,371 @@
{
"metadata": {
"total_size": 24495564800,
"mergekit_version": "0.1.4"
},
"weight_map": {
"lm_head.weight": "model-00001-of-00014.safetensors",
"model.embed_tokens.weight": "model-00002-of-00014.safetensors",
"model.layers.0.input_layernorm.weight": "model-00002-of-00014.safetensors",
"model.layers.0.mlp.down_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.mlp.gate_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.mlp.up_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.post_attention_layernorm.weight": "model-00002-of-00014.safetensors",
"model.layers.0.self_attn.k_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.self_attn.o_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.self_attn.q_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.0.self_attn.v_proj.weight": "model-00002-of-00014.safetensors",
"model.layers.1.input_layernorm.weight": "model-00002-of-00014.safetensors",
"model.layers.1.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.1.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.1.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.input_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.10.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.10.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.10.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.input_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.11.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.mlp.up_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.post_attention_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.11.self_attn.k_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.self_attn.o_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.self_attn.q_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.11.self_attn.v_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.12.input_layernorm.weight": "model-00003-of-00014.safetensors",
"model.layers.12.mlp.down_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.12.mlp.gate_proj.weight": "model-00003-of-00014.safetensors",
"model.layers.12.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.12.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.12.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.12.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.12.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.12.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.input_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.13.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.13.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.13.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.input_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.14.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.14.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.14.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.input_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.15.mlp.down_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.mlp.gate_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.mlp.up_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.post_attention_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.15.self_attn.k_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.self_attn.o_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.self_attn.q_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.15.self_attn.v_proj.weight": "model-00004-of-00014.safetensors",
"model.layers.16.input_layernorm.weight": "model-00004-of-00014.safetensors",
"model.layers.16.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.16.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.16.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.input_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.17.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.17.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.17.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.input_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.18.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.mlp.up_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.post_attention_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.18.self_attn.k_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.self_attn.o_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.self_attn.q_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.18.self_attn.v_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.19.input_layernorm.weight": "model-00005-of-00014.safetensors",
"model.layers.19.mlp.down_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.19.mlp.gate_proj.weight": "model-00005-of-00014.safetensors",
"model.layers.19.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.19.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.19.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.19.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.19.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.19.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.input_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.2.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.2.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.2.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.input_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.20.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.20.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.20.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.input_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.21.mlp.down_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.mlp.gate_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.mlp.up_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.post_attention_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.21.self_attn.k_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.self_attn.o_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.self_attn.q_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.21.self_attn.v_proj.weight": "model-00006-of-00014.safetensors",
"model.layers.22.input_layernorm.weight": "model-00006-of-00014.safetensors",
"model.layers.22.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.22.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.22.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.input_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.23.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.23.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.23.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.input_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.24.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.mlp.up_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.post_attention_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.24.self_attn.k_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.self_attn.o_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.self_attn.q_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.24.self_attn.v_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.25.input_layernorm.weight": "model-00007-of-00014.safetensors",
"model.layers.25.mlp.down_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.25.mlp.gate_proj.weight": "model-00007-of-00014.safetensors",
"model.layers.25.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.25.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.25.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.25.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.25.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.25.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.input_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.26.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.26.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.26.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.input_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.27.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.27.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.27.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.input_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.28.mlp.down_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.mlp.gate_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.mlp.up_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.post_attention_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.28.self_attn.k_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.self_attn.o_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.self_attn.q_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.28.self_attn.v_proj.weight": "model-00008-of-00014.safetensors",
"model.layers.29.input_layernorm.weight": "model-00008-of-00014.safetensors",
"model.layers.29.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.29.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.29.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.input_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.3.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.3.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.3.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.input_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.30.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.mlp.up_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.post_attention_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.30.self_attn.k_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.self_attn.o_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.self_attn.q_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.30.self_attn.v_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.31.input_layernorm.weight": "model-00009-of-00014.safetensors",
"model.layers.31.mlp.down_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.31.mlp.gate_proj.weight": "model-00009-of-00014.safetensors",
"model.layers.31.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.31.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.31.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.31.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.31.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.31.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.input_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.32.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.32.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.32.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.input_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.33.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.33.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.33.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.input_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.34.mlp.down_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.mlp.gate_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.mlp.up_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.post_attention_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.34.self_attn.k_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.self_attn.o_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.self_attn.q_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.34.self_attn.v_proj.weight": "model-00010-of-00014.safetensors",
"model.layers.35.input_layernorm.weight": "model-00010-of-00014.safetensors",
"model.layers.35.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.35.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.35.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.input_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.36.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.36.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.36.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.input_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.37.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.mlp.up_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.post_attention_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.37.self_attn.k_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.self_attn.o_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.self_attn.q_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.37.self_attn.v_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.38.input_layernorm.weight": "model-00011-of-00014.safetensors",
"model.layers.38.mlp.down_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.38.mlp.gate_proj.weight": "model-00011-of-00014.safetensors",
"model.layers.38.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.38.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.38.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.38.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.38.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.38.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.input_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.39.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.39.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.39.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.input_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.4.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.4.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.4.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.input_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.5.mlp.down_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.mlp.gate_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.mlp.up_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.post_attention_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.5.self_attn.k_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.self_attn.o_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.self_attn.q_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.5.self_attn.v_proj.weight": "model-00012-of-00014.safetensors",
"model.layers.6.input_layernorm.weight": "model-00012-of-00014.safetensors",
"model.layers.6.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.6.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.6.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.input_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.7.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.7.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.7.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.input_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.8.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.mlp.up_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.post_attention_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.8.self_attn.k_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.self_attn.o_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.self_attn.q_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.8.self_attn.v_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.9.input_layernorm.weight": "model-00013-of-00014.safetensors",
"model.layers.9.mlp.down_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.9.mlp.gate_proj.weight": "model-00013-of-00014.safetensors",
"model.layers.9.mlp.up_proj.weight": "model-00014-of-00014.safetensors",
"model.layers.9.post_attention_layernorm.weight": "model-00014-of-00014.safetensors",
"model.layers.9.self_attn.k_proj.weight": "model-00014-of-00014.safetensors",
"model.layers.9.self_attn.o_proj.weight": "model-00014-of-00014.safetensors",
"model.layers.9.self_attn.q_proj.weight": "model-00014-of-00014.safetensors",
"model.layers.9.self_attn.v_proj.weight": "model-00014-of-00014.safetensors",
"model.norm.weight": "model-00014-of-00014.safetensors"
}
}

3
tekken.json Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:eccd1665d2e477697c33cb7f0daa6f6dfefc57a0a6bceb66d4be52952f827516
size 14801223

409625
tokenizer.json Normal file

File diff suppressed because it is too large Load Diff

8013
tokenizer_config.json Normal file

File diff suppressed because it is too large Load Diff