初始化项目,由ModelHub XC社区提供模型

Model: Josephgflowers/Cinder-Phi-2-Test-1
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-07-06 17:43:15 +08:00
commit 3b9c1ee745
23 changed files with 52745 additions and 0 deletions

49
.gitattributes vendored Normal file
View File

@@ -0,0 +1,49 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.gguf* filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
tokenizer.json filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:226ca41770c8f1195fc38c9f6d8dfbf4cffc8c96bce2e12b248b1afbf41704fe
size 5563095616

21
README.md Normal file
View File

@@ -0,0 +1,21 @@
---
license: mit
widget:
- text: >
<|system|>
You are a helpful assistant</s>
<|user|>
Can you explain to me how quantum computing works?</s>
<|assistant|>
---
Quick update, 2/20/23, testing is going great. I am really enjoying this version of Cinder. More information coming. Training data similar to openhermes2.5 with some added math, STEM, and reasoning mostly from OpenOrca. As well as Cinder character specific data.
Model Overview Cinder is an AI chatbot tailored for engaging users in scientific and educational conversations, offering companionship, and sparking imaginative exploration.
![image/png](https://cdn-uploads.huggingface.co/production/uploads/6328952f798f8d122ce62a44/obCyZSvfUefEWrOXaeB3o.png)

42
added_tokens.json Normal file
View File

@@ -0,0 +1,42 @@
{
"\t\t": 50294,
"\t\t\t": 50293,
"\t\t\t\t": 50292,
"\t\t\t\t\t": 50291,
"\t\t\t\t\t\t": 50290,
"\t\t\t\t\t\t\t": 50289,
"\t\t\t\t\t\t\t\t": 50288,
"\t\t\t\t\t\t\t\t\t": 50287,
" ": 50286,
" ": 50285,
" ": 50284,
" ": 50283,
" ": 50282,
" ": 50281,
" ": 50280,
" ": 50279,
" ": 50278,
" ": 50277,
" ": 50276,
" ": 50275,
" ": 50274,
" ": 50273,
" ": 50272,
" ": 50271,
" ": 50270,
" ": 50269,
" ": 50268,
" ": 50267,
" ": 50266,
" ": 50265,
" ": 50264,
" ": 50263,
" ": 50262,
" ": 50261,
" ": 50260,
" ": 50259,
" ": 50258,
" ": 50257,
"<|im_end|>": 50295,
"<|im_start|>": 50296
}

34
config.json Normal file
View File

@@ -0,0 +1,34 @@
{
"_name_or_path": "/content/phi",
"architectures": [
"PhiForCausalLM"
],
"attention_dropout": 0.0,
"auto_map": {
"AutoConfig": "configuration_phi.PhiConfig",
"AutoModelForCausalLM": "modeling_phi.PhiForCausalLM"
},
"bos_token_id": 50256,
"embd_pdrop": 0.0,
"eos_token_id": 50297,
"hidden_act": "gelu_new",
"hidden_size": 2560,
"initializer_range": 0.02,
"intermediate_size": 10240,
"layer_norm_eps": 1e-05,
"max_position_embeddings": 2048,
"model_type": "phi",
"num_attention_heads": 32,
"num_hidden_layers": 32,
"num_key_value_heads": 32,
"partial_rotary_factor": 0.4,
"qk_layernorm": false,
"resid_pdrop": 0.1,
"rope_scaling": null,
"rope_theta": 10000.0,
"tie_word_embeddings": false,
"torch_dtype": "float32",
"transformers_version": "4.38.0.dev0",
"use_cache": true,
"vocab_size": 51200
}

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "text-generation", "allow_remote": true}

193
configuration_phi.py Normal file
View File

@@ -0,0 +1,193 @@
# coding=utf-8
# Copyright 2023 Microsoft and the HuggingFace Inc. team. All rights reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
""" Phi model configuration"""
from transformers.configuration_utils import PretrainedConfig
from transformers.utils import logging
logger = logging.get_logger(__name__)
PHI_PRETRAINED_CONFIG_ARCHIVE_MAP = {
"microsoft/phi-2": "https://huggingface.co/microsoft/phi-2/resolve/main/config.json",
}
class PhiConfig(PretrainedConfig):
r"""
This is the configuration class to store the configuration of a [`PhiModel`]. It is used to instantiate an Phi
model according to the specified arguments, defining the model architecture. Instantiating a configuration with the
defaults will yield a similar configuration to that of the Phi
[microsoft/phi-1](https://huggingface.co/microsoft/phi-1).
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
documentation from [`PretrainedConfig`] for more information.
Args:
vocab_size (`int`, *optional*, defaults to 51200):
Vocabulary size of the Phi model. Defines the number of different tokens that can be represented by the
`inputs_ids` passed when calling [`PhiModel`].
hidden_size (`int`, *optional*, defaults to 2048):
Dimension of the hidden representations.
intermediate_size (`int`, *optional*, defaults to 8192):
Dimension of the MLP representations.
num_hidden_layers (`int`, *optional*, defaults to 24):
Number of hidden layers in the Transformer decoder.
num_attention_heads (`int`, *optional*, defaults to 32):
Number of attention heads for each attention layer in the Transformer decoder.
num_key_value_heads (`int`, *optional*):
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
`num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
by meanpooling all the original heads within that group. For more details checkout [this
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to
`num_attention_heads`.
resid_pdrop (`float`, *optional*, defaults to 0.0):
Dropout probability for mlp outputs.
embd_pdrop (`int`, *optional*, defaults to 0.0):
The dropout ratio for the embeddings.
attention_dropout (`float`, *optional*, defaults to 0.0):
The dropout ratio after computing the attention scores.
hidden_act (`str` or `function`, *optional*, defaults to `"gelu_new"`):
The non-linear activation function (function or string) in the decoder.
max_position_embeddings (`int`, *optional*, defaults to 2048):
The maximum sequence length that this model might ever be used with. Phi-1 and Phi-1.5 supports up to 2048
tokens.
initializer_range (`float`, *optional*, defaults to 0.02):
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
layer_norm_eps (`float`, *optional*, defaults to 1e-05):
The epsilon used by the rms normalization layers.
use_cache (`bool`, *optional*, defaults to `True`):
Whether or not the model should return the last key/values attentions (not used by all models). Only
relevant if `config.is_decoder=True`. Whether to tie weight embeddings or not.
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
Whether to tie weight embeddings
rope_theta (`float`, *optional*, defaults to 10000.0):
The base period of the RoPE embeddings.
rope_scaling (`Dict`, *optional*):
Dictionary containing the scaling configuration for the RoPE embeddings. Currently supports two scaling
strategies: linear and dynamic. Their scaling factor must be an float greater than 1. The expected format
is `{"type": strategy name, "factor": scaling factor}`. When using this flag, don't update
`max_position_embeddings` to the expected new maximum. See the following thread for more information on how
these scaling strategies behave:
https://www.reddit.com/r/LocalPersimmon/comments/14mrgpr/dynamically_scaled_rope_further_increases/. This
is an experimental feature, subject to breaking API changes in future versions.
partial_rotary_factor (`float`, *optional*, defaults to 0.5):
Percentage of the query and keys which will have rotary embedding.
qk_layernorm (`bool`, *optional*, defaults to `False`):
Whether or not to normalize the Queries and Keys after projecting the hidden states.
bos_token_id (`int`, *optional*, defaults to 1):
Denotes beginning of sequences token id.
eos_token_id (`int`, *optional*, defaults to 2):
Denotes end of sequences token id.
Example:
```python
>>> from transformers import PhiModel, PhiConfig
>>> # Initializing a Phi-1 style configuration
>>> configuration = PhiConfig.from_pretrained("microsoft/phi-1")
>>> # Initializing a model from the configuration
>>> model = PhiModel(configuration)
>>> # Accessing the model configuration
>>> configuration = model.config
```"""
model_type = "phi"
keys_to_ignore_at_inference = ["past_key_values"]
def __init__(
self,
vocab_size=51200,
hidden_size=2048,
intermediate_size=8192,
num_hidden_layers=24,
num_attention_heads=32,
num_key_value_heads=None,
resid_pdrop=0.0,
embd_pdrop=0.0,
attention_dropout=0.0,
hidden_act="gelu_new",
max_position_embeddings=2048,
initializer_range=0.02,
layer_norm_eps=1e-5,
use_cache=True,
tie_word_embeddings=False,
rope_theta=10000.0,
rope_scaling=None,
partial_rotary_factor=0.5,
qk_layernorm=False,
bos_token_id=1,
eos_token_id=2,
**kwargs,
):
self.vocab_size = vocab_size
self.hidden_size = hidden_size
self.intermediate_size = intermediate_size
self.num_hidden_layers = num_hidden_layers
self.num_attention_heads = num_attention_heads
if num_key_value_heads is None:
num_key_value_heads = num_attention_heads
self.num_key_value_heads = num_key_value_heads
self.resid_pdrop = resid_pdrop
self.embd_pdrop = embd_pdrop
self.attention_dropout = attention_dropout
self.hidden_act = hidden_act
self.max_position_embeddings = max_position_embeddings
self.initializer_range = initializer_range
self.layer_norm_eps = layer_norm_eps
self.use_cache = use_cache
self.rope_theta = rope_theta
self.rope_scaling = rope_scaling
self.partial_rotary_factor = partial_rotary_factor
self.qk_layernorm = qk_layernorm
self._rope_scaling_validation()
super().__init__(
bos_token_id=bos_token_id,
eos_token_id=eos_token_id,
tie_word_embeddings=tie_word_embeddings,
**kwargs,
)
# Copied from transformers.models.llama.configuration_llama.LlamaConfig._rope_scaling_validation
def _rope_scaling_validation(self):
"""
Validate the `rope_scaling` configuration.
"""
if self.rope_scaling is None:
return
if not isinstance(self.rope_scaling, dict) or len(self.rope_scaling) != 2:
raise ValueError(
"`rope_scaling` must be a dictionary with with two fields, `type` and `factor`, "
f"got {self.rope_scaling}"
)
rope_scaling_type = self.rope_scaling.get("type", None)
rope_scaling_factor = self.rope_scaling.get("factor", None)
if rope_scaling_type is None or rope_scaling_type not in ["linear", "dynamic"]:
raise ValueError(
f"`rope_scaling`'s type field must be one of ['linear', 'dynamic'], got {rope_scaling_type}"
)
if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor <= 1.0:
raise ValueError(f"`rope_scaling`'s factor field must be a float > 1, got {rope_scaling_factor}")

6
generation_config.json Normal file
View File

@@ -0,0 +1,6 @@
{
"_from_model_config": true,
"bos_token_id": 50256,
"eos_token_id": 50297,
"transformers_version": "4.38.0.dev0"
}

50001
merges.txt Normal file

File diff suppressed because it is too large Load Diff

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fba8bb640c4502e6b8f8a8aad4d4d7ac1d2696d4ad85422a519a89325bc20d4c
size 4982355512

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:01ae634451b3ef9974477a0701dc2c63dda080e6ffdceffc2616a3b6166c4348
size 4982541984

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3c9537817f6da5721ca58b48977f4c9958c18fe91f1ce0b844650cd741ed80ae
size 1153887616

View File

@@ -0,0 +1,460 @@
{
"metadata": {
"total_size": 11118735360
},
"weight_map": {
"lm_head.bias": "model-00003-of-00003.safetensors",
"lm_head.weight": "model-00003-of-00003.safetensors",
"model.embed_tokens.weight": "model-00001-of-00003.safetensors",
"model.final_layernorm.bias": "model-00003-of-00003.safetensors",
"model.final_layernorm.weight": "model-00003-of-00003.safetensors",
"model.layers.0.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.0.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.0.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.0.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.0.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.0.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.0.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.1.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.1.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.1.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.1.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.1.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.1.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.1.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.10.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.10.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.10.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.10.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.10.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.10.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.10.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.11.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.11.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.11.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.11.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.11.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.11.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.11.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.12.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.12.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.12.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.12.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.12.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.12.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.12.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.13.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.13.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.13.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.13.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.13.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.13.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.13.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.14.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.14.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.14.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.14.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.14.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.14.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.14.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.14.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.14.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.14.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.14.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.14.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.14.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.14.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.15.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.15.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.15.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.15.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.15.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.15.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.15.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.16.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.16.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.16.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.16.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.16.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.16.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.16.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.17.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.17.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.17.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.17.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.17.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.17.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.17.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.18.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.18.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.18.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.18.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.18.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.18.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.18.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.19.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.19.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.19.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.19.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.19.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.19.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.19.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.2.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.2.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.2.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.2.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.2.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.2.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.2.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.20.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.20.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.20.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.20.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.20.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.20.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.20.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.21.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.21.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.21.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.21.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.21.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.21.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.21.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.22.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.22.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.22.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.22.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.22.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.22.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.22.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.23.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.23.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.23.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.23.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.23.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.23.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.23.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.24.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.24.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.24.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.24.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.24.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.24.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.24.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.25.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.25.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.25.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.25.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.25.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.25.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.25.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.26.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.26.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.26.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.26.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.26.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.26.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.26.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.27.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.27.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.27.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.27.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.27.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.27.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.27.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.28.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.28.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.28.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.28.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.28.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.28.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.28.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.29.input_layernorm.bias": "model-00002-of-00003.safetensors",
"model.layers.29.input_layernorm.weight": "model-00002-of-00003.safetensors",
"model.layers.29.mlp.fc1.bias": "model-00002-of-00003.safetensors",
"model.layers.29.mlp.fc1.weight": "model-00002-of-00003.safetensors",
"model.layers.29.mlp.fc2.bias": "model-00002-of-00003.safetensors",
"model.layers.29.mlp.fc2.weight": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.dense.bias": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.dense.weight": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.k_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.k_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.q_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.q_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.v_proj.bias": "model-00002-of-00003.safetensors",
"model.layers.29.self_attn.v_proj.weight": "model-00002-of-00003.safetensors",
"model.layers.3.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.3.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.3.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.3.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.3.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.3.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.3.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.30.input_layernorm.bias": "model-00003-of-00003.safetensors",
"model.layers.30.input_layernorm.weight": "model-00003-of-00003.safetensors",
"model.layers.30.mlp.fc1.bias": "model-00003-of-00003.safetensors",
"model.layers.30.mlp.fc1.weight": "model-00003-of-00003.safetensors",
"model.layers.30.mlp.fc2.bias": "model-00003-of-00003.safetensors",
"model.layers.30.mlp.fc2.weight": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.dense.bias": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.dense.weight": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.k_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.k_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.q_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.q_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.v_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.30.self_attn.v_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.31.input_layernorm.bias": "model-00003-of-00003.safetensors",
"model.layers.31.input_layernorm.weight": "model-00003-of-00003.safetensors",
"model.layers.31.mlp.fc1.bias": "model-00003-of-00003.safetensors",
"model.layers.31.mlp.fc1.weight": "model-00003-of-00003.safetensors",
"model.layers.31.mlp.fc2.bias": "model-00003-of-00003.safetensors",
"model.layers.31.mlp.fc2.weight": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.dense.bias": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.dense.weight": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.k_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.k_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.q_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.q_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.v_proj.bias": "model-00003-of-00003.safetensors",
"model.layers.31.self_attn.v_proj.weight": "model-00003-of-00003.safetensors",
"model.layers.4.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.4.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.4.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.4.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.4.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.4.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.4.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.5.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.5.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.5.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.5.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.5.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.5.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.5.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.6.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.6.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.6.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.6.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.6.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.6.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.6.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.7.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.7.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.7.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.7.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.7.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.7.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.7.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.8.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.8.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.8.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.8.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.8.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.8.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.8.self_attn.v_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.9.input_layernorm.bias": "model-00001-of-00003.safetensors",
"model.layers.9.input_layernorm.weight": "model-00001-of-00003.safetensors",
"model.layers.9.mlp.fc1.bias": "model-00001-of-00003.safetensors",
"model.layers.9.mlp.fc1.weight": "model-00001-of-00003.safetensors",
"model.layers.9.mlp.fc2.bias": "model-00001-of-00003.safetensors",
"model.layers.9.mlp.fc2.weight": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.dense.bias": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.dense.weight": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.k_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.k_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.q_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.q_proj.weight": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.v_proj.bias": "model-00001-of-00003.safetensors",
"model.layers.9.self_attn.v_proj.weight": "model-00001-of-00003.safetensors"
}
}

1369
modeling_phi.py Normal file

File diff suppressed because it is too large Load Diff

3
optimizer.pt Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:95b23cb3b8dba1c128a8a4460d318109946e9b6d674c76c812adb973298169e1
size 10470174

3
rng_state.pth Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c6c9f200a86827343840446efad052c6eb2ded4411d8206c3fa962c6476d382b
size 14244

3
scheduler.pt Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f79db1c73a852c1d531896fa34b507d6845c0017b3119f9df30235a9a2eb6024
size 1064

30
special_tokens_map.json Normal file
View File

@@ -0,0 +1,30 @@
{
"bos_token": {
"content": "<s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"eos_token": {
"content": "</s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"pad_token": {
"content": "</s>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
},
"unk_token": {
"content": "<unk>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false
}
}

3
tokenizer.json Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1d3b61c4bc21578c33581576238dea3ce3a8e95d74109dd6b364723b2211dc97
size 2115299

340
tokenizer_config.json Normal file
View File

@@ -0,0 +1,340 @@
{
"add_prefix_space": false,
"added_tokens_decoder": {
"50256": {
"content": "<|endoftext|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"50257": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50258": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50259": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50260": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50261": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50262": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50263": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50264": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50265": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50266": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50267": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50268": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50269": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50270": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50271": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50272": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50273": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50274": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50275": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50276": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50277": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50278": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50279": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50280": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50281": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50282": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50283": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50284": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50285": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50286": {
"content": " ",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50287": {
"content": "\t\t\t\t\t\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50288": {
"content": "\t\t\t\t\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50289": {
"content": "\t\t\t\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50290": {
"content": "\t\t\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50291": {
"content": "\t\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50292": {
"content": "\t\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50293": {
"content": "\t\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50294": {
"content": "\t\t",
"lstrip": false,
"normalized": true,
"rstrip": false,
"single_word": false,
"special": false
},
"50295": {
"content": "<|im_end|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": true
},
"50296": {
"content": "<|im_start|>",
"lstrip": false,
"normalized": false,
"rstrip": false,
"single_word": false,
"special": false
}
},
"bos_token": "<|endoftext|>",
"clean_up_tokenization_spaces": true,
"eos_token": "<|im_end|>",
"model_max_length": 2048,
"pad_token": "<|endoftext|>",
"tokenizer_class": "CodeGenTokenizer",
"unk_token": "<|endoftext|>"
}

171
trainer_state.json Normal file
View File

@@ -0,0 +1,171 @@
{
"best_metric": null,
"best_model_checkpoint": null,
"epoch": 0.37650602409638556,
"eval_steps": 500,
"global_step": 250,
"is_hyper_param_search": false,
"is_local_process_zero": true,
"is_world_process_zero": true,
"log_history": [
{
"epoch": 0.02,
"learning_rate": 4.924698795180723e-05,
"loss": 1.2267,
"step": 10
},
{
"epoch": 0.03,
"learning_rate": 4.8493975903614455e-05,
"loss": 1.1443,
"step": 20
},
{
"epoch": 0.05,
"learning_rate": 4.774096385542169e-05,
"loss": 1.0681,
"step": 30
},
{
"epoch": 0.06,
"learning_rate": 4.698795180722892e-05,
"loss": 1.0362,
"step": 40
},
{
"epoch": 0.08,
"learning_rate": 4.6234939759036145e-05,
"loss": 1.003,
"step": 50
},
{
"epoch": 0.09,
"learning_rate": 4.5481927710843374e-05,
"loss": 1.1287,
"step": 60
},
{
"epoch": 0.11,
"learning_rate": 4.4728915662650604e-05,
"loss": 1.0379,
"step": 70
},
{
"epoch": 0.12,
"learning_rate": 4.3975903614457834e-05,
"loss": 1.0887,
"step": 80
},
{
"epoch": 0.14,
"learning_rate": 4.3222891566265064e-05,
"loss": 1.0347,
"step": 90
},
{
"epoch": 0.15,
"learning_rate": 4.2469879518072294e-05,
"loss": 1.0294,
"step": 100
},
{
"epoch": 0.17,
"learning_rate": 4.1716867469879523e-05,
"loss": 1.0475,
"step": 110
},
{
"epoch": 0.18,
"learning_rate": 4.0963855421686746e-05,
"loss": 0.9696,
"step": 120
},
{
"epoch": 0.2,
"learning_rate": 4.0210843373493976e-05,
"loss": 1.0094,
"step": 130
},
{
"epoch": 0.21,
"learning_rate": 3.9457831325301206e-05,
"loss": 1.0518,
"step": 140
},
{
"epoch": 0.23,
"learning_rate": 3.8704819277108436e-05,
"loss": 0.9784,
"step": 150
},
{
"epoch": 0.24,
"learning_rate": 3.7951807228915666e-05,
"loss": 0.9207,
"step": 160
},
{
"epoch": 0.26,
"learning_rate": 3.7198795180722895e-05,
"loss": 1.043,
"step": 170
},
{
"epoch": 0.27,
"learning_rate": 3.644578313253012e-05,
"loss": 1.05,
"step": 180
},
{
"epoch": 0.29,
"learning_rate": 3.569277108433735e-05,
"loss": 1.0409,
"step": 190
},
{
"epoch": 0.3,
"learning_rate": 3.4939759036144585e-05,
"loss": 0.9647,
"step": 200
},
{
"epoch": 0.32,
"learning_rate": 3.418674698795181e-05,
"loss": 1.0306,
"step": 210
},
{
"epoch": 0.33,
"learning_rate": 3.343373493975904e-05,
"loss": 0.9644,
"step": 220
},
{
"epoch": 0.35,
"learning_rate": 3.268072289156627e-05,
"loss": 0.9136,
"step": 230
},
{
"epoch": 0.36,
"learning_rate": 3.192771084337349e-05,
"loss": 0.8833,
"step": 240
},
{
"epoch": 0.38,
"learning_rate": 3.117469879518072e-05,
"loss": 1.0202,
"step": 250
}
],
"logging_steps": 10,
"max_steps": 664,
"num_input_tokens_seen": 0,
"num_train_epochs": 1,
"save_steps": 250,
"total_flos": 3.254614228992e+16,
"train_batch_size": 1,
"trial_name": null,
"trial_params": null
}

3
training_args.bin Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7dd1a53407727e6d0716426d780bc4392928d3a16fd38ef4576a16f527a4c9e1
size 4856

1
vocab.json Normal file

File diff suppressed because one or more lines are too long