初始化项目,由ModelHub XC社区提供模型
Model: AI-ModelScope/Qwen2.5-VL-3B-weighted_sft_ratio2-reasoning_and_grounding_changecoord_mixnoreasoning_cpt636 Source: Original Platform
This commit is contained in:
68
.gitattributes
vendored
Normal file
68
.gitattributes
vendored
Normal file
@@ -0,0 +1,68 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zstandard filter=lfs diff=lfs merge=lfs -text
|
||||
*.tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
*.db* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ark* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
|
||||
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.gguf* filter=lfs diff=lfs merge=lfs -text
|
||||
*.ggml filter=lfs diff=lfs merge=lfs -text
|
||||
*.llamafile* filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
merges.txt filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
model-00002-of-00002.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
model-00001-of-00002.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
rng_state_0.pth filter=lfs diff=lfs merge=lfs -text
|
||||
rng_state_1.pth filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
rng_state_2.pth filter=lfs diff=lfs merge=lfs -text
|
||||
rng_state_3.pth filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
rng_state_5.pth filter=lfs diff=lfs merge=lfs -text
|
||||
rng_state_4.pth filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
rng_state_7.pth filter=lfs diff=lfs merge=lfs -text
|
||||
rng_state_6.pth filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
|
||||
vocab.json filter=lfs diff=lfs merge=lfs -text
|
||||
48
README.md
Normal file
48
README.md
Normal file
@@ -0,0 +1,48 @@
|
||||
---
|
||||
license: other
|
||||
tags: []
|
||||
|
||||
#model-type:
|
||||
##如 gpt、phi、llama、chatglm、baichuan 等
|
||||
#- gpt
|
||||
|
||||
#domain:
|
||||
##如 nlp、cv、audio、multi-modal
|
||||
#- nlp
|
||||
|
||||
#language:
|
||||
##语言代码列表 https://help.aliyun.com/document_detail/215387.html?spm=a2c4g.11186623.0.0.9f8d7467kni6Aa
|
||||
#- cn
|
||||
|
||||
#metrics:
|
||||
##如 CIDEr、Blue、ROUGE 等
|
||||
#- CIDEr
|
||||
|
||||
#tags:
|
||||
##各种自定义,包括 pretrained、fine-tuned、instruction-tuned、RL-tuned 等训练方法和其他
|
||||
#- pretrained
|
||||
|
||||
#tools:
|
||||
##如 vllm、fastchat、llamacpp、AdaSeq 等
|
||||
#- vllm
|
||||
---
|
||||
### 当前模型的贡献者未提供更加详细的模型介绍。模型文件和权重,可浏览“模型文件”页面获取。
|
||||
#### 您可以通过如下git clone命令,或者ModelScope SDK来下载模型
|
||||
|
||||
SDK下载
|
||||
```bash
|
||||
#安装ModelScope
|
||||
pip install modelscope
|
||||
```
|
||||
```python
|
||||
#SDK模型下载
|
||||
from modelscope import snapshot_download
|
||||
model_dir = snapshot_download('AI-ModelScope/Qwen2.5-VL-3B-weighted_sft_ratio2-reasoning_and_grounding_changecoord_mixnoreasoning_cpt636')
|
||||
```
|
||||
Git下载
|
||||
```
|
||||
#Git模型下载
|
||||
git clone https://www.modelscope.cn/AI-ModelScope/Qwen2.5-VL-3B-weighted_sft_ratio2-reasoning_and_grounding_changecoord_mixnoreasoning_cpt636.git
|
||||
```
|
||||
|
||||
<p style="color: lightgrey;">如果您是本模型的贡献者,我们邀请您根据<a href="https://modelscope.cn/docs/ModelScope%E6%A8%A1%E5%9E%8B%E6%8E%A5%E5%85%A5%E6%B5%81%E7%A8%8B%E6%A6%82%E8%A7%88" style="color: lightgrey; text-decoration: underline;">模型贡献文档</a>,及时完善模型卡片内容。</p>
|
||||
24
added_tokens.json
Normal file
24
added_tokens.json
Normal file
@@ -0,0 +1,24 @@
|
||||
{
|
||||
"</tool_call>": 151658,
|
||||
"<tool_call>": 151657,
|
||||
"<|box_end|>": 151649,
|
||||
"<|box_start|>": 151648,
|
||||
"<|endoftext|>": 151643,
|
||||
"<|file_sep|>": 151664,
|
||||
"<|fim_middle|>": 151660,
|
||||
"<|fim_pad|>": 151662,
|
||||
"<|fim_prefix|>": 151659,
|
||||
"<|fim_suffix|>": 151661,
|
||||
"<|im_end|>": 151645,
|
||||
"<|im_start|>": 151644,
|
||||
"<|image_pad|>": 151655,
|
||||
"<|object_ref_end|>": 151647,
|
||||
"<|object_ref_start|>": 151646,
|
||||
"<|quad_end|>": 151651,
|
||||
"<|quad_start|>": 151650,
|
||||
"<|repo_name|>": 151663,
|
||||
"<|video_pad|>": 151656,
|
||||
"<|vision_end|>": 151653,
|
||||
"<|vision_pad|>": 151654,
|
||||
"<|vision_start|>": 151652
|
||||
}
|
||||
3
chat_template.json
Normal file
3
chat_template.json
Normal file
@@ -0,0 +1,3 @@
|
||||
{
|
||||
"chat_template": "{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}"
|
||||
}
|
||||
65
config.json
Normal file
65
config.json
Normal file
@@ -0,0 +1,65 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen2_5_VLForConditionalGeneration"
|
||||
],
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 151643,
|
||||
"eos_token_id": 151645,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2048,
|
||||
"image_token_id": 151655,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 11008,
|
||||
"max_position_embeddings": 128000,
|
||||
"max_window_layers": 70,
|
||||
"model_type": "qwen2_5_vl",
|
||||
"num_attention_heads": 16,
|
||||
"num_hidden_layers": 36,
|
||||
"num_key_value_heads": 2,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": {
|
||||
"mrope_section": [
|
||||
16,
|
||||
24,
|
||||
24
|
||||
],
|
||||
"rope_type": "default",
|
||||
"type": "default"
|
||||
},
|
||||
"rope_theta": 1000000.0,
|
||||
"sliding_window": 32768,
|
||||
"tie_word_embeddings": true,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.51.0",
|
||||
"use_cache": false,
|
||||
"use_sliding_window": false,
|
||||
"video_token_id": 151656,
|
||||
"vision_config": {
|
||||
"depth": 32,
|
||||
"fullatt_block_indexes": [
|
||||
7,
|
||||
15,
|
||||
23,
|
||||
31
|
||||
],
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 1280,
|
||||
"in_channels": 3,
|
||||
"in_chans": 3,
|
||||
"intermediate_size": 3420,
|
||||
"model_type": "qwen2_5_vl",
|
||||
"num_heads": 16,
|
||||
"out_hidden_size": 2048,
|
||||
"patch_size": 14,
|
||||
"spatial_merge_size": 2,
|
||||
"spatial_patch_size": 14,
|
||||
"temporal_patch_size": 2,
|
||||
"tokens_per_second": 2,
|
||||
"torch_dtype": "bfloat16",
|
||||
"window_size": 112
|
||||
},
|
||||
"vision_end_token_id": 151653,
|
||||
"vision_start_token_id": 151652,
|
||||
"vision_token_id": 151654,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
||||
{"framework": "pytorch", "task": "others", "allow_remote": true}
|
||||
13
generation_config.json
Normal file
13
generation_config.json
Normal file
@@ -0,0 +1,13 @@
|
||||
{
|
||||
"attn_implementation": "flash_attention_2",
|
||||
"bos_token_id": 151643,
|
||||
"do_sample": true,
|
||||
"eos_token_id": [
|
||||
151645,
|
||||
151643
|
||||
],
|
||||
"pad_token_id": 151643,
|
||||
"repetition_penalty": 1.05,
|
||||
"temperature": 1e-06,
|
||||
"transformers_version": "4.51.0"
|
||||
}
|
||||
3
merges.txt
Normal file
3
merges.txt
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5
|
||||
size 1671853
|
||||
3
model-00001-of-00002.safetensors
Normal file
3
model-00001-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9b6f03beb73b793a2f610e4fd11ee520b8860f114c51f37bdff22747e0cd288b
|
||||
size 4997750760
|
||||
3
model-00002-of-00002.safetensors
Normal file
3
model-00002-of-00002.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d5673f74517778e40d3230f0e02db42ecb77ae5c2504d113b861d1cd39e30d4d
|
||||
size 2511587184
|
||||
831
model.safetensors.index.json
Normal file
831
model.safetensors.index.json
Normal file
@@ -0,0 +1,831 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 7509245952
|
||||
},
|
||||
"weight_map": {
|
||||
"model.embed_tokens.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.0.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.1.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.10.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.11.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.12.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.13.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.14.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.15.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.16.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.17.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.18.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.19.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.19.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.2.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.20.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.20.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.21.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.22.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.23.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.24.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.25.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.26.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.27.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.28.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.29.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.3.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.3.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.30.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.30.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.31.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.32.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.33.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.34.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.input_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.mlp.down_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.mlp.gate_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.mlp.up_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.post_attention_layernorm.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.k_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.k_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.o_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.q_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.q_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.v_proj.bias": "model-00002-of-00002.safetensors",
|
||||
"model.layers.35.self_attn.v_proj.weight": "model-00002-of-00002.safetensors",
|
||||
"model.layers.4.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.4.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.5.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.6.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.7.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.8.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.input_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.post_attention_layernorm.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.k_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.k_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.o_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.q_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.q_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.v_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"model.layers.9.self_attn.v_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"model.norm.weight": "model-00002-of-00002.safetensors",
|
||||
"visual.blocks.0.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.0.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.1.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.10.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.11.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.12.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.13.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.14.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.15.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.16.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.17.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.18.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.19.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.2.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.20.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.21.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.22.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.23.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.24.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.25.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.26.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.27.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.28.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.29.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.3.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.30.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.31.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.4.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.5.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.6.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.7.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.8.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.attn.proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.attn.proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.attn.qkv.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.attn.qkv.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.down_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.down_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.gate_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.gate_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.up_proj.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.mlp.up_proj.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.norm1.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.blocks.9.norm2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.merger.ln_q.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.merger.mlp.0.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.merger.mlp.0.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.merger.mlp.2.bias": "model-00001-of-00002.safetensors",
|
||||
"visual.merger.mlp.2.weight": "model-00001-of-00002.safetensors",
|
||||
"visual.patch_embed.proj.weight": "model-00001-of-00002.safetensors"
|
||||
}
|
||||
}
|
||||
29
preprocessor_config.json
Normal file
29
preprocessor_config.json
Normal file
@@ -0,0 +1,29 @@
|
||||
{
|
||||
"do_convert_rgb": true,
|
||||
"do_normalize": true,
|
||||
"do_rescale": true,
|
||||
"do_resize": true,
|
||||
"image_mean": [
|
||||
0.48145466,
|
||||
0.4578275,
|
||||
0.40821073
|
||||
],
|
||||
"image_processor_type": "Qwen2VLImageProcessor",
|
||||
"image_std": [
|
||||
0.26862954,
|
||||
0.26130258,
|
||||
0.27577711
|
||||
],
|
||||
"max_pixels": 5720064,
|
||||
"merge_size": 2,
|
||||
"min_pixels": 3136,
|
||||
"patch_size": 14,
|
||||
"processor_class": "Qwen2_5_VLProcessor",
|
||||
"resample": 3,
|
||||
"rescale_factor": 0.00392156862745098,
|
||||
"size": {
|
||||
"longest_edge": 12845056,
|
||||
"shortest_edge": 3136
|
||||
},
|
||||
"temporal_patch_size": 2
|
||||
}
|
||||
3
rng_state_0.pth
Normal file
3
rng_state_0.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ff83006d85c789a73e3d5962b60a1ea20ea38374ebe54f6d5e8a5c5777efbb05
|
||||
size 16325
|
||||
3
rng_state_1.pth
Normal file
3
rng_state_1.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:a4405d2b9aa7065e98bf1b965702b69709bef30d31b3257e2445623659424edc
|
||||
size 16325
|
||||
3
rng_state_2.pth
Normal file
3
rng_state_2.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5d63c51a9642e331eb3ab07696e2e11c88c29f4f34cf4f99f8ffa4fb55af453b
|
||||
size 16325
|
||||
3
rng_state_3.pth
Normal file
3
rng_state_3.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ba4e5ea853aa4898edd412a4da34a96b758f0f7218655949614277cb1910eb31
|
||||
size 16325
|
||||
3
rng_state_4.pth
Normal file
3
rng_state_4.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:481092ba472bbb5b59655c5d961d7f139eda9daf3bb24d9d6f7ef405894f2153
|
||||
size 16325
|
||||
3
rng_state_5.pth
Normal file
3
rng_state_5.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:93e9df5970151da7ccada8e10c311cc4a6237c90a1701b6a5187bce10af76bd8
|
||||
size 16325
|
||||
3
rng_state_6.pth
Normal file
3
rng_state_6.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:56e3382b0cee8a47b032dacb4635c6434a00b08b20843e7882b8fafa689b2bf0
|
||||
size 16325
|
||||
3
rng_state_7.pth
Normal file
3
rng_state_7.pth
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:e71863697b5311ef5c1666e4a06796a8f027fab8efa2dfff79a823e493506448
|
||||
size 16325
|
||||
31
special_tokens_map.json
Normal file
31
special_tokens_map.json
Normal file
@@ -0,0 +1,31 @@
|
||||
{
|
||||
"additional_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"eos_token": {
|
||||
"content": "<|im_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
},
|
||||
"pad_token": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false
|
||||
}
|
||||
}
|
||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa
|
||||
size 11421896
|
||||
210
tokenizer_config.json
Normal file
210
tokenizer_config.json
Normal file
@@ -0,0 +1,210 @@
|
||||
{
|
||||
"add_bos_token": false,
|
||||
"add_prefix_space": false,
|
||||
"added_tokens_decoder": {
|
||||
"151643": {
|
||||
"content": "<|endoftext|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151644": {
|
||||
"content": "<|im_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151645": {
|
||||
"content": "<|im_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151646": {
|
||||
"content": "<|object_ref_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151647": {
|
||||
"content": "<|object_ref_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151648": {
|
||||
"content": "<|box_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151649": {
|
||||
"content": "<|box_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151650": {
|
||||
"content": "<|quad_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151651": {
|
||||
"content": "<|quad_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151652": {
|
||||
"content": "<|vision_start|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151653": {
|
||||
"content": "<|vision_end|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151654": {
|
||||
"content": "<|vision_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151655": {
|
||||
"content": "<|image_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151656": {
|
||||
"content": "<|video_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": true
|
||||
},
|
||||
"151657": {
|
||||
"content": "<tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151658": {
|
||||
"content": "</tool_call>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151659": {
|
||||
"content": "<|fim_prefix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151660": {
|
||||
"content": "<|fim_middle|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151661": {
|
||||
"content": "<|fim_suffix|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151662": {
|
||||
"content": "<|fim_pad|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151663": {
|
||||
"content": "<|repo_name|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
},
|
||||
"151664": {
|
||||
"content": "<|file_sep|>",
|
||||
"lstrip": false,
|
||||
"normalized": false,
|
||||
"rstrip": false,
|
||||
"single_word": false,
|
||||
"special": false
|
||||
}
|
||||
},
|
||||
"additional_special_tokens": [
|
||||
"<|im_start|>",
|
||||
"<|im_end|>",
|
||||
"<|object_ref_start|>",
|
||||
"<|object_ref_end|>",
|
||||
"<|box_start|>",
|
||||
"<|box_end|>",
|
||||
"<|quad_start|>",
|
||||
"<|quad_end|>",
|
||||
"<|vision_start|>",
|
||||
"<|vision_end|>",
|
||||
"<|vision_pad|>",
|
||||
"<|image_pad|>",
|
||||
"<|video_pad|>"
|
||||
],
|
||||
"bos_token": null,
|
||||
"chat_template": "{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n{% endif %}<|im_start|>{{ message['role'] }}\n{% if message['content'] is string %}{{ message['content'] }}<|im_end|>\n{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>\n{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant\n{% endif %}",
|
||||
"clean_up_tokenization_spaces": false,
|
||||
"eos_token": "<|im_end|>",
|
||||
"errors": "replace",
|
||||
"extra_special_tokens": {},
|
||||
"model_max_length": 24576,
|
||||
"pad_token": "<|endoftext|>",
|
||||
"padding_side": "right",
|
||||
"processor_class": "Qwen2_5_VLProcessor",
|
||||
"split_special_tokens": false,
|
||||
"tokenizer_class": "Qwen2Tokenizer",
|
||||
"unk_token": null
|
||||
}
|
||||
475
trainer_state.json
Normal file
475
trainer_state.json
Normal file
@@ -0,0 +1,475 @@
|
||||
{
|
||||
"best_global_step": null,
|
||||
"best_metric": null,
|
||||
"best_model_checkpoint": null,
|
||||
"epoch": 0.9994107248084856,
|
||||
"eval_steps": 500,
|
||||
"global_step": 636,
|
||||
"is_hyper_param_search": false,
|
||||
"is_local_process_zero": true,
|
||||
"is_world_process_zero": true,
|
||||
"log_history": [
|
||||
{
|
||||
"epoch": 0.01571400510705166,
|
||||
"grad_norm": 0.92578125,
|
||||
"learning_rate": 6.923076923076923e-06,
|
||||
"loss": 1.183,
|
||||
"step": 10
|
||||
},
|
||||
{
|
||||
"epoch": 0.03142801021410332,
|
||||
"grad_norm": 0.373046875,
|
||||
"learning_rate": 9.999439619928746e-06,
|
||||
"loss": 0.792,
|
||||
"step": 20
|
||||
},
|
||||
{
|
||||
"epoch": 0.04714201532115498,
|
||||
"grad_norm": 0.40234375,
|
||||
"learning_rate": 9.996015529923432e-06,
|
||||
"loss": 0.6339,
|
||||
"step": 30
|
||||
},
|
||||
{
|
||||
"epoch": 0.06285602042820664,
|
||||
"grad_norm": 0.279296875,
|
||||
"learning_rate": 9.989480801510058e-06,
|
||||
"loss": 0.6076,
|
||||
"step": 40
|
||||
},
|
||||
{
|
||||
"epoch": 0.0785700255352583,
|
||||
"grad_norm": 0.3671875,
|
||||
"learning_rate": 9.979839503366368e-06,
|
||||
"loss": 0.5715,
|
||||
"step": 50
|
||||
},
|
||||
{
|
||||
"epoch": 0.09428403064230996,
|
||||
"grad_norm": 0.216796875,
|
||||
"learning_rate": 9.9670976383945e-06,
|
||||
"loss": 0.5996,
|
||||
"step": 60
|
||||
},
|
||||
{
|
||||
"epoch": 0.10999803574936162,
|
||||
"grad_norm": 0.2060546875,
|
||||
"learning_rate": 9.951263139983441e-06,
|
||||
"loss": 0.5625,
|
||||
"step": 70
|
||||
},
|
||||
{
|
||||
"epoch": 0.12571204085641327,
|
||||
"grad_norm": 0.28125,
|
||||
"learning_rate": 9.9323458670695e-06,
|
||||
"loss": 0.569,
|
||||
"step": 80
|
||||
},
|
||||
{
|
||||
"epoch": 0.14142604596346495,
|
||||
"grad_norm": 0.61328125,
|
||||
"learning_rate": 9.910357597997912e-06,
|
||||
"loss": 0.568,
|
||||
"step": 90
|
||||
},
|
||||
{
|
||||
"epoch": 0.1571400510705166,
|
||||
"grad_norm": 0.45703125,
|
||||
"learning_rate": 9.885312023189353e-06,
|
||||
"loss": 0.5343,
|
||||
"step": 100
|
||||
},
|
||||
{
|
||||
"epoch": 0.17285405617756827,
|
||||
"grad_norm": 0.294921875,
|
||||
"learning_rate": 9.857224736615953e-06,
|
||||
"loss": 0.5809,
|
||||
"step": 110
|
||||
},
|
||||
{
|
||||
"epoch": 0.18856806128461992,
|
||||
"grad_norm": 0.2578125,
|
||||
"learning_rate": 9.82611322609213e-06,
|
||||
"loss": 0.5429,
|
||||
"step": 120
|
||||
},
|
||||
{
|
||||
"epoch": 0.20428206639167157,
|
||||
"grad_norm": 0.32421875,
|
||||
"learning_rate": 9.791996862386242e-06,
|
||||
"loss": 0.5571,
|
||||
"step": 130
|
||||
},
|
||||
{
|
||||
"epoch": 0.21999607149872324,
|
||||
"grad_norm": 0.2294921875,
|
||||
"learning_rate": 9.754896887159904e-06,
|
||||
"loss": 0.5516,
|
||||
"step": 140
|
||||
},
|
||||
{
|
||||
"epoch": 0.2357100766057749,
|
||||
"grad_norm": 0.46875,
|
||||
"learning_rate": 9.714836399742406e-06,
|
||||
"loss": 0.5246,
|
||||
"step": 150
|
||||
},
|
||||
{
|
||||
"epoch": 0.25142408171282654,
|
||||
"grad_norm": 0.224609375,
|
||||
"learning_rate": 9.67184034274853e-06,
|
||||
"loss": 0.5639,
|
||||
"step": 160
|
||||
},
|
||||
{
|
||||
"epoch": 0.2671380868198782,
|
||||
"grad_norm": 0.2060546875,
|
||||
"learning_rate": 9.62593548654868e-06,
|
||||
"loss": 0.5282,
|
||||
"step": 170
|
||||
},
|
||||
{
|
||||
"epoch": 0.2828520919269299,
|
||||
"grad_norm": 0.37890625,
|
||||
"learning_rate": 9.577150412601011e-06,
|
||||
"loss": 0.5366,
|
||||
"step": 180
|
||||
},
|
||||
{
|
||||
"epoch": 0.2985660970339815,
|
||||
"grad_norm": 0.369140625,
|
||||
"learning_rate": 9.525515495655935e-06,
|
||||
"loss": 0.5467,
|
||||
"step": 190
|
||||
},
|
||||
{
|
||||
"epoch": 0.3142801021410332,
|
||||
"grad_norm": 0.37890625,
|
||||
"learning_rate": 9.471062884844067e-06,
|
||||
"loss": 0.5126,
|
||||
"step": 200
|
||||
},
|
||||
{
|
||||
"epoch": 0.32999410724808487,
|
||||
"grad_norm": 0.36328125,
|
||||
"learning_rate": 9.413826483659424e-06,
|
||||
"loss": 0.5556,
|
||||
"step": 210
|
||||
},
|
||||
{
|
||||
"epoch": 0.34570811235513654,
|
||||
"grad_norm": 0.38671875,
|
||||
"learning_rate": 9.353841928850286e-06,
|
||||
"loss": 0.5245,
|
||||
"step": 220
|
||||
},
|
||||
{
|
||||
"epoch": 0.36142211746218816,
|
||||
"grad_norm": 0.2890625,
|
||||
"learning_rate": 9.291146568230922e-06,
|
||||
"loss": 0.5356,
|
||||
"step": 230
|
||||
},
|
||||
{
|
||||
"epoch": 0.37713612256923984,
|
||||
"grad_norm": 0.33203125,
|
||||
"learning_rate": 9.22577943742794e-06,
|
||||
"loss": 0.5379,
|
||||
"step": 240
|
||||
},
|
||||
{
|
||||
"epoch": 0.3928501276762915,
|
||||
"grad_norm": 0.4765625,
|
||||
"learning_rate": 9.157781235575774e-06,
|
||||
"loss": 0.5109,
|
||||
"step": 250
|
||||
},
|
||||
{
|
||||
"epoch": 0.40856413278334314,
|
||||
"grad_norm": 0.48828125,
|
||||
"learning_rate": 9.087194299976439e-06,
|
||||
"loss": 0.5472,
|
||||
"step": 260
|
||||
},
|
||||
{
|
||||
"epoch": 0.4242781378903948,
|
||||
"grad_norm": 0.69140625,
|
||||
"learning_rate": 9.014062579739315e-06,
|
||||
"loss": 0.5169,
|
||||
"step": 270
|
||||
},
|
||||
{
|
||||
"epoch": 0.4399921429974465,
|
||||
"grad_norm": 0.263671875,
|
||||
"learning_rate": 8.93843160841738e-06,
|
||||
"loss": 0.5259,
|
||||
"step": 280
|
||||
},
|
||||
{
|
||||
"epoch": 0.4557061481044981,
|
||||
"grad_norm": 0.2080078125,
|
||||
"learning_rate": 8.86034847565694e-06,
|
||||
"loss": 0.5333,
|
||||
"step": 290
|
||||
},
|
||||
{
|
||||
"epoch": 0.4714201532115498,
|
||||
"grad_norm": 0.396484375,
|
||||
"learning_rate": 8.779861797878486e-06,
|
||||
"loss": 0.501,
|
||||
"step": 300
|
||||
},
|
||||
{
|
||||
"epoch": 0.48713415831860146,
|
||||
"grad_norm": 0.2001953125,
|
||||
"learning_rate": 8.697021688006952e-06,
|
||||
"loss": 0.5485,
|
||||
"step": 310
|
||||
},
|
||||
{
|
||||
"epoch": 0.5028481634256531,
|
||||
"grad_norm": 0.55859375,
|
||||
"learning_rate": 8.611879724270222e-06,
|
||||
"loss": 0.5121,
|
||||
"step": 320
|
||||
},
|
||||
{
|
||||
"epoch": 0.5185621685327048,
|
||||
"grad_norm": 0.333984375,
|
||||
"learning_rate": 8.524488918085281e-06,
|
||||
"loss": 0.5235,
|
||||
"step": 330
|
||||
},
|
||||
{
|
||||
"epoch": 0.5342761736397564,
|
||||
"grad_norm": 0.234375,
|
||||
"learning_rate": 8.434903681052058e-06,
|
||||
"loss": 0.5351,
|
||||
"step": 340
|
||||
},
|
||||
{
|
||||
"epoch": 0.5499901787468081,
|
||||
"grad_norm": 0.39453125,
|
||||
"learning_rate": 8.343179791075448e-06,
|
||||
"loss": 0.4959,
|
||||
"step": 350
|
||||
},
|
||||
{
|
||||
"epoch": 0.5657041838538598,
|
||||
"grad_norm": 0.390625,
|
||||
"learning_rate": 8.249374357636678e-06,
|
||||
"loss": 0.5448,
|
||||
"step": 360
|
||||
},
|
||||
{
|
||||
"epoch": 0.5814181889609115,
|
||||
"grad_norm": 0.203125,
|
||||
"learning_rate": 8.153545786235569e-06,
|
||||
"loss": 0.5122,
|
||||
"step": 370
|
||||
},
|
||||
{
|
||||
"epoch": 0.597132194067963,
|
||||
"grad_norm": 1.109375,
|
||||
"learning_rate": 8.05575374202588e-06,
|
||||
"loss": 0.5224,
|
||||
"step": 380
|
||||
},
|
||||
{
|
||||
"epoch": 0.6128461991750147,
|
||||
"grad_norm": 0.5078125,
|
||||
"learning_rate": 7.95605911266637e-06,
|
||||
"loss": 0.5201,
|
||||
"step": 390
|
||||
},
|
||||
{
|
||||
"epoch": 0.6285602042820664,
|
||||
"grad_norm": 0.44140625,
|
||||
"learning_rate": 7.854523970410679e-06,
|
||||
"loss": 0.4947,
|
||||
"step": 400
|
||||
},
|
||||
{
|
||||
"epoch": 0.6442742093891181,
|
||||
"grad_norm": 0.2333984375,
|
||||
"learning_rate": 7.751211533459661e-06,
|
||||
"loss": 0.5422,
|
||||
"step": 410
|
||||
},
|
||||
{
|
||||
"epoch": 0.6599882144961697,
|
||||
"grad_norm": 0.29296875,
|
||||
"learning_rate": 7.646186126600243e-06,
|
||||
"loss": 0.5081,
|
||||
"step": 420
|
||||
},
|
||||
{
|
||||
"epoch": 0.6757022196032214,
|
||||
"grad_norm": 0.3515625,
|
||||
"learning_rate": 7.539513141155253e-06,
|
||||
"loss": 0.5221,
|
||||
"step": 430
|
||||
},
|
||||
{
|
||||
"epoch": 0.6914162247102731,
|
||||
"grad_norm": 0.3203125,
|
||||
"learning_rate": 7.431258994269245e-06,
|
||||
"loss": 0.5198,
|
||||
"step": 440
|
||||
},
|
||||
{
|
||||
"epoch": 0.7071302298173247,
|
||||
"grad_norm": 0.4296875,
|
||||
"learning_rate": 7.321491087555588e-06,
|
||||
"loss": 0.494,
|
||||
"step": 450
|
||||
},
|
||||
{
|
||||
"epoch": 0.7228442349243763,
|
||||
"grad_norm": 0.498046875,
|
||||
"learning_rate": 7.210277765130616e-06,
|
||||
"loss": 0.5377,
|
||||
"step": 460
|
||||
},
|
||||
{
|
||||
"epoch": 0.738558240031428,
|
||||
"grad_norm": 0.240234375,
|
||||
"learning_rate": 7.097688271060956e-06,
|
||||
"loss": 0.5076,
|
||||
"step": 470
|
||||
},
|
||||
{
|
||||
"epoch": 0.7542722451384797,
|
||||
"grad_norm": 0.41796875,
|
||||
"learning_rate": 6.983792706250521e-06,
|
||||
"loss": 0.5094,
|
||||
"step": 480
|
||||
},
|
||||
{
|
||||
"epoch": 0.7699862502455314,
|
||||
"grad_norm": 0.3203125,
|
||||
"learning_rate": 6.8686619847940105e-06,
|
||||
"loss": 0.5155,
|
||||
"step": 490
|
||||
},
|
||||
{
|
||||
"epoch": 0.785700255352583,
|
||||
"grad_norm": 0.609375,
|
||||
"learning_rate": 6.752367789824103e-06,
|
||||
"loss": 0.4946,
|
||||
"step": 500
|
||||
},
|
||||
{
|
||||
"epoch": 0.8014142604596346,
|
||||
"grad_norm": 0.349609375,
|
||||
"learning_rate": 6.6349825288798344e-06,
|
||||
"loss": 0.529,
|
||||
"step": 510
|
||||
},
|
||||
{
|
||||
"epoch": 0.8171282655666863,
|
||||
"grad_norm": 0.439453125,
|
||||
"learning_rate": 6.516579288823938e-06,
|
||||
"loss": 0.4959,
|
||||
"step": 520
|
||||
},
|
||||
{
|
||||
"epoch": 0.832842270673738,
|
||||
"grad_norm": 0.6015625,
|
||||
"learning_rate": 6.3972317903372104e-06,
|
||||
"loss": 0.5176,
|
||||
"step": 530
|
||||
},
|
||||
{
|
||||
"epoch": 0.8485562757807896,
|
||||
"grad_norm": 0.34375,
|
||||
"learning_rate": 6.277014342018273e-06,
|
||||
"loss": 0.5162,
|
||||
"step": 540
|
||||
},
|
||||
{
|
||||
"epoch": 0.8642702808878413,
|
||||
"grad_norm": 0.484375,
|
||||
"learning_rate": 6.156001794117251e-06,
|
||||
"loss": 0.4961,
|
||||
"step": 550
|
||||
},
|
||||
{
|
||||
"epoch": 0.879984285994893,
|
||||
"grad_norm": 0.251953125,
|
||||
"learning_rate": 6.034269491932234e-06,
|
||||
"loss": 0.5362,
|
||||
"step": 560
|
||||
},
|
||||
{
|
||||
"epoch": 0.8956982911019447,
|
||||
"grad_norm": 0.37109375,
|
||||
"learning_rate": 5.911893228897494e-06,
|
||||
"loss": 0.5007,
|
||||
"step": 570
|
||||
},
|
||||
{
|
||||
"epoch": 0.9114122962089962,
|
||||
"grad_norm": 0.330078125,
|
||||
"learning_rate": 5.788949199392679e-06,
|
||||
"loss": 0.5025,
|
||||
"step": 580
|
||||
},
|
||||
{
|
||||
"epoch": 0.9271263013160479,
|
||||
"grad_norm": 0.86328125,
|
||||
"learning_rate": 5.665513951302386e-06,
|
||||
"loss": 0.5092,
|
||||
"step": 590
|
||||
},
|
||||
{
|
||||
"epoch": 0.9428403064230996,
|
||||
"grad_norm": 0.5234375,
|
||||
"learning_rate": 5.541664338355615e-06,
|
||||
"loss": 0.4916,
|
||||
"step": 600
|
||||
},
|
||||
{
|
||||
"epoch": 0.9585543115301512,
|
||||
"grad_norm": 0.2412109375,
|
||||
"learning_rate": 5.417477472274806e-06,
|
||||
"loss": 0.5307,
|
||||
"step": 610
|
||||
},
|
||||
{
|
||||
"epoch": 0.9742683166372029,
|
||||
"grad_norm": 0.349609375,
|
||||
"learning_rate": 5.293030674764243e-06,
|
||||
"loss": 0.5047,
|
||||
"step": 620
|
||||
},
|
||||
{
|
||||
"epoch": 0.9899823217442546,
|
||||
"grad_norm": 0.3671875,
|
||||
"learning_rate": 5.168401429367723e-06,
|
||||
"loss": 0.5158,
|
||||
"step": 630
|
||||
}
|
||||
],
|
||||
"logging_steps": 10,
|
||||
"max_steps": 1272,
|
||||
"num_input_tokens_seen": 0,
|
||||
"num_train_epochs": 2,
|
||||
"save_steps": 500.0,
|
||||
"stateful_callbacks": {
|
||||
"TrainerControl": {
|
||||
"args": {
|
||||
"should_epoch_stop": false,
|
||||
"should_evaluate": false,
|
||||
"should_log": false,
|
||||
"should_save": true,
|
||||
"should_training_stop": false
|
||||
},
|
||||
"attributes": {}
|
||||
}
|
||||
},
|
||||
"total_flos": 2.009693603285382e+19,
|
||||
"train_batch_size": 4,
|
||||
"trial_name": null,
|
||||
"trial_params": null
|
||||
}
|
||||
BIN
vocab.json
(Stored with Git LFS)
Normal file
BIN
vocab.json
(Stored with Git LFS)
Normal file
Binary file not shown.
Reference in New Issue
Block a user