commit 4984c251d5084f5b834c6b1a8a8ca08d06137c01
Author: ModelHub XC
Date: Thu Jun 25 00:32:23 2026 +0800
初始化项目,由ModelHub XC社区提供模型
Model: indischepartij/MiniCPM-3B-Bacchus
Source: Original Platform
diff --git a/.gitattributes b/.gitattributes
new file mode 100644
index 0000000..a6344aa
--- /dev/null
+++ b/.gitattributes
@@ -0,0 +1,35 @@
+*.7z filter=lfs diff=lfs merge=lfs -text
+*.arrow filter=lfs diff=lfs merge=lfs -text
+*.bin filter=lfs diff=lfs merge=lfs -text
+*.bz2 filter=lfs diff=lfs merge=lfs -text
+*.ckpt filter=lfs diff=lfs merge=lfs -text
+*.ftz filter=lfs diff=lfs merge=lfs -text
+*.gz filter=lfs diff=lfs merge=lfs -text
+*.h5 filter=lfs diff=lfs merge=lfs -text
+*.joblib filter=lfs diff=lfs merge=lfs -text
+*.lfs.* filter=lfs diff=lfs merge=lfs -text
+*.mlmodel filter=lfs diff=lfs merge=lfs -text
+*.model filter=lfs diff=lfs merge=lfs -text
+*.msgpack filter=lfs diff=lfs merge=lfs -text
+*.npy filter=lfs diff=lfs merge=lfs -text
+*.npz filter=lfs diff=lfs merge=lfs -text
+*.onnx filter=lfs diff=lfs merge=lfs -text
+*.ot filter=lfs diff=lfs merge=lfs -text
+*.parquet filter=lfs diff=lfs merge=lfs -text
+*.pb filter=lfs diff=lfs merge=lfs -text
+*.pickle filter=lfs diff=lfs merge=lfs -text
+*.pkl filter=lfs diff=lfs merge=lfs -text
+*.pt filter=lfs diff=lfs merge=lfs -text
+*.pth filter=lfs diff=lfs merge=lfs -text
+*.rar filter=lfs diff=lfs merge=lfs -text
+*.safetensors filter=lfs diff=lfs merge=lfs -text
+saved_model/**/* filter=lfs diff=lfs merge=lfs -text
+*.tar.* filter=lfs diff=lfs merge=lfs -text
+*.tar filter=lfs diff=lfs merge=lfs -text
+*.tflite filter=lfs diff=lfs merge=lfs -text
+*.tgz filter=lfs diff=lfs merge=lfs -text
+*.wasm filter=lfs diff=lfs merge=lfs -text
+*.xz filter=lfs diff=lfs merge=lfs -text
+*.zip filter=lfs diff=lfs merge=lfs -text
+*.zst filter=lfs diff=lfs merge=lfs -text
+*tfevents* filter=lfs diff=lfs merge=lfs -text
diff --git a/README.md b/README.md
new file mode 100644
index 0000000..4624664
--- /dev/null
+++ b/README.md
@@ -0,0 +1,315 @@
+---
+license: apache-2.0
+library_name: transformers
+model-index:
+- name: MiniCPM-3B-Bacchus
+ results:
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: AI2 Reasoning Challenge (25-Shot)
+ type: ai2_arc
+ config: ARC-Challenge
+ split: test
+ args:
+ num_few_shot: 25
+ metrics:
+ - type: acc_norm
+ value: 43.52
+ name: normalized accuracy
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: HellaSwag (10-Shot)
+ type: hellaswag
+ split: validation
+ args:
+ num_few_shot: 10
+ metrics:
+ - type: acc_norm
+ value: 70.45
+ name: normalized accuracy
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: MMLU (5-Shot)
+ type: cais/mmlu
+ config: all
+ split: test
+ args:
+ num_few_shot: 5
+ metrics:
+ - type: acc
+ value: 50.49
+ name: accuracy
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: TruthfulQA (0-shot)
+ type: truthful_qa
+ config: multiple_choice
+ split: validation
+ args:
+ num_few_shot: 0
+ metrics:
+ - type: mc2
+ value: 43.52
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: Winogrande (5-shot)
+ type: winogrande
+ config: winogrande_xl
+ split: validation
+ args:
+ num_few_shot: 5
+ metrics:
+ - type: acc
+ value: 66.85
+ name: accuracy
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+ - task:
+ type: text-generation
+ name: Text Generation
+ dataset:
+ name: GSM8k (5-shot)
+ type: gsm8k
+ config: main
+ split: test
+ args:
+ num_few_shot: 5
+ metrics:
+ - type: acc
+ value: 40.49
+ name: accuracy
+ source:
+ url: https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard?query=indischepartij/MiniCPM-3B-Bacchus
+ name: Open LLM Leaderboard
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+This is the model card of a 🤗 transformers model that has been pushed on the Hub. This model card has been automatically generated.
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+# [Open LLM Leaderboard Evaluation Results](https://huggingface.co/spaces/HuggingFaceH4/open_llm_leaderboard)
+Detailed results can be found [here](https://huggingface.co/datasets/open-llm-leaderboard/details_indischepartij__MiniCPM-3B-Bacchus)
+
+| Metric |Value|
+|---------------------------------|----:|
+|Avg. |52.55|
+|AI2 Reasoning Challenge (25-Shot)|43.52|
+|HellaSwag (10-Shot) |70.45|
+|MMLU (5-Shot) |50.49|
+|TruthfulQA (0-shot) |43.52|
+|Winogrande (5-shot) |66.85|
+|GSM8k (5-shot) |40.49|
+
diff --git a/config.json b/config.json
new file mode 100644
index 0000000..e1a9b5d
--- /dev/null
+++ b/config.json
@@ -0,0 +1,31 @@
+{
+ "_name_or_path": "indischepartij/MiniCPM-3B-Hephaestus",
+ "architectures": [
+ "LlamaForCausalLM"
+ ],
+ "attention_bias": false,
+ "attention_dropout": 0.0,
+ "bos_token_id": 1,
+ "dim_model_base": 256,
+ "eos_token_id": 2,
+ "hidden_act": "silu",
+ "hidden_size": 2304,
+ "initializer_range": 0.1,
+ "intermediate_size": 5760,
+ "max_position_embeddings": 2048,
+ "model_type": "llama",
+ "num_attention_heads": 36,
+ "num_hidden_layers": 40,
+ "num_key_value_heads": 36,
+ "pretraining_tp": 1,
+ "rms_norm_eps": 1e-05,
+ "rope_scaling": null,
+ "rope_theta": 10000.0,
+ "scale_depth": 1.4,
+ "scale_emb": 12,
+ "tie_word_embeddings": false,
+ "torch_dtype": "bfloat16",
+ "transformers_version": "4.37.2",
+ "use_cache": true,
+ "vocab_size": 122753
+}
diff --git a/generation_config.json b/generation_config.json
new file mode 100644
index 0000000..69b7806
--- /dev/null
+++ b/generation_config.json
@@ -0,0 +1,6 @@
+{
+ "_from_model_config": true,
+ "bos_token_id": 1,
+ "eos_token_id": 2,
+ "transformers_version": "4.37.2"
+}
diff --git a/model-00001-of-00004.safetensors b/model-00001-of-00004.safetensors
new file mode 100644
index 0000000..164bf1d
--- /dev/null
+++ b/model-00001-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f9c9cbe9ce22abd9dd71c66517dd2f2322f94adf7742fe79da7f9c9f76f461bc
+size 1977797960
diff --git a/model-00002-of-00004.safetensors b/model-00002-of-00004.safetensors
new file mode 100644
index 0000000..fbc7e9f
--- /dev/null
+++ b/model-00002-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8bb67051bcf871dda025a1838b157579fe953e1db155e5a76d01f04f2bcb28b0
+size 1980203376
diff --git a/model-00003-of-00004.safetensors b/model-00003-of-00004.safetensors
new file mode 100644
index 0000000..3f89dc9
--- /dev/null
+++ b/model-00003-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a850b7c192e520f4fee701c9f1d367109e4188b43565d87cc51a3ddd669426ce
+size 1491802192
diff --git a/model-00004-of-00004.safetensors b/model-00004-of-00004.safetensors
new file mode 100644
index 0000000..12954ae
--- /dev/null
+++ b/model-00004-of-00004.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:53218152f94e965eb3b1e73d231a8a18da134c7e149d02fb273fe848c7031afc
+size 565645952
diff --git a/model.safetensors.index.json b/model.safetensors.index.json
new file mode 100644
index 0000000..26f8950
--- /dev/null
+++ b/model.safetensors.index.json
@@ -0,0 +1,370 @@
+{
+ "metadata": {
+ "total_size": 6015407616
+ },
+ "weight_map": {
+ "lm_head.weight": "model-00004-of-00004.safetensors",
+ "model.embed_tokens.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.1.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.10.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.11.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.11.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.11.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.11.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.11.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.11.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.11.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.11.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.11.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.12.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.12.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.13.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.14.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.15.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.16.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.17.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.18.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.19.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.2.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.2.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.20.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.20.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.21.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.22.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.23.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.24.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.25.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.input_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.mlp.down_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.post_attention_layernorm.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.26.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.27.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.27.mlp.gate_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.mlp.up_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.27.self_attn.k_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.self_attn.o_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.self_attn.q_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.27.self_attn.v_proj.weight": "model-00002-of-00004.safetensors",
+ "model.layers.28.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.28.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.29.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.3.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.3.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.30.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.30.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.31.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.32.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.33.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.34.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.35.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.36.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.37.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.38.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.input_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.mlp.down_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.mlp.gate_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.mlp.up_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.post_attention_layernorm.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.self_attn.k_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.self_attn.o_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.self_attn.q_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.39.self_attn.v_proj.weight": "model-00003-of-00004.safetensors",
+ "model.layers.4.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.4.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.5.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.6.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.7.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.8.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.input_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.mlp.down_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.mlp.gate_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.mlp.up_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.post_attention_layernorm.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.self_attn.k_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.self_attn.o_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.self_attn.q_proj.weight": "model-00001-of-00004.safetensors",
+ "model.layers.9.self_attn.v_proj.weight": "model-00001-of-00004.safetensors",
+ "model.norm.weight": "model-00003-of-00004.safetensors"
+ }
+}
diff --git a/special_tokens_map.json b/special_tokens_map.json
new file mode 100644
index 0000000..72ecfee
--- /dev/null
+++ b/special_tokens_map.json
@@ -0,0 +1,24 @@
+{
+ "bos_token": {
+ "content": "",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false
+ },
+ "eos_token": {
+ "content": "",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false
+ },
+ "pad_token": "",
+ "unk_token": {
+ "content": "",
+ "lstrip": false,
+ "normalized": false,
+ "rstrip": false,
+ "single_word": false
+ }
+}
diff --git a/tokenizer.json b/tokenizer.json
new file mode 100644
index 0000000..3250d41
--- /dev/null
+++ b/tokenizer.json
@@ -0,0 +1,294449 @@
+{
+ "version": "1.0",
+ "truncation": {
+ "direction": "Right",
+ "max_length": 512,
+ "strategy": "LongestFirst",
+ "stride": 0
+ },
+ "padding": {
+ "strategy": {
+ "Fixed": 512
+ },
+ "direction": "Left",
+ "pad_to_multiple_of": null,
+ "pad_id": 2,
+ "pad_type_id": 0,
+ "pad_token": ""
+ },
+ "added_tokens": [
+ {
+ "id": 0,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 1,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ },
+ {
+ "id": 2,
+ "content": "",
+ "single_word": false,
+ "lstrip": false,
+ "rstrip": false,
+ "normalized": false,
+ "special": true
+ }
+ ],
+ "normalizer": {
+ "type": "Sequence",
+ "normalizers": [
+ {
+ "type": "Prepend",
+ "prepend": "▁"
+ },
+ {
+ "type": "Replace",
+ "pattern": {
+ "String": " "
+ },
+ "content": "▁"
+ }
+ ]
+ },
+ "pre_tokenizer": null,
+ "post_processor": {
+ "type": "TemplateProcessing",
+ "single": [
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 0
+ }
+ },
+ {
+ "Sequence": {
+ "id": "A",
+ "type_id": 0
+ }
+ }
+ ],
+ "pair": [
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 0
+ }
+ },
+ {
+ "Sequence": {
+ "id": "A",
+ "type_id": 0
+ }
+ },
+ {
+ "SpecialToken": {
+ "id": "",
+ "type_id": 1
+ }
+ },
+ {
+ "Sequence": {
+ "id": "B",
+ "type_id": 1
+ }
+ }
+ ],
+ "special_tokens": {
+ "": {
+ "id": "",
+ "ids": [
+ 1
+ ],
+ "tokens": [
+ ""
+ ]
+ }
+ }
+ },
+ "decoder": {
+ "type": "Sequence",
+ "decoders": [
+ {
+ "type": "Replace",
+ "pattern": {
+ "String": "▁"
+ },
+ "content": " "
+ },
+ {
+ "type": "ByteFallback"
+ },
+ {
+ "type": "Fuse"
+ },
+ {
+ "type": "Strip",
+ "content": " ",
+ "start": 1,
+ "stop": 0
+ }
+ ]
+ },
+ "model": {
+ "type": "BPE",
+ "dropout": null,
+ "unk_token": "",
+ "continuing_subword_prefix": null,
+ "end_of_word_suffix": null,
+ "fuse_unk": true,
+ "byte_fallback": true,
+ "vocab": {
+ "": 0,
+ "": 1,
+ "": 2,
+ "": 3,
+ "": 4,
+ "\n": 5,
+ "\t": 6,
+ "
": 7,
+ "
": 8,
+ "": 9,
+ "": 10,
+ "": 11,
+ "
": 12,
+ "": 13,
+ " | | ": 14,
+ "": 15,
+ "": 16,
+ "": 17,
+ "": 18,
+ "": 21,
+ "": 22,
+ "
": 23,
+ "": 24,
+ "": 25,
+ "": 26,
+ "": 27,
+ "": 28,
+ "": 29,
+ "": 30,
+ "": 31,
+ "": 32,
+ "
": 33,
+ "
": 34,
+ "
": 35,
+ "": 36,
+ "": 37,
+ "": 38,
+ "
": 39,
+ "": 40,
+ "": 41,
+ "
": 42,
+ "": 43,
+ "
": 44,
+ "
": 45,
+ "": 46,
+ "": 47,
+ "
": 48,
+ "": 49,
+ "": 50,
+ "": 51,
+ "0": 52,
+ "1": 53,
+ "2": 54,
+ "3": 55,
+ "4": 56,
+ "5": 57,
+ "6": 58,
+ "7": 59,
+ "8": 60,
+ "9": 61,
+ "+": 62,
+ "-": 63,
+ "=": 64,
+ ",": 65,
+ "。": 66,
+ "!": 67,
+ "?": 68,
+ "、": 69,
+ ":": 70,
+ "¥": 71,
+ ".": 72,
+ "!": 73,
+ "?": 74,
+ "...": 75,
+ "。。。": 76,
+ "。。。。。。": 77,
+ "《": 78,
+ "》": 79,
+ "【": 80,
+ "】": 81,
+ "『": 82,
+ "』": 83,
+ "```": 84,
+ "": 86,
+ "---": 87,
+ "": 88,
+ ";": 89,
+ ".": 90,
+ "=": 91,
+ "<": 92,
+ ">": 93,
+ "-": 94,
+ "+": 95,
+ "%": 96,
+ "‼": 97,
+ "㊣": 98,
+ "/": 99,
+ "|": 100,
+ "": 101,
+ "": 102,
+ "": 103,
+ "": 104,
+ "": 105,
+ "": 106,
+ "": 107,
+ "": 108,
+ "": 109,
+ "": 110,
+ "": 111,
+ "": 112,
+ "": 113,
+ "": 114,
+ "": 115,
+ "": 116,
+ "": 117,
+ "": 118,
+ "": 119,
+ "": 120,
+ "": 121,
+ "": 122,
+ "": 123,
+ "": 124,
+ "": 125,
+ "": 126,
+ "": 127,
+ "": 128,
+ "": 129,
+ "": 130,
+ "": 131,
+ "": 132,
+ "": 133,
+ "": 134,
+ "": 135,
+ "": 136,
+ "": 137,
+ "": 138,
+ "": 139,
+ "": 140,
+ "": 141,
+ "": 142,
+ "": 143,
+ "": 144,
+ "": 145,
+ "": 146,
+ "": 147,
+ "": 148,
+ "": 149,
+ "": 150,
+ "": 151,
+ "": 152,
+ "": 153,
+ "": 154,
+ "": 155,
+ "": 156,
+ "": 157,
+ "": 158,
+ "": 159,
+ "": 160,
+ "": 161,
+ "": 162,
+ "": 163,
+ "": 164,
+ "": 165,
+ "": 166,
+ "": 167,
+ "": 168,
+ "": 169,
+ "": 170,
+ "": 171,
+ "": 172,
+ "": 173,
+ "": 174,
+ "": 175,
+ "": 176,
+ "": 177,
+ "": 178,
+ "": 179,
+ "": 180,
+ "": 181,
+ "": 182,
+ "": 183,
+ "": 184,
+ "": 185,
+ "": 186,
+ "": 187,
+ "": 188,
+ "": 189,
+ "": 190,
+ "": 191,
+ "": 192,
+ "": 193,
+ "": 194,
+ "": 195,
+ "": 196,
+ "": 197,
+ "": 198,
+ "": 199,
+ "": 200,
+ "": 201,
+ "": 202,
+ "": 203,
+ "": 204,
+ "": 205,
+ "": 206,
+ "": 207,
+ "": 208,
+ "": 209,
+ "": 210,
+ "": 211,
+ "": 212,
+ "": 213,
+ "": 214,
+ "": 215,
+ "": 216,
+ "": 217,
+ "": 218,
+ "": 219,
+ "": 220,
+ "": 221,
+ "": 222,
+ "": 223,
+ "": 224,
+ "": 225,
+ "": 226,
+ "": 227,
+ "": 228,
+ "": 229,
+ "": 230,
+ "": 231,
+ "": 232,
+ "": 233,
+ "": 234,
+ "": 235,
+ "": 236,
+ "": 237,
+ "": 238,
+ "": 239,
+ "": 240,
+ "": 241,
+ "": 242,
+ "": 243,
+ "": 244,
+ "": 245,
+ "": 246,
+ "": 247,
+ "": 248,
+ "": 249,
+ "": 250,
+ "": 251,
+ "": 252,
+ "": 253,
+ "": 254,
+ "": 255,
+ "": 256,
+ "": 257,
+ "": 258,
+ "": 259,
+ "": 260,
+ "": 261,
+ "": 262,
+ "": 263,
+ "": 264,
+ "": 265,
+ "": 266,
+ "": 267,
+ "": 268,
+ "": 269,
+ "": 270,
+ "": 271,
+ "": 272,
+ "": 273,
+ "": 274,
+ "": 275,
+ "": 276,
+ "": 277,
+ "": 278,
+ "": 279,
+ "": 280,
+ "": 281,
+ "": 282,
+ "": 283,
+ "": 284,
+ "": 285,
+ "": 286,
+ "": 287,
+ "": 288,
+ "": 289,
+ "": 290,
+ "": 291,
+ "": 292,
+ "": 293,
+ "": 294,
+ "": 295,
+ "": 296,
+ "": 297,
+ "": 298,
+ "": 299,
+ "": 300,
+ "": 301,
+ "": 302,
+ "": 303,
+ "": 304,
+ "": 305,
+ "": 306,
+ "": 307,
+ "": 308,
+ "": 309,
+ "": 310,
+ "": 311,
+ "": 312,
+ "": 313,
+ "": 314,
+ "": 315,
+ "": 316,
+ "": 317,
+ "": 318,
+ "": 319,
+ "": 320,
+ "": 321,
+ "": 322,
+ "": 323,
+ "": 324,
+ "": 325,
+ "": 326,
+ "": 327,
+ "": 328,
+ "": 329,
+ "": 330,
+ "": 331,
+ "": 332,
+ "": 333,
+ "": 334,
+ "": 335,
+ "": 336,
+ "": 337,
+ "": 338,
+ "": 339,
+ "": 340,
+ "": 341,
+ "": 342,
+ "": 343,
+ "": 344,
+ "": 345,
+ "": 346,
+ "": 347,
+ "": 348,
+ "": 349,
+ "": 350,
+ "": 351,
+ "": 352,
+ "": 353,
+ "