From 7eff57c85d2791d8cadbb26c301b71376afecd93 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sun, 9 Aug 2026 04:47:13 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: sakuraumi/Sakura-13B-Novel-Korean Source: Original Platform --- .gitattributes | 34 +++++++++++++++++++++++++++ README.md | 40 ++++++++++++++++++++++++++++++++ config.json | 27 +++++++++++++++++++++ configuration.json | 10 ++++++++ generation_config.json | 7 ++++++ pytorch_model-00001-of-00003.bin | 3 +++ pytorch_model-00002-of-00003.bin | 3 +++ pytorch_model-00003-of-00003.bin | 3 +++ pytorch_model.bin.index.json | 3 +++ special_tokens_map.json | 12 ++++++++++ tokenizer.model | 3 +++ tokenizer_config.json | 34 +++++++++++++++++++++++++++ 12 files changed, 179 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 config.json create mode 100644 configuration.json create mode 100644 generation_config.json create mode 100644 pytorch_model-00001-of-00003.bin create mode 100644 pytorch_model-00002-of-00003.bin create mode 100644 pytorch_model-00003-of-00003.bin create mode 100644 pytorch_model.bin.index.json create mode 100644 special_tokens_map.json create mode 100644 tokenizer.model create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..7bc225d --- /dev/null +++ b/.gitattributes @@ -0,0 +1,34 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bin.* filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zstandard filter=lfs diff=lfs merge=lfs -text +*.tfevents* filter=lfs diff=lfs merge=lfs -text +*.db* filter=lfs diff=lfs merge=lfs -text +*.ark* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text +**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text \ No newline at end of file diff --git a/README.md b/README.md new file mode 100644 index 0000000..6486cdc --- /dev/null +++ b/README.md @@ -0,0 +1,40 @@ +--- +frameworks: +- Pytorch +license: other +tasks: +- text-generation +--- + +#### Clone with HTTP +```bash + git clone https://www.modelscope.cn/sakuraumi/Sakura-13B-Novel-Korean.git +``` + +
+

+ Sakura-13B-Novel-Korean +

+
+ +# 介绍 + +基于LLaMA2-13B,OpenBuddy构建,在约360K轻小说中韩文本数据上微调2个epoch. + +# 模型 + +- Finetuned by [SakuraUmi](https://github.com/pipixia244) +- Finetuned on [Openbuddy-LLaMA2-13B](https://huggingface.co/OpenBuddy/openbuddy-llama2-13b-v11.1-fp16) +- Data support: [CjangCjengh](https://github.com/CjangCjengh) +- Base model: [LLaMA2-13B](https://huggingface.co/meta-llama/Llama-2-13b-chat-hf) +- Languages: Chinese/Korean + +# 推理 + +- Prompt构建 + +```python + input_text = "" # 用户输入 + query = "将下面的韩语文本翻译成中文:" + input_text + prompt = "User: " + query + "\nAssistant: " +``` \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..773c4cd --- /dev/null +++ b/config.json @@ -0,0 +1,27 @@ +{ + "_name_or_path": "OpenBuddy/openbuddy-llama2-13b-v11.1-bf16", + "architectures": [ + "LlamaForCausalLM" + ], + "bos_token_id": 1, + "eos_token_id": 2, + "hidden_act": "silu", + "hidden_size": 5120, + "initializer_range": 0.02, + "intermediate_size": 13824, + "max_position_embeddings": 4096, + "model_type": "llama", + "num_attention_heads": 40, + "num_hidden_layers": 40, + "num_key_value_heads": 40, + "pad_token_id": 0, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": null, + "tie_word_embeddings": false, + "torch_dtype": "bfloat16", + "transformers_version": "4.31.0", + "use_cache": false, + "use_flash_attention": true, + "vocab_size": 37632 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..2b39a5b --- /dev/null +++ b/configuration.json @@ -0,0 +1,10 @@ +{ + "framework": "pytorch", + "task": "text-generation", + "model": { + "type": "llama2" + }, + "pipeline": { + "type": "llama2-text-generation-pipeline" + } +} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..2b10330 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "_from_model_config": true, + "bos_token_id": 1, + "eos_token_id": 2, + "pad_token_id": 0, + "transformers_version": "4.31.0" +} diff --git a/pytorch_model-00001-of-00003.bin b/pytorch_model-00001-of-00003.bin new file mode 100644 index 0000000..92656a2 --- /dev/null +++ b/pytorch_model-00001-of-00003.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10a2c439cfffce02f2014b092661b256cc889b68a0e46410ded41f08195bf2df +size 9953961974 diff --git a/pytorch_model-00002-of-00003.bin b/pytorch_model-00002-of-00003.bin new file mode 100644 index 0000000..e300b86 --- /dev/null +++ b/pytorch_model-00002-of-00003.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8252c54f44fd0930a504b1946ee0448696e7186931bf7739ccb329367245341 +size 9956584547 diff --git a/pytorch_model-00003-of-00003.bin b/pytorch_model-00003-of-00003.bin new file mode 100644 index 0000000..6151611 --- /dev/null +++ b/pytorch_model-00003-of-00003.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e8a75e01177a02854186aaba7c86f467c536699911a77eaa3e40e86c4b068e5f +size 6236649887 diff --git a/pytorch_model.bin.index.json b/pytorch_model.bin.index.json new file mode 100644 index 0000000..b70efce --- /dev/null +++ b/pytorch_model.bin.index.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1ed3f4577b794e89d47cb70baade2f3fb037737616feb86db958e89436b4ae80 +size 29894 diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..5d7b70a --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,12 @@ +{ + "bos_token": "", + "eos_token": "", + "pad_token": "", + "unk_token": { + "content": "", + "lstrip": false, + "normalized": true, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.model b/tokenizer.model new file mode 100644 index 0000000..3109a4d --- /dev/null +++ b/tokenizer.model @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f440c53d2cc6f14a7ed7124dea5f5a7402fb4fc95bccb5d8be6d0f7e74d327ed +size 568229 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..1e4fd8d --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,34 @@ +{ + "add_bos_token": true, + "add_eos_token": false, + "bos_token": { + "__type": "AddedToken", + "content": "", + "lstrip": false, + "normalized": true, + "rstrip": false, + "single_word": false + }, + "clean_up_tokenization_spaces": false, + "eos_token": { + "__type": "AddedToken", + "content": "", + "lstrip": false, + "normalized": true, + "rstrip": false, + "single_word": false + }, + "legacy": true, + "model_max_length": 1000000000000000019884624838656, + "pad_token": null, + "sp_model_kwargs": {}, + "tokenizer_class": "LlamaTokenizer", + "unk_token": { + "__type": "AddedToken", + "content": "", + "lstrip": false, + "normalized": true, + "rstrip": false, + "single_word": false + } +}