From 9df2fa0e39270d49d8e4ad2675bcf0f4ff376a34 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Tue, 23 Jun 2026 11:06:13 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: dphn/dolphin-2.9-llama3-8b-1m Source: Original Platform --- .gitattributes | 35 +++++ README.md | 237 +++++++++++++++++++++++++++++++ config.json | 28 ++++ configuration.json | 1 + generation_config.json | 7 + model-00001-of-00004.safetensors | 3 + model-00002-of-00004.safetensors | 3 + model-00003-of-00004.safetensors | 3 + model-00004-of-00004.safetensors | 3 + model.safetensors.index.json | 3 + special_tokens_map.json | 23 +++ tokenizer.json | 3 + tokenizer_config.json | 3 + 13 files changed, 352 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 config.json create mode 100644 configuration.json create mode 100644 generation_config.json create mode 100644 model-00001-of-00004.safetensors create mode 100644 model-00002-of-00004.safetensors create mode 100644 model-00003-of-00004.safetensors create mode 100644 model-00004-of-00004.safetensors create mode 100644 model.safetensors.index.json create mode 100644 special_tokens_map.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..a6344aa --- /dev/null +++ b/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..fdc7ddd --- /dev/null +++ b/README.md @@ -0,0 +1,237 @@ +--- +license: other +base_model: meta-llama/Meta-Llama-3-8B +tags: +- generated_from_trainer +- axolotl +model-index: +- name: out + results: [] +datasets: +- cognitivecomputations/Dolphin-2.9 +- teknium/OpenHermes-2.5 +- m-a-p/CodeFeedback-Filtered-Instruction +- cognitivecomputations/dolphin-coder +- cognitivecomputations/samantha-data +- HuggingFaceH4/ultrachat_200k +- microsoft/orca-math-word-problems-200k +- abacusai/SystemChat-1.1 +- Locutusque/function-calling-chatml +- internlm/Agent-FLAN +--- + +# Dolphin 2.9 Llama 3 8b 1m 🐬 + +Curated and trained by Eric Hartford, Lucas Atkins, and Fernando Fernandes, and Cognitive Computations + +[![Discord](https://img.shields.io/discord/1156064224225808488?logo=Discord&logoColor=%23ffffff&label=Discord&link=https%3A%2F%2Fdiscord.gg%2FtCMkMDDHwm)](https://discord.gg/cognitivecomputations) +Discord: https://discord.gg/cognitivecomputations + + + +This version of Dolphin has a 1 million token context. I have applied `winglian/llama-3-1m-context-gradient-lora` - created by @gradientai and @winglian and sponsored by @CrusoeCloud + +A bug has been found in the Dolphin 2.9 dataset in SystemConversations that causes the model to overly talk about the "SYSTEM MESSAGE". To counter this, we recommend you add a statement in the system message directing the model not to mention the system message. An example system message is "The assistant is named Dolphin. A helpful and friendly AI assistant, Dolphin avoids discussing the system message unless directly asked about it." + +My appreciation for the sponsors of Dolphin 2.9: +- [Crusoe Cloud](https://crusoe.ai/) - provided excellent on-demand 10xL40S node + +This model is based on Llama-3-8b, and is governed by [META LLAMA 3 COMMUNITY LICENSE AGREEMENT](LICENSE) + +The base model has 8k context, and the full-weight fine-tuning was with 4k sequence length. + +It took 2.5 days on 8x L40S provided by Crusoe Cloud + +This model was trained FFT on all parameters, using ChatML prompt template format. + +example: + +``` +<|im_start|>system +You are Dolphin, a helpful AI assistant.<|im_end|> +<|im_start|>user +{prompt}<|im_end|> +<|im_start|>assistant + +``` + +Dolphin-2.9 has a variety of instruction, conversational, and coding skills. It also has initial agentic abilities and supports function calling. + +Dolphin is uncensored. I have filtered the dataset to remove alignment and bias. This makes the model more compliant. You are advised to implement your own alignment layer before exposing the model as a service. It will be highly compliant with any requests, even unethical ones. Please read my blog post about uncensored models. https://erichartford.com/uncensored-models You are responsible for any content you create using this model. Enjoy responsibly. + +Dolphin is licensed according to Meta's Llama license. I grant permission for any use, including commercial, that falls within accordance with Meta's Llama-3 license. Dolphin was trained on data generated from GPT4, among other models. + +[Built with Axolotl](https://github.com/OpenAccess-AI-Collective/axolotl) +
See axolotl config + +axolotl version: `0.4.0` +```yaml +base_model: meta-llama/Meta-Llama-3-8B +model_type: AutoModelForCausalLM +tokenizer_type: AutoTokenizer +tokenizer_use_fast: false + + +load_in_8bit: false +load_in_4bit: false +strict: false +model_config: + +datasets: + - path: /workspace/datasets/dolphin-2.9/dolphin201-sharegpt2.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/Ultrachat200kunfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/dolphin-coder-translate-sharegpt2.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/dolphin-coder-codegen-sharegpt2.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/m-a-p_Code-Feedback-sharegpt-unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/m-a-p_CodeFeedback-Filtered-Instruction-sharegpt-unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/not_samantha_norefusals.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/Orca-Math-resort-unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/agent_instruct_react_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/toolbench_instruct_j1s1_3k_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/toolbench_negative_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/toolbench_react_10p_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/toolbench_tflan_cot_30p_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/openhermes200k_unfiltered.jsonl + type: sharegpt + conversation: chatml + - path: /workspace/datasets/dolphin-2.9/SystemConversations.jsonl + type: sharegpt + conversation: chatml + +chat_template: chatml + + +dataset_prepared_path: /workspace/datasets/dolphin-2.9/thingy +val_set_size: 0.0002 +output_dir: ./out + +sequence_len: 4096 +sample_packing: true +pad_to_sequence_len: true + +gradient_accumulation_steps: 4 +micro_batch_size: 3 +num_epochs: 3 +logging_steps: 1 +optimizer: adamw_8bit +lr_scheduler: cosine +learning_rate: 2e-5 + +wandb_project: dolphin-2.9-mixtral-8x22b +wandb_watch: +wandb_run_id: +wandb_log_model: + +train_on_inputs: false +group_by_length: false +bf16: auto +fp16: +tf32: false + +gradient_checkpointing: true +gradient_checkpointing_kwargs: + use_reentrant: false +early_stopping_patience: +resume_from_checkpoint: +local_rank: +logging_steps: 1 +xformers_attention: +flash_attention: true +saves_per_epoch: 4 +save_total_limit: 2 +save_steps: +evals_per_epoch: 4 +eval_sample_packing: false +debug: +deepspeed: deepspeed_configs/zero3_bf16.json +weight_decay: 0.05 +fsdp: +fsdp_config: +special_tokens: + eos_token: "<|im_end|>" + pad_token: "<|end_of_text|>" +tokens: + - "<|im_start|>" + - "<|im_end|>" + +``` + +

+ +## Quants + +GGUF : https://huggingface.co/QuantFactory/dolphin-2.9-llama3-8b-GGUF + +GGUF with imatrix: https://huggingface.co/bartowski/dolphin-2.9-llama3-8b-GGUF + +Exllamav2: https://huggingface.co/bartowski/dolphin-2.9-llama3-8b-exl2 + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 2e-05 +- train_batch_size: 3 +- eval_batch_size: 3 +- seed: 42 +- distributed_type: multi-GPU +- num_devices: 8 +- gradient_accumulation_steps: 4 +- total_train_batch_size: 96 +- total_eval_batch_size: 24 +- optimizer: Adam with betas=(0.9,0.999) and epsilon=1e-08 +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 7 +- num_epochs: 3 + +### Training results + +| Training Loss | Epoch | Step | Validation Loss | +|:-------------:|:------:|:----:|:---------------:| +| 1.146 | 0.0005 | 1 | 1.1064 | +| 0.6962 | 0.2501 | 555 | 0.6636 | +| 0.6857 | 0.5001 | 1110 | 0.6503 | +| 0.6592 | 0.7502 | 1665 | 0.6419 | +| 0.6465 | 1.0002 | 2220 | 0.6317 | +| 0.5295 | 1.2395 | 2775 | 0.6408 | +| 0.5302 | 1.4895 | 3330 | 0.6351 | +| 0.5188 | 1.7396 | 3885 | 0.6227 | +| 0.521 | 1.9896 | 4440 | 0.6168 | +| 0.3968 | 2.2289 | 4995 | 0.6646 | +| 0.3776 | 2.4789 | 5550 | 0.6619 | +| 0.3983 | 2.7290 | 6105 | 0.6602 | + + +### Framework versions + +- Transformers 4.40.0 +- Pytorch 2.2.2+cu121 +- Datasets 2.18.0 +- Tokenizers 0.19.1 \ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..35874f8 --- /dev/null +++ b/config.json @@ -0,0 +1,28 @@ +{ + "_name_or_path": "cognitivecomputations/dolphin-2.9-llama3-8b-1m", + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "eos_token_id": 128256, + "hidden_act": "silu", + "hidden_size": 4096, + "initializer_range": 0.02, + "intermediate_size": 14336, + "max_position_embeddings": 1048576, + "model_type": "llama", + "num_attention_heads": 32, + "num_hidden_layers": 32, + "num_key_value_heads": 8, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": null, + "rope_theta": 2804339835.0, + "tie_word_embeddings": false, + "torch_dtype": "float16", + "transformers_version": "4.40.1", + "use_cache": true, + "vocab_size": 128258 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..7f7d1b7 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,7 @@ +{ + "_from_model_config": true, + "bos_token_id": 128000, + "do_sample": true, + "eos_token_id": 128001, + "transformers_version": "4.40.1" +} diff --git a/model-00001-of-00004.safetensors b/model-00001-of-00004.safetensors new file mode 100644 index 0000000..a708547 --- /dev/null +++ b/model-00001-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7e2503d96cd11ef2a3c48dd39aa0a4d57d433ca9ddcaee0f635d81e92c00f7f +size 4976714976 diff --git a/model-00002-of-00004.safetensors b/model-00002-of-00004.safetensors new file mode 100644 index 0000000..6c90991 --- /dev/null +++ b/model-00002-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:186045d3cb140ae78687efa99a1d51f94d9f8f5c987428a5df2be38c44b9e66f +size 4999802616 diff --git a/model-00003-of-00004.safetensors b/model-00003-of-00004.safetensors new file mode 100644 index 0000000..f652ae3 --- /dev/null +++ b/model-00003-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7ef7da7e7e6500c323dc30435c3d631ef2be4a3b0b49f44d46c05cfa23456b3b +size 4915916080 diff --git a/model-00004-of-00004.safetensors b/model-00004-of-00004.safetensors new file mode 100644 index 0000000..4fdd215 --- /dev/null +++ b/model-00004-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d8699df7b577f5017bcfc90804f2754c18ec59960be528b80678ea5f87726fb7 +size 1168155192 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..46605d6 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:95e675e9f23a294baff7e3a614ae39e4cef7f2b7aef18cf8baf18515eec29ef1 +size 23950 diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..44e8cb8 --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,23 @@ +{ + "bos_token": { + "content": "<|begin_of_text|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "eos_token": { + "content": "<|im_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "<|end_of_text|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..3f70237 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:026418a30cded1e472598798da74eeb7d3f402ecff1319f7bfc2ab3be257ea80 +size 9084867 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..d4c61cf --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d801631730dec6198b8f590fbed6006ca1531bd36d48b8690ede742d81d83442 +size 51269