commit e1b5a9a49d26d91463d6774bebf77c2b5a437511 Author: ModelHub XC Date: Tue May 26 23:12:12 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: unsloth/Llama-3.2-1B-Instruct-GGUF Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..5d0e0db --- /dev/null +++ b/.gitattributes @@ -0,0 +1,62 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-BF16.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-1B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/Llama-3.2-1B-Instruct-BF16.gguf b/Llama-3.2-1B-Instruct-BF16.gguf new file mode 100644 index 0000000..e1aba9d --- /dev/null +++ b/Llama-3.2-1B-Instruct-BF16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5d05e90d179fa27c01a049be1574a0201bdc09ba0b345eb85e1dbf3b87610f3 +size 2479595264 diff --git a/Llama-3.2-1B-Instruct-F16.gguf b/Llama-3.2-1B-Instruct-F16.gguf new file mode 100644 index 0000000..be8d441 --- /dev/null +++ b/Llama-3.2-1B-Instruct-F16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e2a46fe01895b16776ed927aac8f56c5661a06d47c8a6f87ec297c805fbe3c2 +size 2479595168 diff --git a/Llama-3.2-1B-Instruct-IQ4_NL.gguf b/Llama-3.2-1B-Instruct-IQ4_NL.gguf new file mode 100644 index 0000000..06eeb4f --- /dev/null +++ b/Llama-3.2-1B-Instruct-IQ4_NL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:705bb011a5f2c46612ee609fe402bbb026f59fff2cf6018252089433196753f4 +size 773025824 diff --git a/Llama-3.2-1B-Instruct-IQ4_XS.gguf b/Llama-3.2-1B-Instruct-IQ4_XS.gguf new file mode 100644 index 0000000..ddf6b06 --- /dev/null +++ b/Llama-3.2-1B-Instruct-IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3cceb8104fdf3e82e063f0f0ca20d90e4b0b933318657741c91c0fffb539297c +size 743141408 diff --git a/Llama-3.2-1B-Instruct-Q2_K.gguf b/Llama-3.2-1B-Instruct-Q2_K.gguf new file mode 100644 index 0000000..94a18ad --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b7091a92bc10d70392a91ebe06cd43e1f5048ae0162e88f8fbe8445447ceae8 +size 580874272 diff --git a/Llama-3.2-1B-Instruct-Q2_K_L.gguf b/Llama-3.2-1B-Instruct-Q2_K_L.gguf new file mode 100644 index 0000000..94a18ad --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q2_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b7091a92bc10d70392a91ebe06cd43e1f5048ae0162e88f8fbe8445447ceae8 +size 580874272 diff --git a/Llama-3.2-1B-Instruct-Q3_K_M.gguf b/Llama-3.2-1B-Instruct-Q3_K_M.gguf new file mode 100644 index 0000000..4198c69 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:38b98a4abebe441bbb52c6364a79abe026593516df25424e0ef2410d38db715e +size 690843680 diff --git a/Llama-3.2-1B-Instruct-Q3_K_S.gguf b/Llama-3.2-1B-Instruct-Q3_K_S.gguf new file mode 100644 index 0000000..ba2c7f3 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fe2f725782ce9e742ecb8fc8ba9a3e1621736297d98bcdb2145e30ad88c0a4f +size 641691680 diff --git a/Llama-3.2-1B-Instruct-Q4_0.gguf b/Llama-3.2-1B-Instruct-Q4_0.gguf new file mode 100644 index 0000000..5797072 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:66bfbb2d48bdb77cd56bd03ef820deff3c4a74b1a09de3b917ae13e72c1a70c2 +size 773025824 diff --git a/Llama-3.2-1B-Instruct-Q4_1.gguf b/Llama-3.2-1B-Instruct-Q4_1.gguf new file mode 100644 index 0000000..0a58b17 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a8bfc798acebfbadf78fb90f21ab5e6464f652dca08f32e188ef03c8b96003e9 +size 831746080 diff --git a/Llama-3.2-1B-Instruct-Q4_K_M.gguf b/Llama-3.2-1B-Instruct-Q4_K_M.gguf new file mode 100644 index 0000000..57bccf3 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3f5a22426976ab26cfe84dba63c1d08391717abb1af893e10f1b2968d862dcc1 +size 807694368 diff --git a/Llama-3.2-1B-Instruct-Q4_K_S.gguf b/Llama-3.2-1B-Instruct-Q4_K_S.gguf new file mode 100644 index 0000000..4c27eb3 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bedf466e28a9a54769a24457bc5ee8526421c9d2653ae52b4bfd56a059fbccf0 +size 775647264 diff --git a/Llama-3.2-1B-Instruct-Q5_K_M.gguf b/Llama-3.2-1B-Instruct-Q5_K_M.gguf new file mode 100644 index 0000000..862435a --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69dce91345442121eb3195370337eefa02cf076c7d84bd39adc0ce9552ccdfef +size 911503392 diff --git a/Llama-3.2-1B-Instruct-Q5_K_S.gguf b/Llama-3.2-1B-Instruct-Q5_K_S.gguf new file mode 100644 index 0000000..86dc2ca --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d4cd7cb5bc691a9d8b178eb077fa31cde1df1fa7182d09cf2fd10feeeeaa8cd +size 892563488 diff --git a/Llama-3.2-1B-Instruct-Q6_K.gguf b/Llama-3.2-1B-Instruct-Q6_K.gguf new file mode 100644 index 0000000..7f62a3f --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:663ac8413e6d64a1412840c76bd1300ac19293454ae6802ff3c9b5543f93b531 +size 1021800480 diff --git a/Llama-3.2-1B-Instruct-Q8_0.gguf b/Llama-3.2-1B-Instruct-Q8_0.gguf new file mode 100644 index 0000000..8447b84 --- /dev/null +++ b/Llama-3.2-1B-Instruct-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3f87a880027e7b9ea8e0da9e4009584336f352af444a0e6e5c20721ac4c7ffd1 +size 1321082528 diff --git a/Llama-3.2-1B-Instruct-UD-IQ1_M.gguf b/Llama-3.2-1B-Instruct-UD-IQ1_M.gguf new file mode 100644 index 0000000..3c72384 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-IQ1_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2296d5ce14318b1c7359e17ab8b86cd0ae93593dc4d98be9d267a60881e1a30c +size 439181344 diff --git a/Llama-3.2-1B-Instruct-UD-IQ1_S.gguf b/Llama-3.2-1B-Instruct-UD-IQ1_S.gguf new file mode 100644 index 0000000..e80529d --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-IQ1_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:60f05b86eb901189dabd946cff4206656395316fcf85aa111d35545e21b0bc65 +size 421617696 diff --git a/Llama-3.2-1B-Instruct-UD-IQ2_M.gguf b/Llama-3.2-1B-Instruct-UD-IQ2_M.gguf new file mode 100644 index 0000000..15c7966 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-IQ2_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b79dd1b66164fe8c3e77e2b56b474f7983482a33eede977c68eb6d3e79a14ac +size 536322080 diff --git a/Llama-3.2-1B-Instruct-UD-IQ2_XXS.gguf b/Llama-3.2-1B-Instruct-UD-IQ2_XXS.gguf new file mode 100644 index 0000000..2b021e4 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-IQ2_XXS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:81cec721583c76fa3b0972a173a28777c76bdcbfa2df66656a95e6198456f660 +size 464330784 diff --git a/Llama-3.2-1B-Instruct-UD-IQ3_XXS.gguf b/Llama-3.2-1B-Instruct-UD-IQ3_XXS.gguf new file mode 100644 index 0000000..2a354b9 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-IQ3_XXS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f39f7695f7e9d8d6b4f6a7278f606e9177ef86ddb10c59d2538e3a7ddc332cb6 +size 575381536 diff --git a/Llama-3.2-1B-Instruct-UD-Q2_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q2_K_XL.gguf new file mode 100644 index 0000000..720ca5e --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q2_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca9b2e5eb5edf80886f1b543231cd122e638d0d524e0654e12d0732d6c78b7bc +size 605532192 diff --git a/Llama-3.2-1B-Instruct-UD-Q3_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q3_K_XL.gguf new file mode 100644 index 0000000..0492327 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q3_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ced2ef04c4219bd88bb852398f06ab9d5e60d356a0ec6260ae0f368e06da95b8 +size 723996704 diff --git a/Llama-3.2-1B-Instruct-UD-Q4_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q4_K_XL.gguf new file mode 100644 index 0000000..5e2ddc1 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q4_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb1e301352132cc8e804ccde1c024f78c060d2da64e260e2123e057cb0540094 +size 834203680 diff --git a/Llama-3.2-1B-Instruct-UD-Q5_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q5_K_XL.gguf new file mode 100644 index 0000000..35f7ae3 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q5_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:183e5fe9a9afaaebe889832f9679473c0ae9120915e5f6e30b297475e55067ee +size 920031264 diff --git a/Llama-3.2-1B-Instruct-UD-Q6_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q6_K_XL.gguf new file mode 100644 index 0000000..e00de2e --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q6_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4b735094c603d9f76f533c6721da2c65d942c6275163948ee622b9c6eec6482 +size 1191059488 diff --git a/Llama-3.2-1B-Instruct-UD-Q8_K_XL.gguf b/Llama-3.2-1B-Instruct-UD-Q8_K_XL.gguf new file mode 100644 index 0000000..774b217 --- /dev/null +++ b/Llama-3.2-1B-Instruct-UD-Q8_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5c890975f8a8a88277e74a875623cb2c410b020b546c4df1ad984732787a1129 +size 1843568672 diff --git a/README.md b/README.md new file mode 100644 index 0000000..0dbbb34 --- /dev/null +++ b/README.md @@ -0,0 +1,71 @@ +--- +base_model: meta-llama/Llama-3.2-1B-Instruct +language: +- en +library_name: transformers +license: llama3.2 +tags: +- llama-3 +- llama +- meta +- facebook +- unsloth +- transformers +--- + +## ***See [our collection](https://huggingface.co/collections/unsloth/llama-32-66f46afde4ca573864321a22) for all versions of Llama 3.2 including GGUF, 4-bit and original 16-bit formats.*** + +# GGUF uploads + +16bit, 8bit, 6bit, 5bit, 4bit, 3bit and 2bit uploads avaliable. + +# Finetune Llama 3.2, Gemma 2, Mistral 2-5x faster with 70% less memory via Unsloth! + +We have a free Google Colab Tesla T4 notebook for Llama 3.2 (3B) here: https://colab.research.google.com/drive/1T5-zKWM_5OD21QHwXHiV9ixTRR7k3iB9?usp=sharing + +[](https://discord.gg/unsloth) +[](https://github.com/unslothai/unsloth) + +# unsloth/Llama-3.2-1B-Instruct +For more details on the model, please go to Meta's original [model card](https://huggingface.co/meta-llama/Llama-3.2-3B-Instruct) + +## ✨ Finetune for Free + +All notebooks are **beginner friendly**! Add your dataset, click "Run All", and you'll get a 2x faster finetuned model which can be exported to GGUF, vLLM or uploaded to Hugging Face. + +| Unsloth supports | Free Notebooks | Performance | Memory use | +|-----------------|--------------------------------------------------------------------------------------------------------------------------|-------------|----------| +| **Llama-3.2 (3B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Llama-3.1 (11B vision)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Llama-3.1 (8B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Phi-3.5 (mini)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1lN6hPQveB_mHSnTOYifygFcrO8C1bxq4?usp=sharing) | 2x faster | 50% less | +| **Gemma 2 (9B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1vIrqH5uYDQwsJ4-OO3DErvuv4pBgVwk4?usp=sharing) | 2.4x faster | 58% less | +| **Mistral (7B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Dyauq4kTZoLewQ1cApceUQVNcnnNTzg_?usp=sharing) | 2.2x faster | 62% less | +| **DPO - Zephyr** | [▶️ Start on Colab](https://colab.research.google.com/drive/15vttTpzzVXv_tJwEk-hIcQ0S9FcEWvwP?usp=sharing) | 1.9x faster | 19% less | + +- This [conversational notebook](https://colab.research.google.com/drive/1Aau3lgPzeZKQ-98h69CCu1UJcvIBLmy2?usp=sharing) is useful for ShareGPT ChatML / Vicuna templates. +- This [text completion notebook](https://colab.research.google.com/drive/1ef-tab5bhkvWmBOObepl1WgJvfvSzn5Q?usp=sharing) is for raw text. This [DPO notebook](https://colab.research.google.com/drive/15vttTpzzVXv_tJwEk-hIcQ0S9FcEWvwP?usp=sharing) replicates Zephyr. +- \* Kaggle has 2x T4s, but we use 1. Due to overhead, 1x T4 is 5x faster. + +## Special Thanks +A huge thank you to the Meta and Llama team for creating and releasing these models. + +## Model Information + +The Meta Llama 3.2 collection of multilingual large language models (LLMs) is a collection of pretrained and instruction-tuned generative models in 1B and 3B sizes (text in/text out). The Llama 3.2 instruction-tuned text only models are optimized for multilingual dialogue use cases, including agentic retrieval and summarization tasks. They outperform many of the available open source and closed chat models on common industry benchmarks. + +**Model developer**: Meta + +**Model Architecture:** Llama 3.2 is an auto-regressive language model that uses an optimized transformer architecture. The tuned versions use supervised fine-tuning (SFT) and reinforcement learning with human feedback (RLHF) to align with human preferences for helpfulness and safety. + +**Supported languages:** English, German, French, Italian, Portuguese, Hindi, Spanish, and Thai are officially supported. Llama 3.2 has been trained on a broader collection of languages than these 8 supported languages. Developers may fine-tune Llama 3.2 models for languages beyond these supported languages, provided they comply with the Llama 3.2 Community License and the Acceptable Use Policy. Developers are always expected to ensure that their deployments, including those that involve additional languages, are completed safely and responsibly. + +**Llama 3.2 family of models** Token counts refer to pretraining data only. All model versions use Grouped-Query Attention (GQA) for improved inference scalability. + +**Model Release Date:** Sept 25, 2024 + +**Status:** This is a static model trained on an offline dataset. Future versions may be released that improve model capabilities and safety. + +**License:** Use of Llama 3.2 is governed by the [Llama 3.2 Community License](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/LICENSE) (a custom, commercial license agreement). + +Where to send questions or comments about the model Instructions on how to provide feedback or comments on the model can be found in the model [README](https://github.com/meta-llama/llama3). For more technical information about generation parameters and recipes for how to use Llama 3.1 in applications, please go [here](https://github.com/meta-llama/llama-recipes). diff --git a/config.json b/config.json new file mode 100644 index 0000000..efa3058 --- /dev/null +++ b/config.json @@ -0,0 +1,37 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "eos_token_id": 128009, + "head_dim": 64, + "hidden_act": "silu", + "hidden_size": 2048, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 32, + "num_hidden_layers": 16, + "num_key_value_heads": 8, + "pad_token_id": 128004, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "torch_dtype": "bfloat16", + "transformers_version": "4.52.0.dev0", + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..159097f --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "others", "allow_remote": true} \ No newline at end of file diff --git a/imatrix_unsloth.dat b/imatrix_unsloth.dat new file mode 100644 index 0000000..f514c47 Binary files /dev/null and b/imatrix_unsloth.dat differ