commit 8c9518d7f5d4bbde3a4483e8269d8249e23eb202 Author: ModelHub XC Date: Sun Jul 5 03:55:14 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: unsloth/Llama-3.2-3B-Instruct-GGUF Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..58db813 --- /dev/null +++ b/.gitattributes @@ -0,0 +1,62 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-BF16.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text +Llama-3.2-3B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/Llama-3.2-3B-Instruct-BF16.gguf b/Llama-3.2-3B-Instruct-BF16.gguf new file mode 100644 index 0000000..d371176 --- /dev/null +++ b/Llama-3.2-3B-Instruct-BF16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9b8dce13b6cbcd8b20037bd6383ee8e747b5034ca32f40b5b8ee2efa9ebf56b8 +size 6433687744 diff --git a/Llama-3.2-3B-Instruct-F16.gguf b/Llama-3.2-3B-Instruct-F16.gguf new file mode 100644 index 0000000..e54a925 --- /dev/null +++ b/Llama-3.2-3B-Instruct-F16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:db0ebda07c65c6cc6d15fb1c28f0f401fe64003aa88cfbf701be58a1c4122d54 +size 6433687616 diff --git a/Llama-3.2-3B-Instruct-IQ4_NL.gguf b/Llama-3.2-3B-Instruct-IQ4_NL.gguf new file mode 100644 index 0000000..22bad1c --- /dev/null +++ b/Llama-3.2-3B-Instruct-IQ4_NL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c559cd2958092a48c7abc7175c9059ef5cdc28336ae737816a0bca8a89a4760b +size 1917190592 diff --git a/Llama-3.2-3B-Instruct-IQ4_XS.gguf b/Llama-3.2-3B-Instruct-IQ4_XS.gguf new file mode 100644 index 0000000..c5e72bc --- /dev/null +++ b/Llama-3.2-3B-Instruct-IQ4_XS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6fca4bfe9ebadd5a4971c8e42b9aae3ec23d7750c95647b797b8803182eab494 +size 1829110208 diff --git a/Llama-3.2-3B-Instruct-Q2_K.gguf b/Llama-3.2-3B-Instruct-Q2_K.gguf new file mode 100644 index 0000000..467afcc --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:162411ff9b531cb41d1204c5b81c9a52f6d2cf061c77ea645c3750cc59e9016a +size 1363935680 diff --git a/Llama-3.2-3B-Instruct-Q2_K_L.gguf b/Llama-3.2-3B-Instruct-Q2_K_L.gguf new file mode 100644 index 0000000..467afcc --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q2_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:162411ff9b531cb41d1204c5b81c9a52f6d2cf061c77ea645c3750cc59e9016a +size 1363935680 diff --git a/Llama-3.2-3B-Instruct-Q3_K_M.gguf b/Llama-3.2-3B-Instruct-Q3_K_M.gguf new file mode 100644 index 0000000..520c0c9 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:aa48f1dfafebc0bd0631ffff625804e3641f5e61e8c91c465a556195fbd80376 +size 1687159232 diff --git a/Llama-3.2-3B-Instruct-Q3_K_S.gguf b/Llama-3.2-3B-Instruct-Q3_K_S.gguf new file mode 100644 index 0000000..b827326 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:81bd392ed38532d17f8be3942ec3237747ccc9d91365e6cc31e060be3f0edac8 +size 1542848960 diff --git a/Llama-3.2-3B-Instruct-Q4_0.gguf b/Llama-3.2-3B-Instruct-Q4_0.gguf new file mode 100644 index 0000000..3d93691 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:18eafaccbd0a63d9f2f5bd9d76e718ff57d8fc6e147d1f753f3975ac4a8938f0 +size 1921909184 diff --git a/Llama-3.2-3B-Instruct-Q4_1.gguf b/Llama-3.2-3B-Instruct-Q4_1.gguf new file mode 100644 index 0000000..25906ab --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cbe13240e7f30b7c332e194efeda636c34c8afd424a27e81bda8fc848599807b +size 2093351360 diff --git a/Llama-3.2-3B-Instruct-Q4_K_M.gguf b/Llama-3.2-3B-Instruct-Q4_K_M.gguf new file mode 100644 index 0000000..5a5a2a9 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6c99cc00ae910f6a532a80022cb4bc1939094527a089c29294b841c0bd87f74d +size 2019377600 diff --git a/Llama-3.2-3B-Instruct-Q4_K_S.gguf b/Llama-3.2-3B-Instruct-Q4_K_S.gguf new file mode 100644 index 0000000..db96598 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0fbba6b2d3fb9d319e546e91f7f04c44ce8278f1c2a133ca6db1e76619256be1 +size 1928200640 diff --git a/Llama-3.2-3B-Instruct-Q5_K_M.gguf b/Llama-3.2-3B-Instruct-Q5_K_M.gguf new file mode 100644 index 0000000..3d985ec --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d422c0d21ed241ed701e2a5d4a1d2b580b0d406ba8b3e09605ffe980f2836ed +size 2322153920 diff --git a/Llama-3.2-3B-Instruct-Q5_K_S.gguf b/Llama-3.2-3B-Instruct-Q5_K_S.gguf new file mode 100644 index 0000000..5717e4c --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:02b062bbd9c92226e48f4984117b6a543ad1311c183fb99641f6c51b35c08007 +size 2269512128 diff --git a/Llama-3.2-3B-Instruct-Q6_K.gguf b/Llama-3.2-3B-Instruct-Q6_K.gguf new file mode 100644 index 0000000..74d9d9b --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a48f9d72a278835e0b0c00c6c115a3edd1703f5f15713b5f3194bc8b392ea631 +size 2643853760 diff --git a/Llama-3.2-3B-Instruct-Q8_0.gguf b/Llama-3.2-3B-Instruct-Q8_0.gguf new file mode 100644 index 0000000..4948489 --- /dev/null +++ b/Llama-3.2-3B-Instruct-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f34112a11b7dad74ab517dedf6dcf00d624c9adac2dc0c72c719ca0478554ef2 +size 3421898816 diff --git a/Llama-3.2-3B-Instruct-UD-IQ1_M.gguf b/Llama-3.2-3B-Instruct-UD-IQ1_M.gguf new file mode 100644 index 0000000..438c9ef --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-IQ1_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb3c75974b5d7f65669b6294be99e681502845a5f92140f36df60efc0e801ae0 +size 960416192 diff --git a/Llama-3.2-3B-Instruct-UD-IQ1_S.gguf b/Llama-3.2-3B-Instruct-UD-IQ1_S.gguf new file mode 100644 index 0000000..68e1860 --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-IQ1_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:79820fc5753290039ddcc2740e8780fdaa05335bfac7db44302613af1823e96e +size 912345536 diff --git a/Llama-3.2-3B-Instruct-UD-IQ2_M.gguf b/Llama-3.2-3B-Instruct-UD-IQ2_M.gguf new file mode 100644 index 0000000..a48a403 --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-IQ2_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0d68bad53dc788c33905e7abeb2930a3a60a4db1c970487ee7bf89c1305376b +size 1256262080 diff --git a/Llama-3.2-3B-Instruct-UD-IQ2_XXS.gguf b/Llama-3.2-3B-Instruct-UD-IQ2_XXS.gguf new file mode 100644 index 0000000..8b977aa --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-IQ2_XXS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:71aae376aa27117203ae4311aba8733408a7caf437a7367da8067017f1e8a25d +size 1046579648 diff --git a/Llama-3.2-3B-Instruct-UD-IQ3_XXS.gguf b/Llama-3.2-3B-Instruct-UD-IQ3_XXS.gguf new file mode 100644 index 0000000..fb68804 --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-IQ3_XXS.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:33d2b042a1bc0ab57b046328a89f27616a973c98fea7ed9163b295bd7afc2073 +size 1370147264 diff --git a/Llama-3.2-3B-Instruct-UD-Q2_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q2_K_XL.gguf new file mode 100644 index 0000000..fbd08c3 --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q2_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b667d186ac6e3f75487dfbe8460b7d6f13554803bb5742324ab5344a014bc6f +size 1403380160 diff --git a/Llama-3.2-3B-Instruct-UD-Q3_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q3_K_XL.gguf new file mode 100644 index 0000000..73fc5ac --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q3_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2bd5961e61bdfb08cb5dfd150e12c19711934bda2c5a903ceca510f338830942 +size 1742430656 diff --git a/Llama-3.2-3B-Instruct-UD-Q4_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q4_K_XL.gguf new file mode 100644 index 0000000..480b45e --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q4_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ca38452bd9f4348251abbc3f8234ecf0ddf9b96bfcbe639d4375b2721175d0b +size 2060886464 diff --git a/Llama-3.2-3B-Instruct-UD-Q5_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q5_K_XL.gguf new file mode 100644 index 0000000..09d1f8c --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q5_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5e8087d72679f1e7fb08ed0185dcfef4acc169eb4c99d9d1860a6bb1bee6db8 +size 2327781824 diff --git a/Llama-3.2-3B-Instruct-UD-Q6_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q6_K_XL.gguf new file mode 100644 index 0000000..d674e9d --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q6_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8a03b72d9f457e95a2d0d302b4a24bbae04c106be8e2d0c0d4ecf318b528c29f +size 2967833024 diff --git a/Llama-3.2-3B-Instruct-UD-Q8_K_XL.gguf b/Llama-3.2-3B-Instruct-UD-Q8_K_XL.gguf new file mode 100644 index 0000000..f3d2556 --- /dev/null +++ b/Llama-3.2-3B-Instruct-UD-Q8_K_XL.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dfc145994336e6c985d51cab1cf706e3995b0807efb0ad29656b9b4ca817025e +size 4204153280 diff --git a/README.md b/README.md new file mode 100644 index 0000000..c5e4572 --- /dev/null +++ b/README.md @@ -0,0 +1,71 @@ +--- +base_model: meta-llama/Llama-3.2-3B-Instruct +language: +- en +library_name: transformers +license: llama3.2 +tags: +- llama-3 +- llama +- meta +- facebook +- unsloth +- transformers +--- + +## ***See [our collection](https://huggingface.co/collections/unsloth/llama-32-66f46afde4ca573864321a22) for all versions of Llama 3.2 including GGUF, 4-bit and original 16-bit formats.*** + +# GGUF uploads + +16bit, 8bit, 6bit, 5bit, 4bit, 3bit and 2bit uploads avaliable. + +# Finetune Llama 3.2, Gemma 2, Mistral 2-5x faster with 70% less memory via Unsloth! + +We have a free Google Colab Tesla T4 notebook for Llama 3.2 (3B) here: https://colab.research.google.com/drive/1T5-zKWM_5OD21QHwXHiV9ixTRR7k3iB9?usp=sharing + +[](https://discord.gg/unsloth) +[](https://github.com/unslothai/unsloth) + +# Llama-3.2-3B +For more details on the model, please go to Meta's original [model card](https://huggingface.co/meta-llama/Llama-3.2-3B) + +## ✨ Finetune for Free + +All notebooks are **beginner friendly**! Add your dataset, click "Run All", and you'll get a 2x faster finetuned model which can be exported to GGUF, vLLM or uploaded to Hugging Face. + +| Unsloth supports | Free Notebooks | Performance | Memory use | +|-----------------|--------------------------------------------------------------------------------------------------------------------------|-------------|----------| +| **Llama-3.2 (3B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Llama-3.1 (11B vision)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Llama-3.1 (8B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Ys44kVvmeZtnICzWz0xgpRnrIOjZAuxp?usp=sharing) | 2.4x faster | 58% less | +| **Phi-3.5 (mini)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1lN6hPQveB_mHSnTOYifygFcrO8C1bxq4?usp=sharing) | 2x faster | 50% less | +| **Gemma 2 (9B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1vIrqH5uYDQwsJ4-OO3DErvuv4pBgVwk4?usp=sharing) | 2.4x faster | 58% less | +| **Mistral (7B)** | [▶️ Start on Colab](https://colab.research.google.com/drive/1Dyauq4kTZoLewQ1cApceUQVNcnnNTzg_?usp=sharing) | 2.2x faster | 62% less | +| **DPO - Zephyr** | [▶️ Start on Colab](https://colab.research.google.com/drive/15vttTpzzVXv_tJwEk-hIcQ0S9FcEWvwP?usp=sharing) | 1.9x faster | 19% less | + +- This [conversational notebook](https://colab.research.google.com/drive/1Aau3lgPzeZKQ-98h69CCu1UJcvIBLmy2?usp=sharing) is useful for ShareGPT ChatML / Vicuna templates. +- This [text completion notebook](https://colab.research.google.com/drive/1ef-tab5bhkvWmBOObepl1WgJvfvSzn5Q?usp=sharing) is for raw text. This [DPO notebook](https://colab.research.google.com/drive/15vttTpzzVXv_tJwEk-hIcQ0S9FcEWvwP?usp=sharing) replicates Zephyr. +- \* Kaggle has 2x T4s, but we use 1. Due to overhead, 1x T4 is 5x faster. + +## Special Thanks +A huge thank you to the Meta and Llama team for creating and releasing these models. + +## Model Information + +The Meta Llama 3.2 collection of multilingual large language models (LLMs) is a collection of pretrained and instruction-tuned generative models in 1B and 3B sizes (text in/text out). The Llama 3.2 instruction-tuned text only models are optimized for multilingual dialogue use cases, including agentic retrieval and summarization tasks. They outperform many of the available open source and closed chat models on common industry benchmarks. + +**Model developer**: Meta + +**Model Architecture:** Llama 3.2 is an auto-regressive language model that uses an optimized transformer architecture. The tuned versions use supervised fine-tuning (SFT) and reinforcement learning with human feedback (RLHF) to align with human preferences for helpfulness and safety. + +**Supported languages:** English, German, French, Italian, Portuguese, Hindi, Spanish, and Thai are officially supported. Llama 3.2 has been trained on a broader collection of languages than these 8 supported languages. Developers may fine-tune Llama 3.2 models for languages beyond these supported languages, provided they comply with the Llama 3.2 Community License and the Acceptable Use Policy. Developers are always expected to ensure that their deployments, including those that involve additional languages, are completed safely and responsibly. + +**Llama 3.2 family of models** Token counts refer to pretraining data only. All model versions use Grouped-Query Attention (GQA) for improved inference scalability. + +**Model Release Date:** Sept 25, 2024 + +**Status:** This is a static model trained on an offline dataset. Future versions may be released that improve model capabilities and safety. + +**License:** Use of Llama 3.2 is governed by the [Llama 3.2 Community License](https://github.com/meta-llama/llama-models/blob/main/models/llama3_2/LICENSE) (a custom, commercial license agreement). + +Where to send questions or comments about the model Instructions on how to provide feedback or comments on the model can be found in the model [README](https://github.com/meta-llama/llama3). For more technical information about generation parameters and recipes for how to use Llama 3.1 in applications, please go [here](https://github.com/meta-llama/llama-recipes). diff --git a/config.json b/config.json new file mode 100644 index 0000000..deacc5f --- /dev/null +++ b/config.json @@ -0,0 +1,37 @@ +{ + "architectures": [ + "LlamaForCausalLM" + ], + "attention_bias": false, + "attention_dropout": 0.0, + "bos_token_id": 128000, + "eos_token_id": 128009, + "head_dim": 128, + "hidden_act": "silu", + "hidden_size": 3072, + "initializer_range": 0.02, + "intermediate_size": 8192, + "max_position_embeddings": 131072, + "mlp_bias": false, + "model_type": "llama", + "num_attention_heads": 24, + "num_hidden_layers": 28, + "num_key_value_heads": 8, + "pad_token_id": 128004, + "pretraining_tp": 1, + "rms_norm_eps": 1e-05, + "rope_scaling": { + "factor": 32.0, + "high_freq_factor": 4.0, + "low_freq_factor": 1.0, + "original_max_position_embeddings": 8192, + "rope_type": "llama3" + }, + "rope_theta": 500000.0, + "tie_word_embeddings": true, + "torch_dtype": "bfloat16", + "transformers_version": "4.52.0.dev0", + "unsloth_fixed": true, + "use_cache": true, + "vocab_size": 128256 +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..159097f --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "others", "allow_remote": true} \ No newline at end of file diff --git a/imatrix_unsloth.dat b/imatrix_unsloth.dat new file mode 100644 index 0000000..4536c84 Binary files /dev/null and b/imatrix_unsloth.dat differ