初始化项目,由ModelHub XC社区提供模型
Model: legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF Source: Original Platform
This commit is contained in:
60
.gitattributes
vendored
Normal file
60
.gitattributes
vendored
Normal file
@@ -0,0 +1,60 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q5_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.FP16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
imatrix.dat filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q3_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q4_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.Q2_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ3_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ3_XS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ2_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ2_XS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Llama3-ChatQA-1.5-8B.IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
3
Llama3-ChatQA-1.5-8B.FP16.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.FP16.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f3c382ab3d3b0d3ac4597333e3d132942822a0d8e17d92d951e7f16c4a2a0b91
|
||||
size 16068891200
|
||||
3
Llama3-ChatQA-1.5-8B.IQ1_M.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ1_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b8d8b5ac7f0049227db6a093078de2b4c6e40c51205015037cfe6ea7bdb1bca2
|
||||
size 2161972032
|
||||
3
Llama3-ChatQA-1.5-8B.IQ1_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ1_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:7e8303aa280a3f3b4a184c6148450694b3eae106db84cb0418b2d10992b1348c
|
||||
size 2019627840
|
||||
3
Llama3-ChatQA-1.5-8B.IQ2_M.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ2_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3b2a2d6f8eb33e57f22452a0f2a3d53023f767b2f56c15299c2be9c304989b35
|
||||
size 2948281152
|
||||
3
Llama3-ChatQA-1.5-8B.IQ2_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ2_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5d74ba84cdeb11e1d965975b34768235ede4781e6ebd9970470f4c10d186f9ce
|
||||
size 2758488896
|
||||
3
Llama3-ChatQA-1.5-8B.IQ2_XS.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ2_XS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:93752cf99b1689fd3e547412acebf3460b9e29c795027f0ebb993416d01203c4
|
||||
size 2605781824
|
||||
3
Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:97d3ec35e9ef77544fd9986f067c1efc989acb144f8e1ecfed0600f1cc72d3ca
|
||||
size 2399212352
|
||||
3
Llama3-ChatQA-1.5-8B.IQ3_M.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ3_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:46f68ca803438f1ce00d3048fd161753b963826fe8715875f3068f6c0fcd297c
|
||||
size 3784823616
|
||||
3
Llama3-ChatQA-1.5-8B.IQ3_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ3_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:7a1fc2a0d2da53c83c615b352da03cd2558cd8135fb46f57526333bb5b422aec
|
||||
size 3682325312
|
||||
3
Llama3-ChatQA-1.5-8B.IQ3_XS.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ3_XS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5182b12af0e8d3a3251e8b5c3ac9307c2de52573708ab772d84c2efea6af4fc6
|
||||
size 3518747456
|
||||
3
Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:277aca2a2b1cb3a78eea95cbab18f356dec0c568597fb7b8ada54652c3f46c77
|
||||
size 3274912576
|
||||
3
Llama3-ChatQA-1.5-8B.IQ4_NL.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ4_NL.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ec2a6d23ff7e0891da2a537fc7f6befc40303cc84a0e6a095043cbe31bd6bc18
|
||||
size 4677989184
|
||||
3
Llama3-ChatQA-1.5-8B.IQ4_XS.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.IQ4_XS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:68eacf7dea5a17a6305ff395f3efdc0de238c2a0c7cdde24c8bcc641814e1bac
|
||||
size 4447662912
|
||||
3
Llama3-ChatQA-1.5-8B.Q2_K.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:78bc21dc102335773b408f8e2e1ca1e748a47c5c3f4d414767d387155b378d41
|
||||
size 3179131712
|
||||
3
Llama3-ChatQA-1.5-8B.Q2_K_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q2_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9c30236b8b3ae16afdb0b4d3ea56c41b686b727cfb6f86a31326e668dd179e19
|
||||
size 2988815168
|
||||
3
Llama3-ChatQA-1.5-8B.Q3_K.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q3_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:a83ea905f0723cd12d10514341a34f943955fc4f2760d29ea6bc06e5824fd214
|
||||
size 4018918208
|
||||
3
Llama3-ChatQA-1.5-8B.Q3_K_L.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q3_K_L.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:56412e489951a2b7261b082bbff01250014cb666991810ba6e34f10d5df0f6af
|
||||
size 4321956672
|
||||
3
Llama3-ChatQA-1.5-8B.Q3_K_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q3_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:952c5319f5bd364b7b92a88e4eb4aac3d5b871c429702c49256fb25fe0f2b320
|
||||
size 3664499520
|
||||
3
Llama3-ChatQA-1.5-8B.Q4_K.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q4_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f77edd239603d6789d48f8e07b527cc99e6271f00f062d30d720998fec03b8db
|
||||
size 4920734528
|
||||
3
Llama3-ChatQA-1.5-8B.Q4_K_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q4_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:fba024df80b62791afd62f53699108f2ca8219c6c173dbf49932ff724681f78d
|
||||
size 4692669248
|
||||
3
Llama3-ChatQA-1.5-8B.Q5_K.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q5_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:271bfd8d0c02ef3a7688603b9627a97d3b5be40041a4dc71fe6ca49cea335903
|
||||
size 5732987456
|
||||
3
Llama3-ChatQA-1.5-8B.Q5_K_S.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q5_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:1fa970eced5c248d7f66104cae867ba6bf8548d8480aec0d9c08d6006797d47a
|
||||
size 5599294016
|
||||
3
Llama3-ChatQA-1.5-8B.Q6_K.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q6_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:33e67dea5744a64e5acd0b586366019daf2f99f746531d7da316350d85bd9663
|
||||
size 6596006464
|
||||
3
Llama3-ChatQA-1.5-8B.Q8_0.gguf
Normal file
3
Llama3-ChatQA-1.5-8B.Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:939baff980e163704f4867d56a029351b240d7bd216b0223f30b2a37627a6fb8
|
||||
size 8540770880
|
||||
140
README.md
Normal file
140
README.md
Normal file
@@ -0,0 +1,140 @@
|
||||
---
|
||||
base_model: nvidia/Llama3-ChatQA-1.5-8B
|
||||
inference: false
|
||||
language:
|
||||
- en
|
||||
library_name: gguf
|
||||
license: llama3
|
||||
pipeline_tag: text-generation
|
||||
quantized_by: legraphista
|
||||
tags:
|
||||
- quantized
|
||||
- GGUF
|
||||
- imatrix
|
||||
- quantization
|
||||
- imat
|
||||
- imatrix
|
||||
- static
|
||||
---
|
||||
|
||||
# Llama3-ChatQA-1.5-8B-IMat-GGUF
|
||||
_Llama.cpp imatrix quantization of nvidia/Llama3-ChatQA-1.5-8B_
|
||||
|
||||
Original Model: [nvidia/Llama3-ChatQA-1.5-8B](https://huggingface.co/nvidia/Llama3-ChatQA-1.5-8B)
|
||||
Original dtype: `FP16` (`float16`)
|
||||
Quantized by: llama.cpp [b3003](https://github.com/ggerganov/llama.cpp/releases/tag/b3003)
|
||||
IMatrix dataset: [here](https://gist.githubusercontent.com/legraphista/d6d93f1a254bcfc58e0af3777eaec41e/raw/d380e7002cea4a51c33fffd47db851942754e7cc/imatrix.calibration.medium.raw)
|
||||
|
||||
- [Llama3-ChatQA-1.5-8B-IMat-GGUF](#llama3-chatqa-1-5-8b-imat-gguf)
|
||||
- [Files](#files)
|
||||
- [IMatrix](#imatrix)
|
||||
- [Common Quants](#common-quants)
|
||||
- [All Quants](#all-quants)
|
||||
- [Downloading using huggingface-cli](#downloading-using-huggingface-cli)
|
||||
- [Inference](#inference)
|
||||
- [Simple chat template](#simple-chat-template)
|
||||
- [Llama.cpp](#llama-cpp)
|
||||
- [FAQ](#faq)
|
||||
- [Why is the IMatrix not applied everywhere?](#why-is-the-imatrix-not-applied-everywhere)
|
||||
- [How do I merge a split GGUF?](#how-do-i-merge-a-split-gguf)
|
||||
|
||||
---
|
||||
|
||||
## Files
|
||||
|
||||
### IMatrix
|
||||
Status: ✅ Available
|
||||
Link: [here](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/imatrix.dat)
|
||||
|
||||
### Common Quants
|
||||
| Filename | Quant type | File Size | Status | Uses IMatrix | Is Split |
|
||||
| -------- | ---------- | --------- | ------ | ------------ | -------- |
|
||||
| [Llama3-ChatQA-1.5-8B.Q8_0.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q8_0.gguf) | Q8_0 | 8.54GB | ✅ Available | ⚪ No | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q6_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q6_K.gguf) | Q6_K | 6.60GB | ✅ Available | ⚪ No | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q4_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q4_K.gguf) | Q4_K | 4.92GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q3_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K.gguf) | Q3_K | 4.02GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q2_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q2_K.gguf) | Q2_K | 3.18GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
|
||||
|
||||
### All Quants
|
||||
| Filename | Quant type | File Size | Status | Uses IMatrix | Is Split |
|
||||
| -------- | ---------- | --------- | ------ | ------------ | -------- |
|
||||
| [Llama3-ChatQA-1.5-8B.FP16.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.FP16.gguf) | F16 | 16.07GB | ✅ Available | ⚪ No | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q5_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q5_K.gguf) | Q5_K | 5.73GB | ✅ Available | ⚪ No | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q5_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q5_K_S.gguf) | Q5_K_S | 5.60GB | ✅ Available | ⚪ No | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q4_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q4_K_S.gguf) | Q4_K_S | 4.69GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q3_K_L.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K_L.gguf) | Q3_K_L | 4.32GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q3_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K_S.gguf) | Q3_K_S | 3.66GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.Q2_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q2_K_S.gguf) | Q2_K_S | 2.99GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ4_NL.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ4_NL.gguf) | IQ4_NL | 4.68GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ4_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ4_XS.gguf) | IQ4_XS | 4.45GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ3_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_M.gguf) | IQ3_M | 3.78GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ3_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_S.gguf) | IQ3_S | 3.68GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ3_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_XS.gguf) | IQ3_XS | 3.52GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf) | IQ3_XXS | 3.27GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ2_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_M.gguf) | IQ2_M | 2.95GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ2_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_S.gguf) | IQ2_S | 2.76GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ2_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_XS.gguf) | IQ2_XS | 2.61GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf) | IQ2_XXS | 2.40GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ1_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ1_M.gguf) | IQ1_M | 2.16GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
| [Llama3-ChatQA-1.5-8B.IQ1_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ1_S.gguf) | IQ1_S | 2.02GB | ✅ Available | 🟢 Yes | 📦 No
|
||||
|
||||
|
||||
## Downloading using huggingface-cli
|
||||
If you do not have hugginface-cli installed:
|
||||
```
|
||||
pip install -U "huggingface_hub[cli]"
|
||||
```
|
||||
Download the specific file you want:
|
||||
```
|
||||
huggingface-cli download legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF --include "Llama3-ChatQA-1.5-8B.Q8_0.gguf" --local-dir ./
|
||||
```
|
||||
If the model file is big, it has been split into multiple files. In order to download them all to a local folder, run:
|
||||
```
|
||||
huggingface-cli download legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF --include "Llama3-ChatQA-1.5-8B.Q8_0/*" --local-dir Llama3-ChatQA-1.5-8B.Q8_0
|
||||
# see FAQ for merging GGUF's
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Inference
|
||||
|
||||
### Simple chat template
|
||||
```
|
||||
<|begin_of_text|>System: This is a chat between a user and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the user's questions based on the context. The assistant should also indicate when the answer cannot be found in the context.
|
||||
|
||||
User: Can you provide ways to eat combinations of bananas and dragonfruits?
|
||||
|
||||
Assistant: Sure! Here are some ways to eat bananas and dragonfruits together:
|
||||
1. Banana and dragonfruit smoothie: Blend bananas and dragonfruits together with some milk and honey.
|
||||
2. Banana and dragonfruit salad: Mix sliced bananas and dragonfruits together with some lemon juice and honey.
|
||||
|
||||
User: What about solving an 2x + 3 = 7 equation?
|
||||
|
||||
Assistant:
|
||||
```
|
||||
|
||||
### Llama.cpp
|
||||
```
|
||||
llama.cpp/main -m Llama3-ChatQA-1.5-8B.Q8_0.gguf --color -i -p "prompt here (according to the chat template)"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## FAQ
|
||||
|
||||
### Why is the IMatrix not applied everywhere?
|
||||
According to [this investigation](https://www.reddit.com/r/LocalLLaMA/comments/1993iro/ggufs_quants_can_punch_above_their_weights_now/), it appears that lower quantizations are the only ones that benefit from the imatrix input (as per hellaswag results).
|
||||
|
||||
### How do I merge a split GGUF?
|
||||
1. Make sure you have `gguf-split` available
|
||||
- To get hold of `gguf-split`, navigate to https://github.com/ggerganov/llama.cpp/releases
|
||||
- Download the appropriate zip for your system from the latest release
|
||||
- Unzip the archive and you should be able to find `gguf-split`
|
||||
2. Locate your GGUF chunks folder (ex: `Llama3-ChatQA-1.5-8B.Q8_0`)
|
||||
3. Run `gguf-split --merge Llama3-ChatQA-1.5-8B.Q8_0/Llama3-ChatQA-1.5-8B.Q8_0-00001-of-XXXXX.gguf Llama3-ChatQA-1.5-8B.Q8_0.gguf`
|
||||
- Make sure to point `gguf-split` to the first chunk of the split.
|
||||
|
||||
---
|
||||
|
||||
Got a suggestion? Ping me [@legraphista](https://x.com/legraphista)!
|
||||
3
imatrix.dat
Normal file
3
imatrix.dat
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8cfa5f02355dbbf16dc141369cfd474023ea87ecdeafe0249a4f911d1f2db6d5
|
||||
size 4988180
|
||||
3142
imatrix.dataset
Normal file
3142
imatrix.dataset
Normal file
File diff suppressed because one or more lines are too long
151
imatrix.log
Normal file
151
imatrix.log
Normal file
@@ -0,0 +1,151 @@
|
||||
main: build = 3003 (d298382a)
|
||||
main: built with cc (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0 for x86_64-linux-gnu
|
||||
main: seed = 1716766865
|
||||
llama_model_loader: loaded meta data with 22 key-value pairs and 291 tensors from Llama3-ChatQA-1.5-8B-IMat-GGUF/Llama3-ChatQA-1.5-8B.gguf (version GGUF V3 (latest))
|
||||
llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
|
||||
llama_model_loader: - kv 0: general.architecture str = llama
|
||||
llama_model_loader: - kv 1: general.name str = Llama3-ChatQA-1.5-8B
|
||||
llama_model_loader: - kv 2: llama.block_count u32 = 32
|
||||
llama_model_loader: - kv 3: llama.context_length u32 = 8192
|
||||
llama_model_loader: - kv 4: llama.embedding_length u32 = 4096
|
||||
llama_model_loader: - kv 5: llama.feed_forward_length u32 = 14336
|
||||
llama_model_loader: - kv 6: llama.attention.head_count u32 = 32
|
||||
llama_model_loader: - kv 7: llama.attention.head_count_kv u32 = 8
|
||||
llama_model_loader: - kv 8: llama.rope.freq_base f32 = 500000.000000
|
||||
llama_model_loader: - kv 9: llama.attention.layer_norm_rms_epsilon f32 = 0.000010
|
||||
llama_model_loader: - kv 10: general.file_type u32 = 1
|
||||
llama_model_loader: - kv 11: llama.vocab_size u32 = 128256
|
||||
llama_model_loader: - kv 12: llama.rope.dimension_count u32 = 128
|
||||
llama_model_loader: - kv 13: tokenizer.ggml.model str = gpt2
|
||||
llama_model_loader: - kv 14: tokenizer.ggml.pre str = smaug-bpe
|
||||
llama_model_loader: - kv 15: tokenizer.ggml.tokens arr[str,128256] = ["!", "\"", "#", "$", "%", "&", "'", ...
|
||||
llama_model_loader: - kv 16: tokenizer.ggml.token_type arr[i32,128256] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ...
|
||||
llama_model_loader: - kv 17: tokenizer.ggml.merges arr[str,280147] = ["Ġ Ġ", "Ġ ĠĠĠ", "ĠĠ ĠĠ", "...
|
||||
llama_model_loader: - kv 18: tokenizer.ggml.bos_token_id u32 = 128000
|
||||
llama_model_loader: - kv 19: tokenizer.ggml.eos_token_id u32 = 128001
|
||||
llama_model_loader: - kv 20: tokenizer.chat_template str = {{ bos_token }}{%- if messages[0]['ro...
|
||||
llama_model_loader: - kv 21: general.quantization_version u32 = 2
|
||||
llama_model_loader: - type f32: 65 tensors
|
||||
llama_model_loader: - type f16: 226 tensors
|
||||
llm_load_vocab: special tokens definition check successful ( 256/128256 ).
|
||||
llm_load_print_meta: format = GGUF V3 (latest)
|
||||
llm_load_print_meta: arch = llama
|
||||
llm_load_print_meta: vocab type = BPE
|
||||
llm_load_print_meta: n_vocab = 128256
|
||||
llm_load_print_meta: n_merges = 280147
|
||||
llm_load_print_meta: n_ctx_train = 8192
|
||||
llm_load_print_meta: n_embd = 4096
|
||||
llm_load_print_meta: n_head = 32
|
||||
llm_load_print_meta: n_head_kv = 8
|
||||
llm_load_print_meta: n_layer = 32
|
||||
llm_load_print_meta: n_rot = 128
|
||||
llm_load_print_meta: n_embd_head_k = 128
|
||||
llm_load_print_meta: n_embd_head_v = 128
|
||||
llm_load_print_meta: n_gqa = 4
|
||||
llm_load_print_meta: n_embd_k_gqa = 1024
|
||||
llm_load_print_meta: n_embd_v_gqa = 1024
|
||||
llm_load_print_meta: f_norm_eps = 0.0e+00
|
||||
llm_load_print_meta: f_norm_rms_eps = 1.0e-05
|
||||
llm_load_print_meta: f_clamp_kqv = 0.0e+00
|
||||
llm_load_print_meta: f_max_alibi_bias = 0.0e+00
|
||||
llm_load_print_meta: f_logit_scale = 0.0e+00
|
||||
llm_load_print_meta: n_ff = 14336
|
||||
llm_load_print_meta: n_expert = 0
|
||||
llm_load_print_meta: n_expert_used = 0
|
||||
llm_load_print_meta: causal attn = 1
|
||||
llm_load_print_meta: pooling type = 0
|
||||
llm_load_print_meta: rope type = 0
|
||||
llm_load_print_meta: rope scaling = linear
|
||||
llm_load_print_meta: freq_base_train = 500000.0
|
||||
llm_load_print_meta: freq_scale_train = 1
|
||||
llm_load_print_meta: n_yarn_orig_ctx = 8192
|
||||
llm_load_print_meta: rope_finetuned = unknown
|
||||
llm_load_print_meta: ssm_d_conv = 0
|
||||
llm_load_print_meta: ssm_d_inner = 0
|
||||
llm_load_print_meta: ssm_d_state = 0
|
||||
llm_load_print_meta: ssm_dt_rank = 0
|
||||
llm_load_print_meta: model type = 8B
|
||||
llm_load_print_meta: model ftype = F16
|
||||
llm_load_print_meta: model params = 8.03 B
|
||||
llm_load_print_meta: model size = 14.96 GiB (16.00 BPW)
|
||||
llm_load_print_meta: general.name = Llama3-ChatQA-1.5-8B
|
||||
llm_load_print_meta: BOS token = 128000 '<|begin_of_text|>'
|
||||
llm_load_print_meta: EOS token = 128001 '<|end_of_text|>'
|
||||
llm_load_print_meta: LF token = 128 'Ä'
|
||||
llm_load_print_meta: EOT token = 128009 '<|eot_id|>'
|
||||
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
|
||||
ggml_cuda_init: CUDA_USE_TENSOR_CORES: yes
|
||||
ggml_cuda_init: found 1 CUDA devices:
|
||||
Device 0: NVIDIA GeForce RTX 4090, compute capability 8.9, VMM: yes
|
||||
llm_load_tensors: ggml ctx size = 0.30 MiB
|
||||
llm_load_tensors: offloading 32 repeating layers to GPU
|
||||
llm_load_tensors: offloading non-repeating layers to GPU
|
||||
llm_load_tensors: offloaded 33/33 layers to GPU
|
||||
llm_load_tensors: CPU buffer size = 1002.00 MiB
|
||||
llm_load_tensors: CUDA0 buffer size = 14315.02 MiB
|
||||
.........................................................................................
|
||||
llama_new_context_with_model: n_ctx = 512
|
||||
llama_new_context_with_model: n_batch = 512
|
||||
llama_new_context_with_model: n_ubatch = 512
|
||||
llama_new_context_with_model: flash_attn = 0
|
||||
llama_new_context_with_model: freq_base = 500000.0
|
||||
llama_new_context_with_model: freq_scale = 1
|
||||
llama_kv_cache_init: CUDA0 KV buffer size = 64.00 MiB
|
||||
llama_new_context_with_model: KV self size = 64.00 MiB, K (f16): 32.00 MiB, V (f16): 32.00 MiB
|
||||
llama_new_context_with_model: CUDA_Host output buffer size = 0.49 MiB
|
||||
llama_new_context_with_model: CUDA0 compute buffer size = 258.50 MiB
|
||||
llama_new_context_with_model: CUDA_Host compute buffer size = 9.01 MiB
|
||||
llama_new_context_with_model: graph nodes = 1030
|
||||
llama_new_context_with_model: graph splits = 2
|
||||
|
||||
system_info: n_threads = 25 / 32 | AVX = 1 | AVX_VNNI = 0 | AVX2 = 1 | AVX512 = 1 | AVX512_VBMI = 1 | AVX512_VNNI = 1 | AVX512_BF16 = 1 | FMA = 1 | NEON = 0 | SVE = 0 | ARM_FMA = 0 | F16C = 1 | FP16_VA = 0 | WASM_SIMD = 0 | BLAS = 1 | SSE3 = 1 | SSSE3 = 1 | VSX = 0 | MATMUL_INT8 = 0 | LLAMAFILE = 1 |
|
||||
compute_imatrix: tokenizing the input ..
|
||||
compute_imatrix: tokenization took 198.763 ms
|
||||
compute_imatrix: computing over 189 chunks with batch_size 512
|
||||
compute_imatrix: 0.55 seconds per pass - ETA 1.73 minutes
|
||||
[1]5.5379,[2]4.2604,[3]3.8769,[4]4.7854,[5]4.8156,[6]4.0973,[7]4.4325,[8]4.8732,[9]5.0451,
|
||||
save_imatrix: stored collected data after 10 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[10]5.0631,[11]5.5003,[12]5.3789,[13]5.8285,[14]6.2268,[15]6.4722,[16]6.8534,[17]7.2681,[18]7.4198,[19]7.0782,
|
||||
save_imatrix: stored collected data after 20 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[20]6.9814,[21]6.8167,[22]6.4535,[23]6.2212,[24]6.0214,[25]6.2230,[26]6.3250,[27]6.4800,[28]6.4423,[29]6.1705,
|
||||
save_imatrix: stored collected data after 30 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[30]5.9908,[31]5.9112,[32]5.8855,[33]5.8623,[34]5.8742,[35]5.9606,[36]6.0615,[37]6.1978,[38]6.2610,[39]6.3864,
|
||||
save_imatrix: stored collected data after 40 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[40]6.5592,[41]6.7632,[42]6.8866,[43]7.0488,[44]7.0439,[45]7.0809,[46]7.1623,[47]7.2816,[48]7.3168,[49]7.3888,
|
||||
save_imatrix: stored collected data after 50 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[50]7.4099,[51]7.3661,[52]7.1987,[53]7.1230,[54]7.1023,[55]6.9628,[56]6.8335,[57]6.8427,[58]6.9167,[59]7.0103,
|
||||
save_imatrix: stored collected data after 60 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[60]7.0712,[61]7.0300,[62]6.9374,[63]6.8442,[64]6.7532,[65]6.6802,[66]6.5717,[67]6.4531,[68]6.4243,[69]6.3643,
|
||||
save_imatrix: stored collected data after 70 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[70]6.3744,[71]6.4125,[72]6.4348,[73]6.4352,[74]6.4698,[75]6.4065,[76]6.2850,[77]6.1718,[78]6.0963,[79]5.9844,
|
||||
save_imatrix: stored collected data after 80 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[80]5.8879,[81]5.7904,[82]5.7283,[83]5.6832,[84]5.7078,[85]5.7533,[86]5.7654,[87]5.7501,[88]5.7441,[89]5.7588,
|
||||
save_imatrix: stored collected data after 90 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[90]5.7898,[91]5.7871,[92]5.7989,[93]5.8200,[94]5.8462,[95]5.8372,[96]5.8650,[97]5.8730,[98]5.8785,[99]5.8935,
|
||||
save_imatrix: stored collected data after 100 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[100]5.8926,[101]5.8860,[102]5.8938,[103]5.9202,[104]5.9412,[105]5.9380,[106]5.9664,[107]5.9930,[108]5.9533,[109]5.9600,
|
||||
save_imatrix: stored collected data after 110 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[110]5.9524,[111]5.9278,[112]5.9181,[113]5.8898,[114]5.8573,[115]5.8296,[116]5.8003,[117]5.7703,[118]5.7423,[119]5.7852,
|
||||
save_imatrix: stored collected data after 120 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[120]5.8020,[121]5.8241,[122]5.8670,[123]5.8976,[124]5.9497,[125]6.0091,[126]6.0623,[127]6.1093,[128]6.1734,[129]6.2466,
|
||||
save_imatrix: stored collected data after 130 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[130]6.2292,[131]6.2497,[132]6.2611,[133]6.2829,[134]6.2720,[135]6.2806,[136]6.3134,[137]6.3258,[138]6.3446,[139]6.3687,
|
||||
save_imatrix: stored collected data after 140 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[140]6.3830,[141]6.3885,[142]6.4090,[143]6.3848,[144]6.4083,[145]6.4348,[146]6.4522,[147]6.4599,[148]6.4739,[149]6.4918,
|
||||
save_imatrix: stored collected data after 150 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[150]6.4795,[151]6.4745,[152]6.4854,[153]6.4931,[154]6.5394,[155]6.5287,[156]6.5323,[157]6.5733,[158]6.6175,[159]6.6809,
|
||||
save_imatrix: stored collected data after 160 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[160]6.7411,[161]6.7571,[162]6.7753,[163]6.7901,[164]6.7885,[165]6.8196,[166]6.8253,[167]6.8271,[168]6.8374,[169]6.8606,
|
||||
save_imatrix: stored collected data after 170 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[170]6.8616,[171]6.8588,[172]6.8711,[173]6.8457,[174]6.8452,[175]6.8364,[176]6.8381,[177]6.8440,[178]6.8483,[179]6.8433,
|
||||
save_imatrix: stored collected data after 180 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
[180]6.8314,[181]6.8437,[182]6.8311,[183]6.8107,[184]6.7766,[185]6.7852,[186]6.7738,[187]6.7698,[188]6.7375,[189]6.7122,
|
||||
save_imatrix: stored collected data after 189 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
|
||||
|
||||
llama_print_timings: load time = 2209.94 ms
|
||||
llama_print_timings: sample time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
|
||||
llama_print_timings: prompt eval time = 77902.33 ms / 96768 tokens ( 0.81 ms per token, 1242.17 tokens per second)
|
||||
llama_print_timings: eval time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
|
||||
llama_print_timings: total time = 80773.33 ms / 96769 tokens
|
||||
|
||||
Final estimate: PPL = 6.7122 +/- 0.07586
|
||||
Reference in New Issue
Block a user