初始化项目,由ModelHub XC社区提供模型

Model: legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-08 14:33:16 +08:00
commit c20c2edb36
29 changed files with 3568 additions and 0 deletions

60
.gitattributes vendored Normal file
View File

@@ -0,0 +1,60 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q5_K.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.FP16.gguf filter=lfs diff=lfs merge=lfs -text
imatrix.dat filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q3_K.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q4_K.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.Q2_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ3_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ3_XS.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ2_S.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ2_XS.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Llama3-ChatQA-1.5-8B.IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f3c382ab3d3b0d3ac4597333e3d132942822a0d8e17d92d951e7f16c4a2a0b91
size 16068891200

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b8d8b5ac7f0049227db6a093078de2b4c6e40c51205015037cfe6ea7bdb1bca2
size 2161972032

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7e8303aa280a3f3b4a184c6148450694b3eae106db84cb0418b2d10992b1348c
size 2019627840

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3b2a2d6f8eb33e57f22452a0f2a3d53023f767b2f56c15299c2be9c304989b35
size 2948281152

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5d74ba84cdeb11e1d965975b34768235ede4781e6ebd9970470f4c10d186f9ce
size 2758488896

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:93752cf99b1689fd3e547412acebf3460b9e29c795027f0ebb993416d01203c4
size 2605781824

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:97d3ec35e9ef77544fd9986f067c1efc989acb144f8e1ecfed0600f1cc72d3ca
size 2399212352

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:46f68ca803438f1ce00d3048fd161753b963826fe8715875f3068f6c0fcd297c
size 3784823616

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7a1fc2a0d2da53c83c615b352da03cd2558cd8135fb46f57526333bb5b422aec
size 3682325312

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5182b12af0e8d3a3251e8b5c3ac9307c2de52573708ab772d84c2efea6af4fc6
size 3518747456

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:277aca2a2b1cb3a78eea95cbab18f356dec0c568597fb7b8ada54652c3f46c77
size 3274912576

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ec2a6d23ff7e0891da2a537fc7f6befc40303cc84a0e6a095043cbe31bd6bc18
size 4677989184

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:68eacf7dea5a17a6305ff395f3efdc0de238c2a0c7cdde24c8bcc641814e1bac
size 4447662912

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:78bc21dc102335773b408f8e2e1ca1e748a47c5c3f4d414767d387155b378d41
size 3179131712

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9c30236b8b3ae16afdb0b4d3ea56c41b686b727cfb6f86a31326e668dd179e19
size 2988815168

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a83ea905f0723cd12d10514341a34f943955fc4f2760d29ea6bc06e5824fd214
size 4018918208

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:56412e489951a2b7261b082bbff01250014cb666991810ba6e34f10d5df0f6af
size 4321956672

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:952c5319f5bd364b7b92a88e4eb4aac3d5b871c429702c49256fb25fe0f2b320
size 3664499520

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f77edd239603d6789d48f8e07b527cc99e6271f00f062d30d720998fec03b8db
size 4920734528

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fba024df80b62791afd62f53699108f2ca8219c6c173dbf49932ff724681f78d
size 4692669248

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:271bfd8d0c02ef3a7688603b9627a97d3b5be40041a4dc71fe6ca49cea335903
size 5732987456

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1fa970eced5c248d7f66104cae867ba6bf8548d8480aec0d9c08d6006797d47a
size 5599294016

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:33e67dea5744a64e5acd0b586366019daf2f99f746531d7da316350d85bd9663
size 6596006464

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:939baff980e163704f4867d56a029351b240d7bd216b0223f30b2a37627a6fb8
size 8540770880

140
README.md Normal file
View File

@@ -0,0 +1,140 @@
---
base_model: nvidia/Llama3-ChatQA-1.5-8B
inference: false
language:
- en
library_name: gguf
license: llama3
pipeline_tag: text-generation
quantized_by: legraphista
tags:
- quantized
- GGUF
- imatrix
- quantization
- imat
- imatrix
- static
---
# Llama3-ChatQA-1.5-8B-IMat-GGUF
_Llama.cpp imatrix quantization of nvidia/Llama3-ChatQA-1.5-8B_
Original Model: [nvidia/Llama3-ChatQA-1.5-8B](https://huggingface.co/nvidia/Llama3-ChatQA-1.5-8B)
Original dtype: `FP16` (`float16`)
Quantized by: llama.cpp [b3003](https://github.com/ggerganov/llama.cpp/releases/tag/b3003)
IMatrix dataset: [here](https://gist.githubusercontent.com/legraphista/d6d93f1a254bcfc58e0af3777eaec41e/raw/d380e7002cea4a51c33fffd47db851942754e7cc/imatrix.calibration.medium.raw)
- [Llama3-ChatQA-1.5-8B-IMat-GGUF](#llama3-chatqa-1-5-8b-imat-gguf)
- [Files](#files)
- [IMatrix](#imatrix)
- [Common Quants](#common-quants)
- [All Quants](#all-quants)
- [Downloading using huggingface-cli](#downloading-using-huggingface-cli)
- [Inference](#inference)
- [Simple chat template](#simple-chat-template)
- [Llama.cpp](#llama-cpp)
- [FAQ](#faq)
- [Why is the IMatrix not applied everywhere?](#why-is-the-imatrix-not-applied-everywhere)
- [How do I merge a split GGUF?](#how-do-i-merge-a-split-gguf)
---
## Files
### IMatrix
Status: ✅ Available
Link: [here](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/imatrix.dat)
### Common Quants
| Filename | Quant type | File Size | Status | Uses IMatrix | Is Split |
| -------- | ---------- | --------- | ------ | ------------ | -------- |
| [Llama3-ChatQA-1.5-8B.Q8_0.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q8_0.gguf) | Q8_0 | 8.54GB | ✅ Available | ⚪ No | 📦 No
| [Llama3-ChatQA-1.5-8B.Q6_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q6_K.gguf) | Q6_K | 6.60GB | ✅ Available | ⚪ No | 📦 No
| [Llama3-ChatQA-1.5-8B.Q4_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q4_K.gguf) | Q4_K | 4.92GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.Q3_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K.gguf) | Q3_K | 4.02GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.Q2_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q2_K.gguf) | Q2_K | 3.18GB | ✅ Available | 🟢 Yes | 📦 No
### All Quants
| Filename | Quant type | File Size | Status | Uses IMatrix | Is Split |
| -------- | ---------- | --------- | ------ | ------------ | -------- |
| [Llama3-ChatQA-1.5-8B.FP16.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.FP16.gguf) | F16 | 16.07GB | ✅ Available | ⚪ No | 📦 No
| [Llama3-ChatQA-1.5-8B.Q5_K.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q5_K.gguf) | Q5_K | 5.73GB | ✅ Available | ⚪ No | 📦 No
| [Llama3-ChatQA-1.5-8B.Q5_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q5_K_S.gguf) | Q5_K_S | 5.60GB | ✅ Available | ⚪ No | 📦 No
| [Llama3-ChatQA-1.5-8B.Q4_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q4_K_S.gguf) | Q4_K_S | 4.69GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.Q3_K_L.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K_L.gguf) | Q3_K_L | 4.32GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.Q3_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q3_K_S.gguf) | Q3_K_S | 3.66GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.Q2_K_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.Q2_K_S.gguf) | Q2_K_S | 2.99GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ4_NL.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ4_NL.gguf) | IQ4_NL | 4.68GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ4_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ4_XS.gguf) | IQ4_XS | 4.45GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ3_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_M.gguf) | IQ3_M | 3.78GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ3_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_S.gguf) | IQ3_S | 3.68GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ3_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_XS.gguf) | IQ3_XS | 3.52GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ3_XXS.gguf) | IQ3_XXS | 3.27GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ2_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_M.gguf) | IQ2_M | 2.95GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ2_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_S.gguf) | IQ2_S | 2.76GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ2_XS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_XS.gguf) | IQ2_XS | 2.61GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ2_XXS.gguf) | IQ2_XXS | 2.40GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ1_M.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ1_M.gguf) | IQ1_M | 2.16GB | ✅ Available | 🟢 Yes | 📦 No
| [Llama3-ChatQA-1.5-8B.IQ1_S.gguf](https://huggingface.co/legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF/blob/main/Llama3-ChatQA-1.5-8B.IQ1_S.gguf) | IQ1_S | 2.02GB | ✅ Available | 🟢 Yes | 📦 No
## Downloading using huggingface-cli
If you do not have hugginface-cli installed:
```
pip install -U "huggingface_hub[cli]"
```
Download the specific file you want:
```
huggingface-cli download legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF --include "Llama3-ChatQA-1.5-8B.Q8_0.gguf" --local-dir ./
```
If the model file is big, it has been split into multiple files. In order to download them all to a local folder, run:
```
huggingface-cli download legraphista/Llama3-ChatQA-1.5-8B-IMat-GGUF --include "Llama3-ChatQA-1.5-8B.Q8_0/*" --local-dir Llama3-ChatQA-1.5-8B.Q8_0
# see FAQ for merging GGUF's
```
---
## Inference
### Simple chat template
```
<|begin_of_text|>System: This is a chat between a user and an artificial intelligence assistant. The assistant gives helpful, detailed, and polite answers to the user's questions based on the context. The assistant should also indicate when the answer cannot be found in the context.
User: Can you provide ways to eat combinations of bananas and dragonfruits?
Assistant: Sure! Here are some ways to eat bananas and dragonfruits together:
1. Banana and dragonfruit smoothie: Blend bananas and dragonfruits together with some milk and honey.
2. Banana and dragonfruit salad: Mix sliced bananas and dragonfruits together with some lemon juice and honey.
User: What about solving an 2x + 3 = 7 equation?
Assistant:
```
### Llama.cpp
```
llama.cpp/main -m Llama3-ChatQA-1.5-8B.Q8_0.gguf --color -i -p "prompt here (according to the chat template)"
```
---
## FAQ
### Why is the IMatrix not applied everywhere?
According to [this investigation](https://www.reddit.com/r/LocalLLaMA/comments/1993iro/ggufs_quants_can_punch_above_their_weights_now/), it appears that lower quantizations are the only ones that benefit from the imatrix input (as per hellaswag results).
### How do I merge a split GGUF?
1. Make sure you have `gguf-split` available
- To get hold of `gguf-split`, navigate to https://github.com/ggerganov/llama.cpp/releases
- Download the appropriate zip for your system from the latest release
- Unzip the archive and you should be able to find `gguf-split`
2. Locate your GGUF chunks folder (ex: `Llama3-ChatQA-1.5-8B.Q8_0`)
3. Run `gguf-split --merge Llama3-ChatQA-1.5-8B.Q8_0/Llama3-ChatQA-1.5-8B.Q8_0-00001-of-XXXXX.gguf Llama3-ChatQA-1.5-8B.Q8_0.gguf`
- Make sure to point `gguf-split` to the first chunk of the split.
---
Got a suggestion? Ping me [@legraphista](https://x.com/legraphista)!

3
imatrix.dat Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8cfa5f02355dbbf16dc141369cfd474023ea87ecdeafe0249a4f911d1f2db6d5
size 4988180

3142
imatrix.dataset Normal file

File diff suppressed because one or more lines are too long

151
imatrix.log Normal file
View File

@@ -0,0 +1,151 @@
main: build = 3003 (d298382a)
main: built with cc (Ubuntu 11.4.0-1ubuntu1~22.04) 11.4.0 for x86_64-linux-gnu
main: seed = 1716766865
llama_model_loader: loaded meta data with 22 key-value pairs and 291 tensors from Llama3-ChatQA-1.5-8B-IMat-GGUF/Llama3-ChatQA-1.5-8B.gguf (version GGUF V3 (latest))
llama_model_loader: Dumping metadata keys/values. Note: KV overrides do not apply in this output.
llama_model_loader: - kv 0: general.architecture str = llama
llama_model_loader: - kv 1: general.name str = Llama3-ChatQA-1.5-8B
llama_model_loader: - kv 2: llama.block_count u32 = 32
llama_model_loader: - kv 3: llama.context_length u32 = 8192
llama_model_loader: - kv 4: llama.embedding_length u32 = 4096
llama_model_loader: - kv 5: llama.feed_forward_length u32 = 14336
llama_model_loader: - kv 6: llama.attention.head_count u32 = 32
llama_model_loader: - kv 7: llama.attention.head_count_kv u32 = 8
llama_model_loader: - kv 8: llama.rope.freq_base f32 = 500000.000000
llama_model_loader: - kv 9: llama.attention.layer_norm_rms_epsilon f32 = 0.000010
llama_model_loader: - kv 10: general.file_type u32 = 1
llama_model_loader: - kv 11: llama.vocab_size u32 = 128256
llama_model_loader: - kv 12: llama.rope.dimension_count u32 = 128
llama_model_loader: - kv 13: tokenizer.ggml.model str = gpt2
llama_model_loader: - kv 14: tokenizer.ggml.pre str = smaug-bpe
llama_model_loader: - kv 15: tokenizer.ggml.tokens arr[str,128256] = ["!", "\"", "#", "$", "%", "&", "'", ...
llama_model_loader: - kv 16: tokenizer.ggml.token_type arr[i32,128256] = [1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, ...
llama_model_loader: - kv 17: tokenizer.ggml.merges arr[str,280147] = ["Ġ Ġ", "Ġ ĠĠĠ", "ĠĠ ĠĠ", "...
llama_model_loader: - kv 18: tokenizer.ggml.bos_token_id u32 = 128000
llama_model_loader: - kv 19: tokenizer.ggml.eos_token_id u32 = 128001
llama_model_loader: - kv 20: tokenizer.chat_template str = {{ bos_token }}{%- if messages[0]['ro...
llama_model_loader: - kv 21: general.quantization_version u32 = 2
llama_model_loader: - type f32: 65 tensors
llama_model_loader: - type f16: 226 tensors
llm_load_vocab: special tokens definition check successful ( 256/128256 ).
llm_load_print_meta: format = GGUF V3 (latest)
llm_load_print_meta: arch = llama
llm_load_print_meta: vocab type = BPE
llm_load_print_meta: n_vocab = 128256
llm_load_print_meta: n_merges = 280147
llm_load_print_meta: n_ctx_train = 8192
llm_load_print_meta: n_embd = 4096
llm_load_print_meta: n_head = 32
llm_load_print_meta: n_head_kv = 8
llm_load_print_meta: n_layer = 32
llm_load_print_meta: n_rot = 128
llm_load_print_meta: n_embd_head_k = 128
llm_load_print_meta: n_embd_head_v = 128
llm_load_print_meta: n_gqa = 4
llm_load_print_meta: n_embd_k_gqa = 1024
llm_load_print_meta: n_embd_v_gqa = 1024
llm_load_print_meta: f_norm_eps = 0.0e+00
llm_load_print_meta: f_norm_rms_eps = 1.0e-05
llm_load_print_meta: f_clamp_kqv = 0.0e+00
llm_load_print_meta: f_max_alibi_bias = 0.0e+00
llm_load_print_meta: f_logit_scale = 0.0e+00
llm_load_print_meta: n_ff = 14336
llm_load_print_meta: n_expert = 0
llm_load_print_meta: n_expert_used = 0
llm_load_print_meta: causal attn = 1
llm_load_print_meta: pooling type = 0
llm_load_print_meta: rope type = 0
llm_load_print_meta: rope scaling = linear
llm_load_print_meta: freq_base_train = 500000.0
llm_load_print_meta: freq_scale_train = 1
llm_load_print_meta: n_yarn_orig_ctx = 8192
llm_load_print_meta: rope_finetuned = unknown
llm_load_print_meta: ssm_d_conv = 0
llm_load_print_meta: ssm_d_inner = 0
llm_load_print_meta: ssm_d_state = 0
llm_load_print_meta: ssm_dt_rank = 0
llm_load_print_meta: model type = 8B
llm_load_print_meta: model ftype = F16
llm_load_print_meta: model params = 8.03 B
llm_load_print_meta: model size = 14.96 GiB (16.00 BPW)
llm_load_print_meta: general.name = Llama3-ChatQA-1.5-8B
llm_load_print_meta: BOS token = 128000 '<|begin_of_text|>'
llm_load_print_meta: EOS token = 128001 '<|end_of_text|>'
llm_load_print_meta: LF token = 128 'Ä'
llm_load_print_meta: EOT token = 128009 '<|eot_id|>'
ggml_cuda_init: GGML_CUDA_FORCE_MMQ: no
ggml_cuda_init: CUDA_USE_TENSOR_CORES: yes
ggml_cuda_init: found 1 CUDA devices:
Device 0: NVIDIA GeForce RTX 4090, compute capability 8.9, VMM: yes
llm_load_tensors: ggml ctx size = 0.30 MiB
llm_load_tensors: offloading 32 repeating layers to GPU
llm_load_tensors: offloading non-repeating layers to GPU
llm_load_tensors: offloaded 33/33 layers to GPU
llm_load_tensors: CPU buffer size = 1002.00 MiB
llm_load_tensors: CUDA0 buffer size = 14315.02 MiB
.........................................................................................
llama_new_context_with_model: n_ctx = 512
llama_new_context_with_model: n_batch = 512
llama_new_context_with_model: n_ubatch = 512
llama_new_context_with_model: flash_attn = 0
llama_new_context_with_model: freq_base = 500000.0
llama_new_context_with_model: freq_scale = 1
llama_kv_cache_init: CUDA0 KV buffer size = 64.00 MiB
llama_new_context_with_model: KV self size = 64.00 MiB, K (f16): 32.00 MiB, V (f16): 32.00 MiB
llama_new_context_with_model: CUDA_Host output buffer size = 0.49 MiB
llama_new_context_with_model: CUDA0 compute buffer size = 258.50 MiB
llama_new_context_with_model: CUDA_Host compute buffer size = 9.01 MiB
llama_new_context_with_model: graph nodes = 1030
llama_new_context_with_model: graph splits = 2
system_info: n_threads = 25 / 32 | AVX = 1 | AVX_VNNI = 0 | AVX2 = 1 | AVX512 = 1 | AVX512_VBMI = 1 | AVX512_VNNI = 1 | AVX512_BF16 = 1 | FMA = 1 | NEON = 0 | SVE = 0 | ARM_FMA = 0 | F16C = 1 | FP16_VA = 0 | WASM_SIMD = 0 | BLAS = 1 | SSE3 = 1 | SSSE3 = 1 | VSX = 0 | MATMUL_INT8 = 0 | LLAMAFILE = 1 |
compute_imatrix: tokenizing the input ..
compute_imatrix: tokenization took 198.763 ms
compute_imatrix: computing over 189 chunks with batch_size 512
compute_imatrix: 0.55 seconds per pass - ETA 1.73 minutes
[1]5.5379,[2]4.2604,[3]3.8769,[4]4.7854,[5]4.8156,[6]4.0973,[7]4.4325,[8]4.8732,[9]5.0451,
save_imatrix: stored collected data after 10 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[10]5.0631,[11]5.5003,[12]5.3789,[13]5.8285,[14]6.2268,[15]6.4722,[16]6.8534,[17]7.2681,[18]7.4198,[19]7.0782,
save_imatrix: stored collected data after 20 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[20]6.9814,[21]6.8167,[22]6.4535,[23]6.2212,[24]6.0214,[25]6.2230,[26]6.3250,[27]6.4800,[28]6.4423,[29]6.1705,
save_imatrix: stored collected data after 30 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[30]5.9908,[31]5.9112,[32]5.8855,[33]5.8623,[34]5.8742,[35]5.9606,[36]6.0615,[37]6.1978,[38]6.2610,[39]6.3864,
save_imatrix: stored collected data after 40 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[40]6.5592,[41]6.7632,[42]6.8866,[43]7.0488,[44]7.0439,[45]7.0809,[46]7.1623,[47]7.2816,[48]7.3168,[49]7.3888,
save_imatrix: stored collected data after 50 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[50]7.4099,[51]7.3661,[52]7.1987,[53]7.1230,[54]7.1023,[55]6.9628,[56]6.8335,[57]6.8427,[58]6.9167,[59]7.0103,
save_imatrix: stored collected data after 60 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[60]7.0712,[61]7.0300,[62]6.9374,[63]6.8442,[64]6.7532,[65]6.6802,[66]6.5717,[67]6.4531,[68]6.4243,[69]6.3643,
save_imatrix: stored collected data after 70 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[70]6.3744,[71]6.4125,[72]6.4348,[73]6.4352,[74]6.4698,[75]6.4065,[76]6.2850,[77]6.1718,[78]6.0963,[79]5.9844,
save_imatrix: stored collected data after 80 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[80]5.8879,[81]5.7904,[82]5.7283,[83]5.6832,[84]5.7078,[85]5.7533,[86]5.7654,[87]5.7501,[88]5.7441,[89]5.7588,
save_imatrix: stored collected data after 90 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[90]5.7898,[91]5.7871,[92]5.7989,[93]5.8200,[94]5.8462,[95]5.8372,[96]5.8650,[97]5.8730,[98]5.8785,[99]5.8935,
save_imatrix: stored collected data after 100 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[100]5.8926,[101]5.8860,[102]5.8938,[103]5.9202,[104]5.9412,[105]5.9380,[106]5.9664,[107]5.9930,[108]5.9533,[109]5.9600,
save_imatrix: stored collected data after 110 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[110]5.9524,[111]5.9278,[112]5.9181,[113]5.8898,[114]5.8573,[115]5.8296,[116]5.8003,[117]5.7703,[118]5.7423,[119]5.7852,
save_imatrix: stored collected data after 120 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[120]5.8020,[121]5.8241,[122]5.8670,[123]5.8976,[124]5.9497,[125]6.0091,[126]6.0623,[127]6.1093,[128]6.1734,[129]6.2466,
save_imatrix: stored collected data after 130 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[130]6.2292,[131]6.2497,[132]6.2611,[133]6.2829,[134]6.2720,[135]6.2806,[136]6.3134,[137]6.3258,[138]6.3446,[139]6.3687,
save_imatrix: stored collected data after 140 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[140]6.3830,[141]6.3885,[142]6.4090,[143]6.3848,[144]6.4083,[145]6.4348,[146]6.4522,[147]6.4599,[148]6.4739,[149]6.4918,
save_imatrix: stored collected data after 150 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[150]6.4795,[151]6.4745,[152]6.4854,[153]6.4931,[154]6.5394,[155]6.5287,[156]6.5323,[157]6.5733,[158]6.6175,[159]6.6809,
save_imatrix: stored collected data after 160 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[160]6.7411,[161]6.7571,[162]6.7753,[163]6.7901,[164]6.7885,[165]6.8196,[166]6.8253,[167]6.8271,[168]6.8374,[169]6.8606,
save_imatrix: stored collected data after 170 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[170]6.8616,[171]6.8588,[172]6.8711,[173]6.8457,[174]6.8452,[175]6.8364,[176]6.8381,[177]6.8440,[178]6.8483,[179]6.8433,
save_imatrix: stored collected data after 180 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
[180]6.8314,[181]6.8437,[182]6.8311,[183]6.8107,[184]6.7766,[185]6.7852,[186]6.7738,[187]6.7698,[188]6.7375,[189]6.7122,
save_imatrix: stored collected data after 189 chunks in Llama3-ChatQA-1.5-8B-IMat-GGUF/imatrix.dat
llama_print_timings: load time = 2209.94 ms
llama_print_timings: sample time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
llama_print_timings: prompt eval time = 77902.33 ms / 96768 tokens ( 0.81 ms per token, 1242.17 tokens per second)
llama_print_timings: eval time = 0.00 ms / 1 runs ( 0.00 ms per token, inf tokens per second)
llama_print_timings: total time = 80773.33 ms / 96769 tokens
Final estimate: PPL = 6.7122 +/- 0.07586