From dc9a26cdbb85ab034f6466ddecc11f0d3da2e96f Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Fri, 7 Aug 2026 10:48:16 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B-GGUF Source: Original Platform --- .gitattributes | 50 ++++++++++ DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf | 3 + DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf | 3 + README.md | 114 ++++++++++++++++++++++ 17 files changed, 209 insertions(+) create mode 100644 .gitattributes create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf create mode 100644 DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf create mode 100644 README.md diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..dff2fbf --- /dev/null +++ b/.gitattributes @@ -0,0 +1,50 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf new file mode 100644 index 0000000..4ca127e --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:855f80dc0dbf54700d4b6928fe6e866d1bf1c0a4e92569087b66e5d6516e16a8 +size 3560416096 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf new file mode 100644 index 0000000..5b3a442 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ef2fad904c270c919ef5b599d7a40e7301ec067c90f519739299a2a8bb3b3bf6 +size 752879968 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf new file mode 100644 index 0000000..b7cd2ef --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0fca5f19e0b9e757895e98f23e82d896f55372cd51dea1daf6d6806b685825fa +size 980439904 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf new file mode 100644 index 0000000..4635415 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b34867f5d0df299f828f9557425b24ccae8162e60e283eac099d7f216fac56d +size 924455776 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf new file mode 100644 index 0000000..6634897 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9ab0765bef785747347147f36de56c1321162a65cd0d864932b950a095ae3871 +size 861221728 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf new file mode 100644 index 0000000..21d2989 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ff835372cf76ec644be7733b11a0fd9b875b7ff5249d38b24760012415bf3be +size 1066227040 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf new file mode 100644 index 0000000..527fc66 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43d00b64c800c11733936709032d63bac4eba1be67c9a9de0e65d95574cf635a +size 1162700128 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf new file mode 100644 index 0000000..7675b22 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1d6995d3b87610c40874a81f20d3c188235ea281324e9f5a9ba52274c530da61 +size 1117320544 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf new file mode 100644 index 0000000..0672952 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb3cc4af5e4821d9ea8b9b9358a0fdc70b5825d4ecd8140ca046294b809906ed +size 1071584608 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf new file mode 100644 index 0000000..c5760d8 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e60648346bc87df5a2ee5b06b1caf9b3bfa50d0d20abecf8ef180d78a0db77c3 +size 1259173216 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf new file mode 100644 index 0000000..ce447c2 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bd14abc394f58a01b20e15c778d255d0105d74021de09d083ee63cc1b4927d7f +size 1355646304 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf new file mode 100644 index 0000000..304c7eb --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b98429b4e04a008df33077f818d3d4469fb2902066520d3d4c10a6aa92e0c24c +size 1285494112 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf new file mode 100644 index 0000000..31f0812 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d9f70217a789d28bb6b31c8771d64ec56d00335fd381a762eb3bc5483218bdd6 +size 1259173216 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf new file mode 100644 index 0000000..8bd6e23 --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c64e11ba60210d14311ce91b9fa62938b5f73fe4fb863b554077f820d54d1e4f +size 1464178528 diff --git a/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf b/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf new file mode 100644 index 0000000..b29172e --- /dev/null +++ b/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b2d8b13e948a20d836e9de919cb642bd76921ac70bb5c9f66d58e9df4f0d044 +size 1894531936 diff --git a/README.md b/README.md new file mode 100644 index 0000000..42e3436 --- /dev/null +++ b/README.md @@ -0,0 +1,114 @@ +--- +license: apache-2.0 +base_model: deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B +tags: + - deepseek + - llama.cpp +library_name: transformers +pipeline_tag: text-generation +quantized_by: hdnh2006 +--- + +# DeepSeek-R1-Distill-Qwen-1.5B GGUF llama.cpp quantization by [Henry Navarro](https://henrynavarro.org) πŸ§ πŸ€– + + +This repository contains GGUF format model files for DeepSeek-R1-Distill-Qwen-1.5B, quantized using [llama.cpp](https://github.com/ggerganov/llama.cpp). + +All the models have been quantized following the [instructions](https://github.com/ggerganov/llama.cpp/blob/master/examples/quantize/README.md#quantize) provided by llama.cpp. This is: +```bash +# obtain the official LLaMA model weights and place them in ./models +ls ./models +llama-2-7b tokenizer_checklist.chk tokenizer.model +# [Optional] for models using BPE tokenizers +ls ./models + vocab.json +# [Optional] for PyTorch .bin models like Mistral-7B +ls ./models + + +# install Python dependencies +python3 -m pip install -r requirements.txt + +# convert the model to ggml FP16 format +python3 convert_hf_to_gguf.py models/mymodel/ + +# quantize the model to 4-bits (using Q4_K_M method) +./llama-quantize ./models/mymodel/ggml-model-f16.gguf ./models/mymodel/ggml-model-Q4_K_M.gguf Q4_K_M + +# update the gguf filetype to current version if older version is now unsupported +./llama-quantize ./models/mymodel/ggml-model-Q4_K_M.gguf ./models/mymodel/ggml-model-Q4_K_M-v2.gguf COPY +``` + + +## Model Details + +Original model: https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B + +## Summary models πŸ“‹ +| Filename | Quant type | Description | +| -------- | ---------- | ----------- | +| [DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-F16.gguf) | F16 | Half precision, no quantization applied | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q8_0.gguf) | Q8_0 | 8-bit quantization, highest quality, largest size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q6_K.gguf) | Q6_K | 6-bit quantization, very high quality | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q5_1.gguf) | Q5_1 | 5-bit quantization, good balance of quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_M.gguf) | Q5_K_M | 5-bit quantization, good balance of quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q5_K_S.gguf) | Q5_K_S | 5-bit quantization, good balance of quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q5_0.gguf) | Q5_0 | 5-bit quantization, good balance of quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q4_1.gguf) | Q4_1 | 4-bit quantization, balanced quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf) | Q4_K_M | 4-bit quantization, balanced quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_S.gguf) | Q4_K_S | 4-bit quantization, balanced quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q4_0.gguf) | Q4_0 | 4-bit quantization, balanced quality and size | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_L.gguf) | Q3_K_L | 3-bit quantization, smaller size, lower quality | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_M.gguf) | Q3_K_M | 3-bit quantization, smaller size, lower quality | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q3_K_S.gguf) | Q3_K_S | 3-bit quantization, smaller size, lower quality | +| [DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf](https://huggingface.co/hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B/blob/main/DeepSeek-R1-Distill-Qwen-1.5B-Q2_K.gguf) | Q2_K | 2-bit quantization, smallest size, lowest quality | + + +## Usage with Ollama πŸ¦™ + +### Direct from Ollama +``` +ollama run hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B +``` + +## Download Models Using huggingface-cli πŸ€— + +### Installation of `huggingface_hub[cli]` +```bash +pip install -U "huggingface_hub[cli]" +``` + +### Downloading Specific Model Files +```bash +huggingface-cli download hdnh2006/DeepSeek-R1-Distill-Qwen-1.5B --include "DeepSeek-R1-Distill-Qwen-1.5B-Q4_K_M.gguf" --local-dir ./ +``` + +## Which File Should I Choose? πŸ“ˆ + +A comprehensive analysis with performance charts is provided by Artefact2 [here](https://gist.github.com/Artefact2/b5f810600771265fc1e39442288e8ec9). + +### Assessing System Capabilities +1. **Determine Your Model Size**: Start by checking the amount of RAM and VRAM available in your system. This will help you decide the largest possible model you can run. +2. **Optimizing for Speed**: + - **GPU Utilization**: To run your model as quickly as possible, aim to fit the entire model into your GPU's VRAM. Pick a version that’s 1-2GB smaller than the total VRAM. +3. **Maximizing Quality**: + - **Combined Memory**: For the highest possible quality, sum your system RAM and GPU's VRAM. Then choose a model that's 1-2GB smaller than this combined total. + +### Deciding Between 'I-Quant' and 'K-Quant' +1. **Simplicity**: + - **K-Quant**: If you prefer a straightforward approach, select a K-quant model. These are labeled as 'QX_K_X', such as Q5_K_M. +2. **Advanced Configuration**: + - **Feature Chart**: For a more nuanced choice, refer to the [llama.cpp feature matrix](https://github.com/ggerganov/llama.cpp/wiki/Feature-matrix). + - **I-Quant Models**: Best suited for configurations below Q4 and for systems running cuBLAS (Nvidia) or rocBLAS (AMD). These are labeled 'IQX_X', such as IQ3_M, and offer better performance for their size. + - **Compatibility Considerations**: + - **I-Quant Models**: While usable on CPU and Apple Metal, they perform slower compared to their K-quant counterparts. The choice between speed and performance becomes a significant tradeoff. + - **AMD Cards**: Verify if you are using the rocBLAS build or the Vulkan build. I-quants are not compatible with Vulkan. + - **Current Support**: At the time of writing, LM Studio offers a preview with ROCm support, and other inference engines provide specific ROCm builds. + +By following these guidelines, you can make an informed decision on which file best suits your system and performance needs. + + +## Contact 🌐 +Website: henrynavarro.org + +Email: public.contact.rerun407@simplelogin.com