初始化项目,由ModelHub XC社区提供模型
Model: geoffmunn/Qwen3-0.6B-f16 Source: Original Platform
This commit is contained in:
90
.gitattributes
vendored
Normal file
90
.gitattributes
vendored
Normal file
@@ -0,0 +1,90 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q3_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q3_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16LQ8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-0.6b-imatrix.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix-5000.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix-4697.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q4_K_M-imatrix.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-Q4_K_S-imatrix.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B:Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-UD:Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-UD-Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-4bit-Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-int4-Q4_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-4B-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-UD-Q4_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-Q4_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q4_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q4_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q3_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q3_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-14B-f16:Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q5_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q5_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix-8843-coder.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix-9343-generic.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q2_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16:Q2_K_HIFI.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
Qwen3-0.6B-f16-imatrix:Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
25
MODELFILE
Normal file
25
MODELFILE
Normal file
@@ -0,0 +1,25 @@
|
||||
# MODELFILE for Qwen3-0.6B-GGUF
|
||||
# Used by LM Studio, OpenWebUI, GPT4All, etc.
|
||||
|
||||
context_length: 32768
|
||||
embedding: false
|
||||
f16: cpu
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
prompt_template: >-
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
|
||||
# Stop sequences help end generation cleanly
|
||||
stop: "<|im_end|>"
|
||||
stop: "<|im_start|>"
|
||||
|
||||
# Default sampling
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
repeat_penalty: 1.1
|
||||
94
Q3_Quantisation_Comparison.md
Normal file
94
Q3_Quantisation_Comparison.md
Normal file
@@ -0,0 +1,94 @@
|
||||
# Qwen3-0.6B Quantization Comparison Summary
|
||||
|
||||
## Q3_HIFI (Adaptive/Custom)
|
||||
**Pros:**
|
||||
- 🏆 **Best quality** with lowest perplexity of 26.43 (16.4% better than Q3_K_M, 26.0% better than Q3_K_S)
|
||||
- 📦 **Smaller than Q3_K_M** (382.37 vs 389.12 MiB) while being significantly better quality
|
||||
- 🎯 Uses intelligent layer-sensitive quantization (Q3_HIFI on sensitive layers, mixed q3_K/q4_K elsewhere)
|
||||
- 📊 Most consistent results (lowest standard deviation in perplexity: ±0.23)
|
||||
|
||||
**Cons:**
|
||||
- 🐢 **Slowest inference** at 601.4 TPS (2.8% slower than Q3_K_M)
|
||||
- 🔧 Custom quantization may have less community support
|
||||
|
||||
**Best for:** Production deployments where output quality matters, tasks requiring accuracy (reasoning, coding, complex instructions), or when you want the best quality-to-size ratio.
|
||||
|
||||
## Performance Comparison (Q3_HIFI vs the others)
|
||||
|
||||
### Q3_K_M
|
||||
|
||||
| Metric | Q3_HIFI | Q3_K_M | Difference |
|
||||
|---------------------|-----------|-----------|-------------------------------|
|
||||
| **Speed (TPS)** | 601.39 | 618.42 | -17.03 (2.8% slower) |
|
||||
| **Perplexity** | 26.43 | 31.64 | **-5.21 (16.4% better)** |
|
||||
| **File Size** | 382.37 MiB| 389.12 MiB| **-6.75 MiB (1.7% smaller)** |
|
||||
| **Bits Per Weight** | 4.27 | 4.34 | -0.07 (1.6% less) |
|
||||
|
||||
**Pros:**
|
||||
- ⚖️ Traditional "balanced" approach between speed and quality
|
||||
- 📚 Well-documented, standard quantization method
|
||||
- ⚡ Fastest inference speed (618.4 TPS)
|
||||
|
||||
**Cons:**
|
||||
- 💾 **Largest file size** at 389.12 MiB despite not being the best quality
|
||||
- ❌ **Outclassed by Q3_HIFI** which is smaller AND better quality
|
||||
|
||||
**Best for:** Legacy compatibility or when you need a proven, standard quantization approach with maximum throughput.
|
||||
**Summary:** Q3_HIFI delivers significantly better quality (16.4% lower perplexity) in a smaller package (1.7% less storage) with only a marginal 2.8% speed penalty.
|
||||
|
||||
### Q3_K_S
|
||||
|
||||
| Metric | Q3_HIFI | Q3_K_S | Difference |
|
||||
|---------------------|-----------|-----------|---------------------------|
|
||||
| **Speed (TPS)** | 601.39 | 612.28 | -10.89 (1.8% slower) |
|
||||
| **Perplexity** | 26.43 | 35.70 | **-9.27 (26.0% better)** |
|
||||
| **File Size** | 382.37 MiB| 366.19 MiB| +16.18 MiB (4.4% larger) |
|
||||
| **Bits Per Weight** | 4.27 | 4.09 | +0.18 (4.4% more) |
|
||||
|
||||
**Pros:**
|
||||
- 💾 **Smallest file size** at 366.19 MiB
|
||||
- ✅ Best choice when storage is the absolute limiting factor
|
||||
|
||||
**Cons:**
|
||||
- ❌ **Worst quality** with perplexity of 35.70 (35% higher than Q3_HIFI)
|
||||
- 🐌 Not actually faster than Q3_K_M (612 vs 618 TPS)
|
||||
- Uses only q3_K quantization throughout (no mixed precision)
|
||||
|
||||
**Best for:** Extremely memory-constrained environments where every megabyte counts.
|
||||
**Summary:** Q3_HIFI trades a 1.8% speed reduction and 4.4% larger file size for a substantial 26.0% improvement in quality (lower perplexity).
|
||||
|
||||
---
|
||||
|
||||
## Recommendation Matrix
|
||||
|
||||
| Priority | Recommended Model | Rationale |
|
||||
|-------------------|-------------------|------------------------------------------------------------------------------|
|
||||
| **Quality First** | Q3_HIFI | 26% better perplexity than Q3_K_S with minimal speed loss |
|
||||
| **Speed First** | Q3_K_M | Fastest inference at 618 TPS, acceptable quality tradeoff |
|
||||
| **Best Balance** | Q3_HIFI | Better quality AND smaller size than Q3_K_M, only 2.8% slower |
|
||||
| **Smallest Size** | Q3_K_S | 4% smaller than Q3_HIFI, 6% smaller than Q3_K_M |
|
||||
|
||||
---
|
||||
|
||||
## Key Insight
|
||||
|
||||
**Q3_HIFI represents a clear advancement** over the traditional Q3_K_M approach. It achieves:
|
||||
- **16.4% lower perplexity** (better accuracy)
|
||||
- **1.7% smaller file size** (382 vs 389 MiB)
|
||||
- Only **2.8% slower** inference (601 vs 618 TPS)
|
||||
|
||||
The Q3_K_M quantization is essentially obsoleted by Q3_HIFI for most use cases. Q3_K_S offers marginal storage savings but with significantly degraded quality and is actually slower than Q3_K_M, making it difficult to recommend.
|
||||
|
||||
The only remaining practical choice is between **Q3_K_M** (maximum speed) and **Q3_HIFI** (maximum quality). Given the 2.8% speed difference is imperceptible at 600+ TPS, **Q3_HIFI is the recommended default**.
|
||||
|
||||
## Appendix (Test Environment Details)
|
||||
|
||||
| Component | Specification |
|
||||
|---------------|---------------------------------|
|
||||
| **OS** | Ubuntu 24.04.3 LTS |
|
||||
| **CPU** | AMD EPYC 9254 24-Core Processor |
|
||||
| **CPU Cores** | 96 cores (2 threads/core) |
|
||||
| **RAM** | 1.0Ti |
|
||||
| **GPU** | NVIDIA L40S × 2 |
|
||||
| **VRAM** | 46068 MiB per GPU |
|
||||
| **CUDA** | 12.9 |
|
||||
201
Q4_Quantization_Comparison.md
Normal file
201
Q4_Quantization_Comparison.md
Normal file
@@ -0,0 +1,201 @@
|
||||
# Qwen3-0.6B Quantization Comparison Summary
|
||||
|
||||
## F16 Baseline Reference
|
||||
| Metric | Value |
|
||||
|--------|-------|
|
||||
| **F16 Perplexity** | 21.8916 |
|
||||
| **File Size** | 1.40 GiB (16.00 BPW) |
|
||||
|
||||
All precision loss percentages below are calculated relative to this F16 baseline.
|
||||
|
||||
---
|
||||
|
||||
## Q4_K_HIFI (INT8 Residuals + Per-Block Scale)
|
||||
**Pros:**
|
||||
- 🏆 **Best quality without imatrix** with perplexity of 23.66 (0.1% better than Q4_K_M, 3.6% better than Q4_K_S)
|
||||
- 📊 **+8.1% PPL vs F16** — moderate precision loss from quantization
|
||||
- 🎯 Uses intelligent outlier preservation with INT8 residuals on critical tensors
|
||||
- 🔬 17 tensors use Q5_K_HIFI_RES8 format for maximum precision on sensitive weights
|
||||
|
||||
**Cons:**
|
||||
- 💾 **Largest file size** at 487.39 MiB (+6.9% vs Q4_K_M)
|
||||
- 🐢 Slower than both Q4_K variants (1.7% slower than Q4_K_M, 3.0% slower than Q4_K_S)
|
||||
|
||||
**Best for:** Production deployments where output quality matters, tasks requiring accuracy (reasoning, coding, complex instructions), especially on smaller models where quantization error has larger impact.
|
||||
|
||||
## Performance Comparison (Q4_K_HIFI vs the others)
|
||||
|
||||
### Q4_K_M
|
||||
|
||||
| Metric | Q4_K_HIFI | Q4_K_M | Difference |
|
||||
|---------------------|------------|------------|-------------------------------|
|
||||
| **Speed (TPS)** | 614.32 | 624.66 | -10.34 (1.7% slower) |
|
||||
| **Perplexity** | 23.6556 | 23.6856 | **-0.03 (0.1% better)** |
|
||||
| **PPL vs F16** | +8.1% | +8.2% | 0.1% less precision loss |
|
||||
| **File Size** | 487.39 MiB | 456.11 MiB | +31.28 MiB (6.9% larger) |
|
||||
| **Bits Per Weight** | 5.44 | 5.09 | +0.35 (6.9% more) |
|
||||
|
||||
**Pros:**
|
||||
- ⚖️ Traditional "balanced" approach between speed and quality
|
||||
- 📚 Well-documented, standard quantization method
|
||||
- 💾 Smaller file size than Q4_K_HIFI
|
||||
- ⚡ 1.7% faster inference than Q4_K_HIFI
|
||||
- 📊 **+8.2% PPL vs F16** — nearly identical precision loss to Q4_K_HIFI
|
||||
|
||||
**Cons:**
|
||||
- ❌ **Slightly lower quality** (essentially equal - only 0.1% higher perplexity than Q4_K_HIFI)
|
||||
|
||||
**Best for:** When storage is constrained but you still need reasonable quality.
|
||||
**Summary:** Without imatrix, Q4_K_HIFI and Q4_K_M are essentially identical in quality (0.1% difference). Q4_K_M is smaller (6.9% less) and faster (1.7%), making it the better choice without imatrix.
|
||||
|
||||
### Q4_K_S
|
||||
|
||||
| Metric | Q4_K_HIFI | Q4_K_S | Difference |
|
||||
|---------------------|------------|------------|-------------------------------|
|
||||
| **Speed (TPS)** | 614.32 | 632.79 | -18.47 (3.0% slower) |
|
||||
| **Perplexity** | 23.6556 | 24.5475 | **-0.89 (3.6% better)** |
|
||||
| **PPL vs F16** | +8.1% | +12.1% | 4.0% less precision loss |
|
||||
| **File Size** | 487.39 MiB | 443.30 MiB | +44.09 MiB (9.9% larger) |
|
||||
| **Bits Per Weight** | 5.44 | 4.95 | +0.49 (9.9% more) |
|
||||
|
||||
**Pros:**
|
||||
- ⚡ **Fastest inference** at 632.79 TPS (3.0% faster than Q4_K_HIFI)
|
||||
- 💾 **Smallest file size** at 443.30 MiB
|
||||
- ✅ Best choice when speed and storage are critical
|
||||
|
||||
**Cons:**
|
||||
- ❌ **Significantly worse quality** with perplexity of 24.55 (3.6% higher than Q4_K_HIFI)
|
||||
- 📊 **+12.1% PPL vs F16** — highest precision loss of all Q4_K variants
|
||||
- ⚠️ Uses minimal q6_K enhancement — impacts quality on sensitive weights
|
||||
|
||||
**Best for:** Extreme resource constraints, quick prototyping, or bulk processing where quality is less important.
|
||||
**Summary:** Q4_K_HIFI trades a 3.0% speed reduction and 9.9% larger file size for a 3.6% improvement in quality.
|
||||
|
||||
---
|
||||
|
||||
## Recommendation Matrix (Without imatrix)
|
||||
|
||||
| Priority | Recommended Model | Rationale |
|
||||
|-------------------|-------------------|--------------------------------------------------------------------------------|
|
||||
| **Quality First** | Q4_K_HIFI | 3.6% better perplexity than Q4_K_S, essentially equal to Q4_K_M |
|
||||
| **Speed First** | Q4_K_S | 3.0% faster, acceptable if quality degradation is tolerable |
|
||||
| **Best Balance** | Q4_K_M | Equal quality to Q4_K_HIFI, smaller (6.9%) and faster (1.7%) |
|
||||
| **Smallest Size** | Q4_K_S | 9.9% smaller than Q4_K_HIFI, 2.8% smaller than Q4_K_M |
|
||||
|
||||
---
|
||||
|
||||
## Key Insight
|
||||
|
||||
**Without imatrix, Q4_K_HIFI provides marginal benefit over Q4_K_M.** At 0.6B scale:
|
||||
- **0.1% lower perplexity** than Q4_K_M (23.66 vs 23.69) — essentially equal
|
||||
- **3.6% lower perplexity** than Q4_K_S (23.66 vs 24.55)
|
||||
- **1.7% slower** than Q4_K_M, **3.0% slower** than Q4_K_S
|
||||
- **All variants lose 8-12% precision vs F16** (21.89 baseline)
|
||||
|
||||
**However, with imatrix, Q4_K_M actually beats Q4_K_HIFI:**
|
||||
- Q4_K_M (imatrix): **22.92** PPL (+4.7% vs F16)
|
||||
- Q4_K_HIFI (imatrix): **22.95** PPL (+4.8% vs F16)
|
||||
|
||||
The INT8 residual format (Q5_K_HIFI_RES8) efficiently preserves outlier precision with moderate overhead. However, imatrix quantization provides similar benefits through a different mechanism.
|
||||
|
||||
💡 **Key Finding:** For 0.6B models, **Q4_K_M + imatrix** is the recommended approach — it achieves slightly better quality than Q4_K_HIFI while being smaller and faster.
|
||||
|
||||
---
|
||||
|
||||
## Precision Loss Summary (vs F16 Baseline: PPL 21.8916)
|
||||
|
||||
| Model | PPL (no imatrix) | vs F16 | PPL (imatrix) | vs F16 |
|
||||
|-----------|------------------|--------|---------------|--------|
|
||||
| Q4_K_HIFI | 23.6556 | **+8.1%** | 22.9451 | **+4.8%** |
|
||||
| Q4_K_M | 23.6856 | **+8.2%** | 22.9210 | **+4.7%** ✅ |
|
||||
| Q4_K_S | 24.5475 | **+12.1%** | 23.3136 | **+6.5%** |
|
||||
|
||||
**Key Observations:**
|
||||
- Without imatrix: All Q4_K variants lose 8-12% precision vs F16
|
||||
- With imatrix: Precision loss drops to 4.7-6.5% — imatrix recovers ~3-5.5% of lost precision
|
||||
- Q4_K_M (imatrix) achieves the **lowest precision loss** at only +4.7% vs F16
|
||||
- Q4_K_S suffers the most, with +12.1% precision loss without imatrix
|
||||
|
||||
---
|
||||
|
||||
## Tensor Distribution
|
||||
|
||||
| Model | q4_K | q5_K | q6_K | Q5_K_HIFI_RES8 | f32 | Total |
|
||||
|-----------|------|------|------|----------------|-----|-------|
|
||||
| Q4_K_S | 190 | 7 | 1 | 0 | 113 | 311 |
|
||||
| Q4_K_M | 169 | 0 | 29 | 0 | 113 | 311 |
|
||||
| Q4_K_HIFI | 158 | 0 | 23 | 17 | 113 | 311 |
|
||||
|
||||
**Q4_K_HIFI Enhancement:** 17 critical tensors (output.weight, token_embd, attn_v layers) use Q5_K_HIFI_RES8 format with INT8 residuals + per-block scale for maximum precision.
|
||||
|
||||
---
|
||||
|
||||
## Addendum: Impact of imatrix on ALL Quantization Types
|
||||
|
||||
All three quantization types can now use imatrix. Here's how they compare:
|
||||
|
||||
### imatrix Perplexity Improvements
|
||||
|
||||
| Model | Without imatrix | With imatrix | Improvement | PPL vs F16 (imatrix) |
|
||||
|-----------|-----------------|--------------|-------------|----------------------|
|
||||
| Q4_K_HIFI | 23.6556 | **22.9451** | **-0.71 (3.0% better)** | **+4.8%** |
|
||||
| Q4_K_M | 23.6856 | **22.9210** | **-0.76 (3.2% better)** | **+4.7%** ✅ |
|
||||
| Q4_K_S | 24.5475 | **23.3136** | **-1.23 (5.0% better)** | **+6.5%** |
|
||||
|
||||
### Revised Comparison (All with imatrix)
|
||||
|
||||
| Model | PPL (imatrix) | vs Q4_K_HIFI | vs F16 | Size |
|
||||
|-----------|---------------|--------------|--------|------|
|
||||
| Q4_K_M | **22.9210** | **-0.024 (-0.1%)** ✅ | **+4.7%** ✅ | 456.11 MiB |
|
||||
| Q4_K_HIFI | 22.9451 | baseline | +4.8% | 487.39 MiB |
|
||||
| Q4_K_S | 23.3136 | +0.369 (+1.6%) | +6.5% | 443.30 MiB |
|
||||
|
||||
### Key Findings
|
||||
|
||||
**With imatrix, Q4_K_M beats Q4_K_HIFI:**
|
||||
|
||||
| Comparison | Without imatrix | With imatrix |
|
||||
|------------|-----------------|--------------|
|
||||
| Q4_K_HIFI vs Q4_K_M | **-0.1%** (essentially equal) | **+0.1%** (Q4_K_M wins!) |
|
||||
| Q4_K_HIFI vs Q4_K_S | **-3.6%** (moderate advantage) | **-1.6%** (small advantage) |
|
||||
|
||||
### Revised Recommendations (When Using imatrix)
|
||||
|
||||
| Priority | Without imatrix | With imatrix |
|
||||
|-------------------|-----------------|--------------|
|
||||
| **Quality First** | Q4_K_M ≈ Q4_K_HIFI | **Q4_K_M** ✅ (best PPL!) |
|
||||
| **Best Balance** | Q4_K_M | **Q4_K_M** ✅ |
|
||||
| **Size/Speed** | Q4_K_S | Q4_K_S |
|
||||
|
||||
### Conclusion
|
||||
|
||||
**If you're using imatrix quantization:**
|
||||
- **Q4_K_M (imatrix) is the winner** — 0.1% better quality than Q4_K_HIFI (imatrix)
|
||||
- Q4_K_M is 6.9% smaller and 1.7% faster than Q4_K_HIFI
|
||||
- The size/speed overhead of Q4_K_HIFI provides **no quality benefit** with imatrix
|
||||
- **Q4_K_M + imatrix is the recommended choice for 0.6B models**
|
||||
|
||||
**If you're NOT using imatrix:**
|
||||
- Q4_K_HIFI and Q4_K_M are essentially equal (0.1% difference)
|
||||
- Q4_K_M is smaller and faster, making it the practical choice
|
||||
- **Q4_K_M without imatrix is recommended** (simpler, smaller, faster, equal quality)
|
||||
|
||||
**Bottom line for 0.6B models:** Q4_K_M (with or without imatrix) is the best choice. Q4_K_HIFI's INT8 residual format doesn't provide meaningful benefits at this model scale.
|
||||
|
||||
---
|
||||
|
||||
## Appendix (Test Environment Details)
|
||||
|
||||
| Component | Specification |
|
||||
|---------------|----------------------------------------|
|
||||
| **OS** | Ubuntu 24.04.3 LTS |
|
||||
| **CPU** | AMD EPYC 9254 24-Core Processor |
|
||||
| **CPU Cores** | 96 cores (2 threads/core) |
|
||||
| **RAM** | 1.0Ti |
|
||||
| **GPU** | NVIDIA L40S × 2 |
|
||||
| **VRAM** | 46068 MiB per GPU |
|
||||
| **CUDA** | 12.9 |
|
||||
| **Test Data** | wikitext-2-raw, 584 chunks |
|
||||
| **Context** | 512 tokens |
|
||||
| **Samples** | 200 per speed benchmark |
|
||||
| **imatrix** | mixed-imatrix-dataset.txt, 4697 chunks |
|
||||
177
Qwen3-0.6B-f16-Q2_K/README.md
Normal file
177
Qwen3-0.6B-f16-Q2_K/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q2
|
||||
- qwen3-0.6b-q2_k
|
||||
- qwen3-0.6b-q2_k-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q2_K
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q2_K** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 347 MB
|
||||
- **Precision**: Q2_K
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|-----------------------------------------------------------------|
|
||||
| **Speed** | ⚡ Fast |
|
||||
| **RAM Required** | ~0.6 GB |
|
||||
| **Recommendation** | 🚨 **DO NOT USE.** Could not provide an answer to any question. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ2_K.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q2_K.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q2_K -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q2_K" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q2_K",
|
||||
"prompt": "Respond exactly as follows: Repeat the word 'hello' five times separated by commas.",
|
||||
"temperature": 0.1,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q3_K_M/README.md
Normal file
177
Qwen3-0.6B-f16-Q3_K_M/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q3
|
||||
- qwen3-0.6b-q3_k_m
|
||||
- qwen3-0.6b-q3_k_m-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q3_K_M
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q3_K_M** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 414 MB
|
||||
- **Precision**: Q3_K_M
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|---------------------------------------------------------------------|
|
||||
| **Speed** | ⚡ Fast |
|
||||
| **RAM Required** | ~0.8 GB |
|
||||
| **Recommendation** | First place in the bat & ball question, no other top 3 appearances. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ3_K_M.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q3_K_M.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q3_K_M -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q3_K_M" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q3_K_M",
|
||||
"prompt": "Respond exactly as follows: Repeat the word 'hello' five times separated by commas.",
|
||||
"temperature": 0.1,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q3_K_S/README.md
Normal file
177
Qwen3-0.6B-f16-Q3_K_S/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q3
|
||||
- qwen3-0.6b-q3_k_s
|
||||
- qwen3-0.6b-q3_k_s-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q3_K_S
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q3_K_S** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 390 MB
|
||||
- **Precision**: Q3_K_S
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|-------------------------------------------------------|
|
||||
| **Speed** | ⚡ Fast |
|
||||
| **RAM Required** | ~0.7 GB |
|
||||
| **Recommendation** | Not recommended, did not appear in any top 3 results. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ3_K_S.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q3_K_S.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q3_K_S -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q3_K_S" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q3_K_S",
|
||||
"prompt": "Respond exactly as follows: Repeat the word 'hello' five times separated by commas.",
|
||||
"temperature": 0.1,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q4_K_M/README.md
Normal file
177
Qwen3-0.6B-f16-Q4_K_M/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q4
|
||||
- qwen3-0.6b-q4_k_m
|
||||
- qwen3-0.6b-q4_k_m-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q4_K_M
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q4_K_M** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 484 MB
|
||||
- **Precision**: Q4_K_M
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|--------------------------------------------------|
|
||||
| **Speed** | 🚀 Fast |
|
||||
| **RAM Required** | ~1.0 GB |
|
||||
| **Recommendation** | Showed up in a few results, but not recommended. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ4_K_M.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q4_K_M.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q4_K_M -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q4_K_M" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q4_K_M",
|
||||
"prompt": "Respond exactly as follows: Write a short joke about cats.",
|
||||
"temperature": 0.8,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q4_K_S/README.md
Normal file
177
Qwen3-0.6B-f16-Q4_K_S/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q4
|
||||
- qwen3-0.6b-q4_k_s
|
||||
- qwen3-0.6b-q4_k_s-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q4_K_S
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q4_K_S** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 471 MB
|
||||
- **Precision**: Q4_K_S
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|---------------------------------------------------------|
|
||||
| **Speed** | 🚀 Fast |
|
||||
| **RAM Required** | ~0.9 GB |
|
||||
| **Recommendation** | A good option for technical, low-temperature questions. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ4_K_S.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-4B-f16:Q4_K_S.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-4B-f16:Q4_K_S -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-4B-f16:Q4_K_S" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q4_K_S",
|
||||
"prompt": "Respond exactly as follows: Repeat the word 'hello' five times separated by commas.",
|
||||
"temperature": 0.1,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q5_K_M/README.md
Normal file
177
Qwen3-0.6B-f16-Q5_K_M/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q5
|
||||
- qwen3-0.6b-q5_k_m
|
||||
- qwen3-0.6b-q5_k_m-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q5_K_M
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q5_K_M** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 551 MB
|
||||
- **Precision**: Q5_K_M
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|--------------------------------------------------------------------|
|
||||
| **Speed** | 🐢 Medium |
|
||||
| **RAM Required** | ~1.2 GB |
|
||||
| **Recommendation** | 🥇 **Best overall model.** Highly recommended for all query types. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ5_K_M.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q5_K_M.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q5_K_M -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q5_K_M" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q5_K_M",
|
||||
"prompt": "Respond exactly as follows: Explain what gravity is in one sentence suitable for a child.",
|
||||
"temperature": 0.6,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q5_K_S/README.md
Normal file
177
Qwen3-0.6B-f16-Q5_K_S/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q5
|
||||
- qwen3-0.6b-q5_k_s
|
||||
- qwen3-0.6b-q5_k_s-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q5_K_S
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q5_K_S** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 544 MB
|
||||
- **Precision**: Q5_K_S
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|---------------------------------------------------------|
|
||||
| **Speed** | 🐢 Medium |
|
||||
| **RAM Required** | ~1.1 GB |
|
||||
| **Recommendation** | 🥈 A very close second place. Good for all query types. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ5_K_S.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q5_K_S.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q5_K_S -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q5_K_S" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q5_K_S",
|
||||
"prompt": "Respond exactly as follows: Write a short joke about cats.",
|
||||
"temperature": 0.8,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q6_K/README.md
Normal file
177
Qwen3-0.6B-f16-Q6_K/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q6
|
||||
- qwen3-0.6b-q6_k
|
||||
- qwen3-0.6b-q6_k-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q6_K
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q6_K** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 623 MB
|
||||
- **Precision**: Q6_K
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|--------------------------------------------------|
|
||||
| **Speed** | 🐌 Slow |
|
||||
| **RAM Required** | ~1.4 GB |
|
||||
| **Recommendation** | Showed up in a few results, but not recommended. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ6_K.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q6_K.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q6_K -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q6_K" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q6_K",
|
||||
"prompt": "Respond exactly as follows: Explain what gravity is in one sentence suitable for a child.",
|
||||
"temperature": 0.6,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
177
Qwen3-0.6B-f16-Q8_0/README.md
Normal file
177
Qwen3-0.6B-f16-Q8_0/README.md
Normal file
@@ -0,0 +1,177 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-q8
|
||||
- qwen3-0.6b-q8_0
|
||||
- qwen3-0.6b-q8_0-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16:Q8_0
|
||||
|
||||
Quantized version of [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B) at **Q8_0** level, derived from **f16** base weights.
|
||||
|
||||
## Model Info
|
||||
|
||||
- **Format**: GGUF (for llama.cpp and compatible runtimes)
|
||||
- **Size**: 805 MB
|
||||
- **Precision**: Q8_0
|
||||
- **Base Model**: [Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)
|
||||
- **Conversion Tool**: [llama.cpp](https://github.com/ggerganov/llama.cpp)
|
||||
|
||||
## Quality & Performance
|
||||
|
||||
| Metric | Value |
|
||||
|--------------------|-----------------------------------------------------------|
|
||||
| **Speed** | 🐌 Slow |
|
||||
| **RAM Required** | ~1.7 GB |
|
||||
| **Recommendation** | 🥉 Very good for non-technical, creative-style questions. |
|
||||
|
||||
## Prompt Template (ChatML)
|
||||
|
||||
This model uses the **ChatML** format used by Qwen:
|
||||
|
||||
```text
|
||||
<|im_start|>system
|
||||
You are a helpful assistant.<|im_end|>
|
||||
<|im_start|>user
|
||||
{prompt}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
```
|
||||
|
||||
Set this in your app (LM Studio, OpenWebUI, etc.) for best results.
|
||||
|
||||
## Generation Parameters
|
||||
|
||||
Recommended defaults:
|
||||
|
||||
| Parameter | Value |
|
||||
|----------------|-------|
|
||||
| Temperature | 0.6 |
|
||||
| Top-P | 0.95 |
|
||||
| Top-K | 20 |
|
||||
| Min-P | 0.0 |
|
||||
| Repeat Penalty | 1.1 |
|
||||
|
||||
Stop sequences: `<|im_end|>`, `<|im_start|>`
|
||||
|
||||
> ⚠️ Due to model size, avoid temperatures above 0.9 — outputs become highly unpredictable.
|
||||
|
||||
## 💡 Usage Tips
|
||||
|
||||
> This model is best suited for lightweight tasks:
|
||||
>
|
||||
> ### ✅ Ideal Uses
|
||||
> - Quick replies and canned responses
|
||||
> - Intent classification (e.g., “Is this user asking for help?”)
|
||||
> - UI prototyping and local AI testing
|
||||
> - Embedded/NPU deployment
|
||||
>
|
||||
> ### ❌ Limitations
|
||||
> - No complex reasoning or multi-step logic
|
||||
> - Poor math and code generation
|
||||
> - Limited world knowledge
|
||||
> - May repeat or hallucinate frequently at higher temps
|
||||
>
|
||||
> ---
|
||||
>
|
||||
> 🔄 **Fast Iteration Friendly**
|
||||
> Perfect for developers building prompt templates or testing UI integrations.
|
||||
>
|
||||
> 🔋 **Runs on Almost Anything**
|
||||
> Even Raspberry Pi Zero W can run Q2_K with swap enabled.
|
||||
>
|
||||
> 📦 **Tiny Footprint**
|
||||
> Fits easily on USB drives, microSD cards, or IoT devices.
|
||||
|
||||
## Customisation & Troubleshooting
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ8_0.gguf`
|
||||
2. `nano Modelfile` and enter these details:
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q8_0.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q8_0 -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q8_0" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## 🖥️ CLI Example Using Ollama or TGI Server
|
||||
|
||||
Here’s how you can query this model via API using `curl` and `jq`. Replace the endpoint with your local server (e.g., Ollama, Text Generation Inference).
|
||||
|
||||
```bash
|
||||
curl http://localhost:11434/api/generate -s -N -d '{
|
||||
"model": "hf.co/geoffmunn/Qwen3-0.6B-f16:Q8_0",
|
||||
"prompt": "Respond exactly as follows: Explain what gravity is in one sentence suitable for a child.",
|
||||
"temperature": 0.6,
|
||||
"top_p": 0.95,
|
||||
"top_k": 20,
|
||||
"min_p": 0.0,
|
||||
"repeat_penalty": 1.1,
|
||||
"stream": false
|
||||
}' | jq -r '.response'
|
||||
```
|
||||
|
||||
🎯 **Why this works well**:
|
||||
- The prompt is meaningful yet achievable for a tiny model.
|
||||
- Temperature tuned appropriately: lower for deterministic output (`0.1`), higher for jokes (`0.8`).
|
||||
- Uses `jq` to extract clean response.
|
||||
|
||||
> 💬 Tip: For ultra-low-latency use, try `Q3_K_M` or `Q4_K_S` on older laptops.
|
||||
|
||||
## Verification
|
||||
|
||||
Check integrity:
|
||||
|
||||
```bash
|
||||
sha256sum -c ../SHA256SUMS.txt
|
||||
```
|
||||
|
||||
## Usage
|
||||
|
||||
Compatible with:
|
||||
- [LM Studio](https://lmstudio.ai) – local AI model runner
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface
|
||||
- [GPT4All](https://gpt4all.io) – private, offline AI chatbot
|
||||
- Directly via `llama.cpp`
|
||||
|
||||
## License
|
||||
|
||||
Apache 2.0 – see base model for full terms.
|
||||
3
Qwen3-0.6B-f16-imatrix-8843-coder.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix-8843-coder.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c45c9b4b6a0116328b8fc846af1860cf4776730d674db8cc13a0a65441fd709a
|
||||
size 1177056
|
||||
3
Qwen3-0.6B-f16-imatrix-9343-generic.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix-9343-generic.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:48c42c29df3e18012e80e3714d221e98fde7bb16cf6a674ef8cafc48e146ba93
|
||||
size 1177056
|
||||
3
Qwen3-0.6B-f16-imatrix:Q2_K.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cbca107e43d1e2c66e64c6f9fb0ac105cafc5b6ee4e04a9495634ee0c6e91a02
|
||||
size 347289088
|
||||
3
Qwen3-0.6B-f16-imatrix:Q2_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q2_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:03013f6995b0fbcdecdfa4c8a2f9147ccc2896a6f57b72fab4b8ce243cbba67f
|
||||
size 387882496
|
||||
3
Qwen3-0.6B-f16-imatrix:Q3_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q3_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:3c821acded439c445e3be29e25add24c1241500a5109cb3e93b75cdcc37c0273
|
||||
size 469736960
|
||||
3
Qwen3-0.6B-f16-imatrix:Q3_K_M.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q3_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c866f9b0d7e6868041c4041cf046ce53911ccd68588d8c68f51dc7cbb7b2e81c
|
||||
size 413979136
|
||||
3
Qwen3-0.6B-f16-imatrix:Q3_K_S.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q3_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:54d992bcfc70d95d1b5f8a7851e99b4566628fe0b4a1777e87413ce2ed0230a2
|
||||
size 389927424
|
||||
3
Qwen3-0.6B-f16-imatrix:Q4_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q4_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:5d9b5f679fc30dc941a5270789ea2b8bbc2af2354cb2ba87de32c7fd85e825fb
|
||||
size 517018176
|
||||
3
Qwen3-0.6B-f16-imatrix:Q4_K_M.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:eb649d990ea34075ac39e439b6da8b64f52489cf34062a955bb2bd5430f1119c
|
||||
size 484220416
|
||||
3
Qwen3-0.6B-f16-imatrix:Q4_K_S.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q4_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:a62cfb4c6e78758ee91e0da8fecd5057c9d17cd87fa858390e63f630b5804d0b
|
||||
size 470785536
|
||||
3
Qwen3-0.6B-f16-imatrix:Q5_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q5_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:257ada8220f533dcb2507abf667dbf3b3e289440a916fd086f58a3346232030e
|
||||
size 572041792
|
||||
3
Qwen3-0.6B-f16-imatrix:Q5_K_M.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q5_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:10dc246a6511ef5d12c0ede09058fa86fc6e2a5abec6ed8d90ce767001d3a30c
|
||||
size 551378432
|
||||
3
Qwen3-0.6B-f16-imatrix:Q5_K_S.gguf
Normal file
3
Qwen3-0.6B-f16-imatrix:Q5_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f2eccfb7b119616095dc731e7b1cff9b3bdb69f5e138710e9232651cca08d3b3
|
||||
size 543579648
|
||||
3
Qwen3-0.6B-f16:Q2_K.gguf
Normal file
3
Qwen3-0.6B-f16:Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:fbc6cb950ba04edecadbe76cc65b139f43be7331c223e305bec29984344304b0
|
||||
size 347288832
|
||||
3
Qwen3-0.6B-f16:Q2_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16:Q2_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:d5e3dbf8e9250e5fca3908b26feda0f68c42f71d72bc53de80ea603eac08cdb1
|
||||
size 387882240
|
||||
3
Qwen3-0.6B-f16:Q3_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16:Q3_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8130fbdc725873d9084f9ebb42171f4e1eb605f368646f8e1fded499dbc4f601
|
||||
size 469736704
|
||||
3
Qwen3-0.6B-f16:Q3_K_M.gguf
Normal file
3
Qwen3-0.6B-f16:Q3_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c69f480b0880981d82e3bf8bb5e8e83ed2da1d47a9c9bb7f2a2345d61edf53c9
|
||||
size 413978880
|
||||
3
Qwen3-0.6B-f16:Q3_K_S.gguf
Normal file
3
Qwen3-0.6B-f16:Q3_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cec5dd044738d87f69050c05ba18d83d86b82a2be086f79ae50bea49c3461ab6
|
||||
size 389927168
|
||||
3
Qwen3-0.6B-f16:Q4_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16:Q4_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6532d4ed281be062e060f1909b6d7ae7a350a005ea99f2001c9e011cd2b1c220
|
||||
size 532947264
|
||||
3
Qwen3-0.6B-f16:Q4_K_M.gguf
Normal file
3
Qwen3-0.6B-f16:Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2dc0ee44eb39790624623cf5e2a8cc21973c4839a67fed406dd3f9b2e6b7f800
|
||||
size 484220160
|
||||
3
Qwen3-0.6B-f16:Q4_K_S.gguf
Normal file
3
Qwen3-0.6B-f16:Q4_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:809ad08fda3b992a0189350f37e17dabfe461b0e54a5b886e393cf168f293d27
|
||||
size 470785280
|
||||
3
Qwen3-0.6B-f16:Q5_K_HIFI.gguf
Normal file
3
Qwen3-0.6B-f16:Q5_K_HIFI.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:805c0d72e8ebc334c2c96815719df0043d03d6b139251355e0109030d7dd155d
|
||||
size 572041536
|
||||
3
Qwen3-0.6B-f16:Q5_K_M.gguf
Normal file
3
Qwen3-0.6B-f16:Q5_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:24dd36130d07b2b3389fd45c9d43c3eb677bac79c60f130452a7ad31052ea807
|
||||
size 551378176
|
||||
3
Qwen3-0.6B-f16:Q5_K_S.gguf
Normal file
3
Qwen3-0.6B-f16:Q5_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:17bacd30d815829d3fa2dbe2becec0b7d0f32b7450bbab0b8f2de34eb79bda47
|
||||
size 543579392
|
||||
3
Qwen3-0.6B-f16:Q6_K.gguf
Normal file
3
Qwen3-0.6B-f16:Q6_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c087e4dedc41c7f56b926f82a63231cbdb9ca42e99fa1e03a918c8ae101d6a5f
|
||||
size 622733568
|
||||
3
Qwen3-0.6B-f16:Q8_0.gguf
Normal file
3
Qwen3-0.6B-f16:Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:01c657652e8f88a397c0a7bf540633b92d2c132fac1dcf88a47d0c86ce34843a
|
||||
size 804753664
|
||||
1337
Qwen3-0.6b-f16-analysis.md
Normal file
1337
Qwen3-0.6b-f16-analysis.md
Normal file
File diff suppressed because it is too large
Load Diff
302
README.md
Normal file
302
README.md
Normal file
@@ -0,0 +1,302 @@
|
||||
---
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- gguf
|
||||
- qwen
|
||||
- qwen3
|
||||
- qwen3-0.6b
|
||||
- qwen3-0.6b-gguf
|
||||
- llama.cpp
|
||||
- quantized
|
||||
- text-generation
|
||||
- chat
|
||||
- edge-ai
|
||||
- tiny-model
|
||||
- imatrix
|
||||
- Q3_HIFI
|
||||
- Q4_HIFI
|
||||
- Q5_HIFI
|
||||
- outlier-aware
|
||||
- high-fidelity
|
||||
datasets:
|
||||
- wikitext
|
||||
- codeparrot
|
||||
- openwebmath
|
||||
quantization: Q3_HIFI
|
||||
base_model: Qwen/Qwen3-0.6B
|
||||
author: geoffmunn
|
||||
pipeline_tag: text-generation
|
||||
language:
|
||||
- en
|
||||
- zh
|
||||
---
|
||||
|
||||
# Qwen3-0.6B-f16-GGUF
|
||||
|
||||
This is a **GGUF-quantized version** of the **[Qwen/Qwen3-0.6B](https://huggingface.co/Qwen/Qwen3-0.6B)** language model — a compact **600-million-parameter** LLM designed for **ultra-fast inference on low-resource devices**.
|
||||
|
||||
Converted for use with `llama.cpp`, [LM Studio](https://lmstudio.ai), [OpenWebUI](https://openwebui.com), and [GPT4All](https://gpt4all.io), enabling private AI anywhere — even offline.
|
||||
|
||||
> ⚠️ **Note**: This is a *very small* model. It will not match larger models (e.g., 4B+) in reasoning, coding, or factual accuracy. However, it shines in **speed, portability, and efficiency**.
|
||||
|
||||
## Why Use a 0.6B Model?
|
||||
|
||||
While limited in capability compared to larger models, **Qwen3-0.6B** excels at:
|
||||
- Running **instantly** on CPUs without GPU
|
||||
- Fitting into **<2GB RAM**, even when quantized
|
||||
- Enabling **offline AI on microcontrollers, phones, or edge devices**
|
||||
- Serving as a **fast baseline** for lightweight NLP tasks (intent detection, short responses)
|
||||
|
||||
It’s ideal for:
|
||||
- Chatbots with simple flows
|
||||
- On-device assistants
|
||||
- Educational demos
|
||||
- Rapid prototyping
|
||||
|
||||
# HIFI Quantization: High-Fidelity Low-Bit Compression
|
||||
|
||||
This is a custom quantization type that was created specifically to test if it was possible to obtain higher precision than the standard options (Q3_K_M for example).
|
||||
|
||||
**HIFI** ("High-Fidelity") quantization intelligently preserves model quality during aggressive weight compression by applying **tiered precision allocation** to critical weights. Instead of uniform bit reduction across all parameters, HIFI:
|
||||
|
||||
1. **Identifies sensitivity**: Uses weight analysis (and optionally imatrix) to locate tensors most vulnerable to quantization error
|
||||
2. **Applies residual correction**: For the most critical 2–6 tensors, stores a secondary 8-bit residual correction term (`*_HIFI_RES8` types) that recovers precision lost in the primary quantization pass
|
||||
3. **Tiered allocation**: Combines base quantization (Q3_K/Q4_K/Q5_K) with elevated precision tensors (Q4_K/Q5_K/Q6_K) on sensitive layers
|
||||
|
||||
This approach delivers near-lossless quality at dramatically reduced memory footprints—typically **64–78% memory reduction** versus F16 with minimal quality degradation.
|
||||
|
||||
# Qwen3 0.6B Quantization Guide: Cross-Bit Summary & Recommendations
|
||||
|
||||
## Executive Summary
|
||||
|
||||
At 0.6B scale, **quantization sensitivity is extreme**—small models lose proportionally more precision than larger ones when compressed. **Q2_K is unusable at any variant** (+88–106% precision loss even with imatrix). Viable options start at Q3_K, with quality improving dramatically through Q4_K and Q5_K:
|
||||
|
||||
| Bit Width | Best Variant (+ imatrix) | Quality vs F16 | File Size | Speed | Memory | Viability |
|
||||
|-----------|--------------------------|----------------|-----------|-------|--------|-----------|
|
||||
| **Q5_K** | Q5_K_M + imatrix | **+2.74%** ✅ | 508 MiB | 603 TPS | 1,103 MiB | Excellent |
|
||||
| **Q4_K** | Q4_K_M + imatrix | +4.82% ✅ | 456 MiB | 624 TPS | 1,038 MiB | Very Good |
|
||||
| **Q3_K** | Q3_K_HIFI + imatrix | +6.4% ✅ | 442 MiB | **632 TPS** (fastest) | 1,167 MiB | Good |
|
||||
| **Q2_K** | Q2_K_HIFI + imatrix | **+88.3%** ❌ | 364 MiB | 638 TPS | 946 MiB | **UNUSABLE** |
|
||||
|
||||
💡 **Critical insight**: Unlike larger models, **0.6B requires imatrix for Q3_K/Q4_K viability** (recovers 9–27% of lost precision). Q5_K variants remain usable without imatrix but still benefit measurably (+0.5–0.9% PPL improvement).
|
||||
|
||||
---
|
||||
|
||||
## Bit-Width Recommendations by Use Case
|
||||
|
||||
### ✅ Quality-Critical Applications
|
||||
**→ Q5_K_M + imatrix**
|
||||
- Only +2.74% precision loss vs F16 (PPL 22.49) — near-lossless for this scale
|
||||
- 45% memory reduction (1,103 MiB vs 2,015 MiB)
|
||||
- 51% faster than F16 (603 TPS)
|
||||
- ⚠️ **Avoid Q5_K_HIFI** — provides *no meaningful advantage* over Q5_K_M (0.02% PPL difference within measurement noise) while requiring custom build and 3.8% larger size
|
||||
|
||||
### ⚖️ Best Overall Balance (Recommended Default)
|
||||
**→ Q4_K_M + imatrix**
|
||||
- Excellent +4.82% precision loss (PPL 22.95) — imperceptible degradation in practice
|
||||
- Strong 624 TPS speed (+56% vs F16)
|
||||
- Compact 456 MiB file size (67% smaller than F16)
|
||||
- Standard llama.cpp compatibility — no custom builds needed
|
||||
- Ideal for most development and production scenarios
|
||||
|
||||
### 🚀 Maximum Speed / Minimum Size
|
||||
**→ Q3_K_HIFI + imatrix**
|
||||
- **Unique win-win at 0.6B scale**: Fastest variant (632 TPS) **AND** best Q3 quality (+6.4% loss)
|
||||
- Smallest viable footprint (442 MiB file, 1,167 MiB runtime)
|
||||
- ⚠️ **Never use Q3_K_S without imatrix** — suffers catastrophic +63.1% quality loss (unusable)
|
||||
|
||||
### 📱 Extreme Memory Constraints (< 450 MiB)
|
||||
**→ Q3_K_S + imatrix**
|
||||
- Absolute smallest (366 MiB file, 1,095 MiB runtime)
|
||||
- Acceptable +36.7% precision loss with imatrix (vs unusable +63.1% without)
|
||||
- Only viable option under 450 MiB budget
|
||||
|
||||
### ⛔ Avoid Entirely
|
||||
**→ All Q2_K variants**
|
||||
- Minimum +88.3% precision loss even with imatrix (PPL 41.22)
|
||||
- Output quality severely compromised — incoherent generations expected
|
||||
- **Minimum viable quantization for 0.6B is Q3_K_S with imatrix**
|
||||
|
||||
---
|
||||
|
||||
## Critical Warnings for 0.6B Scale
|
||||
|
||||
⚠️ **Q2_K is unusable at 0.6B scale** — Do not deploy under any circumstances. Even Q2_K_HIFI + imatrix suffers +88.3% precision loss. This is a hard floor — Q3_K is the minimum viable quantization level.
|
||||
|
||||
⚠️ **imatrix is non-optional for Q3_K/Q4_K** — Without it:
|
||||
- Q3_K variants lose 15.9–63.1% precision (borderline unusable)
|
||||
- Q4_K variants lose 8.1–12.2% precision (significant degradation)
|
||||
- *All recover 9–27% of lost precision with imatrix at zero inference cost*
|
||||
|
||||
⚠️ **HIFI variants provide negligible benefit at 0.6B**:
|
||||
- Q5_K_HIFI differs from Q5_K_M by only **1 tensor** (168 vs 169 q5_K)
|
||||
- Q4_K_HIFI differs from Q4_K_M by marginal tensor allocation changes
|
||||
- Quality differences are **within measurement noise** (±0.20 PPL)
|
||||
- Costs 3.8–6.9% more size and requires custom build — **not worth it**
|
||||
|
||||
⚠️ **Q3_K_HIFI uniquely breaks the quality/speed tradeoff** at 0.6B:
|
||||
- Unlike larger models where HIFI is slower, at 0.6B it's the *fastest* Q3 variant (+2.4% vs Q3_K_M)
|
||||
- This anomaly occurs because tensor allocation differences compress to minimal overhead at tiny scales
|
||||
|
||||
⚠️ **Small models ≠ large models** — Quantization behavior differs fundamentally:
|
||||
- At 0.6B: HIFI variants provide negligible benefit; Q2_K is unusable
|
||||
- At 8B+: HIFI variants deliver measurable quality gains; Q2_K may be viable for specific tasks
|
||||
- Never assume quantization patterns scale linearly across model sizes
|
||||
|
||||
---
|
||||
|
||||
## Memory Budget Guide
|
||||
|
||||
| Available VRAM | Recommended Variant | Expected Quality | Why |
|
||||
|----------------|---------------------|------------------|-----|
|
||||
| **< 450 MiB** | Q3_K_S + imatrix | PPL 29.92, +36.7% loss ⚠️ | Only option that fits; quality acceptable for non-critical tasks |
|
||||
| **450 – 600 MiB** | Q3_K_HIFI + imatrix | PPL 23.29, +6.4% loss ✅ | Best Q3 quality; unique speed/quality win-win |
|
||||
| **600 – 800 MiB** | Q4_K_M + imatrix | PPL 22.95, +4.82% loss ✅ | Best balance of quality/speed/size; standard compatibility |
|
||||
| **800 – 1,200 MiB** | Q5_K_M + imatrix | PPL 22.49, +2.74% loss ✅ | Near-lossless quality; best precision available |
|
||||
| **> 1,200 MiB** | F16 | PPL 21.89, 0% loss | F16 is only 1.4 GiB total — viable baseline for tiny models |
|
||||
|
||||
---
|
||||
|
||||
## Decision Flowchart
|
||||
|
||||
```mermaid
|
||||
Memory < 450 MiB?
|
||||
├─ Yes → Q3_K_S + imatrix (only option; +36.7% loss)
|
||||
└─ No → Need best quality?
|
||||
├─ Yes → Q5_K_M + imatrix (+2.74% loss)
|
||||
└─ No → Need max speed?
|
||||
├─ Yes → Q3_K_HIFI + imatrix (632 TPS, +6.4% loss)
|
||||
└─ No → Q4_K_M + imatrix (best balance, +4.82% loss)
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## Cross-Bit Performance Comparison
|
||||
|
||||
| Priority | Q3_K Best | Q4_K Best | Q5_K Best | Winner |
|
||||
|----------|-----------|-----------|-----------|--------|
|
||||
| **Quality (with imat)** | Q3_K_HIFI (+6.4%) | Q4_K_M (+4.82%) | **Q5_K_M (+2.74%)** ✅ | **Q5_K_M** |
|
||||
| **Speed** | **Q3_K_HIFI (632 TPS)** ✅ | Q4_K_S (624 TPS) | Q5_K_S (607 TPS) | **Q3_K_HIFI** |
|
||||
| **Smallest Size** | **Q3_K_S (366 MiB)** ✅ | Q4_K_S (416 MiB) | Q5_K_S (501 MiB) | **Q3_K_S** |
|
||||
| **Best Balance** | Q3_K_HIFI + imat | **Q4_K_M + imat** ✅ | Q5_K_M + imat | **Q4_K_M** |
|
||||
| **Viability Floor** | Q3_K_S + imat ✅ | Q4_K_S + imat ✅ | Q5_K_S ✅ | **Q3 minimum** |
|
||||
|
||||
✅ = Recommended for general use
|
||||
⚠️ = Context-dependent (see warnings above)
|
||||
❌ = Unusable (Q2_K variants)
|
||||
|
||||
---
|
||||
|
||||
## Bottom Line Recommendations
|
||||
|
||||
| Scenario | Recommended Variant | Rationale |
|
||||
|----------|---------------------|-----------|
|
||||
| **Default / General Purpose** | Q4_K_M + imatrix | Best balance of quality (+4.82%), speed (624 TPS), size (456 MiB), and compatibility |
|
||||
| **Maximum Quality** | Q5_K_M + imatrix | Near-lossless (+2.74% vs F16) with modest size/speed trade-offs |
|
||||
| **Maximum Speed** | Q3_K_HIFI + imatrix | Fastest (632 TPS) with surprisingly good quality (+6.4% loss) |
|
||||
| **Minimum Size** | Q3_K_S + imatrix | Smallest footprint (366 MiB) — only if memory < 450 MiB |
|
||||
| **Avoid Entirely** | All Q2_K variants | Unusable quality even with imatrix (+88%+ loss) |
|
||||
|
||||
⚠️ **Golden rules for 0.6B**:
|
||||
1. **Never use Q2_K** — minimum viable quantization is Q3_K_S *with imatrix*
|
||||
2. **Always use imatrix with Q3_K/Q4_K** — quality degradation without it is severe and avoidable
|
||||
3. **Skip HIFI variants** — provide negligible benefit at this scale while requiring custom builds
|
||||
4. **F16 is viable** — at only 1.4 GiB total, consider F16 as baseline if quality is paramount
|
||||
|
||||
✅ **0.6B quantization reality check**: This scale is highly sensitive to compression. While Q5_K_M + imatrix delivers excellent results (+2.74% loss), the absolute quality floor is much higher than at larger scales. For production work requiring reliable output, **Q4_K_M + imatrix is the pragmatic sweet spot** — excellent quality with robust compatibility and minimal constraints.
|
||||
|
||||
## Non-technical model anaysis and rankings
|
||||
|
||||
**NOTE:** This analysis does not include the HIFI models.
|
||||
|
||||
I have run each of these models across 6 questions, and ranked them all based on the quality of the anwsers.
|
||||
**Qwen3-0.6B-f16:Q5_K_M** is the best model across all question types, but if you want to play it safe with a higher precision model, then you could consider using **Qwen3-0.6B-f16:Q8_0**.
|
||||
|
||||
You can read the results here: [Qwen3-0.6b-f16-analysis.md](Qwen3-0.6b-f16-analysis.md)
|
||||
|
||||
If you find this useful, please give the project a ❤️ like.
|
||||
|
||||
## Non-HIFI recommentation table based on output
|
||||
|
||||
| Level | Speed | Size | Recommendation |
|
||||
|-----------|-----------|------------|--------------------------------------------------------------------|
|
||||
| Q2_K | ⚡ Fastest | 347 MB | 🚨 **DO NOT USE.** Could not provide an answer to any question. |
|
||||
| Q3_K_S | ⚡ Fast | 390 MB | Not recommended, did not appear in any top 3 results. |
|
||||
| Q3_K_M | ⚡ Fast | 414 MB | First place in the bat & ball question, no other top 3 appearances.|
|
||||
| Q4_K_S | 🚀 Fast | 471 MB | A good option for technical, low-temperature questions. |
|
||||
| Q4_K_M | 🚀 Fast | 484 MB | Showed up in a few results, but not recommended. |
|
||||
| 🥈 Q5_K_S | 🐢 Medium | 544 MB | 🥈 A very close second place. Good for all query types. |
|
||||
| 🥇 Q5_K_M | 🐢 Medium | 551 MB | 🥇 **Best overall model.** Highly recommended for all query types. |
|
||||
| Q6_K | 🐌 Slow | 623 MB | Showed up in a few results, but not recommended. |
|
||||
| 🥉 Q8_0 | 🐌 Slow | 805 MB | 🥉 Very good for non-technical, creative-style questions. |
|
||||
|
||||
## Build notes
|
||||
|
||||
You can read the guide for building llama.cpp here: [HIFI_BUILD_GUIDE.md](https://github.com/geoffmunn/llama.cpp/blob/master/HIFI_BUILD_GUIDE.md).
|
||||
|
||||
The HIFI quantization also used a massive 9343 chunk imatrix file for extra precision. You can re-use it here: [Qwen3-0.6B-f16-imatrix-9343-generic.gguf](https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/blob/main/Qwen3-0.6B-f16-imatrix-9343-generic.gguf)
|
||||
|
||||
The imatrix was created as a generic mix of Wikipedia, mathmatics, and coding examples.
|
||||
|
||||
### Source code
|
||||
|
||||
You can use the HIFI GitHub repository to build it from source if you're interested: [https://github.com/geoffmunn/llama.cpp](https://github.com/geoffmunn/llama.cpp).
|
||||
|
||||
Build notes: [HIFI_BUILD_GUIDE.md](https://github.com/geoffmunn/llama.cpp/blob/master/HIFI_BUILD_GUIDE.md)
|
||||
|
||||
Improvements and feedback are welcome.
|
||||
|
||||
## Usage
|
||||
|
||||
Load this model using:
|
||||
- [OpenWebUI](https://openwebui.com) – self-hosted AI interface with RAG & tools
|
||||
- [LM Studio](https://lmstudio.ai) – desktop app with GPU support and chat templates
|
||||
- [GPT4All](https://gpt4all.io) – private, local AI chatbot (offline-first)
|
||||
- Or directly via `llama.cpp`
|
||||
|
||||
Each quantized model includes its own `README.md` and shares a common `MODELFILE` for optimal configuration.
|
||||
|
||||
Importing directly into Ollama should work, but you might encounter this error: `Error: invalid character '<' looking for beginning of value`.
|
||||
In this case try these steps:
|
||||
|
||||
1. `wget https://huggingface.co/geoffmunn/Qwen3-0.6B-f16/resolve/main/Qwen3-0.6B-f16%3AQ3_K_M.gguf` (replace the quantised version with the one you want)
|
||||
2. `nano Modelfile` and enter these details (again, replacing Q3_K_M with the version you want):
|
||||
```text
|
||||
FROM ./Qwen3-0.6B-f16:Q3_K_M.gguf
|
||||
|
||||
# Chat template using ChatML (used by Qwen)
|
||||
SYSTEM You are a helpful assistant
|
||||
|
||||
TEMPLATE "{{ if .System }}<|im_start|>system
|
||||
{{ .System }}<|im_end|>{{ end }}<|im_start|>user
|
||||
{{ .Prompt }}<|im_end|>
|
||||
<|im_start|>assistant
|
||||
"
|
||||
PARAMETER stop <|im_start|>
|
||||
PARAMETER stop <|im_end|>
|
||||
|
||||
# Default sampling
|
||||
PARAMETER temperature 0.6
|
||||
PARAMETER top_p 0.95
|
||||
PARAMETER top_k 20
|
||||
PARAMETER min_p 0.0
|
||||
PARAMETER repeat_penalty 1.1
|
||||
PARAMETER num_ctx 4096
|
||||
```
|
||||
|
||||
The `num_ctx` value has been dropped to increase speed significantly.
|
||||
|
||||
3. Then run this command: `ollama create Qwen3-0.6B-f16:Q3_K_M -f Modelfile`
|
||||
|
||||
You will now see "Qwen3-0.6B-f16:Q3_K_M" in your Ollama model list.
|
||||
|
||||
These import steps are also useful if you want to customise the default parameters or system prompt.
|
||||
|
||||
## Author
|
||||
|
||||
👤 Geoff Munn (@geoffmunn)
|
||||
🔗 [Hugging Face Profile](https://huggingface.co/geoffmunn)
|
||||
|
||||
## Disclaimer
|
||||
|
||||
This is a community conversion for local inference. Not affiliated with Alibaba Cloud or the Qwen team.
|
||||
33333
mixed-imatrix-dataset.txt
Normal file
33333
mixed-imatrix-dataset.txt
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user