初始化项目,由ModelHub XC社区提供模型

Model: ThomasBaruzier/Qwen2.5-3B-Instruct-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-13 04:38:18 +08:00
commit 3564055805
36 changed files with 2839 additions and 0 deletions

67
.gitattributes vendored Normal file
View File

@@ -0,0 +1,67 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
imatrix.dat filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ2_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ2_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q2_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ3_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ3_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q5_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q5_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_0_4_4.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_0_4_8.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-3B-Instruct-Q4_0_8_8.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:25908e409af8c952a17c8fda44b90e699565a90436431574f44c25d140aa5032
size 6178317216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3de5033270911c8bd2f1d9691d361040473b80cb71d9a5a27af6fd40343bd073
size 850027648

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:483fc79cf22b51a1726d5dd90d9fafa161cc4bb005e432a5020f36359f78f086
size 791094400

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7040893e04687d722d2d73f0c6ee9509b1aa2e29d21098da05b24691d74b1073
size 1140515968

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:918ce7280dd943f0e8f01e7c0dc6f83089a9e0e2b762cb1ff8003ac6d06ee9f1
size 1061938304

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:db1cc61b3e3a3e1071d496497915c137f84860b9232ebd8661f3a56d393fd54c
size 1031545984

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3699f107a920bd27b30e9ba0f42502dc2da431a2710862922e90390c77f11ee0
size 948249728

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c38e8000dba7cf4182474cdc4ec53e5ad5d6b8afa161722338138d50598bb4f9
size 1488895104

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:11f528f38f8fab2808f2ce881faafafb312935ba0bfe6701f32973cc85accce5
size 1456864384

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:30fe93fd43692d1f36485a6e6d58b79664f987dccacb9db3552f932ba1b59ab9
size 1391836288

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1fe332d17832332828b38e5f7e4ffb6c807271a2cedae9d76f87d4183d5aec1e
size 1282827392

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:715803f2f2b7bf7742103b3cbbb18f2812446aa3554c9e9192ea74ec81c78009
size 1825209472

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:28ce4a4bcc99a012f85f017cc7dbdddc66bda8d6e555088ec310e752e46b764b
size 1739095168

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8c2003231bf3852c1ec83b9bdc51ba3efa052361b0a4b0615c1b81ed657ac87c
size 1274756224

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:45f5bd70a22aaaaf78e6547bf4beded8934ca1d0f9a6da84606c92b355d3f283
size 1198128256

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6d7074ac93fa80c19e8979ec353af382ef30cd0604247bf1da139bde9f584694
size 1707392128

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cb6ced6cba53a791c2f88d3d06264cddaca9a89afaf4c9751bdfd2e4de819a59
size 1590475904

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5eed1ecd9039a5ba955648d801478b8ded5e7316c8a2ecdd891c9241e5af265d
size 1454357632

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5421020d17cb7a027faf466ca2fefbfb61dce360ece3244be7aae2d6407ee7f8
size 1828486272

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6c136bd63ce2bc66b9940e18cf263b4d8ba06a3e4f7d9c26b2c99ef180b2d7e8
size 1822849952

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:148d087902ef4214b4278b3e917a6065bd61bce3d0fdf1435b77e65807d0b53a
size 1822849952

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cbf1280d75e323eb6117264ff87cf794ce431a15d617459fb3a7a40d3005543e
size 1822849952

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:eae769cb559e8d81b5af1fa402465579f211f8d7755740565c655298b45f95a4
size 1996258432

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1baf35a45b42ea3f5f7aee28bd98f005a215bc93cea4a6272d774f91bfe6e97c
size 1929903232

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d04175567d06bfa9f36bafbcfa273b2e7b6175c5129339219ccb5b7909c13810
size 1834384512

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:eed64928087ab631d645b0d3823f256f69a432fb2af91594208b88646f257544
size 2175302784

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9ef9c274b56d9d75380c049cc87abaae36408a757e548aba30aa134e80852a19
size 2343074944

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:48a87389e978f19da2b1d3c901323e12db2ea03e6659d7d0ecf9dcbb11fcaa71
size 2224815232

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9ac6144e74dad8431ac0e9253011600a5bff9259c5b2d9ec0ceeadb4997a355a
size 2169666688

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:978a3448d884a366bf9a1ab9477649c5f92278e31795914ae1b6769f405e7d43
size 2538159232

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5441226e070ec8d961a838a27c5476688fff210a2d5d64566a867d77a4b7d93c
size 3285476480

171
README.md Normal file
View File

@@ -0,0 +1,171 @@
---
license: other
license_name: qwen-research
license_link: https://huggingface.co/Qwen/Qwen2.5-3B-Instruct/blob/main/LICENSE
language:
- zho
- eng
- fra
- spa
- por
- deu
- ita
- rus
- jpn
- kor
- vie
- tha
- ara
pipeline_tag: text-generation
base_model: Qwen/Qwen2.5-3B
tags:
- chat
---
<hr>
# Llama.cpp imatrix quantizations of Qwen/Qwen2.5-3B-Instruct
<img src="https://cdn-uploads.huggingface.co/production/uploads/646410e04bf9122922289dc7/gDUbZOu1ND0j-th4Q6tep.jpeg" alt="qwen" width="60%"/>
Using llama.cpp commit [eca0fab](https://github.com/ggerganov/llama.cpp/commit/eca0fab) for quantization.
Original model: [Qwen/Qwen2.5-3B-Instruct](https://huggingface.co/Qwen/Qwen2.5-3B-Instruct)
All quants were made using the imatrix option and Bartowski's [calibration file](https://gist.github.com/bartowski1182/eb213dccb3571f863da82e99418f81e8).
<hr>
# Perplexity table (the lower the better)
| Quant | Size (MB) | PPL | Size (%) | Accuracy (%) | PPL error rate |
| ------- | --------- | -------- | -------- | ------------ | -------------- |
| IQ1_S | 755 | 112.0612 | 12.81 | 8.02 | 0.97138 |
| IQ1_M | 811 | 42.7456 | 13.76 | 21.03 | 0.34718 |
| IQ2_XXS | 905 | 25.2117 | 15.36 | 35.65 | 0.20222 |
| IQ2_XS | 984 | 15.9149 | 16.7 | 56.48 | 0.11965 |
| IQ2_S | 1013 | 14.5975 | 17.19 | 61.58 | 0.1082 |
| IQ2_M | 1088 | 12.8779 | 18.46 | 69.8 | 0.09436 |
| Q2_K_S | 1143 | 13.0878 | 19.4 | 68.68 | 0.09636 |
| Q2_K | 1216 | 11.8001 | 20.63 | 76.18 | 0.08674 |
| IQ3_XXS | 1224 | 10.6049 | 20.77 | 84.76 | 0.07572 |
| IQ3_XS | 1328 | 10.0306 | 22.54 | 89.61 | 0.06975 |
| Q3_K_S | 1387 | 15.5457 | 23.54 | 57.82 | 0.11941 |
| IQ3_S | 1390 | 9.9591 | 23.59 | 90.26 | 0.06984 |
| IQ3_M | 1420 | 9.9957 | 24.1 | 89.93 | 0.06962 |
| Q3_K_M | 1517 | 14.0989 | 25.74 | 63.76 | 0.10568 |
| Q3_K_L | 1629 | 13.8579 | 27.64 | 64.86 | 0.10372 |
| IQ4_XS | 1659 | 9.2935 | 28.15 | 96.72 | 0.06517 |
| IQ4_NL | 1741 | 9.2824 | 29.54 | 96.84 | 0.06503 |
| Q4_0 | 1744 | 9.485 | 29.59 | 94.77 | 0.06626 |
| Q4_K_S | 1750 | 9.2573 | 29.7 | 97.1 | 0.06485 |
| Q4_K_M | 1841 | 9.2305 | 31.24 | 97.38 | 0.06475 |
| Q4_1 | 1904 | 9.2746 | 32.31 | 96.92 | 0.06512 |
| Q5_K_S | 2070 | 9.1338 | 35.13 | 98.41 | 0.06402 |
| Q5_0 | 2075 | 9.1513 | 35.21 | 98.22 | 0.06413 |
| Q5_K_M | 2122 | 9.1339 | 36.01 | 98.41 | 0.06407 |
| Q5_1 | 2235 | 9.1231 | 37.93 | 98.53 | 0.06386 |
| Q6_K | 2421 | 9.069 | 41.08 | 99.12 | 0.06342 |
| Q8_0 | 3134 | 9.0114 | 53.18 | 99.75 | 0.06285 |
| F16 | 5893 | 8.9888 | 100 | 100 | 0.06268 |
<hr>
# Qwen2.5-3B-Instruct
## Introduction
Qwen2.5 is the latest series of Qwen large language models. For Qwen2.5, we release a number of base language models and instruction-tuned language models ranging from 0.5 to 72 billion parameters. Qwen2.5 brings the following improvements upon Qwen2:
- Significantly **more knowledge** and has greatly improved capabilities in **coding** and **mathematics**, thanks to our specialized expert models in these domains.
- Significant improvements in **instruction following**, **generating long texts** (over 8K tokens), **understanding structured data** (e.g, tables), and **generating structured outputs** especially JSON. **More resilient to the diversity of system prompts**, enhancing role-play implementation and condition-setting for chatbots.
- **Long-context Support** up to 128K tokens and can generate up to 8K tokens.
- **Multilingual support** for over 29 languages, including Chinese, English, French, Spanish, Portuguese, German, Italian, Russian, Japanese, Korean, Vietnamese, Thai, Arabic, and more.
**This repo contains the instruction-tuned 3B Qwen2.5 model**, which has the following features:
- Type: Causal Language Models
- Training Stage: Pretraining & Post-training
- Architecture: transformers with RoPE, SwiGLU, RMSNorm, Attention QKV bias and tied word embeddings
- Number of Parameters: 3.09B
- Number of Paramaters (Non-Embedding): 2.77B
- Number of Layers: 36
- Number of Attention Heads (GQA): 16 for Q and 2 for KV
- Context Length: Full 32,768 tokens and generation 8192 tokens
For more details, please refer to our [blog](https://qwenlm.github.io/blog/qwen2.5/), [GitHub](https://github.com/QwenLM/Qwen2.5), and [Documentation](https://qwen.readthedocs.io/en/latest/).
## Requirements
The code of Qwen2.5 has been in the latest Hugging face `transformers` and we advise you to use the latest version of `transformers`.
With `transformers<4.37.0`, you will encounter the following error:
```
KeyError: 'qwen2'
```
## Quickstart
Here provides a code snippet with `apply_chat_template` to show you how to load the tokenizer and model and how to generate contents.
```python
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "Qwen/Qwen2.5-3B-Instruct"
model = AutoModelForCausalLM.from_pretrained(
model_name,
torch_dtype="auto",
device_map="auto"
)
tokenizer = AutoTokenizer.from_pretrained(model_name)
prompt = "Give me a short introduction to large language model."
messages = [
{"role": "system", "content": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."},
{"role": "user", "content": prompt}
]
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
model_inputs = tokenizer([text], return_tensors="pt").to(model.device)
generated_ids = model.generate(
**model_inputs,
max_new_tokens=512
)
generated_ids = [
output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids)
]
response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
```
## Evaluation & Performance
Detailed evaluation results are reported in this [📑 blog](https://qwenlm.github.io/blog/qwen2.5/).
For requirements on GPU memory and the respective throughput, see results [here](https://qwen.readthedocs.io/en/latest/benchmark/speed_benchmark.html).
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen2.5,
title = {Qwen2.5: A Party of Foundation Models},
url = {https://qwenlm.github.io/blog/qwen2.5/},
author = {Qwen Team},
month = {September},
year = {2024}
}
@article{qwen2,
title={Qwen2 Technical Report},
author={An Yang and Baosong Yang and Binyuan Hui and Bo Zheng and Bowen Yu and Chang Zhou and Chengpeng Li and Chengyuan Li and Dayiheng Liu and Fei Huang and Guanting Dong and Haoran Wei and Huan Lin and Jialong Tang and Jialin Wang and Jian Yang and Jianhong Tu and Jianwei Zhang and Jianxin Ma and Jin Xu and Jingren Zhou and Jinze Bai and Jinzheng He and Junyang Lin and Kai Dang and Keming Lu and Keqin Chen and Kexin Yang and Mei Li and Mingfeng Xue and Na Ni and Pei Zhang and Peng Wang and Ru Peng and Rui Men and Ruize Gao and Runji Lin and Shijie Wang and Shuai Bai and Sinan Tan and Tianhang Zhu and Tianhao Li and Tianyu Liu and Wenbin Ge and Xiaodong Deng and Xiaohuan Zhou and Xingzhang Ren and Xinyu Zhang and Xipin Wei and Xuancheng Ren and Yang Fan and Yang Yao and Yichang Zhang and Yu Wan and Yunfei Chu and Yuqiong Liu and Zeyu Cui and Zhenru Zhang and Zhihao Fan},
journal={arXiv preprint arXiv:2407.10671},
year={2024}
}
```

2482
calibration_datav3.txt Normal file

File diff suppressed because one or more lines are too long

3
imatrix.dat Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d0e68abe89b59ea10b13d7e0a92c160464c46800d717ca79eefcd1edb18ea786
size 3362981

23
perplexity.md Normal file
View File

@@ -0,0 +1,23 @@
Qwen2.5-3B-Instruct
Quant Size (MB) PPL Size (%) Accuracy (%) PPL error rate
IQ1_S 755 112.0612 0.97138
IQ1_M 811 42.7456 0.34718
IQ2_XXS 905 25.2117 0.20222
IQ2_XS 984 15.9149 0.11965
IQ2_S 1013 14.5975 0.10820
IQ2_M 1088 12.8779 0.09436
Q2_K_S 1143 13.0878 0.09636
Q2_K 1216 11.8001 0.08674
IQ3_XXS 1224 10.6049 0.07572
IQ3_XS 1328 10.0306 0.06975
Q3_K_S 1387 15.5457 0.11941
IQ3_S 1390 9.9591 0.06984
IQ3_M 1420 9.9957 0.06962
Q3_K_M 1517 14.0989 0.10568
Q3_K_L 1629 13.8579 0.10372
IQ4_XS 1659 9.2935 0.06517
IQ4_NL 1741 9.2824 0.06503
Q4_0 1744 9.4850 0.06626
Q4_K_S 1750 9.2573 0.06485
Q4_K_M 1841 9.2305 0.06475