初始化项目,由ModelHub XC社区提供模型

Model: ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-25 19:30:17 +08:00
commit a9beeff36f
32 changed files with 340 additions and 0 deletions

64
.gitattributes vendored Normal file
View File

@@ -0,0 +1,64 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
imatrix.dat filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ2_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q2_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ2_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ3_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ3_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ3_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q5_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q5_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen2.5-Coder-1.5B-Instruct-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:298d758b014dffd713cda0e8d23404e81faea39446c45074ce9470f0a135799a
size 3093669440

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9b46e403370fdda83dba02c9c067252cd3016f08cf36672085958a4e34b9fdd7
size 464461696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7573ae9af61a7e5390146e43a6f226e81dc725d4926f5e0230e250e2819aaa73
size 436528000

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:26dc3613898ef8b747bad674f3af641592683dd04497705b74280916ad428ea2
size 601055104

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ee7b4fde343dd48b530dc5dc13f88dbe6b9b1186b46628727581e35d0b00c4ee
size 563810176

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2a27b948c060d64a0acb38025691223a21e87f734be318edf3f515a021609daf
size 550327168

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:60f5e161ac07fcf10c95bea13c58dea9a6bc69aacf9eff687e512cf71deaa4d2
size 511017856

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a69374037424cd1005b23f5ac6da25738043b8560c06894d3a6fa609bc79bd1c
size 776664448

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:471637540f6fb0d15370ebd6c1eb50cd73bd0202af14440ac9f82429d271fd73
size 762407296

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:06e8a298ac07e0067daedef5ff1fa288298610bfbb17161c684c3c2ab3fb1827
size 731699584

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9b1a71186200e9bbb7bbcaf3236c1f9131e36f3256992cf73a2afe06f2169f2d
size 668792704

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3145e719538378066f2261cd79438d013c9f63369755d9bbe74fce03dafa16f2
size 936331648

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:71d3821a4fa1550e1aec3bbed20b22c981a11f91c70682fcbeed64dbc91b11d1
size 895732096

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e094b3bb56a88ae4a36eea2c4d5de59f9090af878d1656d8d6249fb3e4abe557
size 676305280

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7e47eede3dc15c593b98f2b8fd2b491fe332f9df684e1b74f202bc927bda7eef
size 640135552

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b2cf85a4b9ff715edcaa3e844ca4ff56d02a43f99904dbc1ad137ec2c88f55b9
size 880163200

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:de3152280268c0e241b6c5434c1d1367112f953db514e047085e1319471222f9
size 824179072

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d40358a964cd2f9f234f2aaa0ed24c653ed838a8f40d204684c85b63128b2a4a
size 760945024

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0fa21b833bbfc11e790e5adaf069e49ab5532decad2faded3a0cae7eb1744ec7
size 937535872

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a9990caf52473d20c00f331e00e128953158c8cd2d03e243e9fb63c527a15550
size 1016842624

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:47fd9fefafd5cdcfcb401898aeed7b1fec073616ce2b7a690a04114d16e51781
size 986048896

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:85a89c4990cd8b164e12689d98c3e89ee94333066d747feedab1c1233411eba5
size 940312960

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4894b08c56eb8cdac88f663e028b6276392547a6441e826b857672f51943a770
size 1101310336

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4ff782cdabdfb03d98beab5c9282afae7d3052d792fa673bee69b544b36f4c94
size 1180617088

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d24b4105b5b66af7055d42679f229c3c725fdeddbad591fd1a6a4bed625028ca
size 1125050752

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:dfee6171149dfdcd5ac7a73efb638854b30752e0a1e049c3692dbe88eb8b7395
size 1098729856

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:046d20272cf563b4365ca7187cb65f2e187b792e151717bb0735eb963605547b
size 1272740224

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3e9ad26171a2dad42f3556318cb0fc0d9e77cfca6646f12e5f1dea826f612c93
size 1646573440

159
README.md Normal file
View File

@@ -0,0 +1,159 @@
---
license: apache-2.0
license_link: https://huggingface.co/Qwen/Qwen2.5-Coder-1.5B-Instruct/blob/main/LICENSE
language:
- en
base_model:
- Qwen/Qwen2.5-Coder-1.5B
pipeline_tag: text-generation
library_name: transformers
tags:
- code
- codeqwen
- chat
- qwen
- qwen-coder
---
<br><img src="https://cdn-uploads.huggingface.co/production/uploads/646410e04bf9122922289dc7/wHRc2ZKSErOs92ZcdKQrY.webp" width="720"><br>
# Llama.cpp imatrix quantizations of [Qwen/Qwen2.5-Coder-1.5B-Instruct](https://huggingface.co/Qwen/Qwen2.5-Coder-1.5B-Instruct)
Using llama.cpp commit [3ad5451](https://github.com/ggerganov/llama.cpp/commit/3ad5451) for quantization.
All quants were made using the imatrix option and Bartowski's [calibration file](https://gist.github.com/bartowski1182/eb213dccb3571f863da82e99418f81e8).
<hr>
# Perplexity table (the lower the better)
| Quant | Size (MB) | PPL | Size (%) | Accuracy (%) | PPL error rate |
| ------------------------------------------------------------------------------------------------------------------------------------ | --------- | ------- | -------- | ------------ | -------------- |
| [IQ1_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ1_S.gguf) | 416 | 65.5925 | 14.10 | 18.53 | 1.27 |
| [IQ1_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ1_M.gguf) | 442 | 37.5490 | 14.98 | 32.37 | 0.67 |
| [IQ2_XXS](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ2_XXS.gguf) | 487 | 28.2797 | 16.51 | 42.98 | 0.52 |
| [IQ2_XS](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ2_XS.gguf) | 524 | 19.7773 | 17.76 | 61.45 | 0.34 |
| [IQ2_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ2_S.gguf) | 537 | 18.3488 | 18.20 | 66.24 | 0.31 |
| [IQ2_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ2_M.gguf) | 573 | 16.4023 | 19.42 | 74.10 | 0.28 |
| [Q2_K_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q2_K_S.gguf) | 610 | 16.9464 | 20.68 | 71.72 | 0.28 |
| [IQ3_XXS](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ3_XXS.gguf) | 637 | 14.0319 | 21.59 | 86.61 | 0.23 |
| [Q2_K](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q2_K.gguf) | 644 | 15.4455 | 21.83 | 78.69 | 0.25 |
| [IQ3_XS](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ3_XS.gguf) | 697 | 13.2295 | 23.63 | 91.87 | 0.21 |
| [Q3_K_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q3_K_S.gguf) | 725 | 13.6908 | 24.58 | 88.77 | 0.22 |
| [IQ3_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ3_S.gguf) | 727 | 13.0527 | 24.64 | 93.11 | 0.21 |
| [IQ3_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ3_M.gguf) | 740 | 12.9347 | 25.08 | 93.96 | 0.21 |
| [Q3_K_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q3_K_M.gguf) | 785 | 13.1079 | 26.61 | 92.72 | 0.21 |
| [Q3_K_L](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q3_K_L.gguf) | 839 | 12.9958 | 28.44 | 93.52 | 0.21 |
| [IQ4_XS](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ4_XS.gguf) | 854 | 12.5238 | 28.95 | 97.04 | 0.20 |
| [IQ4_NL](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-IQ4_NL.gguf) | 892 | 12.5165 | 30.24 | 97.10 | 0.20 |
| [Q4_0](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q4_0.gguf) | 894 | 12.6002 | 30.31 | 96.46 | 0.20 |
| [Q4_K_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q4_K_S.gguf) | 896 | 12.4550 | 30.37 | 97.58 | 0.20 |
| [Q4_K_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q4_K_M.gguf) | 940 | 12.4048 | 31.86 | 97.97 | 0.20 |
| [Q4_1](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q4_1.gguf) | 969 | 12.4660 | 32.85 | 97.49 | 0.20 |
| [Q5_K_S](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q5_K_S.gguf) | 1047 | 12.2279 | 35.49 | 99.39 | 0.20 |
| [Q5_0](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q5_0.gguf) | 1050 | 12.2580 | 35.59 | 99.15 | 0.20 |
| [Q5_K_M](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q5_K_M.gguf) | 1072 | 12.2216 | 36.34 | 99.44 | 0.20 |
| [Q5_1](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q5_1.gguf) | 1125 | 12.2391 | 38.13 | 99.30 | 0.20 |
| [Q6_K](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q6_K.gguf) | 1213 | 12.1951 | 41.12 | 99.66 | 0.20 |
| [Q8_0](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-Q8_0.gguf) | 1570 | 12.1583 | 53.22 | 99.96 | 0.20 |
| [F16](https://huggingface.co/ThomasBaruzier/Qwen2.5-Coder-1.5B-Instruct-GGUF/blob/main/Qwen2.5-Coder-1.5B-Instruct-F16.gguf) | 2950 | 12.1537 | 100 | 100 | 0.20 |
---
# Qwen2.5-Coder-1.5B-Instruct
<a href="https://chat.qwenlm.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
## Introduction
Qwen2.5-Coder is the latest series of Code-Specific Qwen large language models (formerly known as CodeQwen). As of now, Qwen2.5-Coder has covered six mainstream model sizes, 0.5, 1.5, 3, 7, 14, 32 billion parameters, to meet the needs of different developers. Qwen2.5-Coder brings the following improvements upon CodeQwen1.5:
- Significantly improvements in **code generation**, **code reasoning** and **code fixing**. Base on the strong Qwen2.5, we scale up the training tokens into 5.5 trillion including source code, text-code grounding, Synthetic data, etc. Qwen2.5-Coder-32B has become the current state-of-the-art open-source codeLLM, with its coding abilities matching those of GPT-4o.
- A more comprehensive foundation for real-world applications such as **Code Agents**. Not only enhancing coding capabilities but also maintaining its strengths in mathematics and general competencies.
**This repo contains the instruction-tuned 1.5B Qwen2.5-Coder model**, which has the following features:
- Type: Causal Language Models
- Training Stage: Pretraining & Post-training
- Architecture: transformers with RoPE, SwiGLU, RMSNorm, Attention QKV bias and tied word embeddings
- Number of Parameters: 1.54B
- Number of Paramaters (Non-Embedding): 1.31B
- Number of Layers: 28
- Number of Attention Heads (GQA): 12 for Q and 2 for KV
- Context Length: Full 32,768 tokens
For more details, please refer to our [blog](https://qwenlm.github.io/blog/qwen2.5-coder-family/), [GitHub](https://github.com/QwenLM/Qwen2.5-Coder), [Documentation](https://qwen.readthedocs.io/en/latest/), [Arxiv](https://arxiv.org/abs/2409.12186).
## Requirements
The code of Qwen2.5-Coder has been in the latest Hugging face `transformers` and we advise you to use the latest version of `transformers`.
With `transformers<4.37.0`, you will encounter the following error:
```
KeyError: 'qwen2'
```
## Quickstart
Here provides a code snippet with `apply_chat_template` to show you how to load the tokenizer and model and how to generate contents.
```python
from transformers import AutoModelForCausalLM, AutoTokenizer
model_name = "Qwen/Qwen2.5-Coder-1.5B-Instruct"
model = AutoModelForCausalLM.from_pretrained(
model_name,
torch_dtype="auto",
device_map="auto"
)
tokenizer = AutoTokenizer.from_pretrained(model_name)
prompt = "write a quick sort algorithm."
messages = [
{"role": "system", "content": "You are Qwen, created by Alibaba Cloud. You are a helpful assistant."},
{"role": "user", "content": prompt}
]
text = tokenizer.apply_chat_template(
messages,
tokenize=False,
add_generation_prompt=True
)
model_inputs = tokenizer([text], return_tensors="pt").to(model.device)
generated_ids = model.generate(
**model_inputs,
max_new_tokens=512
)
generated_ids = [
output_ids[len(input_ids):] for input_ids, output_ids in zip(model_inputs.input_ids, generated_ids)
]
response = tokenizer.batch_decode(generated_ids, skip_special_tokens=True)[0]
```
## Evaluation & Performance
Detailed evaluation results are reported in this [📑 blog](https://qwenlm.github.io/blog/qwen2.5-coder-family/).
For requirements on GPU memory and the respective throughput, see results [here](https://qwen.readthedocs.io/en/latest/benchmark/speed_benchmark.html).
## Citation
If you find our work helpful, feel free to give us a cite.
```
@article{hui2024qwen2,
title={Qwen2. 5-Coder Technical Report},
author={Hui, Binyuan and Yang, Jian and Cui, Zeyu and Yang, Jiaxi and Liu, Dayiheng and Zhang, Lei and Liu, Tianyu and Zhang, Jiajun and Yu, Bowen and Dang, Kai and others},
journal={arXiv preprint arXiv:2409.12186},
year={2024}
}
@article{qwen2,
title={Qwen2 Technical Report},
author={An Yang and Baosong Yang and Binyuan Hui and Bo Zheng and Bowen Yu and Chang Zhou and Chengpeng Li and Chengyuan Li and Dayiheng Liu and Fei Huang and Guanting Dong and Haoran Wei and Huan Lin and Jialong Tang and Jialin Wang and Jian Yang and Jianhong Tu and Jianwei Zhang and Jianxin Ma and Jin Xu and Jingren Zhou and Jinze Bai and Jinzheng He and Junyang Lin and Kai Dang and Keming Lu and Keqin Chen and Kexin Yang and Mei Li and Mingfeng Xue and Na Ni and Pei Zhang and Peng Wang and Ru Peng and Rui Men and Ruize Gao and Runji Lin and Shijie Wang and Shuai Bai and Sinan Tan and Tianhang Zhu and Tianhao Li and Tianyu Liu and Wenbin Ge and Xiaodong Deng and Xiaohuan Zhou and Xingzhang Ren and Xinyu Zhang and Xipin Wei and Xuancheng Ren and Yang Fan and Yang Yao and Yichang Zhang and Yu Wan and Yunfei Chu and Yuqiong Liu and Zeyu Cui and Zhenru Zhang and Zhihao Fan},
journal={arXiv preprint arXiv:2407.10671},
year={2024}
}
```

3
imatrix.dat Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:887916ab6ba33a143cfdc78826fa84c96055078b78c7a6acc724cf1530566dc2
size 2042232

30
perplexity.md Normal file
View File

@@ -0,0 +1,30 @@
| Quant | Size (MB) | PPL | Size (%) | Accuracy (%) | PPL error rate |
| ------ | --------- | ------- | -------- | ------------ | -------------- |
| IQ1_S | 416 | 65.5925 | 14.10 | 18.53 | 1.27 |
| IQ1_M | 442 | 37.5490 | 14.98 | 32.37 | 0.67 |
| IQ2_XXS | 487 | 28.2797 | 16.51 | 42.98 | 0.52 |
| IQ2_XS | 524 | 19.7773 | 17.76 | 61.45 | 0.34 |
| IQ2_S | 537 | 18.3488 | 18.20 | 66.24 | 0.31 |
| IQ2_M | 573 | 16.4023 | 19.42 | 74.10 | 0.28 |
| Q2_K_S | 610 | 16.9464 | 20.68 | 71.72 | 0.28 |
| IQ3_XXS | 637 | 14.0319 | 21.59 | 86.61 | 0.23 |
| Q2_K | 644 | 15.4455 | 21.83 | 78.69 | 0.25 |
| IQ3_XS | 697 | 13.2295 | 23.63 | 91.87 | 0.21 |
| Q3_K_S | 725 | 13.6908 | 24.58 | 88.77 | 0.22 |
| IQ3_S | 727 | 13.0527 | 24.64 | 93.11 | 0.21 |
| IQ3_M | 740 | 12.9347 | 25.08 | 93.96 | 0.21 |
| Q3_K_M | 785 | 13.1079 | 26.61 | 92.72 | 0.21 |
| Q3_K_L | 839 | 12.9958 | 28.44 | 93.52 | 0.21 |
| IQ4_XS | 854 | 12.5238 | 28.95 | 97.04 | 0.20 |
| IQ4_NL | 892 | 12.5165 | 30.24 | 97.10 | 0.20 |
| Q4_0 | 894 | 12.6002 | 30.31 | 96.46 | 0.20 |
| Q4_K_S | 896 | 12.4550 | 30.37 | 97.58 | 0.20 |
| Q4_K_M | 940 | 12.4048 | 31.86 | 97.97 | 0.20 |
| Q4_1 | 969 | 12.4660 | 32.85 | 97.49 | 0.20 |
| Q5_K_S | 1047 | 12.2279 | 35.49 | 99.39 | 0.20 |
| Q5_0 | 1050 | 12.2580 | 35.59 | 99.15 | 0.20 |
| Q5_K_M | 1072 | 12.2216 | 36.34 | 99.44 | 0.20 |
| Q5_1 | 1125 | 12.2391 | 38.13 | 99.30 | 0.20 |
| Q6_K | 1213 | 12.1951 | 41.12 | 99.66 | 0.20 |
| Q8_0 | 1570 | 12.1583 | 53.22 | 99.96 | 0.20 |
| F16 | 2950 | 12.1537 | 100 | 100 | 0.20 |