初始化项目,由ModelHub XC社区提供模型
Model: calcuis/deepseek-r1 Source: Original Platform
This commit is contained in:
50
.gitattributes
vendored
Normal file
50
.gitattributes
vendored
Normal file
@@ -0,0 +1,50 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q2_k.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q3_k_s.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q3_k_l.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q3_k_m.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q4_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q4_1.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q4_k_m.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q4_k_s.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-f16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q5_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q5_1.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q5_k_m.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q5_k_s.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
DeepSeek-R1-q6_k.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
3
DeepSeek-R1-f16.gguf
Normal file
3
DeepSeek-R1-f16.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4e4e7cf0946ab9d63a41316c115c9bd9d68058c1e4b265fc51c387159738088c
|
||||
size 16068893408
|
||||
3
DeepSeek-R1-q2_k.gguf
Normal file
3
DeepSeek-R1-q2_k.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:56107492e36165ecc625c1b899d74397e662d04d2e6f2a95906608188930c598
|
||||
size 3179133664
|
||||
3
DeepSeek-R1-q3_k_l.gguf
Normal file
3
DeepSeek-R1-q3_k_l.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4fb9a49ad298ef79f97315a3d948c3621ae3ed6ac0ee066b5390aa2c39b69cd9
|
||||
size 4321958624
|
||||
3
DeepSeek-R1-q3_k_m.gguf
Normal file
3
DeepSeek-R1-q3_k_m.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4ad15feb7b44505f46ea7d949e6b845d971455eee0b7f1ea2e91139485b1e4c9
|
||||
size 4018920160
|
||||
3
DeepSeek-R1-q3_k_s.gguf
Normal file
3
DeepSeek-R1-q3_k_s.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:586e1d5c6890713fd9cad60eee5404f50bb0c708099932b537165f87fa379377
|
||||
size 3664501472
|
||||
3
DeepSeek-R1-q4_0.gguf
Normal file
3
DeepSeek-R1-q4_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:a9777539d6fd5ed814a04e0278f326bceac92c9b49fc69d293ae0e5a8585728d
|
||||
size 4661213920
|
||||
3
DeepSeek-R1-q4_1.gguf
Normal file
3
DeepSeek-R1-q4_1.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:93da88387ed29985b5def568bcb0700f824ca448bf2e4b47b96ba08e3104bfa8
|
||||
size 5130255072
|
||||
3
DeepSeek-R1-q4_k_m.gguf
Normal file
3
DeepSeek-R1-q4_k_m.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f8eba201522ab44b79bc54166126bfaf836111ff4cbf2d13c59c3b57da10573b
|
||||
size 4920736480
|
||||
3
DeepSeek-R1-q4_k_s.gguf
Normal file
3
DeepSeek-R1-q4_k_s.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:9dc93d5352d65c892be46b9c6011523b71e234197a7cf8e355718c1ad2c5797e
|
||||
size 4692671200
|
||||
3
DeepSeek-R1-q5_0.gguf
Normal file
3
DeepSeek-R1-q5_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:debe9899d69b49538e9f746479451025f130002beb5f1aa702fdf21d6b47d8d9
|
||||
size 5599296224
|
||||
3
DeepSeek-R1-q5_1.gguf
Normal file
3
DeepSeek-R1-q5_1.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:1ac83b73ebbb9881ecc90957d238b7d7c1ad0830b77c95154ce3b1cb65a63372
|
||||
size 6068337376
|
||||
3
DeepSeek-R1-q5_k_m.gguf
Normal file
3
DeepSeek-R1-q5_k_m.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:7cd9eaa42aa09326ac9af59563421ea89bade3922ab6c1e2ef680e1f99d3721a
|
||||
size 5732989664
|
||||
3
DeepSeek-R1-q5_k_s.gguf
Normal file
3
DeepSeek-R1-q5_k_s.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c7cb26a83dc4a0fd9aa01fb2106c01979212d7299510d1897425972a531f441f
|
||||
size 5599296224
|
||||
3
DeepSeek-R1-q6_k.gguf
Normal file
3
DeepSeek-R1-q6_k.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:6277b5d729562f2e01436c81b919b5b69084022560c77c98d72a5d6fc215f13e
|
||||
size 6596008672
|
||||
3
DeepSeek-R1-q8_0.gguf
Normal file
3
DeepSeek-R1-q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8c6e3924d662d3f24a96b228a5c317510c27e91c587e71e78877ed18a875ec82
|
||||
size 8540773088
|
||||
76
README.md
Normal file
76
README.md
Normal file
@@ -0,0 +1,76 @@
|
||||
---
|
||||
license: mit
|
||||
language:
|
||||
- en
|
||||
base_model:
|
||||
- deepseek-ai/DeepSeek-R1
|
||||
pipeline_tag: text-generation
|
||||
tags:
|
||||
- deepseek-r1
|
||||
- gguf-connector
|
||||
---
|
||||
|
||||
# GGUF quantized version of **deepseek-r1**
|
||||
|
||||
### review
|
||||
- no more error loading message: "unknown pre-tokenizer type: deepseek-r1-qwen"
|
||||
- works fine for llama architecture
|
||||
|
||||
### run the model
|
||||
use any gguf connector to interact with gguf file(s), i.e., [connector](https://pypi.org/project/gguf-connector/)
|
||||
|
||||
### reference
|
||||
- base model: deepseek-ai/[DeepSeek-R1](https://huggingface.co/deepseek-ai/DeepSeek-R1)
|
||||
- tool used for quantization: [cutter](https://pypi.org/project/gguf-cutter)
|
||||
|
||||
### citation
|
||||
[DeepSeek-R1](https://arxiv.org/pdf/2501.12948)
|
||||
|
||||
### appendices: model evaluation (written by deekseek-ai)
|
||||
|
||||
#### deepseek-r1-evaluation
|
||||
for all our (here refer to deekseek-ai) models, the maximum generation length is set to 32,768 tokens; for benchmarks requiring sampling, we use a temperature of $0.6$, a top-p value of $0.95$, and generate 64 responses per query to estimate pass@1.
|
||||
|
||||
| Category | Benchmark (Metric) | Claude-3.5-Sonnet-1022 | GPT-4o 0513 | DeepSeek V3 | OpenAI o1-mini | OpenAI o1-1217 | DeepSeek R1 |
|
||||
|----------|-------------------|----------------------|------------|--------------|----------------|------------|--------------|
|
||||
| | Architecture | - | - | MoE | - | - | MoE |
|
||||
| | # Activated Params | - | - | 37B | - | - | 37B |
|
||||
| | # Total Params | - | - | 671B | - | - | 671B |
|
||||
| English | MMLU (Pass@1) | 88.3 | 87.2 | 88.5 | 85.2 | **91.8** | 90.8 |
|
||||
| | MMLU-Redux (EM) | 88.9 | 88.0 | 89.1 | 86.7 | - | **92.9** |
|
||||
| | MMLU-Pro (EM) | 78.0 | 72.6 | 75.9 | 80.3 | - | **84.0** |
|
||||
| | DROP (3-shot F1) | 88.3 | 83.7 | 91.6 | 83.9 | 90.2 | **92.2** |
|
||||
| | IF-Eval (Prompt Strict) | **86.5** | 84.3 | 86.1 | 84.8 | - | 83.3 |
|
||||
| | GPQA-Diamond (Pass@1) | 65.0 | 49.9 | 59.1 | 60.0 | **75.7** | 71.5 |
|
||||
| | SimpleQA (Correct) | 28.4 | 38.2 | 24.9 | 7.0 | **47.0** | 30.1 |
|
||||
| | FRAMES (Acc.) | 72.5 | 80.5 | 73.3 | 76.9 | - | **82.5** |
|
||||
| | AlpacaEval2.0 (LC-winrate) | 52.0 | 51.1 | 70.0 | 57.8 | - | **87.6** |
|
||||
| | ArenaHard (GPT-4-1106) | 85.2 | 80.4 | 85.5 | 92.0 | - | **92.3** |
|
||||
| Code | LiveCodeBench (Pass@1-COT) | 33.8 | 34.2 | - | 53.8 | 63.4 | **65.9** |
|
||||
| | Codeforces (Percentile) | 20.3 | 23.6 | 58.7 | 93.4 | **96.6** | 96.3 |
|
||||
| | Codeforces (Rating) | 717 | 759 | 1134 | 1820 | **2061** | 2029 |
|
||||
| | SWE Verified (Resolved) | **50.8** | 38.8 | 42.0 | 41.6 | 48.9 | 49.2 |
|
||||
| | Aider-Polyglot (Acc.) | 45.3 | 16.0 | 49.6 | 32.9 | **61.7** | 53.3 |
|
||||
| Math | AIME 2024 (Pass@1) | 16.0 | 9.3 | 39.2 | 63.6 | 79.2 | **79.8** |
|
||||
| | MATH-500 (Pass@1) | 78.3 | 74.6 | 90.2 | 90.0 | 96.4 | **97.3** |
|
||||
| | CNMO 2024 (Pass@1) | 13.1 | 10.8 | 43.2 | 67.6 | - | **78.8** |
|
||||
| Chinese | CLUEWSC (EM) | 85.4 | 87.9 | 90.9 | 89.9 | - | **92.8** |
|
||||
| | C-Eval (EM) | 76.7 | 76.0 | 86.5 | 68.9 | - | **91.8** |
|
||||
| | C-SimpleQA (Correct) | 55.4 | 58.7 | **68.0** | 40.3 | - | 63.7 |
|
||||
|
||||
#### distilled model evaluation
|
||||
|
||||
| Model | AIME 2024 pass@1 | AIME 2024 cons@64 | MATH-500 pass@1 | GPQA Diamond pass@1 | LiveCodeBench pass@1 | CodeForces rating |
|
||||
|------------------------------------------|------------------|-------------------|-----------------|----------------------|----------------------|-------------------|
|
||||
| GPT-4o-0513 | 9.3 | 13.4 | 74.6 | 49.9 | 32.9 | 759 |
|
||||
| Claude-3.5-Sonnet-1022 | 16.0 | 26.7 | 78.3 | 65.0 | 38.9 | 717 |
|
||||
| o1-mini | 63.6 | 80.0 | 90.0 | 60.0 | 53.8 | **1820** |
|
||||
| QwQ-32B-Preview | 44.0 | 60.0 | 90.6 | 54.5 | 41.9 | 1316 |
|
||||
| DeepSeek-R1-Distill-Qwen-1.5B | 28.9 | 52.7 | 83.9 | 33.8 | 16.9 | 954 |
|
||||
| DeepSeek-R1-Distill-Qwen-7B | 55.5 | 83.3 | 92.8 | 49.1 | 37.6 | 1189 |
|
||||
| DeepSeek-R1-Distill-Qwen-14B | 69.7 | 80.0 | 93.9 | 59.1 | 53.1 | 1481 |
|
||||
| DeepSeek-R1-Distill-Qwen-32B | **72.6** | 83.3 | 94.3 | 62.1 | 57.2 | 1691 |
|
||||
| DeepSeek-R1-Distill-Llama-8B | 50.4 | 80.0 | 89.1 | 49.0 | 39.6 | 1205 |
|
||||
| DeepSeek-R1-Distill-Llama-70B | 70.0 | **86.7** | **94.5** | **65.2** | **57.5** | 1633 |
|
||||
|
||||
\* these two tables are directly quoted from deepseek-ai
|
||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
||||
{"framework": "pytorch", "task": "text-generation", "allow_remote": true}
|
||||
Reference in New Issue
Block a user