commit bd6361ce2ba16ead20b12a47c427ba593123f58f Author: ModelHub XC Date: Thu Jun 18 07:16:14 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: calcuis/deepseek-r1 Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..34da72b --- /dev/null +++ b/.gitattributes @@ -0,0 +1,50 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q2_k.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q3_k_s.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q3_k_l.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q3_k_m.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q4_0.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q4_1.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q4_k_m.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q4_k_s.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-f16.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q5_0.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q5_1.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q8_0.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q5_k_m.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q5_k_s.gguf filter=lfs diff=lfs merge=lfs -text +DeepSeek-R1-q6_k.gguf filter=lfs diff=lfs merge=lfs -text diff --git a/DeepSeek-R1-f16.gguf b/DeepSeek-R1-f16.gguf new file mode 100644 index 0000000..3b0170f --- /dev/null +++ b/DeepSeek-R1-f16.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4e4e7cf0946ab9d63a41316c115c9bd9d68058c1e4b265fc51c387159738088c +size 16068893408 diff --git a/DeepSeek-R1-q2_k.gguf b/DeepSeek-R1-q2_k.gguf new file mode 100644 index 0000000..1a317f5 --- /dev/null +++ b/DeepSeek-R1-q2_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:56107492e36165ecc625c1b899d74397e662d04d2e6f2a95906608188930c598 +size 3179133664 diff --git a/DeepSeek-R1-q3_k_l.gguf b/DeepSeek-R1-q3_k_l.gguf new file mode 100644 index 0000000..f8299d6 --- /dev/null +++ b/DeepSeek-R1-q3_k_l.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4fb9a49ad298ef79f97315a3d948c3621ae3ed6ac0ee066b5390aa2c39b69cd9 +size 4321958624 diff --git a/DeepSeek-R1-q3_k_m.gguf b/DeepSeek-R1-q3_k_m.gguf new file mode 100644 index 0000000..a9ae1ac --- /dev/null +++ b/DeepSeek-R1-q3_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ad15feb7b44505f46ea7d949e6b845d971455eee0b7f1ea2e91139485b1e4c9 +size 4018920160 diff --git a/DeepSeek-R1-q3_k_s.gguf b/DeepSeek-R1-q3_k_s.gguf new file mode 100644 index 0000000..8dee9ca --- /dev/null +++ b/DeepSeek-R1-q3_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:586e1d5c6890713fd9cad60eee5404f50bb0c708099932b537165f87fa379377 +size 3664501472 diff --git a/DeepSeek-R1-q4_0.gguf b/DeepSeek-R1-q4_0.gguf new file mode 100644 index 0000000..db8dccf --- /dev/null +++ b/DeepSeek-R1-q4_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a9777539d6fd5ed814a04e0278f326bceac92c9b49fc69d293ae0e5a8585728d +size 4661213920 diff --git a/DeepSeek-R1-q4_1.gguf b/DeepSeek-R1-q4_1.gguf new file mode 100644 index 0000000..d04d444 --- /dev/null +++ b/DeepSeek-R1-q4_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93da88387ed29985b5def568bcb0700f824ca448bf2e4b47b96ba08e3104bfa8 +size 5130255072 diff --git a/DeepSeek-R1-q4_k_m.gguf b/DeepSeek-R1-q4_k_m.gguf new file mode 100644 index 0000000..a4a692e --- /dev/null +++ b/DeepSeek-R1-q4_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f8eba201522ab44b79bc54166126bfaf836111ff4cbf2d13c59c3b57da10573b +size 4920736480 diff --git a/DeepSeek-R1-q4_k_s.gguf b/DeepSeek-R1-q4_k_s.gguf new file mode 100644 index 0000000..8b277fa --- /dev/null +++ b/DeepSeek-R1-q4_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9dc93d5352d65c892be46b9c6011523b71e234197a7cf8e355718c1ad2c5797e +size 4692671200 diff --git a/DeepSeek-R1-q5_0.gguf b/DeepSeek-R1-q5_0.gguf new file mode 100644 index 0000000..bced659 --- /dev/null +++ b/DeepSeek-R1-q5_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:debe9899d69b49538e9f746479451025f130002beb5f1aa702fdf21d6b47d8d9 +size 5599296224 diff --git a/DeepSeek-R1-q5_1.gguf b/DeepSeek-R1-q5_1.gguf new file mode 100644 index 0000000..a36ac75 --- /dev/null +++ b/DeepSeek-R1-q5_1.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1ac83b73ebbb9881ecc90957d238b7d7c1ad0830b77c95154ce3b1cb65a63372 +size 6068337376 diff --git a/DeepSeek-R1-q5_k_m.gguf b/DeepSeek-R1-q5_k_m.gguf new file mode 100644 index 0000000..d88f8cd --- /dev/null +++ b/DeepSeek-R1-q5_k_m.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7cd9eaa42aa09326ac9af59563421ea89bade3922ab6c1e2ef680e1f99d3721a +size 5732989664 diff --git a/DeepSeek-R1-q5_k_s.gguf b/DeepSeek-R1-q5_k_s.gguf new file mode 100644 index 0000000..e9164a8 --- /dev/null +++ b/DeepSeek-R1-q5_k_s.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c7cb26a83dc4a0fd9aa01fb2106c01979212d7299510d1897425972a531f441f +size 5599296224 diff --git a/DeepSeek-R1-q6_k.gguf b/DeepSeek-R1-q6_k.gguf new file mode 100644 index 0000000..3416201 --- /dev/null +++ b/DeepSeek-R1-q6_k.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6277b5d729562f2e01436c81b919b5b69084022560c77c98d72a5d6fc215f13e +size 6596008672 diff --git a/DeepSeek-R1-q8_0.gguf b/DeepSeek-R1-q8_0.gguf new file mode 100644 index 0000000..a8314f2 --- /dev/null +++ b/DeepSeek-R1-q8_0.gguf @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c6e3924d662d3f24a96b228a5c317510c27e91c587e71e78877ed18a875ec82 +size 8540773088 diff --git a/README.md b/README.md new file mode 100644 index 0000000..b368f59 --- /dev/null +++ b/README.md @@ -0,0 +1,76 @@ +--- +license: mit +language: +- en +base_model: +- deepseek-ai/DeepSeek-R1 +pipeline_tag: text-generation +tags: +- deepseek-r1 +- gguf-connector +--- + +# GGUF quantized version of **deepseek-r1** + +### review +- no more error loading message: "unknown pre-tokenizer type: deepseek-r1-qwen" +- works fine for llama architecture + +### run the model +use any gguf connector to interact with gguf file(s), i.e., [connector](https://pypi.org/project/gguf-connector/) + +### reference +- base model: deepseek-ai/[DeepSeek-R1](https://huggingface.co/deepseek-ai/DeepSeek-R1) +- tool used for quantization: [cutter](https://pypi.org/project/gguf-cutter) + +### citation +[DeepSeek-R1](https://arxiv.org/pdf/2501.12948) + +### appendices: model evaluation (written by deekseek-ai) + +#### deepseek-r1-evaluation + for all our (here refer to deekseek-ai) models, the maximum generation length is set to 32,768 tokens; for benchmarks requiring sampling, we use a temperature of $0.6$, a top-p value of $0.95$, and generate 64 responses per query to estimate pass@1. + +| Category | Benchmark (Metric) | Claude-3.5-Sonnet-1022 | GPT-4o 0513 | DeepSeek V3 | OpenAI o1-mini | OpenAI o1-1217 | DeepSeek R1 | +|----------|-------------------|----------------------|------------|--------------|----------------|------------|--------------| +| | Architecture | - | - | MoE | - | - | MoE | +| | # Activated Params | - | - | 37B | - | - | 37B | +| | # Total Params | - | - | 671B | - | - | 671B | +| English | MMLU (Pass@1) | 88.3 | 87.2 | 88.5 | 85.2 | **91.8** | 90.8 | +| | MMLU-Redux (EM) | 88.9 | 88.0 | 89.1 | 86.7 | - | **92.9** | +| | MMLU-Pro (EM) | 78.0 | 72.6 | 75.9 | 80.3 | - | **84.0** | +| | DROP (3-shot F1) | 88.3 | 83.7 | 91.6 | 83.9 | 90.2 | **92.2** | +| | IF-Eval (Prompt Strict) | **86.5** | 84.3 | 86.1 | 84.8 | - | 83.3 | +| | GPQA-Diamond (Pass@1) | 65.0 | 49.9 | 59.1 | 60.0 | **75.7** | 71.5 | +| | SimpleQA (Correct) | 28.4 | 38.2 | 24.9 | 7.0 | **47.0** | 30.1 | +| | FRAMES (Acc.) | 72.5 | 80.5 | 73.3 | 76.9 | - | **82.5** | +| | AlpacaEval2.0 (LC-winrate) | 52.0 | 51.1 | 70.0 | 57.8 | - | **87.6** | +| | ArenaHard (GPT-4-1106) | 85.2 | 80.4 | 85.5 | 92.0 | - | **92.3** | +| Code | LiveCodeBench (Pass@1-COT) | 33.8 | 34.2 | - | 53.8 | 63.4 | **65.9** | +| | Codeforces (Percentile) | 20.3 | 23.6 | 58.7 | 93.4 | **96.6** | 96.3 | +| | Codeforces (Rating) | 717 | 759 | 1134 | 1820 | **2061** | 2029 | +| | SWE Verified (Resolved) | **50.8** | 38.8 | 42.0 | 41.6 | 48.9 | 49.2 | +| | Aider-Polyglot (Acc.) | 45.3 | 16.0 | 49.6 | 32.9 | **61.7** | 53.3 | +| Math | AIME 2024 (Pass@1) | 16.0 | 9.3 | 39.2 | 63.6 | 79.2 | **79.8** | +| | MATH-500 (Pass@1) | 78.3 | 74.6 | 90.2 | 90.0 | 96.4 | **97.3** | +| | CNMO 2024 (Pass@1) | 13.1 | 10.8 | 43.2 | 67.6 | - | **78.8** | +| Chinese | CLUEWSC (EM) | 85.4 | 87.9 | 90.9 | 89.9 | - | **92.8** | +| | C-Eval (EM) | 76.7 | 76.0 | 86.5 | 68.9 | - | **91.8** | +| | C-SimpleQA (Correct) | 55.4 | 58.7 | **68.0** | 40.3 | - | 63.7 | + +#### distilled model evaluation + +| Model | AIME 2024 pass@1 | AIME 2024 cons@64 | MATH-500 pass@1 | GPQA Diamond pass@1 | LiveCodeBench pass@1 | CodeForces rating | +|------------------------------------------|------------------|-------------------|-----------------|----------------------|----------------------|-------------------| +| GPT-4o-0513 | 9.3 | 13.4 | 74.6 | 49.9 | 32.9 | 759 | +| Claude-3.5-Sonnet-1022 | 16.0 | 26.7 | 78.3 | 65.0 | 38.9 | 717 | +| o1-mini | 63.6 | 80.0 | 90.0 | 60.0 | 53.8 | **1820** | +| QwQ-32B-Preview | 44.0 | 60.0 | 90.6 | 54.5 | 41.9 | 1316 | +| DeepSeek-R1-Distill-Qwen-1.5B | 28.9 | 52.7 | 83.9 | 33.8 | 16.9 | 954 | +| DeepSeek-R1-Distill-Qwen-7B | 55.5 | 83.3 | 92.8 | 49.1 | 37.6 | 1189 | +| DeepSeek-R1-Distill-Qwen-14B | 69.7 | 80.0 | 93.9 | 59.1 | 53.1 | 1481 | +| DeepSeek-R1-Distill-Qwen-32B | **72.6** | 83.3 | 94.3 | 62.1 | 57.2 | 1691 | +| DeepSeek-R1-Distill-Llama-8B | 50.4 | 80.0 | 89.1 | 49.0 | 39.6 | 1205 | +| DeepSeek-R1-Distill-Llama-70B | 70.0 | **86.7** | **94.5** | **65.2** | **57.5** | 1633 | + +\* these two tables are directly quoted from deepseek-ai diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file