初始化项目,由ModelHub XC社区提供模型
Model: khazarai/Qwen3-4B-Kimi2.5-Reasoning-Distilled-GGUF Source: Original Platform
This commit is contained in:
42
.gitattributes
vendored
Normal file
42
.gitattributes
vendored
Normal file
@@ -0,0 +1,42 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-thinking-2507.BF16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-thinking-2507.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-thinking-2507.Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-thinking-2507.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
benchmark/BaseModel.png filter=lfs diff=lfs merge=lfs -text
|
||||
benchmark/Kimi_distilled.png filter=lfs diff=lfs merge=lfs -text
|
||||
benchmark/kimi2.5-distilled-blue.png filter=lfs diff=lfs merge=lfs -text
|
||||
54
Modelfile
Normal file
54
Modelfile
Normal file
@@ -0,0 +1,54 @@
|
||||
|
||||
FROM qwen3-4b-thinking-2507.BF16.gguf
|
||||
TEMPLATE """
|
||||
{{- $lastUserIdx := -1 -}}
|
||||
{{- range $idx, $msg := .Messages -}}
|
||||
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
|
||||
{{- end }}
|
||||
{{- if or .System .Tools }}<|im_start|>system
|
||||
{{ if .System }}
|
||||
{{ .System }}
|
||||
{{- end }}
|
||||
{{- if .Tools }}
|
||||
|
||||
# Tools
|
||||
|
||||
You may call one or more functions to assist with the user query.
|
||||
|
||||
You are provided with function signatures within <tools></tools> XML tags:
|
||||
<tools>
|
||||
{{- range .Tools }}
|
||||
{"type": "function", "function": {{ .Function }}}
|
||||
{{- end }}
|
||||
</tools>
|
||||
|
||||
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
|
||||
<tool_call>
|
||||
{"name": <function-name>, "arguments": <args-json-object>}
|
||||
</tool_call>
|
||||
{{- end -}}
|
||||
<|im_end|>
|
||||
{{ end }}
|
||||
{{- range $i, $_ := .Messages }}
|
||||
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
|
||||
{{- if eq .Role "user" }}<|im_start|>user
|
||||
{{ .Content }}<|im_end|>
|
||||
{{ else if eq .Role "assistant" }}<|im_start|>assistant
|
||||
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
|
||||
<think>{{ .Thinking }}</think>
|
||||
{{ end -}}
|
||||
{{ if .Content }}{{ .Content }}
|
||||
{{- else if .ToolCalls }}<tool_call>
|
||||
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
|
||||
{{ end }}</tool_call>
|
||||
{{- end }}{{ if not $last }}<|im_end|>
|
||||
{{ end }}
|
||||
{{- else if eq .Role "tool" }}<|im_start|>user
|
||||
<tool_response>
|
||||
{{ .Content }}
|
||||
</tool_response><|im_end|>
|
||||
{{ end }}
|
||||
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
|
||||
{{ end }}
|
||||
{{- end }}
|
||||
"""
|
||||
82
README.md
Normal file
82
README.md
Normal file
@@ -0,0 +1,82 @@
|
||||
---
|
||||
tags:
|
||||
- gguf
|
||||
- llama.cpp
|
||||
- unsloth
|
||||
- reasoning
|
||||
- distillation
|
||||
- sft
|
||||
license: apache-2.0
|
||||
datasets:
|
||||
- khazarai/kimi-2.5-high-reasoning-250x
|
||||
language:
|
||||
- en
|
||||
base_model:
|
||||
- khazarai/Qwen3-4B-Kimi2.5-Reasoning-Distilled
|
||||
pipeline_tag: text-generation
|
||||
---
|
||||
|
||||
# Qwen3-4B-Kimi2.5-Reasoning-Distilled : GGUF
|
||||
|
||||

|
||||
|
||||
|
||||
| Model | Score |
|
||||
| :--- | :--- |
|
||||
| khazarai/Qwen3-4B-Kimi2.5-Reasoning-Distilled | 76.09 |
|
||||
| Qwen/Qwen3-4B-Thinking-2507 | 73.73 |
|
||||
|
||||
- **Benchmark**: khazarai/Multi-Domain-Reasoning-Benchmark
|
||||
- **Total Questions**: 100
|
||||
|
||||
`Qwen3-4B-Kimi2.5-Reasoning-Distilled` is a fine-tuned language model optimized for structured, long-form reasoning. It is derived from the Qwen3-4b-Thinking-2507 base model and fine-tuned using a specialized distillation dataset generated by Kimi-2.5-thinking.
|
||||
|
||||
This model is designed to bridge the gap between small, efficient models (0.6B–4B range) and the complex reasoning capabilities typically found in much larger models. It excels at breaking down problems, self-correcting, and providing detailed analytical answers.
|
||||
|
||||
**Base Model**: Qwen3-4b-Thinking-2507
|
||||
|
||||
**Training Technique**: Unsloth + QLoRa
|
||||
|
||||
|
||||
## Available Model files:
|
||||
- `qwen3-4b-thinking-2507.BF16.gguf`
|
||||
- `qwen3-4b-thinking-2507.Q8_0.gguf`
|
||||
- `qwen3-4b-thinking-2507.Q6_K.gguf`
|
||||
- `qwen3-4b-thinking-2507.Q4_K_M.gguf`
|
||||
|
||||
## Ollama
|
||||
An Ollama Modelfile is included for easy deployment.
|
||||
|
||||
|
||||
## Provided Quants
|
||||
|
||||
(sorted by size, not necessarily quality. IQ-quants are often preferable over similar sized non-IQ quants)
|
||||
|
||||
| Type | Size/GB | Notes |
|
||||
|:-----|--------:|:------|
|
||||
| Q4_K_M | 2.5 | fast, recommended |
|
||||
| Q6_K | 3.3 | very good quality |
|
||||
| Q8_0 | 4.2 | fast, best quality |
|
||||
| f16 | 8.0 | 16 bpw, overkill |
|
||||
|
||||
Here is a handy graph by ikawrakow comparing some lower-quality quant
|
||||
types (lower is better):
|
||||
|
||||

|
||||
|
||||
And here are Artefact2's thoughts on the matter:
|
||||
https://gist.github.com/Artefact2/b5f810600771265fc1e39442288e8ec9
|
||||
|
||||
|
||||
## Dataset
|
||||
|
||||
The model was fine-tuned on the [khazarai/kimi-2.5-high-reasoning-250x](https://huggingface.co/datasets/khazarai/kimi-2.5-high-reasoning-250x)
|
||||
|
||||
Dataset Composition:
|
||||
- Total Samples: 250
|
||||
- Total Tokens: 1,114,407
|
||||
- Teacher Model: Kimi-2.5-Thinking
|
||||
|
||||
## Acknowledgements
|
||||
|
||||
**Unsloth** for the incredibly fast and memory-efficient training framework.
|
||||
3
benchmark/kimi2.5-distilled-blue.png
Normal file
3
benchmark/kimi2.5-distilled-blue.png
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:4177705be7d0aa1bf5d05b52e38117b7438a96871b8e86aa978ed15eea2780c3
|
||||
size 123677
|
||||
72
config.json
Normal file
72
config.json
Normal file
@@ -0,0 +1,72 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": null,
|
||||
"torch_dtype": "float16",
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2560,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 9728,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 262144,
|
||||
"max_window_layers": 36,
|
||||
"model_type": "qwen3",
|
||||
"num_attention_heads": 32,
|
||||
"num_hidden_layers": 36,
|
||||
"num_key_value_heads": 8,
|
||||
"pad_token_id": 151669,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_parameters": {
|
||||
"rope_theta": 5000000,
|
||||
"rope_type": "default"
|
||||
},
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"unsloth_fixed": true,
|
||||
"unsloth_version": "2026.3.4",
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
3
qwen3-4b-thinking-2507.BF16.gguf
Normal file
3
qwen3-4b-thinking-2507.BF16.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:63a98a330dd9b10ff800cf24f320f06569fe2d510d21387f3689b86444189089
|
||||
size 8051284704
|
||||
3
qwen3-4b-thinking-2507.Q4_K_M.gguf
Normal file
3
qwen3-4b-thinking-2507.Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:cece491f80c2c7cc42ed0910a85447ee9ffd454fcec97ad06b7c70de0925d1de
|
||||
size 2497280224
|
||||
3
qwen3-4b-thinking-2507.Q6_K.gguf
Normal file
3
qwen3-4b-thinking-2507.Q6_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:63627f8180991a42d8673c57fe05e61850eaf5f1c976e00b6f3515ae07d2d7b6
|
||||
size 3306260704
|
||||
3
qwen3-4b-thinking-2507.Q8_0.gguf
Normal file
3
qwen3-4b-thinking-2507.Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b11a2fdd36599ba57ccb3e4d0a60251b6d4f9ab3573caef77af08509efbb26fa
|
||||
size 4280404704
|
||||
Reference in New Issue
Block a user