初始化项目,由ModelHub XC社区提供模型
Model: Jackrong/GPT-5-Distill-Qwen3-4B-Instruct-GGUF Source: Original Platform
This commit is contained in:
47
.gitattributes
vendored
Normal file
47
.gitattributes
vendored
Normal file
@@ -0,0 +1,47 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
qwen3-4b-instruct-2507.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
qwen3-4b-instruct-2507.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:dee8ba0d945fca51928f03314fc666065172d34903ef029ad1067bf716024be8
|
||||||
|
size 2286316416
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:69afc332d61f01392b5a2599f14ab29428d10069613d5699706ea5465f97b621
|
||||||
|
size 1669499776
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:ea2bf73abf37f47bf1119bd98f9b02f886dfdec5d46fab8e5d6bb69a3c1c3193
|
||||||
|
size 2239785856
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:0f251bba58a043e53e4f9a2de8a115601462af833c17ebdf0d5160ed31822fd7
|
||||||
|
size 2075618176
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:40ade8c3a9ea6dbdfca4637729f477d7d472d9296d3c8f2bd1897cbbeb4e90d2
|
||||||
|
size 1886997376
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c9854ccc3889f396c401dade873fed9b12d30040e80a6a2e686f40dbf6c30d6c
|
||||||
|
size 2383309696
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:e5ad0e55a0643f20505b11aa4baa3ce43139d5bd3138130a8938e087d273ecc4
|
||||||
|
size 2889513856
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:b17f462120c4080127597b4d8f9099c169ff636f829f9ecf9b444701fe63c463
|
||||||
|
size 2823711616
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:2ff9cb970e741a633e740be02d0f169a248c8fe7d884a77f5da124e0a010a9d2
|
||||||
|
size 3306261376
|
||||||
3
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:0bcd5bf9f910454b87603855a7d58b26cb4c1b2838544beb527ce44247d71c9b
|
||||||
|
size 8051285376
|
||||||
54
Modelfile
Normal file
54
Modelfile
Normal file
@@ -0,0 +1,54 @@
|
|||||||
|
|
||||||
|
FROM qwen3-4b-instruct-2507.Q8_0.gguf
|
||||||
|
TEMPLATE """
|
||||||
|
{{- $lastUserIdx := -1 -}}
|
||||||
|
{{- range $idx, $msg := .Messages -}}
|
||||||
|
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
|
||||||
|
{{- end }}
|
||||||
|
{{- if or .System .Tools }}<|im_start|>system
|
||||||
|
{{ if .System }}
|
||||||
|
{{ .System }}
|
||||||
|
{{- end }}
|
||||||
|
{{- if .Tools }}
|
||||||
|
|
||||||
|
# Tools
|
||||||
|
|
||||||
|
You may call one or more functions to assist with the user query.
|
||||||
|
|
||||||
|
You are provided with function signatures within <tools></tools> XML tags:
|
||||||
|
<tools>
|
||||||
|
{{- range .Tools }}
|
||||||
|
{"type": "function", "function": {{ .Function }}}
|
||||||
|
{{- end }}
|
||||||
|
</tools>
|
||||||
|
|
||||||
|
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
|
||||||
|
<tool_call>
|
||||||
|
{"name": <function-name>, "arguments": <args-json-object>}
|
||||||
|
</tool_call>
|
||||||
|
{{- end -}}
|
||||||
|
<|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- range $i, $_ := .Messages }}
|
||||||
|
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
|
||||||
|
{{- if eq .Role "user" }}<|im_start|>user
|
||||||
|
{{ .Content }}<|im_end|>
|
||||||
|
{{ else if eq .Role "assistant" }}<|im_start|>assistant
|
||||||
|
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
|
||||||
|
<think>{{ .Thinking }}</think>
|
||||||
|
{{ end -}}
|
||||||
|
{{ if .Content }}{{ .Content }}
|
||||||
|
{{- else if .ToolCalls }}<tool_call>
|
||||||
|
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
|
||||||
|
{{ end }}</tool_call>
|
||||||
|
{{- end }}{{ if not $last }}<|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- else if eq .Role "tool" }}<|im_start|>user
|
||||||
|
<tool_response>
|
||||||
|
{{ .Content }}
|
||||||
|
</tool_response><|im_end|>
|
||||||
|
{{ end }}
|
||||||
|
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
|
||||||
|
{{ end }}
|
||||||
|
{{- end }}
|
||||||
|
"""
|
||||||
120
README.md
Normal file
120
README.md
Normal file
@@ -0,0 +1,120 @@
|
|||||||
|
---
|
||||||
|
tags:
|
||||||
|
- gguf
|
||||||
|
- llama.cpp
|
||||||
|
license: apache-2.0
|
||||||
|
datasets:
|
||||||
|
- Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
- zh
|
||||||
|
base_model:
|
||||||
|
- Qwen/Qwen3-4B-Instruct-2507
|
||||||
|
---
|
||||||
|
|
||||||
|
|
||||||
|
# GPT-5-Distill-Qwen3-4B-Instruct-2507
|
||||||
|
|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|

|
||||||
|
|
||||||
|
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/sk5gVFD15S0UNMek3gU0o.png" width="800"/>
|
||||||
|
|
||||||
|
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/vGzi5hSHJJ72ysJuM5EAv.png" width="800"/>
|
||||||
|
|
||||||
|
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/j39PSDVoQmK4EI9pLANpa.png" width="800"/>
|
||||||
|
|
||||||
|
**Model Type**: Instruction-tuned conversational LLM
|
||||||
|
Supports LoRA adapters and full-finetuned models for inference
|
||||||
|
- **Base Model**: `Qwen/Qwen3-4B-Instruct-2507`
|
||||||
|
- **Parameters**: 4B
|
||||||
|
- **Training Method**:
|
||||||
|
- Supervised Fine-Tuning (SFT) on ShareGPT data
|
||||||
|
- Knowledge distillation from LMSYS GPT-5 responses
|
||||||
|
- **Supported Languages**: Chinese, English, mixed inputs/outputs
|
||||||
|
- **Max Context Length**: Up to **32K tokens** (`max_seq_length = 32768`)
|
||||||
|
|
||||||
|
This model is trained on ShareGPT-Qwen3 instruction datasets and distilled toward the conversational style and quality of GPT-5. It aims to achieve high-quality, natural-sounding dialogues with low computational overhead—perfect for lightweight applications without sacrificing responsiveness.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 2. Intended Use Cases
|
||||||
|
|
||||||
|
### ✅ Recommended:
|
||||||
|
|
||||||
|
- Casual chat in Chinese/English
|
||||||
|
- General knowledge explanations & reasoning guidance
|
||||||
|
- Code suggestions and simple debugging tips
|
||||||
|
- Writing assistance: editing, summarizing, rewriting
|
||||||
|
- Role-playing conversations (with well-designed prompts)
|
||||||
|
|
||||||
|
### ⚠️ Not Suitable For:
|
||||||
|
|
||||||
|
- High-risk decision-making:
|
||||||
|
- Medical diagnosis, mental health support
|
||||||
|
- Legal advice, financial investment recommendations
|
||||||
|
- Real-time factual tasks (e.g., news, stock updates)
|
||||||
|
- Authoritative judgment on sensitive topics
|
||||||
|
|
||||||
|
> **Note**: Outputs are for reference only and not intended as the sole basis for critical decisions.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 3. Training Data & Distillation Process
|
||||||
|
|
||||||
|
### Key Datasets:
|
||||||
|
|
||||||
|
#### (1) ds1: ShareGPT-Qwen3 Instruction Dataset
|
||||||
|
- Source: `Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507`
|
||||||
|
- Purpose:
|
||||||
|
- Provides diverse instruction-response pairs
|
||||||
|
- Supports multi-turn dialogues and context awareness
|
||||||
|
- Processing:
|
||||||
|
- Cleaned for quality and relevance
|
||||||
|
- Standardized into `instruction`, `input`, `output` format
|
||||||
|
|
||||||
|
#### (2) ds2: LMSYS GPT-5 Teacher Response Data
|
||||||
|
- Source: `ytz20/LMSYS-Chat-GPT-5-Chat-Response`
|
||||||
|
- Filtering:
|
||||||
|
- Only kept samples with `flaw == "normal"`
|
||||||
|
- Removed hallucinations and inconsistent responses
|
||||||
|
- Purpose:
|
||||||
|
- Distillation target for conversational quality
|
||||||
|
- Enhances clarity, coherence, and fluency
|
||||||
|
|
||||||
|
### Training Flow:
|
||||||
|
|
||||||
|
1. Prepare unified Chat-formatted dataset
|
||||||
|
2. Fine-tune base Qwen3-4B-Instruct-2507 via SFT
|
||||||
|
3. Conduct knowledge distillation using GPT-5's normal responses as teacher outputs
|
||||||
|
4. Balance style imitation with semantic fidelity to ensure robustness
|
||||||
|
|
||||||
|
> ⚖️ **Note**: This work is based on publicly available, non-sensitive datasets and uses them responsibly under fair use principles.
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 4. Key Features Summary
|
||||||
|
|
||||||
|
| Feature | Description |
|
||||||
|
|--------|-------------|
|
||||||
|
| **Lightweight** | ~4B parameter model – fast inference, low resource usage |
|
||||||
|
| **Distillation-Style Responses** | Mimics GPT-5’s conversational fluency and helpfulness |
|
||||||
|
| **Highly Conversational** | Excellent for chatbot-style interactions with rich dialogue flow |
|
||||||
|
| **Multilingual Ready** | Seamless support for Chinese and English |
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
|
## 5. Acknowledgements
|
||||||
|
|
||||||
|
We thank:
|
||||||
|
- LMSYS team for sharing GPT-5 response data
|
||||||
|
- Jackrong for the ShareGPT-Qwen3 dataset
|
||||||
|
- Qwen team for releasing `Qwen3-4B-Instruct`
|
||||||
|
|
||||||
|
This project is an open research effort aimed at making high-quality conversational AI accessible with smaller models.
|
||||||
|
|
||||||
|
---
|
||||||
70
config.json
Normal file
70
config.json
Normal file
@@ -0,0 +1,70 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"Qwen3ForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_bias": false,
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"torch_dtype": "bfloat16",
|
||||||
|
"eos_token_id": 151645,
|
||||||
|
"head_dim": 128,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 2560,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 9728,
|
||||||
|
"layer_types": [
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention",
|
||||||
|
"full_attention"
|
||||||
|
],
|
||||||
|
"max_position_embeddings": 262144,
|
||||||
|
"max_window_layers": 36,
|
||||||
|
"model_type": "qwen3",
|
||||||
|
"num_attention_heads": 32,
|
||||||
|
"num_hidden_layers": 36,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"pad_token_id": 151654,
|
||||||
|
"rms_norm_eps": 1e-06,
|
||||||
|
"rope_scaling": null,
|
||||||
|
"rope_theta": 5000000,
|
||||||
|
"sliding_window": null,
|
||||||
|
"tie_word_embeddings": true,
|
||||||
|
"transformers_version": "4.56.2",
|
||||||
|
"unsloth_fixed": true,
|
||||||
|
"unsloth_version": "2025.11.3",
|
||||||
|
"use_cache": true,
|
||||||
|
"use_sliding_window": false,
|
||||||
|
"vocab_size": 151936
|
||||||
|
}
|
||||||
3
qwen3-4b-instruct-2507.Q4_K_M.gguf
Normal file
3
qwen3-4b-instruct-2507.Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:717904966099905a2cb0f4013bc731d0801e411c53bb61d1d27e9ffe46f65e85
|
||||||
|
size 2497280192
|
||||||
3
qwen3-4b-instruct-2507.Q8_0.gguf
Normal file
3
qwen3-4b-instruct-2507.Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:93642d65627ec796032c76d5401043b9d29fef3c73e3562506be537c383aa482
|
||||||
|
size 4280404672
|
||||||
Reference in New Issue
Block a user