初始化项目,由ModelHub XC社区提供模型
Model: Jackrong/GPT-5-Distill-Qwen3-4B-Instruct-GGUF Source: Original Platform
This commit is contained in:
47
.gitattributes
vendored
Normal file
47
.gitattributes
vendored
Normal file
@@ -0,0 +1,47 @@
|
||||
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-instruct-2507.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
qwen3-4b-instruct-2507.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:dee8ba0d945fca51928f03314fc666065172d34903ef029ad1067bf716024be8
|
||||
size 2286316416
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:69afc332d61f01392b5a2599f14ab29428d10069613d5699706ea5465f97b621
|
||||
size 1669499776
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:ea2bf73abf37f47bf1119bd98f9b02f886dfdec5d46fab8e5d6bb69a3c1c3193
|
||||
size 2239785856
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0f251bba58a043e53e4f9a2de8a115601462af833c17ebdf0d5160ed31822fd7
|
||||
size 2075618176
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:40ade8c3a9ea6dbdfca4637729f477d7d472d9296d3c8f2bd1897cbbeb4e90d2
|
||||
size 1886997376
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c9854ccc3889f396c401dade873fed9b12d30040e80a6a2e686f40dbf6c30d6c
|
||||
size 2383309696
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:e5ad0e55a0643f20505b11aa4baa3ce43139d5bd3138130a8938e087d273ecc4
|
||||
size 2889513856
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b17f462120c4080127597b4d8f9099c169ff636f829f9ecf9b444701fe63c463
|
||||
size 2823711616
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:2ff9cb970e741a633e740be02d0f169a248c8fe7d884a77f5da124e0a010a9d2
|
||||
size 3306261376
|
||||
3
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf
Normal file
3
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:0bcd5bf9f910454b87603855a7d58b26cb4c1b2838544beb527ce44247d71c9b
|
||||
size 8051285376
|
||||
54
Modelfile
Normal file
54
Modelfile
Normal file
@@ -0,0 +1,54 @@
|
||||
|
||||
FROM qwen3-4b-instruct-2507.Q8_0.gguf
|
||||
TEMPLATE """
|
||||
{{- $lastUserIdx := -1 -}}
|
||||
{{- range $idx, $msg := .Messages -}}
|
||||
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
|
||||
{{- end }}
|
||||
{{- if or .System .Tools }}<|im_start|>system
|
||||
{{ if .System }}
|
||||
{{ .System }}
|
||||
{{- end }}
|
||||
{{- if .Tools }}
|
||||
|
||||
# Tools
|
||||
|
||||
You may call one or more functions to assist with the user query.
|
||||
|
||||
You are provided with function signatures within <tools></tools> XML tags:
|
||||
<tools>
|
||||
{{- range .Tools }}
|
||||
{"type": "function", "function": {{ .Function }}}
|
||||
{{- end }}
|
||||
</tools>
|
||||
|
||||
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
|
||||
<tool_call>
|
||||
{"name": <function-name>, "arguments": <args-json-object>}
|
||||
</tool_call>
|
||||
{{- end -}}
|
||||
<|im_end|>
|
||||
{{ end }}
|
||||
{{- range $i, $_ := .Messages }}
|
||||
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
|
||||
{{- if eq .Role "user" }}<|im_start|>user
|
||||
{{ .Content }}<|im_end|>
|
||||
{{ else if eq .Role "assistant" }}<|im_start|>assistant
|
||||
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
|
||||
<think>{{ .Thinking }}</think>
|
||||
{{ end -}}
|
||||
{{ if .Content }}{{ .Content }}
|
||||
{{- else if .ToolCalls }}<tool_call>
|
||||
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
|
||||
{{ end }}</tool_call>
|
||||
{{- end }}{{ if not $last }}<|im_end|>
|
||||
{{ end }}
|
||||
{{- else if eq .Role "tool" }}<|im_start|>user
|
||||
<tool_response>
|
||||
{{ .Content }}
|
||||
</tool_response><|im_end|>
|
||||
{{ end }}
|
||||
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
|
||||
{{ end }}
|
||||
{{- end }}
|
||||
"""
|
||||
120
README.md
Normal file
120
README.md
Normal file
@@ -0,0 +1,120 @@
|
||||
---
|
||||
tags:
|
||||
- gguf
|
||||
- llama.cpp
|
||||
license: apache-2.0
|
||||
datasets:
|
||||
- Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507
|
||||
language:
|
||||
- en
|
||||
- zh
|
||||
base_model:
|
||||
- Qwen/Qwen3-4B-Instruct-2507
|
||||
---
|
||||
|
||||
|
||||
# GPT-5-Distill-Qwen3-4B-Instruct-2507
|
||||
|
||||

|
||||

|
||||

|
||||

|
||||

|
||||

|
||||
|
||||
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/sk5gVFD15S0UNMek3gU0o.png" width="800"/>
|
||||
|
||||
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/vGzi5hSHJJ72ysJuM5EAv.png" width="800"/>
|
||||
|
||||
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/j39PSDVoQmK4EI9pLANpa.png" width="800"/>
|
||||
|
||||
**Model Type**: Instruction-tuned conversational LLM
|
||||
Supports LoRA adapters and full-finetuned models for inference
|
||||
- **Base Model**: `Qwen/Qwen3-4B-Instruct-2507`
|
||||
- **Parameters**: 4B
|
||||
- **Training Method**:
|
||||
- Supervised Fine-Tuning (SFT) on ShareGPT data
|
||||
- Knowledge distillation from LMSYS GPT-5 responses
|
||||
- **Supported Languages**: Chinese, English, mixed inputs/outputs
|
||||
- **Max Context Length**: Up to **32K tokens** (`max_seq_length = 32768`)
|
||||
|
||||
This model is trained on ShareGPT-Qwen3 instruction datasets and distilled toward the conversational style and quality of GPT-5. It aims to achieve high-quality, natural-sounding dialogues with low computational overhead—perfect for lightweight applications without sacrificing responsiveness.
|
||||
|
||||
---
|
||||
|
||||
## 2. Intended Use Cases
|
||||
|
||||
### ✅ Recommended:
|
||||
|
||||
- Casual chat in Chinese/English
|
||||
- General knowledge explanations & reasoning guidance
|
||||
- Code suggestions and simple debugging tips
|
||||
- Writing assistance: editing, summarizing, rewriting
|
||||
- Role-playing conversations (with well-designed prompts)
|
||||
|
||||
### ⚠️ Not Suitable For:
|
||||
|
||||
- High-risk decision-making:
|
||||
- Medical diagnosis, mental health support
|
||||
- Legal advice, financial investment recommendations
|
||||
- Real-time factual tasks (e.g., news, stock updates)
|
||||
- Authoritative judgment on sensitive topics
|
||||
|
||||
> **Note**: Outputs are for reference only and not intended as the sole basis for critical decisions.
|
||||
|
||||
---
|
||||
|
||||
## 3. Training Data & Distillation Process
|
||||
|
||||
### Key Datasets:
|
||||
|
||||
#### (1) ds1: ShareGPT-Qwen3 Instruction Dataset
|
||||
- Source: `Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507`
|
||||
- Purpose:
|
||||
- Provides diverse instruction-response pairs
|
||||
- Supports multi-turn dialogues and context awareness
|
||||
- Processing:
|
||||
- Cleaned for quality and relevance
|
||||
- Standardized into `instruction`, `input`, `output` format
|
||||
|
||||
#### (2) ds2: LMSYS GPT-5 Teacher Response Data
|
||||
- Source: `ytz20/LMSYS-Chat-GPT-5-Chat-Response`
|
||||
- Filtering:
|
||||
- Only kept samples with `flaw == "normal"`
|
||||
- Removed hallucinations and inconsistent responses
|
||||
- Purpose:
|
||||
- Distillation target for conversational quality
|
||||
- Enhances clarity, coherence, and fluency
|
||||
|
||||
### Training Flow:
|
||||
|
||||
1. Prepare unified Chat-formatted dataset
|
||||
2. Fine-tune base Qwen3-4B-Instruct-2507 via SFT
|
||||
3. Conduct knowledge distillation using GPT-5's normal responses as teacher outputs
|
||||
4. Balance style imitation with semantic fidelity to ensure robustness
|
||||
|
||||
> ⚖️ **Note**: This work is based on publicly available, non-sensitive datasets and uses them responsibly under fair use principles.
|
||||
|
||||
---
|
||||
|
||||
## 4. Key Features Summary
|
||||
|
||||
| Feature | Description |
|
||||
|--------|-------------|
|
||||
| **Lightweight** | ~4B parameter model – fast inference, low resource usage |
|
||||
| **Distillation-Style Responses** | Mimics GPT-5’s conversational fluency and helpfulness |
|
||||
| **Highly Conversational** | Excellent for chatbot-style interactions with rich dialogue flow |
|
||||
| **Multilingual Ready** | Seamless support for Chinese and English |
|
||||
|
||||
---
|
||||
|
||||
## 5. Acknowledgements
|
||||
|
||||
We thank:
|
||||
- LMSYS team for sharing GPT-5 response data
|
||||
- Jackrong for the ShareGPT-Qwen3 dataset
|
||||
- Qwen team for releasing `Qwen3-4B-Instruct`
|
||||
|
||||
This project is an open research effort aimed at making high-quality conversational AI accessible with smaller models.
|
||||
|
||||
---
|
||||
70
config.json
Normal file
70
config.json
Normal file
@@ -0,0 +1,70 @@
|
||||
{
|
||||
"architectures": [
|
||||
"Qwen3ForCausalLM"
|
||||
],
|
||||
"attention_bias": false,
|
||||
"attention_dropout": 0.0,
|
||||
"torch_dtype": "bfloat16",
|
||||
"eos_token_id": 151645,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 2560,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 9728,
|
||||
"layer_types": [
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention",
|
||||
"full_attention"
|
||||
],
|
||||
"max_position_embeddings": 262144,
|
||||
"max_window_layers": 36,
|
||||
"model_type": "qwen3",
|
||||
"num_attention_heads": 32,
|
||||
"num_hidden_layers": 36,
|
||||
"num_key_value_heads": 8,
|
||||
"pad_token_id": 151654,
|
||||
"rms_norm_eps": 1e-06,
|
||||
"rope_scaling": null,
|
||||
"rope_theta": 5000000,
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": true,
|
||||
"transformers_version": "4.56.2",
|
||||
"unsloth_fixed": true,
|
||||
"unsloth_version": "2025.11.3",
|
||||
"use_cache": true,
|
||||
"use_sliding_window": false,
|
||||
"vocab_size": 151936
|
||||
}
|
||||
3
qwen3-4b-instruct-2507.Q4_K_M.gguf
Normal file
3
qwen3-4b-instruct-2507.Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:717904966099905a2cb0f4013bc731d0801e411c53bb61d1d27e9ffe46f65e85
|
||||
size 2497280192
|
||||
3
qwen3-4b-instruct-2507.Q8_0.gguf
Normal file
3
qwen3-4b-instruct-2507.Q8_0.gguf
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:93642d65627ec796032c76d5401043b9d29fef3c73e3562506be537c383aa482
|
||||
size 4280404672
|
||||
Reference in New Issue
Block a user