初始化项目,由ModelHub XC社区提供模型

Model: Jackrong/GPT-5-Distill-Qwen3-4B-Instruct-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-24 05:41:18 +08:00
commit 36a6e6e9f8
16 changed files with 327 additions and 0 deletions

47
.gitattributes vendored Normal file
View File

@@ -0,0 +1,47 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.ot filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
qwen3-4b-instruct-2507.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
qwen3-4b-instruct-2507.Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-f16.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q3_K_L.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
GPT-5-Distill-Qwen3-4B-Instruct-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:dee8ba0d945fca51928f03314fc666065172d34903ef029ad1067bf716024be8
size 2286316416

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:69afc332d61f01392b5a2599f14ab29428d10069613d5699706ea5465f97b621
size 1669499776

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ea2bf73abf37f47bf1119bd98f9b02f886dfdec5d46fab8e5d6bb69a3c1c3193
size 2239785856

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0f251bba58a043e53e4f9a2de8a115601462af833c17ebdf0d5160ed31822fd7
size 2075618176

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:40ade8c3a9ea6dbdfca4637729f477d7d472d9296d3c8f2bd1897cbbeb4e90d2
size 1886997376

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c9854ccc3889f396c401dade873fed9b12d30040e80a6a2e686f40dbf6c30d6c
size 2383309696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e5ad0e55a0643f20505b11aa4baa3ce43139d5bd3138130a8938e087d273ecc4
size 2889513856

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b17f462120c4080127597b4d8f9099c169ff636f829f9ecf9b444701fe63c463
size 2823711616

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2ff9cb970e741a633e740be02d0f169a248c8fe7d884a77f5da124e0a010a9d2
size 3306261376

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0bcd5bf9f910454b87603855a7d58b26cb4c1b2838544beb527ce44247d71c9b
size 8051285376

54
Modelfile Normal file
View File

@@ -0,0 +1,54 @@
FROM qwen3-4b-instruct-2507.Q8_0.gguf
TEMPLATE """
{{- $lastUserIdx := -1 -}}
{{- range $idx, $msg := .Messages -}}
{{- if eq $msg.Role "user" }}{{ $lastUserIdx = $idx }}{{ end -}}
{{- end }}
{{- if or .System .Tools }}<|im_start|>system
{{ if .System }}
{{ .System }}
{{- end }}
{{- if .Tools }}
# Tools
You may call one or more functions to assist with the user query.
You are provided with function signatures within <tools></tools> XML tags:
<tools>
{{- range .Tools }}
{"type": "function", "function": {{ .Function }}}
{{- end }}
</tools>
For each function call, return a json object with function name and arguments within <tool_call></tool_call> XML tags:
<tool_call>
{"name": <function-name>, "arguments": <args-json-object>}
</tool_call>
{{- end -}}
<|im_end|>
{{ end }}
{{- range $i, $_ := .Messages }}
{{- $last := eq (len (slice $.Messages $i)) 1 -}}
{{- if eq .Role "user" }}<|im_start|>user
{{ .Content }}<|im_end|>
{{ else if eq .Role "assistant" }}<|im_start|>assistant
{{ if (and $.IsThinkSet (and .Thinking (or $last (gt $i $lastUserIdx)))) -}}
<think>{{ .Thinking }}</think>
{{ end -}}
{{ if .Content }}{{ .Content }}
{{- else if .ToolCalls }}<tool_call>
{{ range .ToolCalls }}{"name": "{{ .Function.Name }}", "arguments": {{ .Function.Arguments }}}
{{ end }}</tool_call>
{{- end }}{{ if not $last }}<|im_end|>
{{ end }}
{{- else if eq .Role "tool" }}<|im_start|>user
<tool_response>
{{ .Content }}
</tool_response><|im_end|>
{{ end }}
{{- if and (ne .Role "assistant") $last }}<|im_start|>assistant
{{ end }}
{{- end }}
"""

120
README.md Normal file
View File

@@ -0,0 +1,120 @@
---
tags:
- gguf
- llama.cpp
license: apache-2.0
datasets:
- Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507
language:
- en
- zh
base_model:
- Qwen/Qwen3-4B-Instruct-2507
---
# GPT-5-Distill-Qwen3-4B-Instruct-2507
![Base Model](https://img.shields.io/badge/Base_Model-Qwen3--4B--Instruct-0088CC?style=flat)
![Distillation](https://img.shields.io/badge/Distillation-GPT--5_Responses-8A2BE2?style=flat)
![Language](https://img.shields.io/badge/Language-English_%7C_Chinese-blue?style=flat)
![Context](https://img.shields.io/badge/Context-32K_Tokens-success?style=flat)
![Format](https://img.shields.io/badge/Format-GGUF_%7C_llama.cpp-yellow?style=flat)
![License](https://img.shields.io/badge/License-Apache_2.0-green?style=flat)
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/sk5gVFD15S0UNMek3gU0o.png" width="800"/>
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/vGzi5hSHJJ72ysJuM5EAv.png" width="800"/>
<img src="https://cdn-uploads.huggingface.co/production/uploads/66309bd090589b7c65950665/j39PSDVoQmK4EI9pLANpa.png" width="800"/>
**Model Type**: Instruction-tuned conversational LLM
Supports LoRA adapters and full-finetuned models for inference
- **Base Model**: `Qwen/Qwen3-4B-Instruct-2507`
- **Parameters**: 4B
- **Training Method**:
- Supervised Fine-Tuning (SFT) on ShareGPT data
- Knowledge distillation from LMSYS GPT-5 responses
- **Supported Languages**: Chinese, English, mixed inputs/outputs
- **Max Context Length**: Up to **32K tokens** (`max_seq_length = 32768`)
This model is trained on ShareGPT-Qwen3 instruction datasets and distilled toward the conversational style and quality of GPT-5. It aims to achieve high-quality, natural-sounding dialogues with low computational overhead—perfect for lightweight applications without sacrificing responsiveness.
---
## 2. Intended Use Cases
### ✅ Recommended:
- Casual chat in Chinese/English
- General knowledge explanations & reasoning guidance
- Code suggestions and simple debugging tips
- Writing assistance: editing, summarizing, rewriting
- Role-playing conversations (with well-designed prompts)
### ⚠️ Not Suitable For:
- High-risk decision-making:
- Medical diagnosis, mental health support
- Legal advice, financial investment recommendations
- Real-time factual tasks (e.g., news, stock updates)
- Authoritative judgment on sensitive topics
> **Note**: Outputs are for reference only and not intended as the sole basis for critical decisions.
---
## 3. Training Data & Distillation Process
### Key Datasets:
#### (1) ds1: ShareGPT-Qwen3 Instruction Dataset
- Source: `Jackrong/ShareGPT-Qwen3-235B-A22B-Instuct-2507`
- Purpose:
- Provides diverse instruction-response pairs
- Supports multi-turn dialogues and context awareness
- Processing:
- Cleaned for quality and relevance
- Standardized into `instruction`, `input`, `output` format
#### (2) ds2: LMSYS GPT-5 Teacher Response Data
- Source: `ytz20/LMSYS-Chat-GPT-5-Chat-Response`
- Filtering:
- Only kept samples with `flaw == "normal"`
- Removed hallucinations and inconsistent responses
- Purpose:
- Distillation target for conversational quality
- Enhances clarity, coherence, and fluency
### Training Flow:
1. Prepare unified Chat-formatted dataset
2. Fine-tune base Qwen3-4B-Instruct-2507 via SFT
3. Conduct knowledge distillation using GPT-5's normal responses as teacher outputs
4. Balance style imitation with semantic fidelity to ensure robustness
> ⚖️ **Note**: This work is based on publicly available, non-sensitive datasets and uses them responsibly under fair use principles.
---
## 4. Key Features Summary
| Feature | Description |
|--------|-------------|
| **Lightweight** | ~4B parameter model – fast inference, low resource usage |
| **Distillation-Style Responses** | Mimics GPT-5’s conversational fluency and helpfulness |
| **Highly Conversational** | Excellent for chatbot-style interactions with rich dialogue flow |
| **Multilingual Ready** | Seamless support for Chinese and English |
---
## 5. Acknowledgements
We thank:
- LMSYS team for sharing GPT-5 response data
- Jackrong for the ShareGPT-Qwen3 dataset
- Qwen team for releasing `Qwen3-4B-Instruct`
This project is an open research effort aimed at making high-quality conversational AI accessible with smaller models.
---

70
config.json Normal file
View File

@@ -0,0 +1,70 @@
{
"architectures": [
"Qwen3ForCausalLM"
],
"attention_bias": false,
"attention_dropout": 0.0,
"torch_dtype": "bfloat16",
"eos_token_id": 151645,
"head_dim": 128,
"hidden_act": "silu",
"hidden_size": 2560,
"initializer_range": 0.02,
"intermediate_size": 9728,
"layer_types": [
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention",
"full_attention"
],
"max_position_embeddings": 262144,
"max_window_layers": 36,
"model_type": "qwen3",
"num_attention_heads": 32,
"num_hidden_layers": 36,
"num_key_value_heads": 8,
"pad_token_id": 151654,
"rms_norm_eps": 1e-06,
"rope_scaling": null,
"rope_theta": 5000000,
"sliding_window": null,
"tie_word_embeddings": true,
"transformers_version": "4.56.2",
"unsloth_fixed": true,
"unsloth_version": "2025.11.3",
"use_cache": true,
"use_sliding_window": false,
"vocab_size": 151936
}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:717904966099905a2cb0f4013bc731d0801e411c53bb61d1d27e9ffe46f65e85
size 2497280192

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:93642d65627ec796032c76d5401043b9d29fef3c73e3562506be537c383aa482
size 4280404672