初始化项目,由ModelHub XC社区提供模型

Model: unsloth/Qwen3-VL-4B-Instruct-1M-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-13 11:48:12 +08:00
commit aa4589985b
33 changed files with 386 additions and 0 deletions

78
.gitattributes vendored Normal file
View File

@@ -0,0 +1,78 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F16.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F32.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
imatrix_unsloth.gguf_file filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-4B-Instruct-1M-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:998ca3414e53b113cac21c1d1f5a7a5ee7bf4cd2ac9a8b17581529a8c8d9ea8f
size 8051286720

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b769903dbcfeb01eb6320b0f92fe72554966e3665474ee48e1b7b6254577d105
size 2381345216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:691f60d8842d8db16b2b95c46355d387c3e73b08b3fc05fa93a10702c91ec046
size 2270753216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:93c544f7e5113e2ba1f257376dfaec4cd001b10747c0b24b55a86624972b3dfa
size 1669501376

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:93c544f7e5113e2ba1f257376dfaec4cd001b10747c0b24b55a86624972b3dfa
size 1669501376

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:bdf31b69b055dc86d100c8b195ce5d6659dfde77dacb9a64cc6cce963ffaf13a
size 2075619776

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:81ff7df09ff7f5db4caa8e4cc50a60dfc942ddd01d528200ea1594dacb5cec5c
size 1886998976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0612188cef469086dd2874ae18ab6a81dba0539ebbdcbc19cd328fba6cb797d1
size 2375774656

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:94ad73608e51ea3f296164fc6422419cb302a4bc3d9070814a34fde42ca48cb8
size 2596630976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8647a69e961c77f8b357ab55117292f8cae2fb25865b56aee4d143b31dfbc6ff
size 2497282496

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:50bba3a480d1a925a964d5bf073b37ff83c3cabccd95b0e5e9bf4566bd90487b
size 2383311296

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:47c2dce87212523d1f8a72c826bb93b09b90d9e43d4d4262231576bb15bc4a16
size 2889515456

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1c07256deb88d877932d6e042138a3953eac49715825f4244b46c709a6837faf
size 2823713216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b7eb83ffc770fe337f85a388d5d8d307d4919e1a23125287a87a71dfd2d4eccc
size 3306262976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5042393776bae4bcd4da81b3cb41cb877f640bd591c9fc6b6a7252d13f7f5d0c
size 4280406976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8407bfe43a03c73dc4fbbb64a3852a080bb3ee7c26518093f6d6489bb6bf7964
size 1143485376

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6e5de7c881d2951a96fc43859c8cc0247a32564b9b5434871eab3cb76023292d
size 1083151296

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:72a45b6d23c616c125ff10ea8a26e7dd3eba5201442122691d64e83009e56a43
size 1530403776

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e0a895a3862844ac6a19bcea7a2dd3b07ef6b0d40388a33907eff66c485872b3
size 1255347136

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9e33c4e70986cd8ab43716ec8f8f1147deb376b1d985ea6dd819f12dd1a1e8f3
size 1674378176

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:33cf23cf3f843c0f370ef926a7d7dd9b3f72a3ac22071e690c187d76cc826bd1
size 1695726016

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:515534b0a3d052d32ed93fb5b7273a6fc7ca0eeb022deb5fc4434a2056196929
size 2128386496

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:45de79dd53d3b12b869f98741a271fca54450dc8c43d5a81a9d35bc0782a51c4
size 2546342336

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ad6afca67cd34eef226f615c7687f983a9013c98d9bdf01130837653b801dec6
size 2899222976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2d4d1fa5b78570c888af94dee7c89784d971a73929fd5f3f4b9e8fcafc12991b
size 3658224576

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:80eb99f884e74898c587b95d90595d92c34c518ca972e3ea33e64615cde40531
size 5056701376

217
README.md Normal file
View File

@@ -0,0 +1,217 @@
---
tags:
- unsloth
base_model:
- Qwen/Qwen3-VL-4B-Instruct
license: apache-2.0
pipeline_tag: image-text-to-text
library_name: transformers
---
> [!NOTE]
> Includes Unsloth **chat template fixes**! <br> For `llama.cpp`, use `--jinja`
>
<div>
<p style="margin-top: 0;margin-bottom: 0;">
<em><a href="https://docs.unsloth.ai/basics/unsloth-dynamic-v2.0-gguf">Unsloth Dynamic 2.0</a> achieves superior accuracy & outperforms other leading quants.</em>
</p>
<div style="display: flex; gap: 5px; align-items: center; ">
<a href="https://github.com/unslothai/unsloth/">
<img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
</a>
<a href="https://discord.gg/unsloth">
<img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
</a>
<a href="https://docs.unsloth.ai/">
<img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
</a>
</div>
</div>
<a href="https://chat.qwenlm.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
# Qwen3-VL-4B-Instruct
Meet Qwen3-VL — the most powerful vision-language model in the Qwen series to date.
This generation delivers comprehensive upgrades across the board: superior text understanding & generation, deeper visual perception & reasoning, extended context length, enhanced spatial and video dynamics comprehension, and stronger agent interaction capabilities.
Available in Dense and MoE architectures that scale from edge to cloud, with Instruct and reasoningenhanced Thinking editions for flexible, ondemand deployment.
#### Key Enhancements:
* **Visual Agent**: Operates PC/mobile GUIs—recognizes elements, understands functions, invokes tools, completes tasks.
* **Visual Coding Boost**: Generates Draw.io/HTML/CSS/JS from images/videos.
* **Advanced Spatial Perception**: Judges object positions, viewpoints, and occlusions; provides stronger 2D grounding and enables 3D grounding for spatial reasoning and embodied AI.
* **Long Context & Video Understanding**: Native 256K context, expandable to 1M; handles books and hours-long video with full recall and second-level indexing.
* **Enhanced Multimodal Reasoning**: Excels in STEM/Math—causal analysis and logical, evidence-based answers.
* **Upgraded Visual Recognition**: Broader, higher-quality pretraining is able to “recognize everything”—celebrities, anime, products, landmarks, flora/fauna, etc.
* **Expanded OCR**: Supports 32 languages (up from 19); robust in low light, blur, and tilt; better with rare/ancient characters and jargon; improved long-document structure parsing.
* **Text Understanding on par with pure LLMs**: Seamless textvision fusion for lossless, unified comprehension.
#### Model Architecture Updates:
<p align="center">
<img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_arc.jpg" width="80%"/>
<p>
1. **Interleaved-MRoPE**: Fullfrequency allocation over time, width, and height via robust positional embeddings, enhancing longhorizon video reasoning.
2. **DeepStack**: Fuses multilevel ViT features to capture finegrained details and sharpen imagetext alignment.
3. **TextTimestamp Alignment:** Moves beyond TRoPE to precise, timestampgrounded event localization for stronger video temporal modeling.
This is the weight repository for Qwen3-VL-4B-Instruct.
---
## Model Performance
**Multimodal performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_vl_instruct.jpg)
**Pure text performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_text_instruct.jpg)
## Quickstart
Below, we provide simple examples to show how to use Qwen3-VL with 🤖 ModelScope and 🤗 Transformers.
The code of Qwen3-VL has been in the latest Hugging Face transformers and we advise you to build from source with command:
```
pip install git+https://github.com/huggingface/transformers
# pip install transformers==4.57.0 # currently, V4.57.0 is not released
```
### Using 🤗 Transformers to Chat
Here we show a code snippet to show how to use the chat model with `transformers`:
```python
from transformers import Qwen3VLForConditionalGeneration, AutoProcessor
# default: Load the model on the available device(s)
model = Qwen3VLForConditionalGeneration.from_pretrained(
"Qwen/Qwen3-VL-4B-Instruct", dtype="auto", device_map="auto"
)
# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
# model = Qwen3VLForConditionalGeneration.from_pretrained(
# "Qwen/Qwen3-VL-4B-Instruct",
# dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# device_map="auto",
# )
processor = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-4B-Instruct")
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
},
{"type": "text", "text": "Describe this image."},
],
}
]
# Preparation for inference
inputs = processor.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_dict=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
# Inference: Generation of the output
generated_ids = model.generate(**inputs, max_new_tokens=128)
generated_ids_trimmed = [
out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
]
output_text = processor.batch_decode(
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
)
print(output_text)
```
### Generation Hyperparameters
#### VL
```bash
export greedy='false'
export top_p=0.8
export top_k=20
export temperature=0.7
export repetition_penalty=1.0
export presence_penalty=1.5
export out_seq_length=16384
```
#### Text
```bash
export greedy='false'
export top_p=1.0
export top_k=40
export repetition_penalty=1.0
export presence_penalty=2.0
export temperature=1.0
export out_seq_length=32768
```
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}
@article{Qwen2VL,
title={Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution},
author={Wang, Peng and Bai, Shuai and Tan, Sinan and Wang, Shijie and Fan, Zhihao and Bai, Jinze and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Fan, Yang and Dang, Kai and Du, Mengfei and Ren, Xuancheng and Men, Rui and Liu, Dayiheng and Zhou, Chang and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2409.12191},
year={2024}
}
@article{Qwen-VL,
title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond},
author={Bai, Jinze and Bai, Shuai and Yang, Shusheng and Wang, Shijie and Tan, Sinan and Wang, Peng and Lin, Junyang and Zhou, Chang and Zhou, Jingren},
journal={arXiv preprint arXiv:2308.12966},
year={2023}
}
```

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:05272066a0551e676e87fb6e4999d306b9ac1c0df22f515104ce9f50d9c0dd0e
size 3872672

3
mmproj-BF16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cf594e140b277b4f4336c870c7ef151bbb07d32e34cc145a8713c6d817ab7cbd
size 839326368

3
mmproj-F16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:90bac17437025aedab5878cc0d0c0a8579c55c97e397a13c96b0b7f0138d6254
size 836180640

3
mmproj-F32.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:889bc4f49c995f67f17165f9fd68bef608a9310367e770e868b799191ab031a7
size 1661409952