初始化项目,由ModelHub XC社区提供模型

Model: unsloth/Qwen3-VL-2B-Instruct-1M-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-09-13 17:44:12 +08:00
commit 8ef3f25a5a
33 changed files with 386 additions and 0 deletions

78
.gitattributes vendored Normal file
View File

@@ -0,0 +1,78 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
imatrix_unsloth.gguf_file filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F32.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Instruct-1M-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:56448092a90b4d059817b5c2173da46150697cd7e5ede48abac0a699def93488
size 3447350880

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:92b8fc5ed8f8137ddef59577c62613d4e53e75eeb03f1b47e9345122d1437804
size 1054424928

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:aa273313d61fd3bcb6b3caf47e2941eaf0b20da27bbd63bf000918d0b386610a
size 1010384736

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:440ead6aafa675a58848f023afa557b97025ee6bf181e800bc049d7954f7093d
size 777797472

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:440ead6aafa675a58848f023afa557b97025ee6bf181e800bc049d7954f7093d
size 777797472

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c75daa3a520eeb48bdf3e4aa4e7927385f717b8c78ebf75c6f2f57bda8909109
size 939540320

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:62f0f69146e845ee47531bd3c923059bfaa63fec348fcb6128050f6467768683
size 867254112

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ea5bbcb58933e241570f3eda42c24f92b38550fbba02c152811bb16ed2d7533b
size 1056784224

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7b0b8f38a237a5898288e483aaecdfbcd347f0cb950e7bf95359fd12934de785
size 1142505312

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0e6ef13c6c41721be6db8ce624111c48a9f26d2592f445ff0e5abf8451d546ff
size 1107410784

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f4737543e8e042077c55fbdce8cca796cbf0ac7e839b8351881567433a7b1551
size 1060192096

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5ca6ec5e136450a5fe3479599d8e30749b5a89dfff5c8b190876745af90d0e93
size 1257881440

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6e488ca63a65633513ad1ee16deb7adb8c6ca188c23066d0bab2b207ba40ed3c
size 1230585696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:af538ed9bda6f5de78ebafbcea6b77235c905948509781ae416c10e241885b37
size 1417756512

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9b12c386a64b3f66233aff7e072062e726613f73e52e524c1a2e368194e215c1
size 1834428256

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d967a973ec4c988a4d4a87b75e9fc16b012a371a10cd49addf4cf79c05ad65af
size 561948512

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9c218484334d1c6c1e576f396ea540f72f9d0fe74ba1d7bfad27fc4f95220302
size 537831264

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:75e2aa59e3d229f872d132e79e7eee913c5b63223d488ea3b39dfcc4df25292e
size 708716384

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2863f969fa71e2301decaecc9e232859e4549bdad038a0ad507ad4bcc317fe4e
size 605792096

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:36840551e5cc89616fbcfc831756588671692252efa52ae72c0a10207f514b0b
size 765175648

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8eff6c7f49c9762aa4da38098eefcec1cd346fc7e3d6441919f1f961bd4c5e29
size 797867872

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f326a16228ed3ce1fda0fb34c1e7f43d41b344fe90728162b9806b845fb70430
size 968900448

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e878b4eb502fd48bddefb2de3427149b90e3f4d2cfbb25c76a8eb34a0a39ae17
size 1129709408

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:dd7ea6888e662c534d34b9fc70894cdacd53912b319c2100523c4e774dab9215
size 1261322080

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:76d15a0e5ff18dfcfa398a0457dfd3a05e3d22b32725770c6809c2fbadcd441a
size 1610950496

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b47edcf5fcd25ecdc0008a10164604c73bddf1d45c1fca3e98c561fd7888edf3
size 2332583776

217
README.md Normal file
View File

@@ -0,0 +1,217 @@
---
tags:
- unsloth
base_model:
- Qwen/Qwen3-VL-2B-Instruct
license: apache-2.0
pipeline_tag: image-text-to-text
library_name: transformers
---
> [!NOTE]
> Includes Unsloth **chat template fixes**! <br> For `llama.cpp`, use `--jinja`
>
<div>
<p style="margin-top: 0;margin-bottom: 0;">
<em><a href="https://docs.unsloth.ai/basics/unsloth-dynamic-v2.0-gguf">Unsloth Dynamic 2.0</a> achieves superior accuracy & outperforms other leading quants.</em>
</p>
<div style="display: flex; gap: 5px; align-items: center; ">
<a href="https://github.com/unslothai/unsloth/">
<img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
</a>
<a href="https://discord.gg/unsloth">
<img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
</a>
<a href="https://docs.unsloth.ai/">
<img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
</a>
</div>
</div>
<a href="https://huggingface.co/spaces/akhaliq/Qwen3-VL-2B-Instruct" target="_blank" style="margin: 2px;">
<img alt="Demo" src="https://img.shields.io/badge/Demo-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
# Qwen3-VL-2B-Instruct
Meet Qwen3-VL — the most powerful vision-language model in the Qwen series to date.
This generation delivers comprehensive upgrades across the board: superior text understanding & generation, deeper visual perception & reasoning, extended context length, enhanced spatial and video dynamics comprehension, and stronger agent interaction capabilities.
Available in Dense and MoE architectures that scale from edge to cloud, with Instruct and reasoningenhanced Thinking editions for flexible, ondemand deployment.
#### Key Enhancements:
* **Visual Agent**: Operates PC/mobile GUIs—recognizes elements, understands functions, invokes tools, completes tasks.
* **Visual Coding Boost**: Generates Draw.io/HTML/CSS/JS from images/videos.
* **Advanced Spatial Perception**: Judges object positions, viewpoints, and occlusions; provides stronger 2D grounding and enables 3D grounding for spatial reasoning and embodied AI.
* **Long Context & Video Understanding**: Native 256K context, expandable to 1M; handles books and hours-long video with full recall and second-level indexing.
* **Enhanced Multimodal Reasoning**: Excels in STEM/Math—causal analysis and logical, evidence-based answers.
* **Upgraded Visual Recognition**: Broader, higher-quality pretraining is able to “recognize everything”—celebrities, anime, products, landmarks, flora/fauna, etc.
* **Expanded OCR**: Supports 32 languages (up from 19); robust in low light, blur, and tilt; better with rare/ancient characters and jargon; improved long-document structure parsing.
* **Text Understanding on par with pure LLMs**: Seamless textvision fusion for lossless, unified comprehension.
#### Model Architecture Updates:
<p align="center">
<img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_arc.jpg" width="80%"/>
<p>
1. **Interleaved-MRoPE**: Fullfrequency allocation over time, width, and height via robust positional embeddings, enhancing longhorizon video reasoning.
2. **DeepStack**: Fuses multilevel ViT features to capture finegrained details and sharpen imagetext alignment.
3. **TextTimestamp Alignment:** Moves beyond TRoPE to precise, timestampgrounded event localization for stronger video temporal modeling.
This is the weight repository for Qwen3-VL-2B-Instruct.
---
## Model Performance
**Multimodal performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_2b_32b_vl_instruct.jpg)
**Pure text performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_2b_32b_text_instruct.jpg)
## Quickstart
Below, we provide simple examples to show how to use Qwen3-VL with 🤖 ModelScope and 🤗 Transformers.
The code of Qwen3-VL has been in the latest Hugging Face transformers and we advise you to build from source with command:
```
pip install git+https://github.com/huggingface/transformers
# pip install transformers==4.57.0 # currently, V4.57.0 is not released
```
### Using 🤗 Transformers to Chat
Here we show a code snippet to show how to use the chat model with `transformers`:
```python
from transformers import Qwen3VLForConditionalGeneration, AutoProcessor
# default: Load the model on the available device(s)
model = Qwen3VLForConditionalGeneration.from_pretrained(
"Qwen/Qwen3-VL-2B-Instruct", dtype="auto", device_map="auto"
)
# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
# model = Qwen3VLForConditionalGeneration.from_pretrained(
# "Qwen/Qwen3-VL-2B-Instruct",
# dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# device_map="auto",
# )
processor = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-2B-Instruct")
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
},
{"type": "text", "text": "Describe this image."},
],
}
]
# Preparation for inference
inputs = processor.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_dict=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
# Inference: Generation of the output
generated_ids = model.generate(**inputs, max_new_tokens=128)
generated_ids_trimmed = [
out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
]
output_text = processor.batch_decode(
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
)
print(output_text)
```
### Generation Hyperparameters
#### VL
```bash
export greedy='false'
export top_p=0.8
export top_k=20
export temperature=0.7
export repetition_penalty=1.0
export presence_penalty=1.5
export out_seq_length=16384
```
#### Text
```bash
export greedy='false'
export top_p=1.0
export top_k=40
export repetition_penalty=1.0
export presence_penalty=2.0
export temperature=1.0
export out_seq_length=32768
```
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}
@article{Qwen2VL,
title={Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution},
author={Wang, Peng and Bai, Shuai and Tan, Sinan and Wang, Shijie and Fan, Zhihao and Bai, Jinze and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Fan, Yang and Dang, Kai and Du, Mengfei and Ren, Xuancheng and Men, Rui and Liu, Dayiheng and Zhou, Chang and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2409.12191},
year={2024}
}
@article{Qwen-VL,
title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond},
author={Bai, Jinze and Bai, Shuai and Yang, Shusheng and Wang, Shijie and Tan, Sinan and Wang, Peng and Lin, Junyang and Zhou, Chang and Zhou, Jingren},
journal={arXiv preprint arXiv:2308.12966},
year={2023}
}
```

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0189a223986310f6efdfbc2b1b9160bd64a00e41da814471488d368733e5d9bb
size 2094592

3
mmproj-BF16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:94b834ee8fbaf50730240199ea40db2f41c6ba3d4ee1e5fbf3ed17be34fbf8c2
size 822540960

3
mmproj-F16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ee8af4e9a983e4f690b3969bf61f4d36b3f3647285e56a1e7b6f6de9b41ea499
size 819395232

3
mmproj-F32.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:95f0e4800b77d4697710c340a82e281a9eb06237bac04e00c9f940b5ce6f62fd
size 1627847328