初始化项目,由ModelHub XC社区提供模型

Model: unsloth/Qwen3-VL-2B-Thinking-1M-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-07-29 22:46:13 +08:00
commit 91eb98bbf1
33 changed files with 386 additions and 0 deletions

78
.gitattributes vendored Normal file
View File

@@ -0,0 +1,78 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
imatrix_unsloth.gguf_file filter=lfs diff=lfs merge=lfs -text
mmproj-F32.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-2B-Thinking-1M-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5e33088aecb4fc1f2b7afb0d165ac02cd6d57c82791cbb3f95fb6fcbae6d2306
size 3447351360

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8514a02a089b5373006ed29a17bcac9efa8cb1923ec7170c2549f98c278d42e7
size 1054425408

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b54ff80dc7c0b2a41e2b31c5b8b50c99345b3f89185a4a7bf88e1d9cfc80a0be
size 1010385216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6d17d50bf59f251f5f24285483a27fa637e1ebc8e00ca7d57c386dc89d4376f8
size 777797952

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6d17d50bf59f251f5f24285483a27fa637e1ebc8e00ca7d57c386dc89d4376f8
size 777797952

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:92665f90249c1ba05eef55b39836c60fc2b802a7eec4fa98e78fb74c7651942f
size 939540800

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c0ed3c4fed77a1ce0cb822b0d306774a64f230995209bd99a3cec9ec73b8484b
size 867254592

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:53a6247630b80b9410772b07db9e1c30e63fb4ca470732445fbc5c6d04c201e5
size 1056784704

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:65dd811cb3a3abb44ca6c402be83e609c981243ef1ffc0f468c2320bf3bb9712
size 1142505792

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9cc3f9bc29979fa36582b65b4f12ff9172d4bc79b776b0c0ce20c1f52ca26d3f
size 1107411264

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:401b92a06aecbfd177e10d9dccb6a6bf94f3fbebdff254fd71daae75b7c55fe8
size 1060192576

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d874494ba77041c4daf083ab4cd0dbe928d4ac8409190713b79cbbd2401d12ee
size 1257881920

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b9fb8c7db37d3000fc4738c28b1ae781a146e91c3255b2431c4e7ee3b652831c
size 1230586176

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:000a8a1655e643df381ba2e4f78106a911cb3370d655d2e720c1d9f289d07078
size 1417756992

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f5615ed62a5aab9a99ede0670127658625250922cd6535ab2dbe1c2b9653ea8d
size 1834428736

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2106644b61b6123d735801926bd501eb4b0755244b41fe7eb82fc41bf2cf60be
size 561948992

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a7d3d1ba3d6106f17808ae80d5f5d71a910011478f7d9a634a72216e4189c191
size 537831744

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5adbb324c713d17cb682cbe8a2effe03c05d90784f170e9c468348d132343572
size 708716864

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0fb8d26bb5c2f197f2bc5af93da5507a8552b599238fde6e86c7fc416bc38cad
size 605792576

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1e0a2ae180f878689a25d991e2fc479f40ee3fc08c1ee3b6d21255b7cec9f830
size 765176128

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:47ff9f29ff6ecde550b5a490d76d92c4899d4897f0b7726951b5354f79bb8428
size 797868352

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f774406a7d73c72776f093a16036fd4c75536bae3c33151407596df668d53d60
size 968900928

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:57a58b4c98fdea679fae3da89fdce1a06714a22c4419142cf5db0e8c34944dc5
size 1132953920

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:41ac8785f6e2738f3a244e16ce8611d7836cda16d0ce3bb3b47a73e4943669a3
size 1262993728

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e5a91a894e92114310b3accccdfde2ce7342de4d4ea72ae28aba10e756a376b8
size 1610950976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:226585a8fc77af077cfa66d0c5c85126dce9bab423ad3b965763a3bf1874216f
size 2332584256

217
README.md Normal file
View File

@@ -0,0 +1,217 @@
---
tags:
- unsloth
base_model:
- Qwen/Qwen3-VL-2B-Thinking
license: apache-2.0
pipeline_tag: image-text-to-text
library_name: transformers
---
> [!NOTE]
> Includes Unsloth **chat template fixes**! <br> For `llama.cpp`, use `--jinja`
>
<div>
<p style="margin-top: 0;margin-bottom: 0;">
<em><a href="https://docs.unsloth.ai/basics/unsloth-dynamic-v2.0-gguf">Unsloth Dynamic 2.0</a> achieves superior accuracy & outperforms other leading quants.</em>
</p>
<div style="display: flex; gap: 5px; align-items: center; ">
<a href="https://github.com/unslothai/unsloth/">
<img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
</a>
<a href="https://discord.gg/unsloth">
<img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
</a>
<a href="https://docs.unsloth.ai/">
<img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
</a>
</div>
</div>
<a href="https://chat.qwenlm.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
# Qwen3-VL-2B-Thinking
Meet Qwen3-VL — the most powerful vision-language model in the Qwen series to date.
This generation delivers comprehensive upgrades across the board: superior text understanding & generation, deeper visual perception & reasoning, extended context length, enhanced spatial and video dynamics comprehension, and stronger agent interaction capabilities.
Available in Dense and MoE architectures that scale from edge to cloud, with Instruct and reasoningenhanced Thinking editions for flexible, ondemand deployment.
#### Key Enhancements:
* **Visual Agent**: Operates PC/mobile GUIs—recognizes elements, understands functions, invokes tools, completes tasks.
* **Visual Coding Boost**: Generates Draw.io/HTML/CSS/JS from images/videos.
* **Advanced Spatial Perception**: Judges object positions, viewpoints, and occlusions; provides stronger 2D grounding and enables 3D grounding for spatial reasoning and embodied AI.
* **Long Context & Video Understanding**: Native 256K context, expandable to 1M; handles books and hours-long video with full recall and second-level indexing.
* **Enhanced Multimodal Reasoning**: Excels in STEM/Math—causal analysis and logical, evidence-based answers.
* **Upgraded Visual Recognition**: Broader, higher-quality pretraining is able to “recognize everything”—celebrities, anime, products, landmarks, flora/fauna, etc.
* **Expanded OCR**: Supports 32 languages (up from 19); robust in low light, blur, and tilt; better with rare/ancient characters and jargon; improved long-document structure parsing.
* **Text Understanding on par with pure LLMs**: Seamless textvision fusion for lossless, unified comprehension.
#### Model Architecture Updates:
<p align="center">
<img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_arc.jpg" width="80%"/>
<p>
1. **Interleaved-MRoPE**: Fullfrequency allocation over time, width, and height via robust positional embeddings, enhancing longhorizon video reasoning.
2. **DeepStack**: Fuses multilevel ViT features to capture finegrained details and sharpen imagetext alignment.
3. **TextTimestamp Alignment:** Moves beyond TRoPE to precise, timestampgrounded event localization for stronger video temporal modeling.
This is the weight repository for Qwen3-VL-2B-Thinking.
---
## Model Performance
**Multimodal performance**
![](https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen3-VL/qwen3vl_2b_32b_vl_thinking.jpg)
**Pure text performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_2b_32b_text_thinking.jpg)
## Quickstart
Below, we provide simple examples to show how to use Qwen3-VL with 🤖 ModelScope and 🤗 Transformers.
The code of Qwen3-VL has been in the latest Hugging face transformers and we advise you to build from source with command:
```
pip install git+https://github.com/huggingface/transformers
# pip install transformers==4.57.0 # currently, V4.57.0 is not released
```
### Using 🤗 Transformers to Chat
Here we show a code snippet to show you how to use the chat model with `transformers`:
```python
from transformers import Qwen3VLForConditionalGeneration, AutoProcessor
# default: Load the model on the available device(s)
model = Qwen3VLForConditionalGeneration.from_pretrained(
"Qwen/Qwen3-VL-2B-Thinking", dtype="auto", device_map="auto"
)
# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
# model = Qwen3VLForConditionalGeneration.from_pretrained(
# "Qwen/Qwen3-VL-2B-Thinking",
# dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# device_map="auto",
# )
processor = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-2B-Thinking")
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
},
{"type": "text", "text": "Describe this image."},
],
}
]
# Preparation for inference
inputs = processor.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_dict=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
# Inference: Generation of the output
generated_ids = model.generate(**inputs, max_new_tokens=128)
generated_ids_trimmed = [
out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
]
output_text = processor.batch_decode(
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
)
print(output_text)
```
### Generation Hyperparameters
#### VL
```bash
export greedy='false'
export top_p=0.95
export top_k=20
export repetition_penalty=1.0
export presence_penalty=0.0
export temperature=1.0
export out_seq_length=40960
```
#### Text
```bash
export greedy='false'
export top_p=0.95
export top_k=20
export repetition_penalty=1.0
export presence_penalty=1.5
export temperature=1.0
export out_seq_length=32768 (for aime, lcb, and gpqa, it is recommended to set to 81920)
```
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}
@article{Qwen2VL,
title={Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution},
author={Wang, Peng and Bai, Shuai and Tan, Sinan and Wang, Shijie and Fan, Zhihao and Bai, Jinze and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Fan, Yang and Dang, Kai and Du, Mengfei and Ren, Xuancheng and Men, Rui and Liu, Dayiheng and Zhou, Chang and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2409.12191},
year={2024}
}
@article{Qwen-VL,
title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond},
author={Bai, Jinze and Bai, Shuai and Yang, Shusheng and Wang, Shijie and Tan, Sinan and Wang, Peng and Lin, Junyang and Zhou, Chang and Zhou, Jingren},
journal={arXiv preprint arXiv:2308.12966},
year={2023}
}
```

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2410de491ac07d8510671fbecc754b1377d299831e2c195a79ad31512aae64b7
size 2094592

3
mmproj-BF16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:15df250b894cc8e5ace0b7b0065c69e1a9333eabd70bb6bec03ef361310b1272
size 822540960

3
mmproj-F16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:64f67101ba66b6ec5c2f4abb9f889a86f3c02e8e6efd1d2133daf503518c6cc4
size 819395232

3
mmproj-F32.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:921c77c5ec1ec39a0342afe08145bbdb0464db4ee6c2f500284e8bb41401b0a1
size 1627847328