初始化项目,由ModelHub XC社区提供模型

Model: unsloth/Qwen3-VL-8B-Thinking-1M-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-06-30 20:56:12 +08:00
commit 74426c5dc5
33 changed files with 386 additions and 0 deletions

78
.gitattributes vendored Normal file
View File

@@ -0,0 +1,78 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text
imatrix_unsloth.gguf_file filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F32.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-BF16.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Thinking-1M-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-BF16.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:01abfc9c7674e491be251d9e1e63cb9c4e7594cd684972fe5790217933c69f6a
size 16388046016

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:44bd5e2757cec85fcd3cb5c10d71e1813bc64b67cc820001526e08c1948f79a6
size 4793626080

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:70aca37bf1841abea6e2af0a683d4ee742ca478b5f627ae18416c10a28293b6a
size 4581289440

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cec1c76b8bf3320a023b31e089e57e009e2dab858ebdb798764a81b2d493d565
size 3281735136

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:bfd1cb316051052c25a512a2ae223437f6589b90f66d1ad6f856e197376035c5
size 3427593696

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:282523d239b9dc785fde8a7f9e1939688ed1c49accdacae1a6494d1d61562a5c
size 4124163552

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0b3ce9cdfb32fbc7cd9d9f6aafce7481fd741f7a9122e62ce43dde03e8a1a744
size 3769613792

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2954c5e658551e0a9d3c8281ce29cd4cddb23032852a38953ecaba491923eb7b
size 4787334624

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:773b105d5e55c2c4a5392a999fd60a920c4d2574d1426db4aac0e108394e4093
size 5247757792

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5ee4cba0e2474bc27972cfeeb611aedfa15f1e383f92f15f74c44608978311fb
size 5027786208

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:54170ec98dbb74786d6f526aa5a37083cfc24b8552840f00fc765ea2ed596b72
size 4802014688

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a7c8948556d5be1d3818e9532bf8c32158e7124106e753b391093febdb463f4d
size 5851114976

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:cbb8bcfc3cd71567288043aa2eb2895afce90ab663192eefeb48cec524d64e73
size 5720763872

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9a45b0271ec7cfbf775e0869050cb3124af94a2a9b70b4d7b5cb7d52a3ee1c91
size 6725901792

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:e368ff03c14f57ff2360efc4c86ea1e5882f1a8db8d1418697a04b4bb2e8d38c
size 8709520864

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:20101fd22b963bb986b343e64c5eaca41ea3fda6b1abb104e09ffa07f75d1fa5
size 2396491232

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:6659916eaa2fe86045209f5b020d137f5e7e2263d0520b2b000f5db6a9409beb
size 2275380704

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:23fbeba885f0639f19daa8a82749dca7b5466df14adeecea64f9402d8ff2f479
size 3110899168

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fb191a98a2f406f8c62788b73c3f0c98476e740c29e2551199fd81bdf4b4533b
size 2605223392

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:5000792c3b1b1dbea193536e4f0601ae6813d3457986274b301153f3d449d709
size 3410267616

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:989e528aabbc77ad728eadbed90ec8f3e4a8787205760b85f0d4d6cd2744aa09
size 3501977056

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8dcec735639021e45dbf896f548088b1d70d26a0fed2177a91271ed543ae8dff
size 4307054048

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a5490f8eed6212ccfd693598afc50900d71766ed9e348dfcff4d445a88f8defe
size 5148700128

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:df2f30e68d1a8441e7a8457d735d5c588e5bbb038b53a467a120b44ea73c4235
size 5884767712

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ee9c6f5b2194bc43039acedf566b4e6f5010bd02122fae33793fc2b8783a71fb
size 7490551264

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b09ee59fc208f2819856df0ac1c817637edc8998e02a692c0475683411577148
size 10824039904

217
README.md Normal file
View File

@@ -0,0 +1,217 @@
---
tags:
- unsloth
base_model:
- Qwen/Qwen3-VL-8B-Thinking
license: apache-2.0
pipeline_tag: image-text-to-text
---
> [!NOTE]
> Includes Unsloth **chat template fixes**! <br> For `llama.cpp`, use `--jinja`
>
<div>
<p style="margin-top: 0;margin-bottom: 0;">
<em><a href="https://docs.unsloth.ai/basics/unsloth-dynamic-v2.0-gguf">Unsloth Dynamic 2.0</a> achieves superior accuracy & outperforms other leading quants.</em>
</p>
<div style="display: flex; gap: 5px; align-items: center; ">
<a href="https://github.com/unslothai/unsloth/">
<img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
</a>
<a href="https://discord.gg/unsloth">
<img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
</a>
<a href="https://docs.unsloth.ai/">
<img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
</a>
</div>
</div>
<a href="https://chat.qwenlm.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
# Qwen3-VL-8B-Thinking
Meet Qwen3-VL — the most powerful vision-language model in the Qwen series to date.
This generation delivers comprehensive upgrades across the board: superior text understanding & generation, deeper visual perception & reasoning, extended context length, enhanced spatial and video dynamics comprehension, and stronger agent interaction capabilities.
Available in Dense and MoE architectures that scale from edge to cloud, with Instruct and reasoningenhanced Thinking editions for flexible, ondemand deployment.
#### Key Enhancements:
* **Visual Agent**: Operates PC/mobile GUIs—recognizes elements, understands functions, invokes tools, completes tasks.
* **Visual Coding Boost**: Generates Draw.io/HTML/CSS/JS from images/videos.
* **Advanced Spatial Perception**: Judges object positions, viewpoints, and occlusions; provides stronger 2D grounding and enables 3D grounding for spatial reasoning and embodied AI.
* **Long Context & Video Understanding**: Native 256K context, expandable to 1M; handles books and hours-long video with full recall and second-level indexing.
* **Enhanced Multimodal Reasoning**: Excels in STEM/Math—causal analysis and logical, evidence-based answers.
* **Upgraded Visual Recognition**: Broader, higher-quality pretraining is able to “recognize everything”—celebrities, anime, products, landmarks, flora/fauna, etc.
* **Expanded OCR**: Supports 32 languages (up from 19); robust in low light, blur, and tilt; better with rare/ancient characters and jargon; improved long-document structure parsing.
* **Text Understanding on par with pure LLMs**: Seamless textvision fusion for lossless, unified comprehension.
#### Model Architecture Updates:
<p align="center">
<img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_arc.jpg" width="80%"/>
<p>
1. **Interleaved-MRoPE**: Fullfrequency allocation over time, width, and height via robust positional embeddings, enhancing longhorizon video reasoning.
2. **DeepStack**: Fuses multilevel ViT features to capture finegrained details and sharpen imagetext alignment.
3. **TextTimestamp Alignment:** Moves beyond TRoPE to precise, timestampgrounded event localization for stronger video temporal modeling.
This is the weight repository for Qwen3-VL-8B-Thinking.
---
## Model Performance
**Multimodal performance**
![](https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_vl_thinking.jpg)
**Pure text performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_text_thinking.jpg)
## Quickstart
Below, we provide simple examples to show how to use Qwen3-VL with 🤖 ModelScope and 🤗 Transformers.
The code of Qwen3-VL has been in the latest Hugging face transformers and we advise you to build from source with command:
```
pip install git+https://github.com/huggingface/transformers
# pip install transformers==4.57.0 # currently, V4.57.0 is not released
```
### Using 🤗 Transformers to Chat
Here we show a code snippet to show you how to use the chat model with `transformers`:
```python
from transformers import Qwen3VLForConditionalGeneration, AutoProcessor
# default: Load the model on the available device(s)
model = Qwen3VLForConditionalGeneration.from_pretrained(
"Qwen/Qwen3-VL-8B-Thinking", dtype="auto", device_map="auto"
)
# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
# model = Qwen3VLForConditionalGeneration.from_pretrained(
# "Qwen/Qwen3-VL-8B-Thinking",
# dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# device_map="auto",
# )
processor = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-8B-Thinking")
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
},
{"type": "text", "text": "Describe this image."},
],
}
]
# Preparation for inference
inputs = processor.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_dict=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
# Inference: Generation of the output
generated_ids = model.generate(**inputs, max_new_tokens=128)
generated_ids_trimmed = [
out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
]
output_text = processor.batch_decode(
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
)
print(output_text)
```
### Generation Hyperparameters
#### VL
```bash
export greedy='false'
export top_p=0.95
export top_k=20
export repetition_penalty=1.0
export presence_penalty=0.0
export temperature=1.0
export out_seq_length=40960
```
#### Text
```bash
export greedy='false'
export top_p=0.95
export top_k=20
export repetition_penalty=1.0
export presence_penalty=1.5
export temperature=1.0
export out_seq_length=32768 (for aime, lcb, and gpqa, it is recommended to set to 81920)
```
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}
@article{Qwen2VL,
title={Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution},
author={Wang, Peng and Bai, Shuai and Tan, Sinan and Wang, Shijie and Fan, Zhihao and Bai, Jinze and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Fan, Yang and Dang, Kai and Du, Mengfei and Ren, Xuancheng and Men, Rui and Liu, Dayiheng and Zhou, Chang and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2409.12191},
year={2024}
}
@article{Qwen-VL,
title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond},
author={Bai, Jinze and Bai, Shuai and Yang, Shusheng and Wang, Shijie and Tan, Sinan and Wang, Peng and Lin, Junyang and Zhou, Chang and Zhou, Jingren},
journal={arXiv preprint arXiv:2308.12966},
year={2023}
}
```

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4c55d4ce671c0409ae7534b86aba6dd0a8629875b1ab61f041f3280d16f68b18
size 5347232

3
mmproj-BF16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:15ea38c2e158682e7a2d1241fa512a373b2265b69b2ce71f1fe1353f55787341
size 1162569280

3
mmproj-F16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2749b039eed112d162ca96d366ae70c3dfe8ca3f56eaec3448af75d224997d47
size 1159030336

3
mmproj-F32.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:9de3b112a64176fd45554a76a529f58cf83f8e9c46e1a999bf01ae04155c036d
size 2305574464