初始化项目,由ModelHub XC社区提供模型

Model: unsloth/Qwen3-VL-8B-Instruct-1M-GGUF
Source: Original Platform
This commit is contained in:
ModelHub XC
2026-08-08 20:09:12 +08:00
commit 8f25fb3e11
33 changed files with 386 additions and 0 deletions

78
.gitattributes vendored Normal file
View File

@@ -0,0 +1,78 @@
*.7z filter=lfs diff=lfs merge=lfs -text
*.arrow filter=lfs diff=lfs merge=lfs -text
*.bin filter=lfs diff=lfs merge=lfs -text
*.bin.* filter=lfs diff=lfs merge=lfs -text
*.bz2 filter=lfs diff=lfs merge=lfs -text
*.ftz filter=lfs diff=lfs merge=lfs -text
*.gz filter=lfs diff=lfs merge=lfs -text
*.h5 filter=lfs diff=lfs merge=lfs -text
*.joblib filter=lfs diff=lfs merge=lfs -text
*.lfs.* filter=lfs diff=lfs merge=lfs -text
*.model filter=lfs diff=lfs merge=lfs -text
*.msgpack filter=lfs diff=lfs merge=lfs -text
*.onnx filter=lfs diff=lfs merge=lfs -text
*.parquet filter=lfs diff=lfs merge=lfs -text
*.pb filter=lfs diff=lfs merge=lfs -text
*.pt filter=lfs diff=lfs merge=lfs -text
*.pth filter=lfs diff=lfs merge=lfs -text
*.rar filter=lfs diff=lfs merge=lfs -text
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.tar.* filter=lfs diff=lfs merge=lfs -text
*.tflite filter=lfs diff=lfs merge=lfs -text
*.tgz filter=lfs diff=lfs merge=lfs -text
*.xz filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zstandard filter=lfs diff=lfs merge=lfs -text
*.tfevents* filter=lfs diff=lfs merge=lfs -text
*.db* filter=lfs diff=lfs merge=lfs -text
*.ark* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*data* filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.meta filter=lfs diff=lfs merge=lfs -text
**/*ckpt*.index filter=lfs diff=lfs merge=lfs -text
*.safetensors filter=lfs diff=lfs merge=lfs -text
*.ckpt filter=lfs diff=lfs merge=lfs -text
*.ggml filter=lfs diff=lfs merge=lfs -text
*.llamafile* filter=lfs diff=lfs merge=lfs -text
*.pt2 filter=lfs diff=lfs merge=lfs -text
*.mlmodel filter=lfs diff=lfs merge=lfs -text
*.npy filter=lfs diff=lfs merge=lfs -text
*.npz filter=lfs diff=lfs merge=lfs -text
*.pickle filter=lfs diff=lfs merge=lfs -text
*.pkl filter=lfs diff=lfs merge=lfs -text
*.tar filter=lfs diff=lfs merge=lfs -text
*.wasm filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q3_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-IQ4_NL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-IQ3_XXS.gguf filter=lfs diff=lfs merge=lfs -text
imatrix_unsloth.gguf_file filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q4_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q4_1.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q3_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q5_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-IQ4_XS.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q5_K_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q4_K_S.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q2_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q2_K_L.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q5_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q4_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-IQ1_S.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-BF16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q6_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q3_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F16.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-IQ2_M.gguf filter=lfs diff=lfs merge=lfs -text
mmproj-F32.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q8_0.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-Q6_K.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-IQ1_M.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q8_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-Q2_K_XL.gguf filter=lfs diff=lfs merge=lfs -text
Qwen3-VL-8B-Instruct-1M-UD-IQ2_XXS.gguf filter=lfs diff=lfs merge=lfs -text

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:4897c9bacbbe084fc3b4e20d9bccf9bd42e1b939ee25abe3659c976c155849ac
size 16388045568

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a2e6bae5ca93f8269911cddc31299a416caf651483215606389dd9ac16929248
size 4793625600

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:07f877b56e95f51f20289fa1a70e4e00d9fb8fdf482538ea5d15df47cae79056
size 4581288960

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:842224d378b1ec78ce22dd0c3325f62aefeda226772e7c031a378d00c1e4b6e6
size 3281734656

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a3e7425666bce763ff479f7f380b490634867b9fef6d4ad2787fb14e8750e13e
size 3427593216

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1345b08462b0d647f963de719361f50ea39ca5d569f67ccb5d282d216afd7d3c
size 4124163072

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d895fffcf6273694e880410b5b79e246b39da4fb042328d70ad2361fe00045b8
size 3769613312

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:2797468ae2d9583c31b9e70260fc7b2421170bfbd1712e50620c4460837e6131
size 4787334144

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:03833075960d9c14218a3e0adce9b6fa7c42ec87ea5cc0f70f4aad419577f517
size 5247757312

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:0848c5094555a7f94bd54ceee7440b4c1813f0be376cc4db589037ab93ff1424
size 5027785728

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:65e5cb76c0c598aec3ab8e14078b23490697be13394e4b904d17249af1f8a7ad
size 4802014208

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:75ee88c9d18e6120a09ae9167a223fc5bc1864a9edb003370e7cb36344023215
size 5851114496

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:fe4f0c7f9f3b601f46f5d1844d6bdd10bea6a6505539c029bb9613c18535ba32
size 5720763392

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d006ad800dca31efd1f9927e1bae9b3924ef6fefa4eb6f8e791e7eec2b80c4f7
size 6725901312

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c681b394d300bf279bf1ad8e5cc7fdfd574b96f666141c6e13f41ba9f77d30c9
size 8709520384

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f62cf356e350d07a9eb033c89e2ab4c89fb53d27bcee17a67f268d769f114ae8
size 2396490752

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:41b21fd101a30e875b97447146b748612d1249bf23acd6b3a35a42885e4a4df9
size 2275380224

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7cae1380eb4a679268b741fd27c1c3277dc2c6ea1e6c4398d4ee720e3c7549e3
size 3110898688

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:b822b0be24ce9abeb712821c9615ded722c1864e6a5bc194c477289e749c3a29
size 2605222912

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:7481eb33890c59a5d7b02cb8adb03242ea96fd25a06286b3d5c3d5617257cf29
size 3410267136

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:47869476dfcafa665b467a0ad9a8621cf5a12e1b3baaa34f9eb9af7fcc79a833
size 3501976576

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:ecf3e53f3b756d5c4279e7a919ac90c65ac3e4d21950b8346710c84dcd8b812f
size 4307053568

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:3da22ca47db302edd8761119c1deb7e41bb9b0e621691aea905e41f7a8d6222f
size 5135723520

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:1100620ee3c9921a993c545ebf10824ecaed627e2e5320a79ec1a84f029c35c0
size 5878082560

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:638eadb461d77e750b73d5134cf7429e484a3e831a0cc7dc63668ce3aad464e9
size 7490550784

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:94d2056dd3e712a70404f02f6a37929cf379a3a65e7c601c08a6471e0eda583e
size 10824039424

217
README.md Normal file
View File

@@ -0,0 +1,217 @@
---
tags:
- unsloth
base_model:
- Qwen/Qwen3-VL-8B-Instruct
license: apache-2.0
pipeline_tag: image-text-to-text
library_name: transformers
---
> [!NOTE]
> Includes Unsloth **chat template fixes**! <br> For `llama.cpp`, use `--jinja`
>
<div>
<p style="margin-top: 0;margin-bottom: 0;">
<em><a href="https://docs.unsloth.ai/basics/unsloth-dynamic-v2.0-gguf">Unsloth Dynamic 2.0</a> achieves superior accuracy & outperforms other leading quants.</em>
</p>
<div style="display: flex; gap: 5px; align-items: center; ">
<a href="https://github.com/unslothai/unsloth/">
<img src="https://github.com/unslothai/unsloth/raw/main/images/unsloth%20new%20logo.png" width="133">
</a>
<a href="https://discord.gg/unsloth">
<img src="https://github.com/unslothai/unsloth/raw/main/images/Discord%20button.png" width="173">
</a>
<a href="https://docs.unsloth.ai/">
<img src="https://raw.githubusercontent.com/unslothai/unsloth/refs/heads/main/images/documentation%20green%20button.png" width="143">
</a>
</div>
</div>
<a href="https://chat.qwenlm.ai/" target="_blank" style="margin: 2px;">
<img alt="Chat" src="https://img.shields.io/badge/%F0%9F%92%9C%EF%B8%8F%20Qwen%20Chat%20-536af5" style="display: inline-block; vertical-align: middle;"/>
</a>
# Qwen3-VL-8B-Instruct
Meet Qwen3-VL — the most powerful vision-language model in the Qwen series to date.
This generation delivers comprehensive upgrades across the board: superior text understanding & generation, deeper visual perception & reasoning, extended context length, enhanced spatial and video dynamics comprehension, and stronger agent interaction capabilities.
Available in Dense and MoE architectures that scale from edge to cloud, with Instruct and reasoningenhanced Thinking editions for flexible, ondemand deployment.
#### Key Enhancements:
* **Visual Agent**: Operates PC/mobile GUIs—recognizes elements, understands functions, invokes tools, completes tasks.
* **Visual Coding Boost**: Generates Draw.io/HTML/CSS/JS from images/videos.
* **Advanced Spatial Perception**: Judges object positions, viewpoints, and occlusions; provides stronger 2D grounding and enables 3D grounding for spatial reasoning and embodied AI.
* **Long Context & Video Understanding**: Native 256K context, expandable to 1M; handles books and hours-long video with full recall and second-level indexing.
* **Enhanced Multimodal Reasoning**: Excels in STEM/Math—causal analysis and logical, evidence-based answers.
* **Upgraded Visual Recognition**: Broader, higher-quality pretraining is able to “recognize everything”—celebrities, anime, products, landmarks, flora/fauna, etc.
* **Expanded OCR**: Supports 32 languages (up from 19); robust in low light, blur, and tilt; better with rare/ancient characters and jargon; improved long-document structure parsing.
* **Text Understanding on par with pure LLMs**: Seamless textvision fusion for lossless, unified comprehension.
#### Model Architecture Updates:
<p align="center">
<img src="https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_arc.jpg" width="80%"/>
<p>
1. **Interleaved-MRoPE**: Fullfrequency allocation over time, width, and height via robust positional embeddings, enhancing longhorizon video reasoning.
2. **DeepStack**: Fuses multilevel ViT features to capture finegrained details and sharpen imagetext alignment.
3. **TextTimestamp Alignment:** Moves beyond TRoPE to precise, timestampgrounded event localization for stronger video temporal modeling.
This is the weight repository for Qwen3-VL-8B-Instruct.
---
## Model Performance
**Multimodal performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_vl_instruct.jpg)
**Pure text performance**
![](https://qianwen-res.oss-accelerate.aliyuncs.com/Qwen3-VL/qwen3vl_4b_8b_text_instruct.jpg)
## Quickstart
Below, we provide simple examples to show how to use Qwen3-VL with 🤖 ModelScope and 🤗 Transformers.
The code of Qwen3-VL has been in the latest Hugging Face transformers and we advise you to build from source with command:
```
pip install git+https://github.com/huggingface/transformers
# pip install transformers==4.57.0 # currently, V4.57.0 is not released
```
### Using 🤗 Transformers to Chat
Here we show a code snippet to show how to use the chat model with `transformers`:
```python
from transformers import Qwen3VLForConditionalGeneration, AutoProcessor
# default: Load the model on the available device(s)
model = Qwen3VLForConditionalGeneration.from_pretrained(
"Qwen/Qwen3-VL-8B-Instruct", dtype="auto", device_map="auto"
)
# We recommend enabling flash_attention_2 for better acceleration and memory saving, especially in multi-image and video scenarios.
# model = Qwen3VLForConditionalGeneration.from_pretrained(
# "Qwen/Qwen3-VL-8B-Instruct",
# dtype=torch.bfloat16,
# attn_implementation="flash_attention_2",
# device_map="auto",
# )
processor = AutoProcessor.from_pretrained("Qwen/Qwen3-VL-8B-Instruct")
messages = [
{
"role": "user",
"content": [
{
"type": "image",
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
},
{"type": "text", "text": "Describe this image."},
],
}
]
# Preparation for inference
inputs = processor.apply_chat_template(
messages,
tokenize=True,
add_generation_prompt=True,
return_dict=True,
return_tensors="pt"
)
inputs = inputs.to(model.device)
# Inference: Generation of the output
generated_ids = model.generate(**inputs, max_new_tokens=128)
generated_ids_trimmed = [
out_ids[len(in_ids) :] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
]
output_text = processor.batch_decode(
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
)
print(output_text)
```
### Generation Hyperparameters
#### VL
```bash
export greedy='false'
export top_p=0.8
export top_k=20
export temperature=0.7
export repetition_penalty=1.0
export presence_penalty=1.5
export out_seq_length=16384
```
#### Text
```bash
export greedy='false'
export top_p=1.0
export top_k=40
export repetition_penalty=1.0
export presence_penalty=2.0
export temperature=1.0
export out_seq_length=32768
```
## Citation
If you find our work helpful, feel free to give us a cite.
```
@misc{qwen3technicalreport,
title={Qwen3 Technical Report},
author={Qwen Team},
year={2025},
eprint={2505.09388},
archivePrefix={arXiv},
primaryClass={cs.CL},
url={https://arxiv.org/abs/2505.09388},
}
@article{Qwen2.5-VL,
title={Qwen2.5-VL Technical Report},
author={Bai, Shuai and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Song, Sibo and Dang, Kai and Wang, Peng and Wang, Shijie and Tang, Jun and Zhong, Humen and Zhu, Yuanzhi and Yang, Mingkun and Li, Zhaohai and Wan, Jianqiang and Wang, Pengfei and Ding, Wei and Fu, Zheren and Xu, Yiheng and Ye, Jiabo and Zhang, Xi and Xie, Tianbao and Cheng, Zesen and Zhang, Hang and Yang, Zhibo and Xu, Haiyang and Lin, Junyang},
journal={arXiv preprint arXiv:2502.13923},
year={2025}
}
@article{Qwen2VL,
title={Qwen2-VL: Enhancing Vision-Language Model's Perception of the World at Any Resolution},
author={Wang, Peng and Bai, Shuai and Tan, Sinan and Wang, Shijie and Fan, Zhihao and Bai, Jinze and Chen, Keqin and Liu, Xuejing and Wang, Jialin and Ge, Wenbin and Fan, Yang and Dang, Kai and Du, Mengfei and Ren, Xuancheng and Men, Rui and Liu, Dayiheng and Zhou, Chang and Zhou, Jingren and Lin, Junyang},
journal={arXiv preprint arXiv:2409.12191},
year={2024}
}
@article{Qwen-VL,
title={Qwen-VL: A Versatile Vision-Language Model for Understanding, Localization, Text Reading, and Beyond},
author={Bai, Jinze and Bai, Shuai and Yang, Shusheng and Wang, Shijie and Tan, Sinan and Wang, Peng and Lin, Junyang and Zhou, Chang and Zhou, Jingren},
journal={arXiv preprint arXiv:2308.12966},
year={2023}
}
```

1
configuration.json Normal file
View File

@@ -0,0 +1 @@
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}

View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:d1ace0dbfb4a328899f248b5ff7092029f3a46737deec7011cba34f33e6aa95b
size 5347232

3
mmproj-BF16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:be380cab808bd677e306528e884e08a68f821cb1186ca48478a6dcdbf31f63a3
size 1162569280

3
mmproj-F16.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:a64a9e44dc06cfaca2b3854a0921e9768a094fac7eb7c7066acdd55f10c59440
size 1159030336

3
mmproj-F32.gguf Normal file
View File

@@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:aa4189fb551ac2da887f295b7c5bd2154f5a77c2c66d2894361dd486027ab214
size 2305574464