初始化项目,由ModelHub XC社区提供模型
Model: prithivMLmods/Lh41-1042-Magellanic-7B-0711 Source: Original Platform
This commit is contained in:
36
.gitattributes
vendored
Normal file
36
.gitattributes
vendored
Normal file
@@ -0,0 +1,36 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||||
130
README.md
Normal file
130
README.md
Normal file
@@ -0,0 +1,130 @@
|
|||||||
|
---
|
||||||
|
license: apache-2.0
|
||||||
|
tags:
|
||||||
|
- trl
|
||||||
|
- text-generation-inference
|
||||||
|
- image-captioning
|
||||||
|
- optical-character-recognition
|
||||||
|
- intelligent-character-recognition
|
||||||
|
- caption
|
||||||
|
- ocr
|
||||||
|
- visual-understanding
|
||||||
|
- art
|
||||||
|
- icr
|
||||||
|
- image-to-text
|
||||||
|
- vlm
|
||||||
|
- science
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
- zh
|
||||||
|
library_name: transformers
|
||||||
|
pipeline_tag: image-text-to-text
|
||||||
|
base_model:
|
||||||
|
- Qwen/Qwen2.5-VL-7B-Instruct
|
||||||
|
---
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
# **Lh41-1042-Magellanic-7B-0711**
|
||||||
|
|
||||||
|
> The **Lh41-1042-Magellanic-7B-0711** model is a fine-tuned version of **Qwen2.5-VL-7B-Instruct**, optimized for **Image Captioning**, **Visual Analysis**, and **Image Reasoning**. Built on top of the Qwen2.5-VL architecture, this experimental model enhances visual comprehension capabilities with focused training on 3,000K image pairs for superior image understanding and reasoning tasks across all categories of images with variational dimensions.
|
||||||
|
|
||||||
|
# Key Enhancements
|
||||||
|
|
||||||
|
* **Advanced Image Captioning**: Superior capability for generating detailed and contextually accurate descriptions of images across diverse categories and dimensions.
|
||||||
|
|
||||||
|
* **Enhanced Visual Analysis**: Designed to efficiently analyze and interpret complex visual content, patterns, and relationships within images.
|
||||||
|
|
||||||
|
* **Superior Image Reasoning**: Optimized for logical reasoning and inference based on visual information, enabling complex visual question answering.
|
||||||
|
|
||||||
|
* **Multi-Category Image Support**: Specialized in handling all categories of images with variational dimensions, from simple objects to complex scenes.
|
||||||
|
|
||||||
|
* **State-of-the-Art Performance Across Resolutions**: Achieves competitive results on OCR and visual QA benchmarks such as DocVQA, MathVista, RealWorldQA, and MTVQA.
|
||||||
|
|
||||||
|
* **Video Understanding up to 20+ minutes**: Supports detailed comprehension of long-duration videos for content summarization, Q&A, and multi-modal reasoning.
|
||||||
|
|
||||||
|
* **Visually-Grounded Device Interaction**: Enables mobile/robotic device operation via visual inputs and text-based instructions using contextual understanding and decision-making logic.
|
||||||
|
|
||||||
|
# Quick Start with Transformers
|
||||||
|
|
||||||
|
```python
|
||||||
|
from transformers import Qwen2_5_VLForConditionalGeneration, AutoTokenizer, AutoProcessor
|
||||||
|
from qwen_vl_utils import process_vision_info
|
||||||
|
|
||||||
|
model = Qwen2_5_VLForConditionalGeneration.from_pretrained(
|
||||||
|
"prithivMLmods/Lh41-1042-Magellanic-7B-0711", torch_dtype="auto", device_map="auto"
|
||||||
|
)
|
||||||
|
|
||||||
|
processor = AutoProcessor.from_pretrained("prithivMLmods/Lh41-1042-Magellanic-7B-0711")
|
||||||
|
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{
|
||||||
|
"type": "image",
|
||||||
|
"image": "https://qianwen-res.oss-cn-beijing.aliyuncs.com/Qwen-VL/assets/demo.jpeg",
|
||||||
|
},
|
||||||
|
{"type": "text", "text": "Describe this image."},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
text = processor.apply_chat_template(
|
||||||
|
messages, tokenize=False, add_generation_prompt=True
|
||||||
|
)
|
||||||
|
image_inputs, video_inputs = process_vision_info(messages)
|
||||||
|
inputs = processor(
|
||||||
|
text=[text],
|
||||||
|
images=image_inputs,
|
||||||
|
videos=video_inputs,
|
||||||
|
padding=True,
|
||||||
|
return_tensors="pt",
|
||||||
|
)
|
||||||
|
inputs = inputs.to("cuda")
|
||||||
|
|
||||||
|
generated_ids = model.generate(**inputs, max_new_tokens=128)
|
||||||
|
generated_ids_trimmed = [
|
||||||
|
out_ids[len(in_ids):] for in_ids, out_ids in zip(inputs.input_ids, generated_ids)
|
||||||
|
]
|
||||||
|
output_text = processor.batch_decode(
|
||||||
|
generated_ids_trimmed, skip_special_tokens=True, clean_up_tokenization_spaces=False
|
||||||
|
)
|
||||||
|
print(output_text)
|
||||||
|
```
|
||||||
|
|
||||||
|
# Intended Use
|
||||||
|
|
||||||
|
This model is intended for:
|
||||||
|
|
||||||
|
* Advanced image captioning with contextually rich and detailed descriptions.
|
||||||
|
* High-fidelity visual analysis and interpretation of complex visual content.
|
||||||
|
* Image reasoning tasks requiring logical inference and pattern recognition.
|
||||||
|
* Visual question answering for educational and enterprise applications.
|
||||||
|
* Multi-modal content understanding across diverse image categories and dimensions.
|
||||||
|
* Automated image description generation for accessibility and content management.
|
||||||
|
* Visual content analysis for creative and professional applications.
|
||||||
|
* Robotic or mobile automation with vision-guided contextual interaction.
|
||||||
|
|
||||||
|
## Training Details
|
||||||
|
|
||||||
|
| Parameter | Value |
|
||||||
|
|-------------------------|-----------------------------------------------------|
|
||||||
|
| **Dataset Size** | 3,000K image pairs |
|
||||||
|
| **Model Architecture** | `Qwen2_5_VLForConditionalGeneration` |
|
||||||
|
| **Total Disk Volume** | 600,000 MB |
|
||||||
|
| **Training Time** | approx. 16,488 seconds (~4.58 hours) |
|
||||||
|
| **Model Stage** | Experimental |
|
||||||
|
| **Hardware** | 3 × NVIDIA A40 (29 vCPUs) |
|
||||||
|
| **Warmup Steps** | 750 |
|
||||||
|
| **Precision** | bfloat16 |
|
||||||
|
|
||||||
|
# Limitations
|
||||||
|
|
||||||
|
* May show degraded performance on extremely low-quality or occluded images.
|
||||||
|
* Not optimized for real-time applications on low-resource or edge devices due to computational demands.
|
||||||
|
* Variable accuracy on uncommon visual patterns or highly specialized domain images.
|
||||||
|
* Long video processing may require substantial memory and is not optimized for streaming applications.
|
||||||
|
* Visual token settings affect performance; suboptimal configurations can impact results.
|
||||||
|
* In rare cases, outputs may contain hallucinated or contextually misaligned information.
|
||||||
|
* As an experimental model, performance may vary across different use cases and requires further validation.
|
||||||
24
added_tokens.json
Normal file
24
added_tokens.json
Normal file
@@ -0,0 +1,24 @@
|
|||||||
|
{
|
||||||
|
"</tool_call>": 151658,
|
||||||
|
"<tool_call>": 151657,
|
||||||
|
"<|box_end|>": 151649,
|
||||||
|
"<|box_start|>": 151648,
|
||||||
|
"<|endoftext|>": 151643,
|
||||||
|
"<|file_sep|>": 151664,
|
||||||
|
"<|fim_middle|>": 151660,
|
||||||
|
"<|fim_pad|>": 151662,
|
||||||
|
"<|fim_prefix|>": 151659,
|
||||||
|
"<|fim_suffix|>": 151661,
|
||||||
|
"<|im_end|>": 151645,
|
||||||
|
"<|im_start|>": 151644,
|
||||||
|
"<|image_pad|>": 151655,
|
||||||
|
"<|object_ref_end|>": 151647,
|
||||||
|
"<|object_ref_start|>": 151646,
|
||||||
|
"<|quad_end|>": 151651,
|
||||||
|
"<|quad_start|>": 151650,
|
||||||
|
"<|repo_name|>": 151663,
|
||||||
|
"<|video_pad|>": 151656,
|
||||||
|
"<|vision_end|>": 151653,
|
||||||
|
"<|vision_pad|>": 151654,
|
||||||
|
"<|vision_start|>": 151652
|
||||||
|
}
|
||||||
7
chat_template.jinja
Normal file
7
chat_template.jinja
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
{% set image_count = namespace(value=0) %}{% set video_count = namespace(value=0) %}{% for message in messages %}{% if loop.first and message['role'] != 'system' %}<|im_start|>system
|
||||||
|
You are a helpful assistant.<|im_end|>
|
||||||
|
{% endif %}<|im_start|>{{ message['role'] }}
|
||||||
|
{% if message['content'] is string %}{{ message['content'] }}<|im_end|>
|
||||||
|
{% else %}{% for content in message['content'] %}{% if content['type'] == 'image' or 'image' in content or 'image_url' in content %}{% set image_count.value = image_count.value + 1 %}{% if add_vision_id %}Picture {{ image_count.value }}: {% endif %}<|vision_start|><|image_pad|><|vision_end|>{% elif content['type'] == 'video' or 'video' in content %}{% set video_count.value = video_count.value + 1 %}{% if add_vision_id %}Video {{ video_count.value }}: {% endif %}<|vision_start|><|video_pad|><|vision_end|>{% elif 'text' in content %}{{ content['text'] }}{% endif %}{% endfor %}<|im_end|>
|
||||||
|
{% endif %}{% endfor %}{% if add_generation_prompt %}<|im_start|>assistant
|
||||||
|
{% endif %}
|
||||||
3
config.json
Normal file
3
config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:1c07f7141ea63074a4124ac0e55300662f68b0c0d61d00a50479081eaf8947a2
|
||||||
|
size 3336
|
||||||
1
configuration.json
Normal file
1
configuration.json
Normal file
@@ -0,0 +1 @@
|
|||||||
|
{"framework": "pytorch", "task": "image-text-to-text", "allow_remote": true}
|
||||||
15
generation_config.json
Normal file
15
generation_config.json
Normal file
@@ -0,0 +1,15 @@
|
|||||||
|
{
|
||||||
|
"bos_token_id": 151643,
|
||||||
|
"do_sample": true,
|
||||||
|
"eos_token_id": [
|
||||||
|
151645,
|
||||||
|
151643
|
||||||
|
],
|
||||||
|
"max_length": 128000,
|
||||||
|
"pad_token_id": 151643,
|
||||||
|
"repetition_penalty": 1.05,
|
||||||
|
"temperature": 0.1,
|
||||||
|
"top_k": 1,
|
||||||
|
"top_p": 0.001,
|
||||||
|
"transformers_version": "4.53.2"
|
||||||
|
}
|
||||||
BIN
merges.txt
(Stored with Git LFS)
Normal file
BIN
merges.txt
(Stored with Git LFS)
Normal file
Binary file not shown.
3
model-00001-of-00004.safetensors
Normal file
3
model-00001-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:4848c2700b0009832898ed1fd79753642e9a1989173179709176411863a0623a
|
||||||
|
size 4968243304
|
||||||
3
model-00002-of-00004.safetensors
Normal file
3
model-00002-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:2c36219815c148d68879a61a32c2410ce9a427ab7c16a32de1bfadbe766b885a
|
||||||
|
size 4991495816
|
||||||
3
model-00003-of-00004.safetensors
Normal file
3
model-00003-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:dffff8396d71356e6b3e8c243fbb0d6bf7c7d8e9ab7531dc32811fdda4f1e43c
|
||||||
|
size 4932751040
|
||||||
3
model-00004-of-00004.safetensors
Normal file
3
model-00004-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:27d6efac4583bf6f0b7fe860514f7b5e3fbef2a480cbd038dbec788d6f381ae6
|
||||||
|
size 1691924384
|
||||||
3
model.safetensors.index.json
Normal file
3
model.safetensors.index.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:3067e9b0f35596ff3426a0d0ec8c982a51fa1e110c4fc30dcf3be9ea37409df6
|
||||||
|
size 57619
|
||||||
29
preprocessor_config.json
Normal file
29
preprocessor_config.json
Normal file
@@ -0,0 +1,29 @@
|
|||||||
|
{
|
||||||
|
"do_convert_rgb": true,
|
||||||
|
"do_normalize": true,
|
||||||
|
"do_rescale": true,
|
||||||
|
"do_resize": true,
|
||||||
|
"image_mean": [
|
||||||
|
0.48145466,
|
||||||
|
0.4578275,
|
||||||
|
0.40821073
|
||||||
|
],
|
||||||
|
"image_processor_type": "Qwen2VLImageProcessor",
|
||||||
|
"image_std": [
|
||||||
|
0.26862954,
|
||||||
|
0.26130258,
|
||||||
|
0.27577711
|
||||||
|
],
|
||||||
|
"max_pixels": 12845056,
|
||||||
|
"merge_size": 2,
|
||||||
|
"min_pixels": 3136,
|
||||||
|
"patch_size": 14,
|
||||||
|
"processor_class": "Qwen2_5_VLProcessor",
|
||||||
|
"resample": 3,
|
||||||
|
"rescale_factor": 0.00392156862745098,
|
||||||
|
"size": {
|
||||||
|
"longest_edge": 12845056,
|
||||||
|
"shortest_edge": 3136
|
||||||
|
},
|
||||||
|
"temporal_patch_size": 2
|
||||||
|
}
|
||||||
31
special_tokens_map.json
Normal file
31
special_tokens_map.json
Normal file
@@ -0,0 +1,31 @@
|
|||||||
|
{
|
||||||
|
"additional_special_tokens": [
|
||||||
|
"<|im_start|>",
|
||||||
|
"<|im_end|>",
|
||||||
|
"<|object_ref_start|>",
|
||||||
|
"<|object_ref_end|>",
|
||||||
|
"<|box_start|>",
|
||||||
|
"<|box_end|>",
|
||||||
|
"<|quad_start|>",
|
||||||
|
"<|quad_end|>",
|
||||||
|
"<|vision_start|>",
|
||||||
|
"<|vision_end|>",
|
||||||
|
"<|vision_pad|>",
|
||||||
|
"<|image_pad|>",
|
||||||
|
"<|video_pad|>"
|
||||||
|
],
|
||||||
|
"eos_token": {
|
||||||
|
"content": "<|im_end|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"pad_token": {
|
||||||
|
"content": "<|endoftext|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
}
|
||||||
|
}
|
||||||
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:e04081d680d5bb294b2e57aea5b3aa1256d9e06263e907917fc241c5adc2fbe4
|
||||||
|
size 11422163
|
||||||
3
tokenizer_config.json
Normal file
3
tokenizer_config.json
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:8f80888dc06a02ae0ae3cd43b69ba9f971132938bbf73b36d6da170f3ad7d64b
|
||||||
|
size 4755
|
||||||
43
video_preprocessor_config.json
Normal file
43
video_preprocessor_config.json
Normal file
@@ -0,0 +1,43 @@
|
|||||||
|
{
|
||||||
|
"crop_size": null,
|
||||||
|
"data_format": "channels_first",
|
||||||
|
"default_to_square": true,
|
||||||
|
"device": null,
|
||||||
|
"do_center_crop": null,
|
||||||
|
"do_convert_rgb": true,
|
||||||
|
"do_normalize": true,
|
||||||
|
"do_pad": null,
|
||||||
|
"do_rescale": true,
|
||||||
|
"do_resize": true,
|
||||||
|
"do_sample_frames": false,
|
||||||
|
"fps": null,
|
||||||
|
"image_mean": [
|
||||||
|
0.48145466,
|
||||||
|
0.4578275,
|
||||||
|
0.40821073
|
||||||
|
],
|
||||||
|
"image_std": [
|
||||||
|
0.26862954,
|
||||||
|
0.26130258,
|
||||||
|
0.27577711
|
||||||
|
],
|
||||||
|
"input_data_format": null,
|
||||||
|
"max_frames": 768,
|
||||||
|
"max_pixels": 12845056,
|
||||||
|
"merge_size": 2,
|
||||||
|
"min_frames": 4,
|
||||||
|
"min_pixels": 3136,
|
||||||
|
"num_frames": null,
|
||||||
|
"patch_size": 14,
|
||||||
|
"processor_class": "Qwen2_5_VLProcessor",
|
||||||
|
"resample": 3,
|
||||||
|
"rescale_factor": 0.00392156862745098,
|
||||||
|
"size": {
|
||||||
|
"longest_edge": 12845056,
|
||||||
|
"shortest_edge": 3136
|
||||||
|
},
|
||||||
|
"size_divisor": null,
|
||||||
|
"temporal_patch_size": 2,
|
||||||
|
"video_metadata": null,
|
||||||
|
"video_processor_type": "Qwen2VLVideoProcessor"
|
||||||
|
}
|
||||||
BIN
vocab.json
(Stored with Git LFS)
Normal file
BIN
vocab.json
(Stored with Git LFS)
Normal file
Binary file not shown.
Reference in New Issue
Block a user