初始化项目,由ModelHub XC社区提供模型
Model: YouLiXiya/tinyllava-v1.0-1.1b-hf Source: Original Platform
This commit is contained in:
35
.gitattributes
vendored
Normal file
35
.gitattributes
vendored
Normal file
@@ -0,0 +1,35 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
117
README.md
Normal file
117
README.md
Normal file
@@ -0,0 +1,117 @@
|
|||||||
|
---
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
pipeline_tag: image-to-text
|
||||||
|
inference: false
|
||||||
|
arxiv: 2304.08485
|
||||||
|
license: apache-2.0
|
||||||
|
datasets:
|
||||||
|
- liuhaotian/LLaVA-Pretrain
|
||||||
|
- liuhaotian/LLaVA-Instruct-150K
|
||||||
|
---
|
||||||
|
# LLaVA Model Card
|
||||||
|
|
||||||
|

|
||||||
|
|
||||||
|
Below is the model card of TinyLlava model 1.1b.
|
||||||
|
|
||||||
|
Check out also the Google Colab demo to run Llava on a free-tier Google Colab instance: [](https://colab.research.google.com/drive/1XtdA_UoyNzqiEYVR-iWA-xmit8Y2tKV2#scrollTo=DFVZgElEQk3x)
|
||||||
|
|
||||||
|
|
||||||
|
## Model details
|
||||||
|
|
||||||
|
**Model type:**
|
||||||
|
TinyLLaVA is an open-source chatbot trained by fine-tuning TinyLlama on GPT-generated multimodal instruction-following data.
|
||||||
|
It is an auto-regressive language model, based on the transformer architecture.
|
||||||
|
|
||||||
|
**Paper or resources for more information:**
|
||||||
|
https://llava-vl.github.io/
|
||||||
|
|
||||||
|
## How to use the model
|
||||||
|
|
||||||
|
First, make sure to have `transformers >= 4.35.3`.
|
||||||
|
The model supports multi-image and multi-prompt generation. Meaning that you can pass multiple images in your prompt. Make sure also to follow the correct prompt template (`USER: xxx\nASSISTANT:`) and add the token `<image>` to the location where you want to query images:
|
||||||
|
|
||||||
|
### Using `pipeline`:
|
||||||
|
|
||||||
|
Below we used [`"YouLiXiya/tinyllava-v1.0-1.1b-hf"`](https://huggingface.co/YouLiXiya/tinyllava-v1.0-1.1b-hf) checkpoint.
|
||||||
|
|
||||||
|
```python
|
||||||
|
from transformers import pipeline
|
||||||
|
from PIL import Image
|
||||||
|
import requests
|
||||||
|
|
||||||
|
model_id = "YouLiXiya/tinyllava-v1.0-1.1b-hf"
|
||||||
|
pipe = pipeline("image-to-text", model=model_id)
|
||||||
|
url = "https://huggingface.co/datasets/huggingface/documentation-images/resolve/main/transformers/tasks/ai2d-demo.jpg"
|
||||||
|
|
||||||
|
image = Image.open(requests.get(url, stream=True).raw)
|
||||||
|
prompt = "USER: <image>\nWhat does the label 15 represent? (1) lava (2) core (3) tunnel (4) ash cloud\nASSISTANT:"
|
||||||
|
|
||||||
|
outputs = pipe(image, prompt=prompt, generate_kwargs={"max_new_tokens": 200})
|
||||||
|
print(outputs)
|
||||||
|
{'generated_text': 'USER: \nWhat does the label 15 represent? (1) lava (2) core (3) tunnel (4) ash cloud\nASSISTANT: The label 15 represents lava, which is the type of rock that is formed from molten magma. '}
|
||||||
|
```
|
||||||
|
|
||||||
|
### Using pure `transformers`:
|
||||||
|
|
||||||
|
Below is an example script to run generation in `float16` precision on a GPU device:
|
||||||
|
|
||||||
|
```python
|
||||||
|
import requests
|
||||||
|
from PIL import Image
|
||||||
|
|
||||||
|
import torch
|
||||||
|
from transformers import AutoProcessor, LlavaForConditionalGeneration
|
||||||
|
|
||||||
|
model_id = "YouLiXiya/tinyllava-v1.0-1.1b-hf"
|
||||||
|
|
||||||
|
prompt = "USER: <image>\nWhat are these?\nASSISTANT:"
|
||||||
|
image_file = "http://images.cocodataset.org/val2017/000000039769.jpg"
|
||||||
|
|
||||||
|
model = LlavaForConditionalGeneration.from_pretrained(
|
||||||
|
model_id,
|
||||||
|
torch_dtype=torch.float16,
|
||||||
|
low_cpu_mem_usage=True,
|
||||||
|
).to(0)
|
||||||
|
|
||||||
|
processor = AutoProcessor.from_pretrained(model_id)
|
||||||
|
|
||||||
|
raw_image = Image.open(requests.get(image_file, stream=True).raw)
|
||||||
|
inputs = processor(prompt, raw_image, return_tensors='pt').to(0, torch.float16)
|
||||||
|
|
||||||
|
output = model.generate(**inputs, max_new_tokens=200, do_sample=False)
|
||||||
|
print(processor.decode(output[0][2:], skip_special_tokens=True))
|
||||||
|
```
|
||||||
|
|
||||||
|
### Model optimization
|
||||||
|
|
||||||
|
#### 4-bit quantization through `bitsandbytes` library
|
||||||
|
|
||||||
|
First make sure to install `bitsandbytes`, `pip install bitsandbytes` and make sure to have access to a CUDA compatible GPU device. Simply change the snippet above with:
|
||||||
|
|
||||||
|
```diff
|
||||||
|
model = LlavaForConditionalGeneration.from_pretrained(
|
||||||
|
model_id,
|
||||||
|
torch_dtype=torch.float16,
|
||||||
|
low_cpu_mem_usage=True,
|
||||||
|
+ load_in_4bit=True
|
||||||
|
)
|
||||||
|
```
|
||||||
|
|
||||||
|
#### Use Flash-Attention 2 to further speed-up generation
|
||||||
|
|
||||||
|
First make sure to install `flash-attn`. Refer to the [original repository of Flash Attention](https://github.com/Dao-AILab/flash-attention) regarding that package installation. Simply change the snippet above with:
|
||||||
|
|
||||||
|
```diff
|
||||||
|
model = LlavaForConditionalGeneration.from_pretrained(
|
||||||
|
model_id,
|
||||||
|
torch_dtype=torch.float16,
|
||||||
|
low_cpu_mem_usage=True,
|
||||||
|
+ use_flash_attention_2=True
|
||||||
|
).to(0)
|
||||||
|
```
|
||||||
|
|
||||||
|
## License
|
||||||
|
Llama 2 is licensed under the LLAMA 2 Community License,
|
||||||
|
Copyright (c) Meta Platforms, Inc. All Rights Reserved.
|
||||||
4
added_tokens.json
Normal file
4
added_tokens.json
Normal file
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"<image>": 32000,
|
||||||
|
"<pad>": 32001
|
||||||
|
}
|
||||||
40
config.json
Normal file
40
config.json
Normal file
@@ -0,0 +1,40 @@
|
|||||||
|
{
|
||||||
|
"architectures": [
|
||||||
|
"LlavaForConditionalGeneration"
|
||||||
|
],
|
||||||
|
"ignore_index": -100,
|
||||||
|
"image_token_index": 32000,
|
||||||
|
"model_type": "llava",
|
||||||
|
"pad_token_id": 32001,
|
||||||
|
"projector_hidden_act": "gelu",
|
||||||
|
"text_config": {
|
||||||
|
"_name_or_path": "TinyLlama/TinyLlama-1.1B-Chat-V1.0",
|
||||||
|
"architectures": [
|
||||||
|
"LlamaForCausalLM"
|
||||||
|
],
|
||||||
|
"hidden_size": 2048,
|
||||||
|
"intermediate_size": 5632,
|
||||||
|
"model_type": "llama",
|
||||||
|
"num_hidden_layers": 22,
|
||||||
|
"num_key_value_heads": 4,
|
||||||
|
"rms_norm_eps": 1e-05,
|
||||||
|
"torch_dtype": "bfloat16",
|
||||||
|
"vocab_size": 32064
|
||||||
|
},
|
||||||
|
"torch_dtype": "float16",
|
||||||
|
"transformers_version": "4.37.0.dev0",
|
||||||
|
"vision_config": {
|
||||||
|
"dropout": 0.0,
|
||||||
|
"hidden_size": 1024,
|
||||||
|
"image_size": 224,
|
||||||
|
"intermediate_size": 4096,
|
||||||
|
"model_type": "clip_vision_model",
|
||||||
|
"num_attention_heads": 16,
|
||||||
|
"num_hidden_layers": 24,
|
||||||
|
"patch_size": 14,
|
||||||
|
"projection_dim": 768
|
||||||
|
},
|
||||||
|
"vision_feature_layer": -2,
|
||||||
|
"vision_feature_select_strategy": "default",
|
||||||
|
"vocab_size": 32064
|
||||||
|
}
|
||||||
7
generation_config.json
Normal file
7
generation_config.json
Normal file
@@ -0,0 +1,7 @@
|
|||||||
|
{
|
||||||
|
"_from_model_config": true,
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"pad_token_id": 32001,
|
||||||
|
"transformers_version": "4.37.0.dev0"
|
||||||
|
}
|
||||||
3
model.safetensors
Normal file
3
model.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:d151cbb50192fde78dec713c2cd15d68c73183aab3be97ebbb2f5865bac08611
|
||||||
|
size 2819651632
|
||||||
28
preprocessor_config.json
Normal file
28
preprocessor_config.json
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
{
|
||||||
|
"crop_size": {
|
||||||
|
"height": 224,
|
||||||
|
"width": 224
|
||||||
|
},
|
||||||
|
"do_center_crop": true,
|
||||||
|
"do_convert_rgb": true,
|
||||||
|
"do_normalize": true,
|
||||||
|
"do_rescale": true,
|
||||||
|
"do_resize": true,
|
||||||
|
"image_mean": [
|
||||||
|
0.48145466,
|
||||||
|
0.4578275,
|
||||||
|
0.40821073
|
||||||
|
],
|
||||||
|
"image_processor_type": "CLIPImageProcessor",
|
||||||
|
"image_std": [
|
||||||
|
0.26862954,
|
||||||
|
0.26130258,
|
||||||
|
0.27577711
|
||||||
|
],
|
||||||
|
"processor_class": "LlavaProcessor",
|
||||||
|
"resample": 3,
|
||||||
|
"rescale_factor": 0.00392156862745098,
|
||||||
|
"size": {
|
||||||
|
"shortest_edge": 224
|
||||||
|
}
|
||||||
|
}
|
||||||
30
special_tokens_map.json
Normal file
30
special_tokens_map.json
Normal file
@@ -0,0 +1,30 @@
|
|||||||
|
{
|
||||||
|
"bos_token": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"eos_token": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"pad_token": {
|
||||||
|
"content": "<pad>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"unk_token": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
}
|
||||||
|
}
|
||||||
93409
tokenizer.json
Normal file
93409
tokenizer.json
Normal file
File diff suppressed because it is too large
Load Diff
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
Binary file not shown.
59
tokenizer_config.json
Normal file
59
tokenizer_config.json
Normal file
@@ -0,0 +1,59 @@
|
|||||||
|
{
|
||||||
|
"add_bos_token": true,
|
||||||
|
"add_eos_token": false,
|
||||||
|
"added_tokens_decoder": {
|
||||||
|
"0": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"1": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"2": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"32000": {
|
||||||
|
"content": "<image>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"32001": {
|
||||||
|
"content": "<pad>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"bos_token": "<s>",
|
||||||
|
"chat_template": "{% for message in messages %}\n{% if message['role'] == 'user' %}\n{{ '<|user|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'system' %}\n{{ '<|system|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'assistant' %}\n{{ '<|assistant|>\n' + message['content'] + eos_token }}\n{% endif %}\n{% if loop.last and add_generation_prompt %}\n{{ '<|assistant|>' }}\n{% endif %}\n{% endfor %}",
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "</s>",
|
||||||
|
"legacy": false,
|
||||||
|
"model_max_length": 2048,
|
||||||
|
"pad_token": "<pad>",
|
||||||
|
"padding_side": "right",
|
||||||
|
"processor_class": "LlavaProcessor",
|
||||||
|
"sp_model_kwargs": {},
|
||||||
|
"tokenizer_class": "LlamaTokenizer",
|
||||||
|
"unk_token": "<unk>",
|
||||||
|
"use_default_system_prompt": false
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user