初始化项目,由ModelHub XC社区提供模型
Model: LeroyDyer/SpydazWebAI_QuietStar_Project Source: Original Platform
This commit is contained in:
39
.gitattributes
vendored
Normal file
39
.gitattributes
vendored
Normal file
@@ -0,0 +1,39 @@
|
|||||||
|
*.7z filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.arrow filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.bz2 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ckpt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ftz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.gz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.h5 filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.joblib filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.lfs.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.mlmodel filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.model filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.msgpack filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npy filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.npz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.onnx filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.ot filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.parquet filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pb filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pickle filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pkl filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.pth filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.rar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||||
|
saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar.* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tar filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tflite filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.tgz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.wasm filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.xz filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zip filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*.zst filter=lfs diff=lfs merge=lfs -text
|
||||||
|
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
||||||
|
Mixtral_AI_CyberBrain_3_0-unsloth.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
Mixtral_AI_CyberBrain_3_0-unsloth.F16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
Mixtral_AI_CyberBrain_3_0.F16.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
|
Mixtral_AI_CyberBrain_3_0.Q4_K_M.gguf filter=lfs diff=lfs merge=lfs -text
|
||||||
3
Mixtral_AI_CyberBrain_3_0.F16.gguf
Normal file
3
Mixtral_AI_CyberBrain_3_0.F16.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:c72478c90fcb4c0dd76e9aee2ba32af44654cda07400a9f7be1decc7ab6e9952
|
||||||
|
size 14484764640
|
||||||
3
Mixtral_AI_CyberBrain_3_0.Q4_K_M.gguf
Normal file
3
Mixtral_AI_CyberBrain_3_0.Q4_K_M.gguf
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:37b109558d71a486bbd580541620821fcf5fba61c16fa50416a0fcfb5c37b309
|
||||||
|
size 4368450624
|
||||||
179
README.md
Normal file
179
README.md
Normal file
@@ -0,0 +1,179 @@
|
|||||||
|
---
|
||||||
|
base_model:
|
||||||
|
- LeroyDyer/Mixtral_AI_Multi_TEST
|
||||||
|
- LeroyDyer/Mixtral_AI_Cyber_Dolphin_2.0
|
||||||
|
- LeroyDyer/Mixtral_AI_CyberLAW
|
||||||
|
- LeroyDyer/Mixtral_AI_CyberBrain_3_0
|
||||||
|
- LeroyDyer/Mixtral_AI_Cyber_5.0
|
||||||
|
- LeroyDyer/Mixtral_AI_CyberBrain_2.0
|
||||||
|
- ezelikman/quietstar-8-ahead
|
||||||
|
language:
|
||||||
|
- en
|
||||||
|
license: mit
|
||||||
|
library_name: transformers
|
||||||
|
tags:
|
||||||
|
- mergekit
|
||||||
|
- merge
|
||||||
|
- unsloth
|
||||||
|
- Cyber-Series
|
||||||
|
---
|
||||||
|
|
||||||
|
ActulLLY ITS woRKING IT JUST NEEDS TRAINING DATA!! .... Personally i found models run better in gpt4all! - (served better by lmstudio)
|
||||||
|
|
||||||
|
|
||||||
|
This project is implemented by simply patching the base Mistral implementation in Huggingface transformers using a new modeling_mistral.py and a new configuration_mistral.py and otherwise applying standard transformers features (e.g. the default Trainer).
|
||||||
|
|
||||||
|
IE: First Clone the latest transformers
|
||||||
|
enter the models\mistral folder and upload the modelling_mistral.py
|
||||||
|
then cd transformers and install frot he folder pip install ./transformers
|
||||||
|
|
||||||
|
after it can be loaded normally for training;
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
from unsloth import FastLanguageModel
|
||||||
|
import torch
|
||||||
|
max_seq_length = 2048 # Choose any! We auto support RoPE Scaling internally!
|
||||||
|
dtype = None # None for auto detection. Float16 for Tesla T4, V100, Bfloat16 for Ampere+
|
||||||
|
load_in_4bit = True # Use 4bit quantization to reduce memory usage. Can be False.
|
||||||
|
|
||||||
|
# 4bit pre quantized models we support for 4x faster downloading + no OOMs.
|
||||||
|
fourbit_models = [
|
||||||
|
"unsloth/mistral-7b-bnb-4bit",
|
||||||
|
"unsloth/mistral-7b-instruct-v0.2-bnb-4bit",
|
||||||
|
"unsloth/llama-2-7b-bnb-4bit",
|
||||||
|
"unsloth/llama-2-13b-bnb-4bit",
|
||||||
|
"unsloth/codellama-34b-bnb-4bit",
|
||||||
|
"unsloth/tinyllama-bnb-4bit",
|
||||||
|
"unsloth/gemma-7b-bnb-4bit", # New Google 6 trillion tokens model 2.5x faster!
|
||||||
|
"unsloth/gemma-2b-bnb-4bit",
|
||||||
|
] # More models at https://huggingface.co/unsloth
|
||||||
|
|
||||||
|
model = FastLanguageModel.from_pretrained(
|
||||||
|
model_name = "LeroyDyer/Mixtral_AI_CyberBrain_3.0", # Choose ANY! eg teknium/OpenHermes-2.5-Mistral-7B
|
||||||
|
max_seq_length = 2048,
|
||||||
|
dtype = dtype,
|
||||||
|
load_in_4bit = load_in_4bit,
|
||||||
|
# trust_remote_code = True,
|
||||||
|
ignore_mismatched_sizes = True,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=True,
|
||||||
|
|
||||||
|
# token = "hf_...", # use one if using gated models like meta-llama/Llama-2-7b-hf
|
||||||
|
)
|
||||||
|
|
||||||
|
tokenizer = AutoTokenizer.from_pretrained(tokenizer_id,truncation=True,padding_side="right")
|
||||||
|
tokenizer.pad_token_id = tokenizer.eos_token_id
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
model.tokenizer = tokenizer
|
||||||
|
|
||||||
|
model.train
|
||||||
|
|
||||||
|
|
||||||
|
```
|
||||||
|
|
||||||
|
|
||||||
|
right now the modelling_mistral.py s still havng problems loading remotely hence the hacky way... but after its fixed it will be fine.
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
# merge
|
||||||
|
|
||||||
|
This is a merge of pre-trained language models created using [mergekit](https://github.com/cg123/mergekit).
|
||||||
|
yes multiple verions of this model was merged in attempts to grab the neccasary tensors ...
|
||||||
|
but some how it did not build as some parameters was not loading. ie it would not load the config file! hopefully this will be rectified soon. so remote loading will be fine ... enabling for enhanced training.
|
||||||
|
the model was trained to perfection so it still works fine!
|
||||||
|
the lora was made so tat later it can be loaded with the model for further training of the effected tensors...
|
||||||
|
|
||||||
|
## Extended capabilities:
|
||||||
|
* mistralai/Mistral-7B-Instruct-v0.1 - Prime-Base
|
||||||
|
|
||||||
|
* ChaoticNeutrals/Eris-LelantaclesV2-7b - role play
|
||||||
|
|
||||||
|
* ChaoticNeutrals/Eris_PrimeV3-Vision-7B - vision
|
||||||
|
|
||||||
|
* rvv-karma/BASH-Coder-Mistral-7B - coding
|
||||||
|
|
||||||
|
* Locutusque/Hercules-3.1-Mistral-7B - Unhinging
|
||||||
|
|
||||||
|
* KoboldAI/Mistral-7B-Erebus-v3 - NSFW
|
||||||
|
|
||||||
|
* Locutusque/Hyperion-2.1-Mistral-7B - CHAT
|
||||||
|
|
||||||
|
* Severian/Nexus-IKM-Mistral-7B-Pytorch - Thinking
|
||||||
|
|
||||||
|
* NousResearch/Hermes-2-Pro-Mistral-7B - Generalizing
|
||||||
|
|
||||||
|
* mistralai/Mistral-7B-Instruct-v0.2 - BASE
|
||||||
|
|
||||||
|
* Nitral-AI/ProdigyXBioMistral_7B - medical
|
||||||
|
|
||||||
|
* Nitral-AI/Infinite-Mika-7b - 128k - Context Expansion enforcement
|
||||||
|
|
||||||
|
* Nous-Yarn-Mistral-7b-128k - 128k - Context Expansion
|
||||||
|
|
||||||
|
* yanismiraoui/Yarn-Mistral-7b-128k-sharded
|
||||||
|
|
||||||
|
* ChaoticNeutrals/Eris_Prime-V2-7B - Roleplay
|
||||||
|
|
||||||
|
|
||||||
|
his Expert is a companon to the MEGA_MIND 24b CyberSeries represents a groundbreaking leap in the realm of language models, integrating a diverse array of expert models into a unified framework. At its core lies the Mistral-7B-Instruct-v0.2, a refined instructional model designed for versatility and efficiency.
|
||||||
|
|
||||||
|
Enhanced with an expanded context window and advanced routing mechanisms, the Mistral-7B-Instruct-v0.2 exemplifies the power of Mixture of Experts, allowing seamless integration of specialized sub-models. This architecture facilitates unparalleled performance and scalability, enabling the CyberSeries to tackle a myriad of tasks with unparalleled speed and accuracy.
|
||||||
|
|
||||||
|
Among its illustrious sub-models, the OpenOrca - Mistral-7B-8k shines as a testament to fine-tuning excellence, boasting top-ranking performance in its class. Meanwhile, the Hermes 2 Pro introduces cutting-edge capabilities such as Function Calling and JSON Mode, catering to diverse application needs.
|
||||||
|
|
||||||
|
Driven by Reinforcement Learning from AI Feedback, the Starling-LM-7B-beta demonstrates remarkable adaptability and optimization, while the Phi-1.5 Transformer model stands as a beacon of excellence across various domains, from common sense reasoning to medical inference.
|
||||||
|
|
||||||
|
With models like BioMistral tailored specifically for medical applications and Nous-Yarn-Mistral-7b-128k excelling in handling long-context data, the MEGA_MIND 24b CyberSeries emerges as a transformative force in the landscape of language understanding and artificial intelligence.
|
||||||
|
|
||||||
|
Experience the future of language models with the MEGA_MIND 24b CyberSeries, where innovation meets performance, and possibilities are limitless.
|
||||||
|
|
||||||
|
### Models Merged
|
||||||
|
|
||||||
|
The following models were included in the merge:
|
||||||
|
* [LeroyDyer/Mixtral_AI_CyberBrain_2.0](https://huggingface.co/LeroyDyer/Mixtral_AI_CyberBrain_2.0)
|
||||||
|
* [ezelikman/quietstar-8-ahead](https://huggingface.co/ezelikman/quietstar-8-ahead)
|
||||||
|
|
||||||
|
### Configuration
|
||||||
|
|
||||||
|
The following YAML configuration was used to produce this model:
|
||||||
|
|
||||||
|
```yaml
|
||||||
|
|
||||||
|
slices:
|
||||||
|
- sources:
|
||||||
|
- model: LeroyDyer/Mixtral_AI_CyberBrain_2.0
|
||||||
|
layer_range: [0, 32]
|
||||||
|
- model: ezelikman/quietstar-8-ahead
|
||||||
|
layer_range: [0, 32]
|
||||||
|
# or, the equivalent models: syntax:
|
||||||
|
# models:
|
||||||
|
# - model: mistralai/Mistral-7B-Instruct-v0.2
|
||||||
|
# LaRGER MODEL MUST BE BASE or
|
||||||
|
# BASE MODEL MUST BE THE TOKENIZER YOU WISH TO ADOPT
|
||||||
|
# so for models with customized processes they must be the base model
|
||||||
|
# If the base model has remote code then this must be collected and added
|
||||||
|
# to the repo after and the config file adusted to allow for automapping to your new repo
|
||||||
|
# - model: yanismiraoui/Yarn-Mistral-7b-128k-sharded
|
||||||
|
merge_method: slerp
|
||||||
|
base_model: ezelikman/quietstar-8-ahead
|
||||||
|
parameters:
|
||||||
|
t:
|
||||||
|
- filter: self_attn
|
||||||
|
value: [0.3, 0.6, 0.3786, 0.6, 0.6]
|
||||||
|
- filter: mlp
|
||||||
|
value: [0.7, 0.4, 0.6, 0.4, 0.7]
|
||||||
|
- value: 0.5 # fallback for rest of tensors
|
||||||
|
dtype: float16
|
||||||
|
|
||||||
|
```
|
||||||
4
added_tokens.json
Normal file
4
added_tokens.json
Normal file
@@ -0,0 +1,4 @@
|
|||||||
|
{
|
||||||
|
"<|endthought|>": 32000,
|
||||||
|
"<|startthought|>": 32001
|
||||||
|
}
|
||||||
107
config.json
Normal file
107
config.json
Normal file
@@ -0,0 +1,107 @@
|
|||||||
|
{
|
||||||
|
"_name_or_path": "LeroyDyer/Mixtral_AI_CyberBrain_3_0",
|
||||||
|
"architectures": [
|
||||||
|
"MistralForCausalLM","QuietForCausalLM"
|
||||||
|
],
|
||||||
|
"attention_dropout": 0.0,
|
||||||
|
"auto_map": {
|
||||||
|
"AutoConfig":"LeroyDyer/Mixtral_AI_CyberBrain_3_0--configuration_quiet.QuietConfig",
|
||||||
|
"AutoModel": "LeroyDyer/Mixtral_AI_CyberBrain_3_0--modeling_mistral.MistralModel",
|
||||||
|
"AutoModelForCausalLM": "LeroyDyer/Mixtral_AI_CyberBrain_3_0--modeling_quiet.QuietForCausalLM",
|
||||||
|
"QuietForCausalLM": "LeroyDyer/Mixtral_AI_CyberBrain_3_0--modeling_quiet.QuietForCausalLM",
|
||||||
|
"QuietConfig": "LeroyDyer/Mixtral_AI_CyberBrain_3_0--configuration_quiet.QuietConfig"
|
||||||
|
},
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"hidden_act": "silu",
|
||||||
|
"hidden_size": 4096,
|
||||||
|
"initializer_range": 0.02,
|
||||||
|
"intermediate_size": 14336,
|
||||||
|
"max_position_embeddings": 32768,
|
||||||
|
"max_thoughts": 10,
|
||||||
|
"merged_lm_and_talk_heads": false,
|
||||||
|
"merged_lm_and_think_heads": true,
|
||||||
|
"merged_talk_heads": true,
|
||||||
|
"model_type": "mistral",
|
||||||
|
"num_attention_heads": 32,
|
||||||
|
"num_hidden_layers": 32,
|
||||||
|
"num_key_value_heads": 8,
|
||||||
|
"rms_norm_eps": 1e-05,
|
||||||
|
"rope_theta": 10000.0,
|
||||||
|
"sliding_window": 4096,
|
||||||
|
"tie_word_embeddings": false,
|
||||||
|
"torch_dtype": "bfloat16",
|
||||||
|
"transformers_version": "4.37.0.dev0",
|
||||||
|
"use_cache": true,
|
||||||
|
"use_complex_talk_head": true,
|
||||||
|
"use_complex_think_head": false,
|
||||||
|
"use_concat_talk_head": true,
|
||||||
|
"use_shallow_talk": false,
|
||||||
|
"use_shallow_think": true,
|
||||||
|
"use_weighted_talk_head": true,
|
||||||
|
"vocab_size": 32002,
|
||||||
|
"freeze_mm_mlp_adapter": false,
|
||||||
|
"freeze_mm_vision_resampler": false,
|
||||||
|
"ignore_index": -100,
|
||||||
|
"image_aspect_ratio": "anyres",
|
||||||
|
"image_crop_resolution": 224,
|
||||||
|
"image_grid_pinpoints": [
|
||||||
|
[
|
||||||
|
336,
|
||||||
|
672
|
||||||
|
],
|
||||||
|
[
|
||||||
|
672,
|
||||||
|
336
|
||||||
|
],
|
||||||
|
[
|
||||||
|
672,
|
||||||
|
672
|
||||||
|
],
|
||||||
|
[
|
||||||
|
1008,
|
||||||
|
336
|
||||||
|
],
|
||||||
|
[
|
||||||
|
336,
|
||||||
|
1008
|
||||||
|
]
|
||||||
|
],
|
||||||
|
"image_split_resolution": 224,
|
||||||
|
"image_token_index": 32002,
|
||||||
|
"mm_hidden_size": 1024,
|
||||||
|
"mm_patch_merge_type": "spatial_unpad",
|
||||||
|
"mm_projector_lr": null,
|
||||||
|
"mm_projector_type": "mlp2x_gelu",
|
||||||
|
"mm_resampler_type": null,
|
||||||
|
"mm_use_im_patch_token": false,
|
||||||
|
"mm_use_im_start_end": false,
|
||||||
|
"mm_vision_select_feature": "patch",
|
||||||
|
"mm_vision_select_layer": -2,
|
||||||
|
"mm_vision_tower": "openai/clip-vit-large-patch14-336",
|
||||||
|
"mm_vision_tower_lr": 2e-06,
|
||||||
|
"projector_hidden_act": "gelu",
|
||||||
|
"text_config": {
|
||||||
|
"model_type": "llama"
|
||||||
|
},
|
||||||
|
"tokenizer_model_max_length": 4096,
|
||||||
|
"tokenizer_padding_side": "right",
|
||||||
|
"tune_mm_mlp_adapter": false,
|
||||||
|
"tune_mm_vision_resampler": false,
|
||||||
|
"unfreeze_mm_vision_tower": true,
|
||||||
|
"use_mm_proj": true,
|
||||||
|
"vision_config": {
|
||||||
|
"hidden_size": 1024,
|
||||||
|
"image_size": 336,
|
||||||
|
"intermediate_size": 4096,
|
||||||
|
"model_type": "clip_vision_model",
|
||||||
|
"num_attention_heads": 16,
|
||||||
|
"num_hidden_layers": 24,
|
||||||
|
"patch_size": 14,
|
||||||
|
"projection_dim": 768,
|
||||||
|
"vocab_size": 32000
|
||||||
|
},
|
||||||
|
"vision_feature_layer": -2,
|
||||||
|
"vision_feature_select_strategy": "default"
|
||||||
|
}
|
||||||
|
|
||||||
390
configuration_mistral.py
Normal file
390
configuration_mistral.py
Normal file
@@ -0,0 +1,390 @@
|
|||||||
|
# coding=utf-8
|
||||||
|
# Copyright 2023 Mistral AI and the HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
""" Mistral model configuration"""
|
||||||
|
|
||||||
|
from transformers.configuration_utils import PretrainedConfig
|
||||||
|
from transformers.utils import logging
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
QUIET_PRETRAINED_CONFIG_ARCHIVE_MAP = {
|
||||||
|
"LeroyDyer/Mixtral_AI_CyberBrain_3_0": "https://huggingface.co/LeroyDyer/Mixtral_AI_CyberBrain_3_0/resolve/main/config.json",
|
||||||
|
}
|
||||||
|
|
||||||
|
logger = logging.get_logger(__name__)
|
||||||
|
|
||||||
|
MISTRAL_PRETRAINED_CONFIG_ARCHIVE_MAP = {
|
||||||
|
"mistralai/Mistral-7B-v0.1": "https://huggingface.co/mistralai/Mistral-7B-v0.1/resolve/main/config.json",
|
||||||
|
"mistralai/Mistral-7B-Instruct-v0.1": "https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.1/resolve/main/config.json",
|
||||||
|
"LeroyDyer/Mixtral_AI_CyberBrain_3_0": "https://huggingface.co/LeroyDyer/Mixtral_AI_CyberBrain_3_0/resolve/main/config.json",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class MistralConfig(PretrainedConfig):
|
||||||
|
r"""
|
||||||
|
This is the configuration class to store the configuration of a [`MistralModel`]. It is used to instantiate an
|
||||||
|
Mistral model according to the specified arguments, defining the model architecture. Instantiating a configuration
|
||||||
|
with the defaults will yield a similar configuration to that of the Mistral-7B-v0.1 or Mistral-7B-Instruct-v0.1.
|
||||||
|
|
||||||
|
[mistralai/Mistral-7B-v0.1](https://huggingface.co/mistralai/Mistral-7B-v0.1)
|
||||||
|
[mistralai/Mistral-7B-Instruct-v0.1](https://huggingface.co/mistralai/Mistral-7B-Instruct-v0.1)
|
||||||
|
|
||||||
|
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
|
||||||
|
documentation from [`PretrainedConfig`] for more information.
|
||||||
|
|
||||||
|
|
||||||
|
Args:
|
||||||
|
vocab_size (`int`, *optional*, defaults to 32000):
|
||||||
|
Vocabulary size of the Mistral model. Defines the number of different tokens that can be represented by the
|
||||||
|
`inputs_ids` passed when calling [`MistralModel`]
|
||||||
|
hidden_size (`int`, *optional*, defaults to 4096):
|
||||||
|
Dimension of the hidden representations.
|
||||||
|
intermediate_size (`int`, *optional*, defaults to 14336):
|
||||||
|
Dimension of the MLP representations.
|
||||||
|
num_hidden_layers (`int`, *optional*, defaults to 32):
|
||||||
|
Number of hidden layers in the Transformer encoder.
|
||||||
|
num_attention_heads (`int`, *optional*, defaults to 32):
|
||||||
|
Number of attention heads for each attention layer in the Transformer encoder.
|
||||||
|
num_key_value_heads (`int`, *optional*, defaults to 8):
|
||||||
|
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
|
||||||
|
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
|
||||||
|
`num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
|
||||||
|
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
|
||||||
|
by meanpooling all the original heads within that group. For more details checkout [this
|
||||||
|
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to `8`.
|
||||||
|
hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
|
||||||
|
The non-linear activation function (function or string) in the decoder.
|
||||||
|
max_position_embeddings (`int`, *optional*, defaults to `4096*32`):
|
||||||
|
The maximum sequence length that this model might ever be used with. Mistral's sliding window attention
|
||||||
|
allows sequence of up to 4096*32 tokens.
|
||||||
|
initializer_range (`float`, *optional*, defaults to 0.02):
|
||||||
|
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
|
||||||
|
rms_norm_eps (`float`, *optional*, defaults to 1e-06):
|
||||||
|
The epsilon used by the rms normalization layers.
|
||||||
|
use_cache (`bool`, *optional*, defaults to `True`):
|
||||||
|
Whether or not the model should return the last key/values attentions (not used by all models). Only
|
||||||
|
relevant if `config.is_decoder=True`.
|
||||||
|
pad_token_id (`int`, *optional*):
|
||||||
|
The id of the padding token.
|
||||||
|
bos_token_id (`int`, *optional*, defaults to 1):
|
||||||
|
The id of the "beginning-of-sequence" token.
|
||||||
|
eos_token_id (`int`, *optional*, defaults to 2):
|
||||||
|
The id of the "end-of-sequence" token.
|
||||||
|
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
|
||||||
|
Whether the model's input and output word embeddings should be tied.
|
||||||
|
rope_scaling (`Dict`, *optional*):
|
||||||
|
Dictionary containing the scaling configuration for the RoPE embeddings. Currently supports three scaling
|
||||||
|
strategies: linear and dynamic. Their scaling factor must be an float greater than 1. The expected format
|
||||||
|
is `{"type": strategy name, "factor": scaling factor}`.
|
||||||
|
rope_theta (`float`, *optional*, defaults to 10000.0):
|
||||||
|
The base period of the RoPE embeddings.
|
||||||
|
sliding_window (`int`, *optional*, defaults to 4096):
|
||||||
|
Sliding window attention window size. If not specified, will default to `4096`.
|
||||||
|
|
||||||
|
|
||||||
|
```python
|
||||||
|
>>> from transformers import MistralModel, MistralConfig
|
||||||
|
|
||||||
|
>>> # Initializing a Mistral 7B style configuration
|
||||||
|
>>> configuration = MistralConfig()
|
||||||
|
|
||||||
|
>>> # Initializing a model from the Mistral 7B style configuration
|
||||||
|
>>> model = MistralModel(configuration)
|
||||||
|
|
||||||
|
>>> # Accessing the model configuration
|
||||||
|
>>> configuration = model.config
|
||||||
|
```"""
|
||||||
|
|
||||||
|
model_type = "mistral"
|
||||||
|
keys_to_ignore_at_inference = ["past_key_values"]
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
vocab_size=32000,
|
||||||
|
hidden_size=4096,
|
||||||
|
intermediate_size=14336,
|
||||||
|
num_hidden_layers=32,
|
||||||
|
num_attention_heads=32,
|
||||||
|
num_key_value_heads=8,
|
||||||
|
hidden_act="silu",
|
||||||
|
max_position_embeddings=4096 * 32,
|
||||||
|
initializer_range=0.02,
|
||||||
|
rms_norm_eps=1e-6,
|
||||||
|
use_cache=True,
|
||||||
|
pad_token_id=None,
|
||||||
|
bos_token_id=1,
|
||||||
|
eos_token_id=2,
|
||||||
|
tie_word_embeddings=False,
|
||||||
|
rope_scaling=None,
|
||||||
|
rope_theta=10000.0,
|
||||||
|
sliding_window=4096,
|
||||||
|
attention_dropout=0.0,
|
||||||
|
max_thoughts=16,
|
||||||
|
max_temperature=10,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=False,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
self.vocab_size = vocab_size
|
||||||
|
self.max_position_embeddings = max_position_embeddings
|
||||||
|
self.hidden_size = hidden_size
|
||||||
|
self.intermediate_size = intermediate_size
|
||||||
|
self.num_hidden_layers = num_hidden_layers
|
||||||
|
self.num_attention_heads = num_attention_heads
|
||||||
|
self.sliding_window = sliding_window
|
||||||
|
attention_dropout=0.0,
|
||||||
|
max_thoughts=16,
|
||||||
|
max_temperature=10,
|
||||||
|
complexity_factor = 0.5,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=True,
|
||||||
|
# for backward compatibility
|
||||||
|
if num_key_value_heads is None:
|
||||||
|
num_key_value_heads = num_attention_heads
|
||||||
|
|
||||||
|
self.num_key_value_heads = num_key_value_heads
|
||||||
|
self.hidden_act = hidden_act
|
||||||
|
self.initializer_range = initializer_range
|
||||||
|
self.rms_norm_eps = rms_norm_eps
|
||||||
|
self.use_cache = use_cache
|
||||||
|
self.rope_scaling = rope_scaling
|
||||||
|
self._rope_scaling_validation()
|
||||||
|
self.rope_theta = rope_theta
|
||||||
|
self.attention_dropout = attention_dropout
|
||||||
|
self.max_thoughts = max_thoughts
|
||||||
|
self.complexity_factor = complexity_factor
|
||||||
|
self.max_temperature = max_temperature
|
||||||
|
self.merged_talk_heads = merged_talk_heads
|
||||||
|
self.merged_lm_and_talk_heads = merged_lm_and_talk_heads
|
||||||
|
self.merged_lm_and_think_heads = merged_lm_and_think_heads
|
||||||
|
self.use_concat_talk_head = use_concat_talk_head
|
||||||
|
self.use_shallow_think = use_shallow_think
|
||||||
|
self.use_shallow_talk = use_shallow_talk
|
||||||
|
self.use_complex_think_head = use_complex_think_head
|
||||||
|
self.use_complex_talk_head = use_complex_talk_head
|
||||||
|
self.use_weighted_talk_head = use_weighted_talk_head
|
||||||
|
|
||||||
|
super().__init__(
|
||||||
|
pad_token_id=pad_token_id,
|
||||||
|
bos_token_id=bos_token_id,
|
||||||
|
eos_token_id=eos_token_id,
|
||||||
|
tie_word_embeddings=tie_word_embeddings,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
def _rope_scaling_validation(self):
|
||||||
|
"""
|
||||||
|
Validate the `rope_scaling` configuration.
|
||||||
|
"""
|
||||||
|
if self.rope_scaling is None:
|
||||||
|
return
|
||||||
|
|
||||||
|
if not isinstance(self.rope_scaling, dict):
|
||||||
|
raise ValueError(
|
||||||
|
"`rope_scaling` must be a dictionary, "
|
||||||
|
f"got {self.rope_scaling}"
|
||||||
|
)
|
||||||
|
rope_scaling_type = self.rope_scaling.get("type", None)
|
||||||
|
rope_scaling_factor = self.rope_scaling.get("factor", None)
|
||||||
|
if rope_scaling_type is None or rope_scaling_type not in ["linear", "dynamic", "yarn", "dynamic-yarn"]:
|
||||||
|
raise ValueError(
|
||||||
|
f"`rope_scaling`'s name field must be one of ['linear', 'dynamic', 'yarn', 'dynamic-yarn'], got {rope_scaling_type}"
|
||||||
|
)
|
||||||
|
if rope_scaling_factor is None or not isinstance(rope_scaling_factor, float) or rope_scaling_factor <= 1.0:
|
||||||
|
raise ValueError(f"`rope_scaling`'s factor field must be an float > 1, got {rope_scaling_factor}")
|
||||||
|
if rope_scaling_type == "yarn" or rope_scaling_type == "dynamic-yarn":
|
||||||
|
original_max_position_embeddings = self.rope_scaling.get("original_max_position_embeddings", None)
|
||||||
|
if original_max_position_embeddings is None or not isinstance(original_max_position_embeddings, int):
|
||||||
|
raise ValueError(f"`rope_scaling.original_max_position_embeddings` must be set to an int when using yarn, and dynamic-yarn")
|
||||||
|
|
||||||
|
|
||||||
|
# coding=utf-8
|
||||||
|
# Copyright 2023 Quiet AI and the HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
""" Quiet model configuration"""
|
||||||
|
|
||||||
|
from transformers.configuration_utils import PretrainedConfig
|
||||||
|
from transformers.utils import logging
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
|
class QuietConfig(PretrainedConfig):
|
||||||
|
r"""
|
||||||
|
This is the configuration class to store the configuration of a [`QuietModel`]. It is used to instantiate an
|
||||||
|
Quiet model according to the specified arguments, defining the model architecture. Instantiating a configuration
|
||||||
|
with the defaults will yield a similar configuration to that of the Quiet-7B-v0.1 or Quiet-7B-Instruct-v0.1.
|
||||||
|
[quietai/Quiet-7B-v0.1](https://huggingface.co/quietai/Quiet-7B-v0.1)
|
||||||
|
[quietai/Quiet-7B-Instruct-v0.1](https://huggingface.co/quietai/Quiet-7B-Instruct-v0.1)
|
||||||
|
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
|
||||||
|
documentation from [`PretrainedConfig`] for more information.
|
||||||
|
Args:
|
||||||
|
vocab_size (`int`, *optional*, defaults to 32000):
|
||||||
|
Vocabulary size of the Quiet model. Defines the number of different tokens that can be represented by the
|
||||||
|
`inputs_ids` passed when calling [`QuietModel`]
|
||||||
|
hidden_size (`int`, *optional*, defaults to 4096):
|
||||||
|
Dimension of the hidden representations.
|
||||||
|
intermediate_size (`int`, *optional*, defaults to 14336):
|
||||||
|
Dimension of the MLP representations.
|
||||||
|
num_hidden_layers (`int`, *optional*, defaults to 32):
|
||||||
|
Number of hidden layers in the Transformer encoder.
|
||||||
|
num_attention_heads (`int`, *optional*, defaults to 32):
|
||||||
|
Number of attention heads for each attention layer in the Transformer encoder.
|
||||||
|
num_key_value_heads (`int`, *optional*, defaults to 8):
|
||||||
|
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
|
||||||
|
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
|
||||||
|
`num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
|
||||||
|
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
|
||||||
|
by meanpooling all the original heads within that group. For more details checkout [this
|
||||||
|
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to `8`.
|
||||||
|
hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
|
||||||
|
The non-linear activation function (function or string) in the decoder.
|
||||||
|
max_position_embeddings (`int`, *optional*, defaults to `4096*32`):
|
||||||
|
The maximum sequence length that this model might ever be used with. Quiet's sliding window attention
|
||||||
|
allows sequence of up to 4096*32 tokens.
|
||||||
|
initializer_range (`float`, *optional*, defaults to 0.02):
|
||||||
|
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
|
||||||
|
rms_norm_eps (`float`, *optional*, defaults to 1e-06):
|
||||||
|
The epsilon used by the rms normalization layers.
|
||||||
|
use_cache (`bool`, *optional*, defaults to `True`):
|
||||||
|
Whether or not the model should return the last key/values attentions (not used by all models). Only
|
||||||
|
relevant if `config.is_decoder=True`.
|
||||||
|
pad_token_id (`int`, *optional*):
|
||||||
|
The id of the padding token.
|
||||||
|
bos_token_id (`int`, *optional*, defaults to 1):
|
||||||
|
The id of the "beginning-of-sequence" token.
|
||||||
|
eos_token_id (`int`, *optional*, defaults to 2):
|
||||||
|
The id of the "end-of-sequence" token.
|
||||||
|
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
|
||||||
|
Whether the model's input and output word embeddings should be tied.
|
||||||
|
rope_theta (`float`, *optional*, defaults to 10000.0):
|
||||||
|
The base period of the RoPE embeddings.
|
||||||
|
sliding_window (`int`, *optional*, defaults to 4096):
|
||||||
|
Sliding window attention window size. If not specified, will default to `4096`.
|
||||||
|
attention_dropout (`float`, *optional*, defaults to 0.0):
|
||||||
|
The dropout ratio for the attention probabilities.
|
||||||
|
```python
|
||||||
|
>>> from transformers import QuietModel, QuietConfig
|
||||||
|
>>> # Initializing a Quiet 7B style configuration
|
||||||
|
>>> configuration = QuietConfig()
|
||||||
|
>>> # Initializing a model from the Quiet 7B style configuration
|
||||||
|
>>> model = QuietModel(configuration)
|
||||||
|
>>> # Accessing the model configuration
|
||||||
|
>>> configuration = model.config
|
||||||
|
```"""
|
||||||
|
|
||||||
|
model_type = "quiet"
|
||||||
|
keys_to_ignore_at_inference = ["past_key_values"]
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
vocab_size=32000,
|
||||||
|
hidden_size=4096,
|
||||||
|
intermediate_size=14336,
|
||||||
|
num_hidden_layers=32,
|
||||||
|
num_attention_heads=32,
|
||||||
|
num_key_value_heads=8,
|
||||||
|
hidden_act="silu",
|
||||||
|
max_position_embeddings=4096 * 32,
|
||||||
|
initializer_range=0.02,
|
||||||
|
rms_norm_eps=1e-6,
|
||||||
|
use_cache=True,
|
||||||
|
pad_token_id=None,
|
||||||
|
bos_token_id=1,
|
||||||
|
eos_token_id=2,
|
||||||
|
tie_word_embeddings=False,
|
||||||
|
rope_theta=10000.0,
|
||||||
|
complexity_factor = 0.5,
|
||||||
|
sliding_window=4096,
|
||||||
|
attention_dropout=0.0,
|
||||||
|
max_thoughts=16,
|
||||||
|
max_temperature=10,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=True,
|
||||||
|
hidden_dropout_prob=0.0,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
self.vocab_size = vocab_size
|
||||||
|
self.max_position_embeddings = max_position_embeddings
|
||||||
|
self.hidden_size = hidden_size
|
||||||
|
self.intermediate_size = intermediate_size
|
||||||
|
self.num_hidden_layers = num_hidden_layers
|
||||||
|
self.num_attention_heads = num_attention_heads
|
||||||
|
self.sliding_window = sliding_window
|
||||||
|
|
||||||
|
# for backward compatibility
|
||||||
|
if num_key_value_heads is None:
|
||||||
|
num_key_value_heads = num_attention_heads
|
||||||
|
|
||||||
|
self.num_key_value_heads = num_key_value_heads
|
||||||
|
self.hidden_act = hidden_act
|
||||||
|
self.initializer_range = initializer_range
|
||||||
|
self.rms_norm_eps = rms_norm_eps
|
||||||
|
self.use_cache = use_cache
|
||||||
|
self.rope_theta = rope_theta
|
||||||
|
self.attention_dropout = attention_dropout
|
||||||
|
self.max_thoughts = max_thoughts
|
||||||
|
self.complexity_factor = complexity_factor
|
||||||
|
self.max_temperature = max_temperature
|
||||||
|
self.merged_talk_heads = merged_talk_heads
|
||||||
|
self.merged_lm_and_talk_heads = merged_lm_and_talk_heads
|
||||||
|
self.merged_lm_and_think_heads = merged_lm_and_think_heads
|
||||||
|
self.use_concat_talk_head = use_concat_talk_head
|
||||||
|
self.use_shallow_think = use_shallow_think
|
||||||
|
self.use_shallow_talk = use_shallow_talk
|
||||||
|
self.use_complex_think_head = use_complex_think_head
|
||||||
|
self.use_complex_talk_head = use_complex_talk_head
|
||||||
|
self.use_weighted_talk_head = use_weighted_talk_head
|
||||||
|
self.hidden_dropout_prob = hidden_dropout_prob
|
||||||
|
|
||||||
|
super().__init__(
|
||||||
|
pad_token_id=pad_token_id,
|
||||||
|
bos_token_id=bos_token_id,
|
||||||
|
eos_token_id=eos_token_id,
|
||||||
|
tie_word_embeddings=tie_word_embeddings,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
170
configuration_quiet.py
Normal file
170
configuration_quiet.py
Normal file
@@ -0,0 +1,170 @@
|
|||||||
|
# coding=utf-8
|
||||||
|
# Copyright 2023 Quiet AI and the HuggingFace Inc. team. All rights reserved.
|
||||||
|
#
|
||||||
|
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||||
|
# you may not use this file except in compliance with the License.
|
||||||
|
# You may obtain a copy of the License at
|
||||||
|
#
|
||||||
|
# http://www.apache.org/licenses/LICENSE-2.0
|
||||||
|
#
|
||||||
|
# Unless required by applicable law or agreed to in writing, software
|
||||||
|
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||||
|
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||||
|
# See the License for the specific language governing permissions and
|
||||||
|
# limitations under the License.
|
||||||
|
""" Quiet model configuration"""
|
||||||
|
|
||||||
|
from transformers.configuration_utils import PretrainedConfig
|
||||||
|
from transformers.utils import logging
|
||||||
|
|
||||||
|
|
||||||
|
logger = logging.get_logger(__name__)
|
||||||
|
|
||||||
|
QUIET_PRETRAINED_CONFIG_ARCHIVE_MAP = {
|
||||||
|
"quietai/Quiet-7B-v0.1": "https://huggingface.co/quietai/Quiet-7B-v0.1/resolve/main/config.json",
|
||||||
|
"quietai/Quiet-7B-Instruct-v0.1": "https://huggingface.co/quietai/Quiet-7B-Instruct-v0.1/resolve/main/config.json",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
|
class QuietConfig(PretrainedConfig):
|
||||||
|
r"""
|
||||||
|
This is the configuration class to store the configuration of a [`QuietModel`]. It is used to instantiate an
|
||||||
|
Quiet model according to the specified arguments, defining the model architecture. Instantiating a configuration
|
||||||
|
with the defaults will yield a similar configuration to that of the Quiet-7B-v0.1 or Quiet-7B-Instruct-v0.1.
|
||||||
|
[quietai/Quiet-7B-v0.1](https://huggingface.co/quietai/Quiet-7B-v0.1)
|
||||||
|
[quietai/Quiet-7B-Instruct-v0.1](https://huggingface.co/quietai/Quiet-7B-Instruct-v0.1)
|
||||||
|
Configuration objects inherit from [`PretrainedConfig`] and can be used to control the model outputs. Read the
|
||||||
|
documentation from [`PretrainedConfig`] for more information.
|
||||||
|
Args:
|
||||||
|
vocab_size (`int`, *optional*, defaults to 32000):
|
||||||
|
Vocabulary size of the Quiet model. Defines the number of different tokens that can be represented by the
|
||||||
|
`inputs_ids` passed when calling [`QuietModel`]
|
||||||
|
hidden_size (`int`, *optional*, defaults to 4096):
|
||||||
|
Dimension of the hidden representations.
|
||||||
|
intermediate_size (`int`, *optional*, defaults to 14336):
|
||||||
|
Dimension of the MLP representations.
|
||||||
|
num_hidden_layers (`int`, *optional*, defaults to 32):
|
||||||
|
Number of hidden layers in the Transformer encoder.
|
||||||
|
num_attention_heads (`int`, *optional*, defaults to 32):
|
||||||
|
Number of attention heads for each attention layer in the Transformer encoder.
|
||||||
|
num_key_value_heads (`int`, *optional*, defaults to 8):
|
||||||
|
This is the number of key_value heads that should be used to implement Grouped Query Attention. If
|
||||||
|
`num_key_value_heads=num_attention_heads`, the model will use Multi Head Attention (MHA), if
|
||||||
|
`num_key_value_heads=1 the model will use Multi Query Attention (MQA) otherwise GQA is used. When
|
||||||
|
converting a multi-head checkpoint to a GQA checkpoint, each group key and value head should be constructed
|
||||||
|
by meanpooling all the original heads within that group. For more details checkout [this
|
||||||
|
paper](https://arxiv.org/pdf/2305.13245.pdf). If it is not specified, will default to `8`.
|
||||||
|
hidden_act (`str` or `function`, *optional*, defaults to `"silu"`):
|
||||||
|
The non-linear activation function (function or string) in the decoder.
|
||||||
|
max_position_embeddings (`int`, *optional*, defaults to `4096*32`):
|
||||||
|
The maximum sequence length that this model might ever be used with. Quiet's sliding window attention
|
||||||
|
allows sequence of up to 4096*32 tokens.
|
||||||
|
initializer_range (`float`, *optional*, defaults to 0.02):
|
||||||
|
The standard deviation of the truncated_normal_initializer for initializing all weight matrices.
|
||||||
|
rms_norm_eps (`float`, *optional*, defaults to 1e-06):
|
||||||
|
The epsilon used by the rms normalization layers.
|
||||||
|
use_cache (`bool`, *optional*, defaults to `True`):
|
||||||
|
Whether or not the model should return the last key/values attentions (not used by all models). Only
|
||||||
|
relevant if `config.is_decoder=True`.
|
||||||
|
pad_token_id (`int`, *optional*):
|
||||||
|
The id of the padding token.
|
||||||
|
bos_token_id (`int`, *optional*, defaults to 1):
|
||||||
|
The id of the "beginning-of-sequence" token.
|
||||||
|
eos_token_id (`int`, *optional*, defaults to 2):
|
||||||
|
The id of the "end-of-sequence" token.
|
||||||
|
tie_word_embeddings (`bool`, *optional*, defaults to `False`):
|
||||||
|
Whether the model's input and output word embeddings should be tied.
|
||||||
|
rope_theta (`float`, *optional*, defaults to 10000.0):
|
||||||
|
The base period of the RoPE embeddings.
|
||||||
|
sliding_window (`int`, *optional*, defaults to 4096):
|
||||||
|
Sliding window attention window size. If not specified, will default to `4096`.
|
||||||
|
attention_dropout (`float`, *optional*, defaults to 0.0):
|
||||||
|
The dropout ratio for the attention probabilities.
|
||||||
|
```python
|
||||||
|
>>> from transformers import QuietModel, QuietConfig
|
||||||
|
>>> # Initializing a Quiet 7B style configuration
|
||||||
|
>>> configuration = QuietConfig()
|
||||||
|
>>> # Initializing a model from the Quiet 7B style configuration
|
||||||
|
>>> model = QuietModel(configuration)
|
||||||
|
>>> # Accessing the model configuration
|
||||||
|
>>> configuration = model.config
|
||||||
|
```"""
|
||||||
|
|
||||||
|
model_type = "quiet"
|
||||||
|
keys_to_ignore_at_inference = ["past_key_values"]
|
||||||
|
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
vocab_size=32000,
|
||||||
|
hidden_size=4096,
|
||||||
|
intermediate_size=14336,
|
||||||
|
num_hidden_layers=32,
|
||||||
|
num_attention_heads=32,
|
||||||
|
num_key_value_heads=8,
|
||||||
|
hidden_act="silu",
|
||||||
|
max_position_embeddings=4096 * 32,
|
||||||
|
initializer_range=0.02,
|
||||||
|
rms_norm_eps=1e-6,
|
||||||
|
use_cache=True,
|
||||||
|
pad_token_id=None,
|
||||||
|
bos_token_id=1,
|
||||||
|
eos_token_id=2,
|
||||||
|
tie_word_embeddings=False,
|
||||||
|
rope_theta=10000.0,
|
||||||
|
complexity_factor = 0.5,
|
||||||
|
sliding_window=4096,
|
||||||
|
attention_dropout=0.0,
|
||||||
|
max_thoughts=16,
|
||||||
|
max_temperature=10,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=True,
|
||||||
|
hidden_dropout_prob=0.0,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
self.vocab_size = vocab_size
|
||||||
|
self.max_position_embeddings = max_position_embeddings
|
||||||
|
self.hidden_size = hidden_size
|
||||||
|
self.intermediate_size = intermediate_size
|
||||||
|
self.num_hidden_layers = num_hidden_layers
|
||||||
|
self.num_attention_heads = num_attention_heads
|
||||||
|
self.sliding_window = sliding_window
|
||||||
|
|
||||||
|
# for backward compatibility
|
||||||
|
if num_key_value_heads is None:
|
||||||
|
num_key_value_heads = num_attention_heads
|
||||||
|
|
||||||
|
self.num_key_value_heads = num_key_value_heads
|
||||||
|
self.hidden_act = hidden_act
|
||||||
|
self.initializer_range = initializer_range
|
||||||
|
self.rms_norm_eps = rms_norm_eps
|
||||||
|
self.use_cache = use_cache
|
||||||
|
self.rope_theta = rope_theta
|
||||||
|
self.attention_dropout = attention_dropout
|
||||||
|
self.max_thoughts = max_thoughts
|
||||||
|
self.complexity_factor = complexity_factor
|
||||||
|
self.max_temperature = max_temperature
|
||||||
|
self.merged_talk_heads = merged_talk_heads
|
||||||
|
self.merged_lm_and_talk_heads = merged_lm_and_talk_heads
|
||||||
|
self.merged_lm_and_think_heads = merged_lm_and_think_heads
|
||||||
|
self.use_concat_talk_head = use_concat_talk_head
|
||||||
|
self.use_shallow_think = use_shallow_think
|
||||||
|
self.use_shallow_talk = use_shallow_talk
|
||||||
|
self.use_complex_think_head = use_complex_think_head
|
||||||
|
self.use_complex_talk_head = use_complex_talk_head
|
||||||
|
self.use_weighted_talk_head = use_weighted_talk_head
|
||||||
|
self.hidden_dropout_prob = hidden_dropout_prob
|
||||||
|
|
||||||
|
super().__init__(
|
||||||
|
pad_token_id=pad_token_id,
|
||||||
|
bos_token_id=bos_token_id,
|
||||||
|
eos_token_id=eos_token_id,
|
||||||
|
tie_word_embeddings=tie_word_embeddings,
|
||||||
|
**kwargs,
|
||||||
|
)
|
||||||
1133
configuration_utils.py
Normal file
1133
configuration_utils.py
Normal file
File diff suppressed because it is too large
Load Diff
219
generate.py
Normal file
219
generate.py
Normal file
@@ -0,0 +1,219 @@
|
|||||||
|
def custom_generate(
|
||||||
|
self,
|
||||||
|
input_ids,
|
||||||
|
attention_mask=None,
|
||||||
|
max_new_tokens=None,
|
||||||
|
min_length=None,
|
||||||
|
do_sample=None,
|
||||||
|
early_stopping=None,
|
||||||
|
num_beams=None,
|
||||||
|
temperature=None,
|
||||||
|
top_k=None,
|
||||||
|
top_p=None,
|
||||||
|
repetition_penalty=None,
|
||||||
|
bad_words_ids=None,
|
||||||
|
bos_token_id=None,
|
||||||
|
pad_token_id=None,
|
||||||
|
eos_token_id=None,
|
||||||
|
streamer=None,
|
||||||
|
length_penalty=None,
|
||||||
|
no_repeat_ngram_size=None,
|
||||||
|
num_return_sequences=None,
|
||||||
|
decoder_start_token_id=None,
|
||||||
|
use_cache=None,
|
||||||
|
num_beam_groups=None,
|
||||||
|
diversity_penalty=None,
|
||||||
|
prefix_allowed_tokens_fn=None,
|
||||||
|
output_attentions=None,
|
||||||
|
output_hidden_states=None,
|
||||||
|
output_scores=None,
|
||||||
|
return_dict_in_generate=None,
|
||||||
|
forced_bos_token_id=None,
|
||||||
|
forced_eos_token_id=None,
|
||||||
|
remove_invalid_values=None,
|
||||||
|
synced_gpus=None,
|
||||||
|
**kwargs,
|
||||||
|
):
|
||||||
|
if input_ids is None or input_ids.nelement() == 0:
|
||||||
|
# If input_ids is None or an empty tensor, create a default input tensor
|
||||||
|
input_ids = torch.LongTensor([[self.tokenizer.bos_token_id]]).to(self.device)
|
||||||
|
attention_mask = torch.ones_like(input_ids).to(self.device)
|
||||||
|
|
||||||
|
device = input_ids.device
|
||||||
|
with torch.no_grad():
|
||||||
|
batch_size = input_ids.shape[0]
|
||||||
|
finished_generating = torch.zeros(batch_size, dtype=torch.bool, device=device)
|
||||||
|
generated_token_ids = torch.full((batch_size, max_new_tokens), self.tokenizer.pad_token_id, dtype=torch.long, device=device)
|
||||||
|
|
||||||
|
for cur_token_idx in range(max_new_tokens):
|
||||||
|
# Sample the next token
|
||||||
|
new_ids = self(
|
||||||
|
input_ids[~finished_generating],
|
||||||
|
attention_mask=attention_mask[~finished_generating] if attention_mask is not None else None,
|
||||||
|
**kwargs
|
||||||
|
)['logits']
|
||||||
|
|
||||||
|
# Mask out the start and end thought tokens so we don't accidentally sample them
|
||||||
|
new_ids[:, :, self.tokenizer.vocab_size:] = -float("inf")
|
||||||
|
|
||||||
|
for list_idx, answer_idx in enumerate((~finished_generating).nonzero(as_tuple=True)[0]):
|
||||||
|
# Find the index of the last token that is not padding
|
||||||
|
base_answer_ids = input_ids[answer_idx]
|
||||||
|
new_answer_ids = new_ids[list_idx]
|
||||||
|
last_token_idx = (base_answer_ids != self.tokenizer.pad_token_id).nonzero(as_tuple=True)[0].max()
|
||||||
|
|
||||||
|
new_ids_sampled = torch.multinomial(
|
||||||
|
torch.nn.functional.softmax(new_answer_ids[last_token_idx] / temperature, dim=-1), 1)
|
||||||
|
|
||||||
|
# Assign the new id to the last token
|
||||||
|
if last_token_idx + 1 >= len(base_answer_ids):
|
||||||
|
# Add padding everywhere
|
||||||
|
new_padding = torch.full((batch_size, 1), self.tokenizer.pad_token_id, dtype=torch.long,
|
||||||
|
device=device)
|
||||||
|
input_ids = torch.cat([input_ids, new_padding], dim=-1)
|
||||||
|
if attention_mask is not None:
|
||||||
|
attention_mask = torch.cat([attention_mask, torch.zeros_like(new_padding)], dim=-1)
|
||||||
|
|
||||||
|
if attention_mask is not None:
|
||||||
|
attention_mask[answer_idx, last_token_idx + 1] = 1
|
||||||
|
input_ids[answer_idx, last_token_idx + 1] = new_ids_sampled
|
||||||
|
generated_token_ids[answer_idx, cur_token_idx] = new_ids_sampled
|
||||||
|
|
||||||
|
if new_ids_sampled == self.tokenizer.eos_token_id or new_ids_sampled == self.tokenizer.bos_token_id or new_ids_sampled == self.tokenizer.pad_token_id:
|
||||||
|
finished_generating[answer_idx] = 1
|
||||||
|
|
||||||
|
# Check if the end token is generated
|
||||||
|
if new_ids_sampled == self.tokenizer.convert_tokens_to_ids("</s>"):
|
||||||
|
finished_generating[answer_idx] = 1
|
||||||
|
|
||||||
|
if finished_generating.all():
|
||||||
|
break
|
||||||
|
|
||||||
|
if streamer is not None:
|
||||||
|
streamer.put(new_ids_sampled)
|
||||||
|
|
||||||
|
return generated_token_ids
|
||||||
|
|
||||||
|
|
||||||
|
def generate(
|
||||||
|
self,
|
||||||
|
input_ids,
|
||||||
|
attention_mask=None,
|
||||||
|
max_new_tokens=None,
|
||||||
|
min_length=None,
|
||||||
|
do_sample=None,
|
||||||
|
early_stopping=None,
|
||||||
|
num_beams=None,
|
||||||
|
temperature=1.1,
|
||||||
|
streamer=None,
|
||||||
|
top_k=None,
|
||||||
|
top_p=None,
|
||||||
|
repetition_penalty=None,
|
||||||
|
bad_words_ids=None,
|
||||||
|
bos_token_id=None,
|
||||||
|
pad_token_id=None,
|
||||||
|
eos_token_id=None,
|
||||||
|
length_penalty=None,
|
||||||
|
no_repeat_ngram_size=None,
|
||||||
|
num_return_sequences=None,
|
||||||
|
decoder_start_token_id=None,
|
||||||
|
use_cache=None,
|
||||||
|
num_beam_groups=None,
|
||||||
|
diversity_penalty=None,
|
||||||
|
prefix_allowed_tokens_fn=None,
|
||||||
|
output_attentions=None,
|
||||||
|
output_hidden_states=None,
|
||||||
|
output_scores=None,
|
||||||
|
return_dict_in_generate=None,
|
||||||
|
forced_bos_token_id=None,
|
||||||
|
forced_eos_token_id=None,
|
||||||
|
remove_invalid_values=None,
|
||||||
|
synced_gpus=None,
|
||||||
|
n_ahead=4,
|
||||||
|
n_ahead_talk=4,
|
||||||
|
merged_talk_heads=True,
|
||||||
|
merged_lm_and_talk_heads=False,
|
||||||
|
merged_lm_and_think_heads=True,
|
||||||
|
use_concat_talk_head=True,
|
||||||
|
use_shallow_think=True,
|
||||||
|
use_shallow_talk=False,
|
||||||
|
use_complex_think_head=False,
|
||||||
|
use_complex_talk_head=True,
|
||||||
|
use_weighted_talk_head=True,
|
||||||
|
trust_remote_code=True,
|
||||||
|
torch_dtype=torch.bfloat16,
|
||||||
|
**model_kwargs,
|
||||||
|
):
|
||||||
|
|
||||||
|
if max_new_tokens is None:
|
||||||
|
max_new_tokens = 128
|
||||||
|
|
||||||
|
# Set model attributes
|
||||||
|
self.max_thoughts = n_ahead + n_ahead_talk + 1
|
||||||
|
self.merged_talk_heads = merged_talk_heads
|
||||||
|
self.merged_lm_and_talk_heads = merged_lm_and_talk_heads
|
||||||
|
self.merged_lm_and_think_heads = merged_lm_and_think_heads
|
||||||
|
self.use_concat_talk_head = use_concat_talk_head
|
||||||
|
self.use_shallow_think = use_shallow_think
|
||||||
|
self.use_shallow_talk = use_shallow_talk
|
||||||
|
self.use_complex_think_head = use_complex_think_head
|
||||||
|
self.use_complex_talk_head = use_complex_talk_head
|
||||||
|
self.use_weighted_talk_head = use_weighted_talk_head
|
||||||
|
|
||||||
|
# Set model properties
|
||||||
|
self.use_end_thought_token = True
|
||||||
|
self.use_start_thought_token = True
|
||||||
|
self.n_ahead = n_ahead
|
||||||
|
self.n_passes = 1
|
||||||
|
self.eval_mode = True
|
||||||
|
self.first_run = False
|
||||||
|
self.rm_initialized = True
|
||||||
|
self.original_mode = False
|
||||||
|
|
||||||
|
# Check if the input is a string (for compatibility with text-generation-webui)
|
||||||
|
if isinstance(input_ids, str):
|
||||||
|
input_ids = self.tokenizer.encode(input_ids, return_tensors='pt')
|
||||||
|
|
||||||
|
# Move input_ids and attention_mask to the same device as the model
|
||||||
|
input_ids = input_ids.to(self.device)
|
||||||
|
if attention_mask is not None:
|
||||||
|
attention_mask = attention_mask.to(self.device)
|
||||||
|
|
||||||
|
generated_token_ids = custom_generate(
|
||||||
|
self,
|
||||||
|
input_ids=input_ids,
|
||||||
|
attention_mask=attention_mask,
|
||||||
|
max_new_tokens=max_new_tokens,
|
||||||
|
min_length=min_length,
|
||||||
|
do_sample=do_sample,
|
||||||
|
early_stopping=early_stopping,
|
||||||
|
num_beams=num_beams,
|
||||||
|
temperature=temperature,
|
||||||
|
top_k=top_k,
|
||||||
|
top_p=top_p,
|
||||||
|
repetition_penalty=repetition_penalty,
|
||||||
|
bad_words_ids=bad_words_ids,
|
||||||
|
bos_token_id=bos_token_id,
|
||||||
|
pad_token_id=pad_token_id,
|
||||||
|
eos_token_id=eos_token_id,
|
||||||
|
length_penalty=length_penalty,
|
||||||
|
no_repeat_ngram_size=no_repeat_ngram_size,
|
||||||
|
num_return_sequences=num_return_sequences,
|
||||||
|
decoder_start_token_id=decoder_start_token_id,
|
||||||
|
use_cache=use_cache,
|
||||||
|
num_beam_groups=num_beam_groups,
|
||||||
|
diversity_penalty=diversity_penalty,
|
||||||
|
prefix_allowed_tokens_fn=prefix_allowed_tokens_fn,
|
||||||
|
output_attentions=output_attentions,
|
||||||
|
output_hidden_states=output_hidden_states,
|
||||||
|
output_scores=output_scores,
|
||||||
|
return_dict_in_generate=return_dict_in_generate,
|
||||||
|
forced_bos_token_id=forced_bos_token_id,
|
||||||
|
forced_eos_token_id=forced_eos_token_id,
|
||||||
|
remove_invalid_values=remove_invalid_values,
|
||||||
|
synced_gpus=synced_gpus,
|
||||||
|
streamer=streamer,
|
||||||
|
**model_kwargs,
|
||||||
|
)
|
||||||
|
|
||||||
|
return generated_token_ids
|
||||||
6
generation_config.json
Normal file
6
generation_config.json
Normal file
@@ -0,0 +1,6 @@
|
|||||||
|
{
|
||||||
|
"_from_model_config": true,
|
||||||
|
"bos_token_id": 1,
|
||||||
|
"eos_token_id": 2,
|
||||||
|
"transformers_version": "4.40.0.dev0"
|
||||||
|
}
|
||||||
22
mergekit_config.yml
Normal file
22
mergekit_config.yml
Normal file
@@ -0,0 +1,22 @@
|
|||||||
|
|
||||||
|
slices:
|
||||||
|
- sources:
|
||||||
|
- model: LeroyDyer/Mixtral_AI_CyberBrain_3_0
|
||||||
|
layer_range: [0, 32]
|
||||||
|
- model: ezelikman/quietstar-8-ahead
|
||||||
|
layer_range: [0, 32]
|
||||||
|
# or, the equivalent models: syntax:
|
||||||
|
# models:
|
||||||
|
# - model: mistralai/Mistral-7B-Instruct-v0.2
|
||||||
|
# LaRGER MODEL MUST BE BASE
|
||||||
|
# - model: yanismiraoui/Yarn-Mistral-7b-128k-sharded
|
||||||
|
merge_method: slerp
|
||||||
|
base_model: ezelikman/quietstar-8-ahead
|
||||||
|
parameters:
|
||||||
|
t:
|
||||||
|
- filter: self_attn
|
||||||
|
value: [0.3, 0.6, 0.3786, 0.6, 0.6]
|
||||||
|
- filter: mlp
|
||||||
|
value: [0.7, 0.4, 0.6, 0.4, 0.7]
|
||||||
|
- value: 0.5 # fallback for rest of tensors
|
||||||
|
dtype: float16
|
||||||
3
mm_projector.bin
Normal file
3
mm_projector.bin
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:be33a3b477e091832a3c8a9fabf5769c43b4cb2c161fc8c464b7bb214f16143a
|
||||||
|
size 41961085
|
||||||
3
model-00001-of-00008.safetensors
Normal file
3
model-00001-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:9aa1b16f05743b2f339f6e70521c166198221a90edc9a9559a2d666416020ca6
|
||||||
|
size 1973489936
|
||||||
3
model-00002-of-00008.safetensors
Normal file
3
model-00002-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:356d76c09d93a81c26368bb2fee8698d036ef21b6ee7007cf88e71463adc642e
|
||||||
|
size 1979798000
|
||||||
3
model-00003-of-00008.safetensors
Normal file
3
model-00003-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:907a3d64fc26868db58b658da99d2aa49f979d8db0d459c703637416b671368c
|
||||||
|
size 1946227312
|
||||||
3
model-00004-of-00008.safetensors
Normal file
3
model-00004-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:68dd86eba610c49dbdf57c29721f7d7040358eaf990fc3d6b1c0ce3d6fddb153
|
||||||
|
size 1979798016
|
||||||
3
model-00005-of-00008.safetensors
Normal file
3
model-00005-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:7abb6a3613474b4eae84e76fd4636149738b693d10eb12242250b37cd2af600b
|
||||||
|
size 1946227336
|
||||||
3
model-00006-of-00008.safetensors
Normal file
3
model-00006-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:22c18413958dacaa1f724841221f89a772a9f2aca52ec92c375e78a054e30813
|
||||||
|
size 1979798016
|
||||||
3
model-00007-of-00008.safetensors
Normal file
3
model-00007-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:1725e5ba912edc98426340511e1984347d123ca4d92c66cbd6ba7e9d3ee1309a
|
||||||
|
size 1862340784
|
||||||
3
model-00008-of-00008.safetensors
Normal file
3
model-00008-of-00008.safetensors
Normal file
@@ -0,0 +1,3 @@
|
|||||||
|
version https://git-lfs.github.com/spec/v1
|
||||||
|
oid sha256:95e0b00098431ab705f87191f852793a1082b2f5e373c060f22c2ea3804ff51a
|
||||||
|
size 815851048
|
||||||
1
model.safetensors.index.json
Normal file
1
model.safetensors.index.json
Normal file
File diff suppressed because one or more lines are too long
4627
modeling_mistral.py
Normal file
4627
modeling_mistral.py
Normal file
File diff suppressed because it is too large
Load Diff
1489
modeling_mistral_yarn.py
Normal file
1489
modeling_mistral_yarn.py
Normal file
File diff suppressed because it is too large
Load Diff
2369
modeling_quiet.py
Normal file
2369
modeling_quiet.py
Normal file
File diff suppressed because it is too large
Load Diff
28
preprocessor_config.json
Normal file
28
preprocessor_config.json
Normal file
@@ -0,0 +1,28 @@
|
|||||||
|
{
|
||||||
|
"crop_size": {
|
||||||
|
"height": 336,
|
||||||
|
"width": 336
|
||||||
|
},
|
||||||
|
"do_center_crop": true,
|
||||||
|
"do_convert_rgb": true,
|
||||||
|
"do_normalize": true,
|
||||||
|
"do_rescale": true,
|
||||||
|
"do_resize": true,
|
||||||
|
"image_mean": [
|
||||||
|
0.48145466,
|
||||||
|
0.4578275,
|
||||||
|
0.40821073
|
||||||
|
],
|
||||||
|
"image_processor_type": "CLIPImageProcessor",
|
||||||
|
"image_std": [
|
||||||
|
0.26862954,
|
||||||
|
0.26130258,
|
||||||
|
0.27577711
|
||||||
|
],
|
||||||
|
"processor_class": "LlavaProcessor",
|
||||||
|
"resample": 3,
|
||||||
|
"rescale_factor": 0.00392156862745098,
|
||||||
|
"size": {
|
||||||
|
"shortest_edge": 336
|
||||||
|
}
|
||||||
|
}
|
||||||
34
special_tokens_map.json
Normal file
34
special_tokens_map.json
Normal file
@@ -0,0 +1,34 @@
|
|||||||
|
{
|
||||||
|
"additional_special_tokens": [
|
||||||
|
"<|endthought|>",
|
||||||
|
"<|startthought|>"
|
||||||
|
],
|
||||||
|
"bos_token": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"eos_token": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"pad_token": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
},
|
||||||
|
"unk_token": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false
|
||||||
|
}
|
||||||
|
}
|
||||||
91140
tokenizer.json
Normal file
91140
tokenizer.json
Normal file
File diff suppressed because it is too large
Load Diff
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
BIN
tokenizer.model
(Stored with Git LFS)
Normal file
Binary file not shown.
61
tokenizer_config.json
Normal file
61
tokenizer_config.json
Normal file
@@ -0,0 +1,61 @@
|
|||||||
|
{
|
||||||
|
"add_bos_token": true,
|
||||||
|
"add_eos_token": false,
|
||||||
|
"added_tokens_decoder": {
|
||||||
|
"0": {
|
||||||
|
"content": "<unk>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"1": {
|
||||||
|
"content": "<s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"2": {
|
||||||
|
"content": "</s>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"32000": {
|
||||||
|
"content": "<|endthought|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
},
|
||||||
|
"32001": {
|
||||||
|
"content": "<|startthought|>",
|
||||||
|
"lstrip": false,
|
||||||
|
"normalized": false,
|
||||||
|
"rstrip": false,
|
||||||
|
"single_word": false,
|
||||||
|
"special": true
|
||||||
|
}
|
||||||
|
},
|
||||||
|
"additional_special_tokens": [
|
||||||
|
"<|endthought|>",
|
||||||
|
"<|startthought|>"
|
||||||
|
],
|
||||||
|
"bos_token": "<s>",
|
||||||
|
"clean_up_tokenization_spaces": false,
|
||||||
|
"eos_token": "</s>",
|
||||||
|
"legacy": true,
|
||||||
|
"model_max_length": 1000000000000000019884624838656,
|
||||||
|
"pad_token": "</s>",
|
||||||
|
"sp_model_kwargs": {},
|
||||||
|
"spaces_between_special_tokens": false,
|
||||||
|
"tokenizer_class": "LlamaTokenizer",
|
||||||
|
"unk_token": "<unk>",
|
||||||
|
"use_default_system_prompt": false
|
||||||
|
}
|
||||||
Reference in New Issue
Block a user