初始化项目,由ModelHub XC社区提供模型
Model: llmat/Mistral-Small-24B-Instruct-2501-NVFP4 Source: Original Platform
This commit is contained in:
6
.gitattributes
vendored
Normal file
6
.gitattributes
vendored
Normal file
@@ -0,0 +1,6 @@
|
||||
*.bin filter=lfs diff=lfs merge=lfs -text
|
||||
*.safetensors filter=lfs diff=lfs merge=lfs -text
|
||||
*.pt filter=lfs diff=lfs merge=lfs -text
|
||||
*.model filter=lfs diff=lfs merge=lfs -text
|
||||
*.gguf filter=lfs diff=lfs merge=lfs -text
|
||||
tokenizer.json filter=lfs diff=lfs merge=lfs -text
|
||||
62
README.md
Normal file
62
README.md
Normal file
@@ -0,0 +1,62 @@
|
||||
---
|
||||
language:
|
||||
- en
|
||||
- de
|
||||
- fr
|
||||
- it
|
||||
- pt
|
||||
- hi
|
||||
- es
|
||||
- th
|
||||
pipeline_tag: text-generation
|
||||
license: apache-2.0
|
||||
tags:
|
||||
- quantization
|
||||
- nvfp4
|
||||
- vllm
|
||||
model_name: Mistral-Small-24B-Instruct-2501-NVFP4
|
||||
base_model: mistralai/Mistral-Small-24B-Instruct-2501
|
||||
---
|
||||
|
||||
# Mistral-Small-24B-Instruct-2501-NVFP4
|
||||
|
||||
NVFP4-quantized version of `mistralai/Mistral-Small-24B-Instruct-2501` produced with [llmcompressor](https://github.com/neuralmagic/llm-compressor).
|
||||
|
||||
## Notes
|
||||
- Quantization scheme: NVFP4 (linear layers, `lm_head` excluded)
|
||||
- Calibration samples: 512
|
||||
- Max sequence length during calibration: 2048
|
||||
|
||||
## Deployment
|
||||
|
||||
### Use with vLLM
|
||||
|
||||
This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/latest/) backend, as shown in the example below.
|
||||
|
||||
```python
|
||||
from vllm import LLM, SamplingParams
|
||||
from transformers import AutoTokenizer
|
||||
|
||||
model_id = "llmat/Mistral-Small-24B-Instruct-2501-NVFP4"
|
||||
number_gpus = 1
|
||||
|
||||
sampling_params = SamplingParams(temperature=0.6, top_p=0.9, max_tokens=256)
|
||||
|
||||
tokenizer = AutoTokenizer.from_pretrained(model_id)
|
||||
|
||||
messages = [
|
||||
{"role": "system", "content": "You are a pirate chatbot who always responds in pirate speak!"},
|
||||
{"role": "user", "content": "Who are you?"},
|
||||
]
|
||||
|
||||
prompts = tokenizer.apply_chat_template(messages, add_generation_prompt=True, tokenize=False)
|
||||
|
||||
llm = LLM(model=model_id, tensor_parallel_size=number_gpus)
|
||||
|
||||
outputs = llm.generate(prompts, sampling_params)
|
||||
|
||||
generated_text = outputs[0].outputs[0].text
|
||||
print(generated_text)
|
||||
```
|
||||
|
||||
vLLM aslo supports OpenAI-compatible serving. See the [documentation](https://docs.vllm.ai/en/latest/) for more details.
|
||||
25
chat_template.jinja
Normal file
25
chat_template.jinja
Normal file
@@ -0,0 +1,25 @@
|
||||
{%- set today = strftime_now("%Y-%m-%d") %}
|
||||
{%- set default_system_message = "You are Mistral Small 3, a Large Language Model (LLM) created by Mistral AI, a French startup headquartered in Paris.\nYour knowledge base was last updated on 2023-10-01. The current date is " + today + ".\n\nWhen you're not sure about some information, you say that you don't have the information and don't make up anything.\nIf the user's question is not clear, ambiguous, or does not provide enough context for you to accurately answer the question, you do not try to answer it right away and you rather ask the user to clarify their request (e.g. \"What are some good restaurants around me?\" => \"Where are you?\" or \"When is the next flight to Tokyo\" => \"Where do you travel from?\")" %}
|
||||
|
||||
{{- bos_token }}
|
||||
|
||||
{%- if messages[0]['role'] == 'system' %}
|
||||
{%- set system_message = messages[0]['content'] %}
|
||||
{%- set loop_messages = messages[1:] %}
|
||||
{%- else %}
|
||||
{%- set system_message = default_system_message %}
|
||||
{%- set loop_messages = messages %}
|
||||
{%- endif %}
|
||||
{{- '[SYSTEM_PROMPT]' + system_message + '[/SYSTEM_PROMPT]' }}
|
||||
|
||||
{%- for message in loop_messages %}
|
||||
{%- if message['role'] == 'user' %}
|
||||
{{- '[INST]' + message['content'] + '[/INST]' }}
|
||||
{%- elif message['role'] == 'system' %}
|
||||
{{- '[SYSTEM_PROMPT]' + message['content'] + '[/SYSTEM_PROMPT]' }}
|
||||
{%- elif message['role'] == 'assistant' %}
|
||||
{{- message['content'] + eos_token }}
|
||||
{%- else %}
|
||||
{{- raise_exception('Only user, system and assistant roles are supported!') }}
|
||||
{%- endif %}
|
||||
{%- endfor %}
|
||||
72
config.json
Normal file
72
config.json
Normal file
@@ -0,0 +1,72 @@
|
||||
{
|
||||
"architectures": [
|
||||
"MistralForCausalLM"
|
||||
],
|
||||
"attention_dropout": 0.0,
|
||||
"bos_token_id": 1,
|
||||
"eos_token_id": 2,
|
||||
"head_dim": 128,
|
||||
"hidden_act": "silu",
|
||||
"hidden_size": 5120,
|
||||
"initializer_range": 0.02,
|
||||
"intermediate_size": 32768,
|
||||
"max_position_embeddings": 32768,
|
||||
"model_type": "mistral",
|
||||
"num_attention_heads": 32,
|
||||
"num_hidden_layers": 40,
|
||||
"num_key_value_heads": 8,
|
||||
"quantization_config": {
|
||||
"config_groups": {
|
||||
"group_0": {
|
||||
"format": "nvfp4-pack-quantized",
|
||||
"input_activations": {
|
||||
"actorder": null,
|
||||
"block_structure": null,
|
||||
"dynamic": "local",
|
||||
"group_size": 16,
|
||||
"num_bits": 4,
|
||||
"observer": "minmax",
|
||||
"observer_kwargs": {},
|
||||
"strategy": "tensor_group",
|
||||
"symmetric": true,
|
||||
"type": "float"
|
||||
},
|
||||
"output_activations": null,
|
||||
"targets": [
|
||||
"Linear"
|
||||
],
|
||||
"weights": {
|
||||
"actorder": null,
|
||||
"block_structure": null,
|
||||
"dynamic": false,
|
||||
"group_size": 16,
|
||||
"num_bits": 4,
|
||||
"observer": "minmax",
|
||||
"observer_kwargs": {},
|
||||
"strategy": "tensor_group",
|
||||
"symmetric": true,
|
||||
"type": "float"
|
||||
}
|
||||
}
|
||||
},
|
||||
"format": "nvfp4-pack-quantized",
|
||||
"global_compression_ratio": null,
|
||||
"ignore": [
|
||||
"lm_head"
|
||||
],
|
||||
"kv_cache_scheme": null,
|
||||
"quant_method": "compressed-tensors",
|
||||
"quantization_status": "compressed",
|
||||
"sparsity_config": {},
|
||||
"transform_config": {},
|
||||
"version": "0.11.0"
|
||||
},
|
||||
"rms_norm_eps": 1e-05,
|
||||
"rope_theta": 100000000.0,
|
||||
"sliding_window": null,
|
||||
"tie_word_embeddings": false,
|
||||
"torch_dtype": "bfloat16",
|
||||
"transformers_version": "4.55.4",
|
||||
"use_cache": true,
|
||||
"vocab_size": 131072
|
||||
}
|
||||
8
generation_config.json
Normal file
8
generation_config.json
Normal file
@@ -0,0 +1,8 @@
|
||||
{
|
||||
"_from_model_config": true,
|
||||
"bos_token_id": 1,
|
||||
"do_sample": true,
|
||||
"eos_token_id": 2,
|
||||
"temperature": 0.15,
|
||||
"transformers_version": "4.55.4"
|
||||
}
|
||||
3
model-00001-of-00004.safetensors
Normal file
3
model-00001-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:896e73dae49172891f466d17bf74e11d1268c8f16d1cc21696b17154d4b7fa5e
|
||||
size 4999352440
|
||||
3
model-00002-of-00004.safetensors
Normal file
3
model-00002-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b3d311183ee633899a786d0ff4de1fc26a8a8869faa17804ebf39ec95a244e98
|
||||
size 4991604736
|
||||
3
model-00003-of-00004.safetensors
Normal file
3
model-00003-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:31dd56a12bf9dd16e1d65b77ce291824ce0bcc997345b6d59ff40a66500b537b
|
||||
size 3856457144
|
||||
3
model-00004-of-00004.safetensors
Normal file
3
model-00004-of-00004.safetensors
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:58034435463df5dfb748849bb5002d84b1ad1a1b308dfe26f5420e3e0a340cba
|
||||
size 1342177408
|
||||
1211
model.safetensors.index.json
Normal file
1211
model.safetensors.index.json
Normal file
File diff suppressed because it is too large
Load Diff
6
recipe.yaml
Normal file
6
recipe.yaml
Normal file
@@ -0,0 +1,6 @@
|
||||
default_stage:
|
||||
default_modifiers:
|
||||
QuantizationModifier:
|
||||
targets: [Linear]
|
||||
ignore: [lm_head]
|
||||
scheme: NVFP4
|
||||
1025
special_tokens_map.json
Normal file
1025
special_tokens_map.json
Normal file
File diff suppressed because it is too large
Load Diff
3
tokenizer.json
Normal file
3
tokenizer.json
Normal file
@@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:b639fcaeefa983aa47e9108527764e3b0e20cf34bae9bf19c6ea8406b35b8428
|
||||
size 17078136
|
||||
9018
tokenizer_config.json
Normal file
9018
tokenizer_config.json
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user