From 9074d725facf92fed7612e6d8dd169296e7b7ba3 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Sun, 30 Aug 2026 12:46:13 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: RedHatAI/Qwen2.5-7B-Instruct-quantized.w8a8 Source: Original Platform --- .gitattributes | 36 +++ README.md | 448 +++++++++++++++++++++++++++++++ added_tokens.json | 24 ++ config.json | 3 + configuration.json | 1 + generation_config.json | 14 + merges.txt | 3 + model-00001-of-00002.safetensors | 3 + model-00002-of-00002.safetensors | 3 + model.safetensors.index.json | 3 + recipe.yaml | 18 ++ special_tokens_map.json | 31 +++ tokenizer.json | 3 + tokenizer_config.json | 3 + vocab.json | 3 + 15 files changed, 596 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 added_tokens.json create mode 100644 config.json create mode 100644 configuration.json create mode 100644 generation_config.json create mode 100644 merges.txt create mode 100644 model-00001-of-00002.safetensors create mode 100644 model-00002-of-00002.safetensors create mode 100644 model.safetensors.index.json create mode 100644 recipe.yaml create mode 100644 special_tokens_map.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json create mode 100644 vocab.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..b846b69 --- /dev/null +++ b/README.md @@ -0,0 +1,448 @@ +--- +language: +- zh +- en +- fr +- es +- pt +- de +- it +- ru +- ja +- ko +- vi +- th +- ar +- id +- tr +- fa +- nl +- pl +- cs +- he +- sv +- fi +- da +- no +- el +- bg +- uk +- ur +- sr +- ms +- zsm +- nld +base_model: +- Qwen/Qwen2.5-7B-Instruct +pipeline_tag: text-generation +tags: +- qwen +- qwen2_5 +- qwen2_5_instruct +- w8a8 +- int8 +- vllm +- conversational +- text-generation-inference +- compressed-tensors +license: apache-2.0 +license_name: apache-2.0 +name: RedHatAI/Qwen2.5-7B-Instruct-quantized.w8a8 +description: This model was obtained by quantizing the weights and activations of Qwen2.5-7B-Instruct to INT8 data type. +readme: https://huggingface.co/RedHatAI/Qwen2.5-7B-Instruct-quantized.w8a8/main/README.md +tasks: +- text-to-text +provider: Alibaba Cloud +license_link: https://www.apache.org/licenses/LICENSE-2.0 +validated_on: + - RHOAI 2.20 + - RHAIIS 3.0 + - RHELAI 1.5 + - vLLM 0.8.4 +--- + +

+ Qwen2.5-7B-Instruct-quantized.w8a8 + Model Icon +

+ + +Validated Badge + + +## Model Overview +- **Model Architecture:** Qwen2 + - **Input:** Text + - **Output:** Text +- **Model Optimizations:** + - **Activation quantization:** INT8 + - **Weight quantization:** INT8 +- **Intended Use Cases:** Intended for commercial and research use multiple languages. Similarly to [Qwen2.5-7B](https://huggingface.co/Qwen/Qwen2.5-7B), this models is intended for assistant-like chat. +- **Out-of-scope:** Use in any manner that violates applicable laws or regulations (including trade compliance laws). +- **Release Date:** 10/09/2024 +- **Version:** 1.0 +- **Validated on:** RHOAI 2.20, RHAIIS 3.0, RHELAI 1.5 +- **License(s):** [apache-2.0](https://huggingface.co/Qwen/Qwen2.5-7B/blob/main/LICENSE) +- **Model Developers:** Neural Magic + +### Model Optimizations + +This model was obtained by quantizing activations and weights of [Qwen2.5-7B-Instruct](https://huggingface.co/Qwen/Qwen2.5-7B-Instruct) to INT8 data type. +This optimization reduces the number of bits used to represent weights and activations from 16 to 8, reducing GPU memory requirements (by approximately 50%) and increasing matrix-multiply compute throughput (by approximately 2x). +Weight quantization also reduces disk size requirements by approximately 50%. + +Only weights and activations of the linear operators within transformers blocks are quantized. +Weights are quantized with a symmetric static per-channel scheme, whereas activations are quantized with a symmetric dynamic per-token scheme. +A combination of the [SmoothQuant](https://arxiv.org/abs/2211.10438) and [GPTQ](https://arxiv.org/abs/2210.17323) algorithms is applied for quantization, as implemented in the [llm-compressor](https://github.com/vllm-project/llm-compressor) library. + +## Deployment + +This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/latest/) backend, as shown in the example below. + +```python +from vllm import LLM, SamplingParams +from transformers import AutoTokenizer + +model_id = "RedHatAI/Qwen2.5-7B-Instruct-quantized.w8a8" +number_gpus = 1 +max_model_len = 8192 + +sampling_params = SamplingParams(temperature=0.7, top_p=0.8, max_tokens=256) + +tokenizer = AutoTokenizer.from_pretrained(model_id) + +messages = [ + {"role": "user", "content": "Give me a short introduction to large language model."}, +] + +prompts = tokenizer.apply_chat_template(messages, tokenize=False) + +llm = LLM(model=model_id, tensor_parallel_size=number_gpus, max_model_len=max_model_len) + +outputs = llm.generate(prompts, sampling_params) + +generated_text = outputs[0].outputs[0].text +print(generated_text) +``` + +vLLM aslo supports OpenAI-compatible serving. See the [documentation](https://docs.vllm.ai/en/latest/) for more details. + +
+ Deploy on Red Hat AI Inference Server + +```bash +podman run --rm -it --device nvidia.com/gpu=all -p 8000:8000 \ + --ipc=host \ +--env "HUGGING_FACE_HUB_TOKEN=$HF_TOKEN" \ +--env "HF_HUB_OFFLINE=0" -v ~/.cache/vllm:/home/vllm/.cache \ +--name=vllm \ +registry.access.redhat.com/rhaiis/rh-vllm-cuda \ +vllm serve \ +--tensor-parallel-size 8 \ +--max-model-len 32768 \ +--enforce-eager --model RedHatAI/Qwen2.5-7B-Instruct-quantized.w8a8 +``` +​​See [Red Hat AI Inference Server documentation](https://docs.redhat.com/en/documentation/red_hat_ai_inference_server/) for more details. +
+ +
+ Deploy on Red Hat Enterprise Linux AI + +```bash +# Download model from Red Hat Registry via docker +# Note: This downloads the model to ~/.cache/instructlab/models unless --model-dir is specified. +ilab model download --repository docker://registry.redhat.io/rhelai1/qwen2-5-7b-instruct-quantized-w8a8:1.5 +``` + +```bash +# Serve model via ilab +ilab model serve --model-path ~/.cache/instructlab/models/qwen2-5-7b-instruct-quantized-w8a8 + +# Chat with model +ilab model chat --model ~/.cache/instructlab/models/qwen2-5-7b-instruct-quantized-w8a8 +``` +See [Red Hat Enterprise Linux AI documentation](https://docs.redhat.com/en/documentation/red_hat_enterprise_linux_ai/1.4) for more details. +
+ +
+ Deploy on Red Hat Openshift AI + +```python +# Setting up vllm server with ServingRuntime +# Save as: vllm-servingruntime.yaml +apiVersion: serving.kserve.io/v1alpha1 +kind: ServingRuntime +metadata: + name: vllm-cuda-runtime # OPTIONAL CHANGE: set a unique name + annotations: + openshift.io/display-name: vLLM NVIDIA GPU ServingRuntime for KServe + opendatahub.io/recommended-accelerators: '["nvidia.com/gpu"]' + labels: + opendatahub.io/dashboard: 'true' +spec: + annotations: + prometheus.io/port: '8080' + prometheus.io/path: '/metrics' + multiModel: false + supportedModelFormats: + - autoSelect: true + name: vLLM + containers: + - name: kserve-container + image: quay.io/modh/vllm:rhoai-2.20-cuda # CHANGE if needed. If AMD: quay.io/modh/vllm:rhoai-2.20-rocm + command: + - python + - -m + - vllm.entrypoints.openai.api_server + args: + - "--port=8080" + - "--model=/mnt/models" + - "--served-model-name={{.Name}}" + env: + - name: HF_HOME + value: /tmp/hf_home + ports: + - containerPort: 8080 + protocol: TCP +``` + +```python +# Attach model to vllm server. This is an NVIDIA template +# Save as: inferenceservice.yaml +apiVersion: serving.kserve.io/v1beta1 +kind: InferenceService +metadata: + annotations: + openshift.io/display-name: Qwen2.5-7B-Instruct-quantized.w8a8 # OPTIONAL CHANGE + serving.kserve.io/deploymentMode: RawDeployment + name: Qwen2.5-7B-Instruct-quantized.w8a8 # specify model name. This value will be used to invoke the model in the payload + labels: + opendatahub.io/dashboard: 'true' +spec: + predictor: + maxReplicas: 1 + minReplicas: 1 + model: + modelFormat: + name: vLLM + name: '' + resources: + limits: + cpu: '2' # this is model specific + memory: 8Gi # this is model specific + nvidia.com/gpu: '1' # this is accelerator specific + requests: # same comment for this block + cpu: '1' + memory: 4Gi + nvidia.com/gpu: '1' + runtime: vllm-cuda-runtime # must match the ServingRuntime name above + storageUri: oci://registry.redhat.io/rhelai1/modelcar-qwen2-5-7b-instruct-quantized-w8a8:1.5 + tolerations: + - effect: NoSchedule + key: nvidia.com/gpu + operator: Exists +``` + +```bash +# make sure first to be in the project where you want to deploy the model +# oc project +# apply both resources to run model +# Apply the ServingRuntime +oc apply -f vllm-servingruntime.yaml +# Apply the InferenceService +oc apply -f qwen-inferenceservice.yaml +``` + +```python +# Replace and below: +# - Run `oc get inferenceservice` to find your URL if unsure. +# Call the server using curl: +curl https://-predictor-default./v1/chat/completions + -H "Content-Type: application/json" \ + -d '{ + "model": "Qwen2.5-7B-Instruct-quantized.w8a8", + "stream": true, + "stream_options": { + "include_usage": true + }, + "max_tokens": 1, + "messages": [ + { + "role": "user", + "content": "How can a bee fly when its wings are so small?" + } + ] +}' +``` + +See [Red Hat Openshift AI documentation](https://docs.redhat.com/en/documentation/red_hat_openshift_ai/2025) for more details. +
+ +## Creation + +
+ Creation details + This model was created with [llm-compressor](https://github.com/vllm-project/llm-compressor) by running the code snippet below. + + + ```python + from transformers import AutoModelForCausalLM, AutoTokenizer + from llmcompressor.modifiers.quantization import GPTQModifier + from llmcompressor.modifiers.smoothquant import SmoothQuantModifier + from llmcompressor.transformers import oneshot + from datasets import load_dataset + + # Load model + model_stub = "Qwen/Qwen2.5-7B-Instruct" + model_name = model_stub.split("/")[-1] + + num_samples = 512 + max_seq_len = 8192 + + tokenizer = AutoTokenizer.from_pretrained(model_stub) + + model = AutoModelForCausalLM.from_pretrained( + model_stub, + device_map="auto", + torch_dtype="auto", + ) + + def preprocess_fn(example): + return {"text": tokenizer.apply_chat_template(example["messages"], add_generation_prompt=False, tokenize=False)} + + ds = load_dataset("neuralmagic/LLM_compression_calibration", split="train") + ds = ds.map(preprocess_fn) + + # Configure the quantization algorithm and scheme + recipe = [ + SmoothQuantModifier( + smoothing_strength=0.8, + mappings=[ + [["re:.*q_proj", "re:.*k_proj", "re:.*v_proj"], "re:.*input_layernorm"], + [["re:.*gate_proj", "re:.*up_proj"], "re:.*post_attention_layernorm"], + [["re:.*down_proj"], "re:.*up_proj"], + ], + ), + GPTQModifier( + ignore=["lm_head"], + sequential_targets=["Qwen2DecoderLayer"], + dampening_frac=0.01, + targets="Linear", + scheme="W8A8", + ), + ] + + # Apply quantization + oneshot( + model=model, + dataset=ds, + recipe=recipe, + max_seq_length=max_seq_len, + num_calibration_samples=num_samples, + ) + + # Save to disk in compressed-tensors format + save_path = model_name + "-quantized.w8a8" + model.save_pretrained(save_path) + tokenizer.save_pretrained(save_path) + print(f"Model and tokenizer saved to: {save_path}") + ``` +
+ +## Evaluation + +The model was evaluated on the [OpenLLM](https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard) leaderboard tasks (version 1) with the [lm-evaluation-harness](https://github.com/EleutherAI/lm-evaluation-harness/tree/387Bbd54bc621086e05aa1b030d8d4d5635b25e6) (commit 387Bbd54bc621086e05aa1b030d8d4d5635b25e6) and the [vLLM](https://docs.vllm.ai/en/stable/) engine, using the following command: +``` +lm_eval \ + --model vllm \ + --model_args pretrained="neuralmagic/Qwen2.5-7B-Instruct-quantized.w8a8",dtype=auto,gpu_memory_utilization=0.5,max_model_len=4096,add_bos_token=True,enable_chunk_prefill=True,tensor_parallel_size=1 \ + --tasks openllm \ + --batch_size auto +``` + +### Accuracy + +#### Open LLM Leaderboard evaluation scores + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Benchmark + Qwen2.5-7B-Instruct + Qwen2.5-7B-Instruct-quantized.w8a8
(this model) +
Recovery +
MMLU (5-shot) + 74.24 + 73.87 + 99.5% +
ARC Challenge (25-shot) + 63.40 + 63.23 + 99.7% +
GSM-8K (5-shot, strict-match) + 80.36 + 80.74 + 100.5% +
Hellaswag (10-shot) + 81.52 + 81.06 + 99.4% +
Winogrande (5-shot) + 74.66 + 74.82 + 100.2% +
TruthfulQA (0-shot, mc2) + 64.76 + 64.58 + 99.7% +
Average + 73.16 + 73.05 + 99.4% +
+ diff --git a/added_tokens.json b/added_tokens.json new file mode 100644 index 0000000..482ced4 --- /dev/null +++ b/added_tokens.json @@ -0,0 +1,24 @@ +{ + "": 151658, + "": 151657, + "<|box_end|>": 151649, + "<|box_start|>": 151648, + "<|endoftext|>": 151643, + "<|file_sep|>": 151664, + "<|fim_middle|>": 151660, + "<|fim_pad|>": 151662, + "<|fim_prefix|>": 151659, + "<|fim_suffix|>": 151661, + "<|im_end|>": 151645, + "<|im_start|>": 151644, + "<|image_pad|>": 151655, + "<|object_ref_end|>": 151647, + "<|object_ref_start|>": 151646, + "<|quad_end|>": 151651, + "<|quad_start|>": 151650, + "<|repo_name|>": 151663, + "<|video_pad|>": 151656, + "<|vision_end|>": 151653, + "<|vision_pad|>": 151654, + "<|vision_start|>": 151652 +} diff --git a/config.json b/config.json new file mode 100644 index 0000000..5933afc --- /dev/null +++ b/config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0744eaffc6b3c050da145fd9d2f3b511b75aea1c1ba8b5988ec876aae1e8a3c3 +size 1921 diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..8ab8c29 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,14 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "repetition_penalty": 1.05, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "4.45.1" +} diff --git a/merges.txt b/merges.txt new file mode 100644 index 0000000..80c1a19 --- /dev/null +++ b/merges.txt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5 +size 1671853 diff --git a/model-00001-of-00002.safetensors b/model-00001-of-00002.safetensors new file mode 100644 index 0000000..b2678c9 --- /dev/null +++ b/model-00001-of-00002.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c75a04f4e867253bd2340d1f0968c90dc388d463383baea97f2a8a0c71fb5155 +size 4985986104 diff --git a/model-00002-of-00002.safetensors b/model-00002-of-00002.safetensors new file mode 100644 index 0000000..fb5cba2 --- /dev/null +++ b/model-00002-of-00002.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:159e10054aed8340a387b82fed0f2ce7891c2d53a2e9f23c2a9d13b8da17a945 +size 3722801480 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..2f59f70 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9032d296d5fdc3b749bdd1d5e5353e4d6545b908376d2e40c6f0c7081d786329 +size 44817 diff --git a/recipe.yaml b/recipe.yaml new file mode 100644 index 0000000..c64846d --- /dev/null +++ b/recipe.yaml @@ -0,0 +1,18 @@ +quant_stage: + quant_modifiers: + SmoothQuantModifier: + smoothing_strength: 0.8 + mappings: + - - ['re:.*q_proj', 're:.*k_proj', 're:.*v_proj'] + - re:.*input_layernorm + - - ['re:.*gate_proj', 're:.*up_proj'] + - re:.*post_attention_layernorm + - - ['re:.*down_proj'] + - re:.*up_proj + GPTQModifier: + sequential_update: true + dampening_frac: 0.01 + ignore: [lm_head] + scheme: W8A8 + targets: Linear + observer: mse diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..ac23c0a --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,31 @@ +{ + "additional_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "eos_token": { + "content": "<|im_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "<|endoftext|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..33d22a4 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb73a25aba3c83c6c815a03a334b0440bd549f9a54fa3673e005f5532f6b32fe +size 11421995 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..a12c302 --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e88129d9769a0b14b1587a7d5e829fe93ac0e1511636471fdfc0811951418e6 +size 7306 diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..6c49fc6 --- /dev/null +++ b/vocab.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910 +size 2776833