commit 6c095722385bf5037178e69a3043f86262d52130 Author: ModelHub XC Date: Wed Sep 16 12:04:12 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: RedHatAI/QwQ-32B-Preview-quantized.w8a8 Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..52373fe --- /dev/null +++ b/.gitattributes @@ -0,0 +1,36 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text +tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..fde1d6d --- /dev/null +++ b/README.md @@ -0,0 +1,205 @@ +--- +tags: +- w8a8 +- int8 +- vllm +license: apache-2.0 +license_link: https://huggingface.co/Qwen/QwQ-32B-Preview/blob/main/LICENSE +language: + - en +base_model: Qwen/Qwen2.5-32B-Instruct +library_name: transformers +--- + +# QwQ-32B-Preview-quantized.w8a8 + +## Model Overview +- **Model Architecture:** QwQ-32B-Preview + - **Input:** Text + - **Output:** Text +- **Model Optimizations:** + - **Weight quantization:** INT8 + - **Activation quantization:** INT8 +- **Release Date:** 3/1/2025 +- **Version:** 1.0 +- **Model Developers:** Neural Magic + +Quantized version of [QwQ-32B-Preview](https://huggingface.co/Qwen/QwQ-32B-Preview). +It achieves an average score of 76.49 on the [OpenLLM](https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard) benchmark (version 1), whereas the unquantized model achieves 77.20. + +### Model Optimizations + +This model was obtained by quantizing the weights and activations of [QwQ-32B-Preview](https://huggingface.co/Qwen/QwQ-32B-Preview) to INT8 data type, ready for inference with vLLM >= 0.5.2. +This optimization reduces the number of bits per parameter from 16 to 8, reducing the disk size and GPU memory requirements by approximately 50%. Only the weights and activations of the linear operators within transformers blocks are quantized. + +## Deployment + +### Use with vLLM + +This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/latest/) backend, as shown in the example below. + +```python +from transformers import AutoTokenizer +from vllm import LLM, SamplingParams + +max_model_len, tp_size = 4096, 1 +model_name = "neuralmagic-ent/QwQ-32B-Preview-quantized.w8a8" +tokenizer = AutoTokenizer.from_pretrained(model_name) +llm = LLM(model=model_name, tensor_parallel_size=tp_size, max_model_len=max_model_len, trust_remote_code=True) +sampling_params = SamplingParams(temperature=0.3, max_tokens=256, stop_token_ids=[tokenizer.eos_token_id]) + +messages_list = [ + [{"role": "user", "content": "Who are you? Please respond in pirate speak!"}], +] + +prompt_token_ids = [tokenizer.apply_chat_template(messages, add_generation_prompt=True) for messages in messages_list] + +outputs = llm.generate(prompt_token_ids=prompt_token_ids, sampling_params=sampling_params) + +generated_text = [output.outputs[0].text for output in outputs] +print(generated_text) +``` + +vLLM also supports OpenAI-compatible serving. See the [documentation](https://docs.vllm.ai/en/latest/) for more details. + +## Creation + +This model was created with [llm-compressor](https://github.com/vllm-project/llm-compressor) by running the code snippet below with the following arguments: + + +```bash +python quantize.py --model_path Qwen/QwQ-32B-Preview --quant_path "output_dir/QwQ-32B-Preview-quantized.w8a8" --calib_size 1024 --dampening_frac 0.1 --observer mse +``` + + +```python +from datasets import load_dataset +from transformers import AutoTokenizer +from llmcompressor.modifiers.quantization import GPTQModifier +from llmcompressor.transformers import SparseAutoModelForCausalLM, oneshot, apply +import argparse +from compressed_tensors.quantization import QuantizationScheme, QuantizationArgs, QuantizationType, QuantizationStrategy + + +parser = argparse.ArgumentParser() +parser.add_argument('--model_path', type=str) +parser.add_argument('--quant_path', type=str) +parser.add_argument('--calib_size', type=int, default=256) +parser.add_argument('--dampening_frac', type=float, default=0.1) +parser.add_argument('--observer', type=str, default="minmax") +args = parser.parse_args() + +model = SparseAutoModelForCausalLM.from_pretrained( + args.model_path, + device_map="auto", + torch_dtype="auto", + use_cache=False, + trust_remote_code=True, +) +tokenizer = AutoTokenizer.from_pretrained(args.model_path) + +NUM_CALIBRATION_SAMPLES = args.calib_size +DATASET_ID = "garage-bAInd/Open-Platypus" +DATASET_SPLIT = "train" +ds = load_dataset(DATASET_ID, split=DATASET_SPLIT) +ds = ds.shuffle(seed=42).select(range(NUM_CALIBRATION_SAMPLES)) + +def preprocess(example): + concat_txt = example["instruction"] + "\n" + example["output"] + return {"text": concat_txt} + +ds = ds.map(preprocess) + +def tokenize(sample): + return tokenizer( + sample["text"], + padding=False, + truncation=False, + add_special_tokens=True, + ) + + +ds = ds.map(tokenize, remove_columns=ds.column_names) + +recipe = [ + GPTQModifier( + targets=["Linear"], + ignore=["lm_head"], + scheme="W8A8", + dampening_frac=args.dampening_frac, + observer=args.observer, + ) +] +oneshot( + model=model, + dataset=ds, + recipe=recipe, + num_calibration_samples=args.calib_size, + max_seq_length=8192, +) + +# Save to disk compressed. +SAVE_DIR = args.quant_path +model.save_pretrained(SAVE_DIR, save_compressed=True) +tokenizer.save_pretrained(SAVE_DIR) +``` + +## Evaluation + +The model was evaluated on OpenLLM Leaderboard [V1](https://huggingface.co/spaces/open-llm-leaderboard-old/open_llm_leaderboard) and [V2](https://huggingface.co/spaces/open-llm-leaderboard/open_llm_leaderboard#/), using the following commands: + +OpenLLM Leaderboard V1: +``` +lm_eval \ + --model vllm \ + --model_args pretrained="neuralmagic-ent/QwQ-32B-Preview-quantized.w8a8",dtype=auto,add_bos_token=True,max_model_len=4096,tensor_parallel_size=1,gpu_memory_utilization=0.8,enable_chunked_prefill=True,trust_remote_code=True \ + --tasks openllm \ + --write_out \ + --batch_size auto \ + --output_path output_dir \ + --show_config +``` + +OpenLLM Leaderboard V2: +``` +lm_eval \ + --model vllm \ + --model_args pretrained="neuralmagic-ent/QwQ-32B-Preview-quantized.w8a8",dtype=auto,add_bos_token=False,max_model_len=4096,tensor_parallel_size=1,gpu_memory_utilization=0.8,enable_chunked_prefill=True,trust_remote_code=True \ + --apply_chat_template \ + --fewshot_as_multiturn \ + --tasks leaderboard \ + --write_out \ + --batch_size auto \ + --output_path output_dir \ + --show_config + +``` + +### Accuracy + +#### OpenLLM Leaderboard V1 evaluation scores + +| Metric | Qwen/QwQ-32B-Preview | neuralmagic-ent/QwQ-32B-Preview-quantized.w8a8 | +|-----------------------------------------|:---------------------------------:|:-------------------------------------------:| +| ARC-Challenge (Acc-Norm, 25-shot) | 70.73 | 70.73 | +| GSM8K (Strict-Match, 5-shot) | 83.09 | 79.91 | +| HellaSwag (Acc-Norm, 10-shot) | 85.77 | 85.75 | +| MMLU (Acc, 5-shot) | 82.67 | 82.24 | +| TruthfulQA (MC2, 0-shot) | 60.88 | 59.18 | +| Winogrande (Acc, 5-shot) | 80.03 | 81.14 | +| **Average Score** | **77.20** | **76.49** | +| **Recovery** | **100.00** | **99.08** | + +#### OpenLLM Leaderboard V2 evaluation scores + +| Metric | Qwen/QwQ-32B-Preview | neuralmagic-ent/QwQ-32B-Preview-quantized.w8a8 | +|---------------------------------------------------------|:---------------------------------:|:-------------------------------------------:| +| IFEval (Inst-and-Prompt Level Strict Acc, 0-shot) | 42.34 | 43.49 | +| BBH (Acc-Norm, 3-shot) | 53.03 | 52.95 | +| Math-Hard (Exact-Match, 4-shot) | 21.15 | 22.36 | +| GPQA (Acc-Norm, 0-shot) | 2.97 | 3.5 | +| MUSR (Acc-Norm, 0-shot) | 9.57 | 10.87 | +| MMLU-Pro (Acc, 5-shot) | 52.00 | 51.4 | +| **Average Score** | **30.18** | **30.76** | +| **Recovery** | **100.00** | **101.92** | + diff --git a/added_tokens.json b/added_tokens.json new file mode 100644 index 0000000..482ced4 --- /dev/null +++ b/added_tokens.json @@ -0,0 +1,24 @@ +{ + "": 151658, + "": 151657, + "<|box_end|>": 151649, + "<|box_start|>": 151648, + "<|endoftext|>": 151643, + "<|file_sep|>": 151664, + "<|fim_middle|>": 151660, + "<|fim_pad|>": 151662, + "<|fim_prefix|>": 151659, + "<|fim_suffix|>": 151661, + "<|im_end|>": 151645, + "<|im_start|>": 151644, + "<|image_pad|>": 151655, + "<|object_ref_end|>": 151647, + "<|object_ref_start|>": 151646, + "<|quad_end|>": 151651, + "<|quad_start|>": 151650, + "<|repo_name|>": 151663, + "<|video_pad|>": 151656, + "<|vision_end|>": 151653, + "<|vision_pad|>": 151654, + "<|vision_start|>": 151652 +} diff --git a/config.json b/config.json new file mode 100644 index 0000000..6d9798d --- /dev/null +++ b/config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fbf730ee074bf28ebe2299a4f389040123dfe234c78bb9137806c5f16e17ad44 +size 1796 diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..6b350ba --- /dev/null +++ b/generation_config.json @@ -0,0 +1,13 @@ +{ + "bos_token_id": 151643, + "do_sample": true, + "eos_token_id": [ + 151645, + 151643 + ], + "pad_token_id": 151643, + "temperature": 0.7, + "top_k": 20, + "top_p": 0.8, + "transformers_version": "4.47.1" +} diff --git a/merges.txt b/merges.txt new file mode 100644 index 0000000..80c1a19 --- /dev/null +++ b/merges.txt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8831e4f1a044471340f7c0a83d7bd71306a5b867e95fd870f74d0c5308a904d5 +size 1671853 diff --git a/model-00001-of-00007.safetensors b/model-00001-of-00007.safetensors new file mode 100644 index 0000000..2b8305d --- /dev/null +++ b/model-00001-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de0111c56f51428f9beda9d632521bb9ce79a6fac84a67558d32761cb2a412d3 +size 4997762496 diff --git a/model-00002-of-00007.safetensors b/model-00002-of-00007.safetensors new file mode 100644 index 0000000..a95ea25 --- /dev/null +++ b/model-00002-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cbc72d4ddd26edf5e1a65b28a21589312e881fb6fc4b1bbae7b149965b89cad2 +size 4914421312 diff --git a/model-00003-of-00007.safetensors b/model-00003-of-00007.safetensors new file mode 100644 index 0000000..c45044e --- /dev/null +++ b/model-00003-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7ef88f31d10359ebf3d940b873d271723d5a0d41fcfb23edbbdc4b14d8e978d6 +size 4877701896 diff --git a/model-00004-of-00007.safetensors b/model-00004-of-00007.safetensors new file mode 100644 index 0000000..b8a0d3c --- /dev/null +++ b/model-00004-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8ac4a9378b7116b360ccb67385be46dc0e896c717c0b3bc11a86dad94eaf9c02 +size 4877701896 diff --git a/model-00005-of-00007.safetensors b/model-00005-of-00007.safetensors new file mode 100644 index 0000000..991d8ec --- /dev/null +++ b/model-00005-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:085323cfeb249569a23ee657411a7d3f48464273decc8d50ae08391bba24329c +size 4877701896 diff --git a/model-00006-of-00007.safetensors b/model-00006-of-00007.safetensors new file mode 100644 index 0000000..9a0f190 --- /dev/null +++ b/model-00006-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c3dd932c6e3ef4dbdfd5bdd27c32ae0b9cede00033c5c025c09127aafd14fcf0 +size 4877701896 diff --git a/model-00007-of-00007.safetensors b/model-00007-of-00007.safetensors new file mode 100644 index 0000000..376d115 --- /dev/null +++ b/model-00007-of-00007.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea954cbeae92e996cff924211e5c53e13798ed570ced2ac212b417c951f55614 +size 4908582968 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..0d02816 --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ecb9438a33bfc025000cd8575afc73dec022f58d448a1ac920cdbe1adca60e82 +size 102346 diff --git a/quant_w8a8.sh b/quant_w8a8.sh new file mode 100644 index 0000000..f33c5f4 --- /dev/null +++ b/quant_w8a8.sh @@ -0,0 +1,22 @@ +#!/bin/bash + + # for DAMPENING_FRAC 0.01 0.1; + # do + # for OBSERVER minmax mse; + # do + # for ACTORDER False group; + # do + +# export DAMPENING_FRAC=0.01 +# export OBSERVER="minmax" + +export CUDA_VISIBLE_DEVICES=${1} +export MDL=${2} +export OBSERVER=${3} # minmax mse +export DAMPENING_FRAC=${4} # 0.01 0.1 + +for CALIB_SIZE in 128 512 1024; +do + python w8a8.py --model_path ${MDL} --quant_path "output_dir_w8a8/${MDL}/calib${CALIB_SIZE}_eosFalse_damp${DAMPENING_FRAC}_obs${OBSERVER}" --calib_size ${CALIB_SIZE} --dampening_frac ${DAMPENING_FRAC} --observer ${OBSERVER} +done + diff --git a/recipe.yaml b/recipe.yaml new file mode 100644 index 0000000..d312a39 --- /dev/null +++ b/recipe.yaml @@ -0,0 +1,7 @@ +DEFAULT_stage: + DEFAULT_modifiers: + GPTQModifier: + targets: [Linear] + dampening_frac: 0.1 + ignore: [lm_head] + scheme: W8A8 diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..ac23c0a --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,31 @@ +{ + "additional_special_tokens": [ + "<|im_start|>", + "<|im_end|>", + "<|object_ref_start|>", + "<|object_ref_end|>", + "<|box_start|>", + "<|box_end|>", + "<|quad_start|>", + "<|quad_end|>", + "<|vision_start|>", + "<|vision_end|>", + "<|vision_pad|>", + "<|image_pad|>", + "<|video_pad|>" + ], + "eos_token": { + "content": "<|im_end|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + }, + "pad_token": { + "content": "<|endoftext|>", + "lstrip": false, + "normalized": false, + "rstrip": false, + "single_word": false + } +} diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..51ebb3b --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c5ae00e602b8860cbd784ba82a8aa14e8feecec692e7076590d014d7b7fdafa +size 11421896 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..42d36fd --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cd4a4256823870b3684d0e86980f0936f67d2987bebbb1a7be0830b9655d9af8 +size 7413 diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..6c49fc6 --- /dev/null +++ b/vocab.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ca10d7e9fb3ed18575dd1e277a2579c16d108e32f27439684afa0e10b1440910 +size 2776833 diff --git a/w8a8.py b/w8a8.py new file mode 100644 index 0000000..3151d8c --- /dev/null +++ b/w8a8.py @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:37066992e9fd64dc41ab6add690c06ebc27ac7bafd4e199ba85d4c75e924d4d0 +size 4264