From 3a7156a0d5a92d87016a2c5de514ff99c81ec110 Mon Sep 17 00:00:00 2001 From: ModelHub XC Date: Thu, 10 Sep 2026 05:14:12 +0800 Subject: [PATCH] =?UTF-8?q?=E5=88=9D=E5=A7=8B=E5=8C=96=E9=A1=B9=E7=9B=AE?= =?UTF-8?q?=EF=BC=8C=E7=94=B1ModelHub=20XC=E7=A4=BE=E5=8C=BA=E6=8F=90?= =?UTF-8?q?=E4=BE=9B=E6=A8=A1=E5=9E=8B?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Model: RedHatAI/starcoder2-15b-quantized.w8a8 Source: Original Platform --- .gitattributes | 35 +++++ README.md | 212 +++++++++++++++++++++++++++++++ config.json | 71 +++++++++++ configuration.json | 1 + generation_config.json | 6 + merges.txt | 3 + model-00001-of-00004.safetensors | 3 + model-00002-of-00004.safetensors | 3 + model-00003-of-00004.safetensors | 3 + model-00004-of-00004.safetensors | 3 + model.safetensors.index.json | 3 + recipe.yaml | 8 ++ special_tokens_map.json | 3 + tokenizer.json | 3 + tokenizer_config.json | 3 + vocab.json | 3 + 16 files changed, 363 insertions(+) create mode 100644 .gitattributes create mode 100644 README.md create mode 100644 config.json create mode 100644 configuration.json create mode 100644 generation_config.json create mode 100644 merges.txt create mode 100644 model-00001-of-00004.safetensors create mode 100644 model-00002-of-00004.safetensors create mode 100644 model-00003-of-00004.safetensors create mode 100644 model-00004-of-00004.safetensors create mode 100644 model.safetensors.index.json create mode 100644 recipe.yaml create mode 100644 special_tokens_map.json create mode 100644 tokenizer.json create mode 100644 tokenizer_config.json create mode 100644 vocab.json diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..a6344aa --- /dev/null +++ b/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..a26cbcc --- /dev/null +++ b/README.md @@ -0,0 +1,212 @@ +--- +pipeline_tag: text-generation +datasets: +- bigcode/the-stack-v2-train +license: bigcode-openrail-m +library_name: transformers +tags: +- code +model-index: +- name: starcoder2-15b-quantized.w8a8 + results: + - task: + type: text-generation + dataset: + name: HumanEval+ + type: humanevalplus + metrics: + - type: pass@1 + value: 38.1 + - task: + type: text-generation + dataset: + name: HumanEval + type: humaneval + metrics: + - type: pass@1 + value: 44.6 +--- + +# starcoder2-3b-quantized.w8a8 + +## Model Overview +- **Model Architecture:** StarCoder2 + - **Input:** Text + - **Output:** Text +- **Model Optimizations:** + - **Activation quantization:** INT8 + - **Weight quantization:** INT8 +- **Intended Use Cases:** Intended for commercial and research use. Similarly to [starcoder2-15b](https://huggingface.co/bigcode/starcoder2-15b), this model is intended for code generation and is _not_ an instruction model. Commands like "Write a function that computes the square root." do not work well. +- **Out-of-scope:** Use in any manner that violates applicable laws or regulations (including trade compliance laws). +- **Release Date:** 8/1/2024 +- **Version:** 1.0 +- **License(s):** bigcode-openrail-m +- **Model Developers:** Neural Magic + +Quantized version of [starcoder2-15b](https://huggingface.co/bigcode/starcoder2-15b). +It achieves a HumanEval pass@1 of 44.6, whereas the unquantized model achieves 44.8 when evaluated under the same conditions. + +### Model Optimizations + +This model was obtained by quantizing the weights of [starcoder2-15b](https://huggingface.co/bigcode/starcoder2-15b) to INT8 data type. +This optimization reduces the number of bits used to represent weights and activations from 16 to 8, reducing GPU memory requirements (by approximately 50%) and increasing matrix-multiply compute throughput (by approximately 2x). +Weight quantization also reduces disk size requirements by approximately 50%. + +Only weights and activations of the linear operators within transformers blocks are quantized. +Weights are quantized with a symmetric static per-channel scheme, where a fixed linear scaling factor is applied between INT8 and floating point representations for each output channel dimension. +Activations are quantized with a symmetric dynamic per-token scheme, computing a linear scaling factor at runtime for each token between INT8 and floating point representations. +The [GPTQ](https://arxiv.org/abs/2210.17323) algorithm is applied for quantization, as implemented in the [llm-compressor](https://github.com/vllm-project/llm-compressor) library. +GPTQ used a 1% damping factor and 256 sequences of 8,192 random tokens. + + +## Deployment + +### Use with vLLM + +This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/latest/) backend, as shown in the example below. + +```python +from vllm import LLM, SamplingParams +from transformers import AutoTokenizer + +model_id = "neuralmagic/starcoder2-15b-quantized.w8a8" +number_gpus = 1 + +sampling_params = SamplingParams(temperature=0.2, top_p=0.95, max_tokens=256) + +tokenizer = AutoTokenizer.from_pretrained(model_id) + +prompts = ["def print_hello_world():"] + +llm = LLM(model=model_id, tensor_parallel_size=number_gpus) + +outputs = llm.generate(prompts, sampling_params) + +generated_text = outputs[0].outputs[0].text +print(generated_text) +``` + +vLLM aslo supports OpenAI-compatible serving. See the [documentation](https://docs.vllm.ai/en/latest/) for more details. + + +## Creation + +This model was created by using the [llm-compressor](https://github.com/vllm-project/llm-compressor) library as presented in the code snipet below. + +```python +from transformers import AutoTokenizer +from datasets import Dataset +from llmcompressor.transformers import SparseAutoModelForCausalLM, oneshot +from llmcompressor.modifiers.quantization import GPTQModifier +import random + +model_id = "bigcode/starcoder2-15b" + +num_samples = 256 +max_seq_len = 8192 + +tokenizer = AutoTokenizer.from_pretrained(model_id) + +max_token_id = len(tokenizer.get_vocab()) - 1 +input_ids = [[random.randint(0, max_token_id) for _ in range(max_seq_len)] for _ in range(num_samples)] +attention_mask = num_samples * [max_seq_len * [1]] +ds = Dataset.from_dict({"input_ids": input_ids, "attention_mask": attention_mask}) + +recipe = GPTQModifier( + targets="Linear", + scheme="W8A8", + ignore=["lm_head"], + dampening_frac=0.01, +) + +model = SparseAutoModelForCausalLM.from_pretrained( + model_id, + device_map="auto", + trust_remote_code=True, +) + +oneshot( + model=model, + dataset=ds, + recipe=recipe, + max_seq_length=max_seq_len, + num_calibration_samples=num_samples, +) +model.save_pretrained("starcoder2-15b-quantized.w8a8") +``` + + +## Evaluation + +The model was evaluated on the [HumanEval](https://arxiv.org/abs/2107.03374) and [HumanEval+](https://arxiv.org/abs/2305.01210) benchmarks, using the generation configuration from [Big Code Models Leaderboard](https://huggingface.co/spaces/bigcode/bigcode-models-leaderboard). +We used Neural Magic's fork of [evalplus](https://github.com/neuralmagic/evalplus) and the [vLLM](https://docs.vllm.ai/en/stable/) engine, using the following commands: + +``` +python codegen/generate.py \ + --model neuralmagic/starcoder2-15b-quantized.w8a8 \ + --bs 16 \ + --temperature 0.2 \ + --n_samples 50 \ + --dataset humaneval \ + -- root "." + +python3 evalplus/sanitize.py humaneval/neuralmagic--starcoder2-15b-quantized.w8a8_vllm_temp_0.2 + +evalplus.evaluate --dataset humaneval --samples humaneval/neuralmagic--starcoder2-15b-quantized.w8a8_vllm_temp_0.2-sanitized +``` + +### Accuracy + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Benchmark + starcoder2-15b + starcoder2-15b-quantized.w8a8 (this model) + Recovery +
HumanEval pass@1 + 44.8 + 44.6 + 99.6% +
HumanEval pass@10 + 62.7 + 63.3 + 101.0% +
HumanEval+ pass@1 + 38.6 + 38.1 + 98.7% +
HumanEval+ pass@10 + 54.9 + 55.5 + 101.1% +
\ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..04ed889 --- /dev/null +++ b/config.json @@ -0,0 +1,71 @@ +{ + "_name_or_path": "/root/.cache/huggingface/hub/models--bigcode--starcoder2-15b/snapshots/46d44742909c03ac8cee08eb03fdebce02e193ec", + "architectures": [ + "Starcoder2ForCausalLM" + ], + "attention_dropout": 0.1, + "bos_token_id": 0, + "embedding_dropout": 0.1, + "eos_token_id": 0, + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 6144, + "initializer_range": 0.01275, + "intermediate_size": 24576, + "max_position_embeddings": 16384, + "mlp_type": "default", + "model_type": "starcoder2", + "norm_epsilon": 1e-05, + "norm_type": "layer_norm", + "num_attention_heads": 48, + "num_hidden_layers": 40, + "num_key_value_heads": 4, + "residual_dropout": 0.1, + "rope_theta": 100000, + "sliding_window": 4096, + "tie_word_embeddings": false, + "torch_dtype": "float32", + "transformers_version": "4.43.3", + "use_bias": true, + "use_cache": true, + "vocab_size": 49152, + "quantization_config": { + "config_groups": { + "group_0": { + "input_activations": { + "block_structure": null, + "dynamic": true, + "group_size": null, + "num_bits": 8, + "observer": "memoryless", + "observer_kwargs": {}, + "strategy": "token", + "symmetric": true, + "type": "int" + }, + "output_activations": null, + "targets": [ + "Linear" + ], + "weights": { + "block_structure": null, + "dynamic": false, + "group_size": null, + "num_bits": 8, + "observer": "minmax", + "observer_kwargs": {}, + "strategy": "channel", + "symmetric": true, + "type": "int" + } + } + }, + "format": "int-quantized", + "global_compression_ratio": 1.3887540031660834, + "ignore": [ + "lm_head" + ], + "kv_cache_scheme": null, + "quant_method": "compressed-tensors", + "quantization_status": "compressed" + } +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..f46fcf7 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "bos_token_id": 0, + "eos_token_id": 0, + "transformers_version": "4.43.3" +} diff --git a/merges.txt b/merges.txt new file mode 100644 index 0000000..966b9fd --- /dev/null +++ b/merges.txt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac9dfb2c5bd3b10bd46aa7fdaa8c54e8e1bcfe1618d9288320ebc82c518442fc +size 441705 diff --git a/model-00001-of-00004.safetensors b/model-00001-of-00004.safetensors new file mode 100644 index 0000000..d768d34 --- /dev/null +++ b/model-00001-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9707549168f9120dbade171c86e09e96b59bc86a8a2f9345dd83c5bbd148914a +size 4899134952 diff --git a/model-00002-of-00004.safetensors b/model-00002-of-00004.safetensors new file mode 100644 index 0000000..37662d1 --- /dev/null +++ b/model-00002-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d64cf3a9a2d323e815498759a1c8e5ec8cb492f51ab1624f18e6765ac9e5b8a7 +size 4995012936 diff --git a/model-00003-of-00004.safetensors b/model-00003-of-00004.safetensors new file mode 100644 index 0000000..32c6754 --- /dev/null +++ b/model-00003-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c0722cd9632595986b906171fb8ad280991ca959a35f1d2ed6c61c79935d417c +size 4995012944 diff --git a/model-00004-of-00004.safetensors b/model-00004-of-00004.safetensors new file mode 100644 index 0000000..b9a20c5 --- /dev/null +++ b/model-00004-of-00004.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7c369bbf90a54d8df9d02e7f27721c32db2cb30272727aac3600f84e5351c5d6 +size 2896079664 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000..75daf2f --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76a2090af5574d280cfa705efc7b6388d603f1302095aa041c4c968aa2ee74bb +size 72823 diff --git a/recipe.yaml b/recipe.yaml new file mode 100644 index 0000000..a9d4aa2 --- /dev/null +++ b/recipe.yaml @@ -0,0 +1,8 @@ +quant_stage: + quant_modifiers: + GPTQModifier: + sequential_update: false + dampening_frac: 0.01 + ignore: [lm_head] + scheme: W8A8 + targets: Linear diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..8d8f8dd --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b8d9e1e1efd775a22a19ee45b85a31b524766c0db89de00ed249a7433794762 +size 1332 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..16c4611 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:17fa145258b20c18287f1e3bd804e074cc13333f11984a2f5a2f11c5110437aa +size 2060947 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..a33707a --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:967231a56db74a08ced2afea94492b831e581da94a411436cce9ec40a0541f00 +size 7909 diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..56972d7 --- /dev/null +++ b/vocab.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:542ee3f8b27b3a825754728b4dc59eff31065fecce843a93f09b98f13dbd7de9 +size 777202