commit 379614e73d803cf3baad601ab12a11413c7d3a72 Author: ModelHub XC Date: Thu Sep 3 17:26:13 2026 +0800 初始化项目,由ModelHub XC社区提供模型 Model: RedHatAI/starcoder2-3b-quantized.w8a8 Source: Original Platform diff --git a/.gitattributes b/.gitattributes new file mode 100644 index 0000000..a6344aa --- /dev/null +++ b/.gitattributes @@ -0,0 +1,35 @@ +*.7z filter=lfs diff=lfs merge=lfs -text +*.arrow filter=lfs diff=lfs merge=lfs -text +*.bin filter=lfs diff=lfs merge=lfs -text +*.bz2 filter=lfs diff=lfs merge=lfs -text +*.ckpt filter=lfs diff=lfs merge=lfs -text +*.ftz filter=lfs diff=lfs merge=lfs -text +*.gz filter=lfs diff=lfs merge=lfs -text +*.h5 filter=lfs diff=lfs merge=lfs -text +*.joblib filter=lfs diff=lfs merge=lfs -text +*.lfs.* filter=lfs diff=lfs merge=lfs -text +*.mlmodel filter=lfs diff=lfs merge=lfs -text +*.model filter=lfs diff=lfs merge=lfs -text +*.msgpack filter=lfs diff=lfs merge=lfs -text +*.npy filter=lfs diff=lfs merge=lfs -text +*.npz filter=lfs diff=lfs merge=lfs -text +*.onnx filter=lfs diff=lfs merge=lfs -text +*.ot filter=lfs diff=lfs merge=lfs -text +*.parquet filter=lfs diff=lfs merge=lfs -text +*.pb filter=lfs diff=lfs merge=lfs -text +*.pickle filter=lfs diff=lfs merge=lfs -text +*.pkl filter=lfs diff=lfs merge=lfs -text +*.pt filter=lfs diff=lfs merge=lfs -text +*.pth filter=lfs diff=lfs merge=lfs -text +*.rar filter=lfs diff=lfs merge=lfs -text +*.safetensors filter=lfs diff=lfs merge=lfs -text +saved_model/**/* filter=lfs diff=lfs merge=lfs -text +*.tar.* filter=lfs diff=lfs merge=lfs -text +*.tar filter=lfs diff=lfs merge=lfs -text +*.tflite filter=lfs diff=lfs merge=lfs -text +*.tgz filter=lfs diff=lfs merge=lfs -text +*.wasm filter=lfs diff=lfs merge=lfs -text +*.xz filter=lfs diff=lfs merge=lfs -text +*.zip filter=lfs diff=lfs merge=lfs -text +*.zst filter=lfs diff=lfs merge=lfs -text +*tfevents* filter=lfs diff=lfs merge=lfs -text diff --git a/README.md b/README.md new file mode 100644 index 0000000..807121e --- /dev/null +++ b/README.md @@ -0,0 +1,212 @@ +--- +pipeline_tag: text-generation +datasets: +- bigcode/the-stack-v2-train +license: bigcode-openrail-m +library_name: transformers +tags: +- code +model-index: +- name: starcoder2-3b-quantized.w8a8 + results: + - task: + type: text-generation + dataset: + name: HumanEval+ + type: humanevalplus + metrics: + - type: pass@1 + value: 26.8 + - task: + type: text-generation + dataset: + name: HumanEval + type: humaneval + metrics: + - type: pass@1 + value: 31.4 +--- + +# starcoder2-3b-quantized.w8a8 + +## Model Overview +- **Model Architecture:** StarCoder2 + - **Input:** Text + - **Output:** Text +- **Model Optimizations:** + - **Activation quantization:** INT8 + - **Weight quantization:** INT8 +- **Intended Use Cases:** Intended for commercial and research use. Similarly to [starcoder2-3b](https://huggingface.co/bigcode/starcoder2-3b), this model is intended for code generation and is _not_ an instruction model. Commands like "Write a function that computes the square root." do not work well. +- **Out-of-scope:** Use in any manner that violates applicable laws or regulations (including trade compliance laws). +- **Release Date:** 8/1/2024 +- **Version:** 1.0 +- **License(s):** bigcode-openrail-m +- **Model Developers:** Neural Magic + +Quantized version of [starcoder2-3b](https://huggingface.co/bigcode/starcoder2-3b). +It achieves a HumanEval pass@1 of 31.4, whereas the unquantized model achieves 30.7 when evaluated under the same conditions. + +### Model Optimizations + +This model was obtained by quantizing the weights of [starcoder2-3b](https://huggingface.co/bigcode/starcoder2-3b) to INT8 data type. +This optimization reduces the number of bits used to represent weights and activations from 16 to 8, reducing GPU memory requirements (by approximately 50%) and increasing matrix-multiply compute throughput (by approximately 2x). +Weight quantization also reduces disk size requirements by approximately 50%. + +Only weights and activations of the linear operators within transformers blocks are quantized. +Weights are quantized with a symmetric static per-channel scheme, where a fixed linear scaling factor is applied between INT8 and floating point representations for each output channel dimension. +Activations are quantized with a symmetric dynamic per-token scheme, computing a linear scaling factor at runtime for each token between INT8 and floating point representations. +The [GPTQ](https://arxiv.org/abs/2210.17323) algorithm is applied for quantization, as implemented in the [llm-compressor](https://github.com/vllm-project/llm-compressor) library. +GPTQ used a 1% damping factor and 256 sequences of 8,192 random tokens. + + +## Deployment + +### Use with vLLM + +This model can be deployed efficiently using the [vLLM](https://docs.vllm.ai/en/latest/) backend, as shown in the example below. + +```python +from vllm import LLM, SamplingParams +from transformers import AutoTokenizer + +model_id = "neuralmagic/starcoder2-3b-quantized.w8a8" +number_gpus = 1 + +sampling_params = SamplingParams(temperature=0.2, top_p=0.95, max_tokens=256) + +tokenizer = AutoTokenizer.from_pretrained(model_id) + +prompts = ["def print_hello_world():"] + +llm = LLM(model=model_id, tensor_parallel_size=number_gpus) + +outputs = llm.generate(prompts, sampling_params) + +generated_text = outputs[0].outputs[0].text +print(generated_text) +``` + +vLLM aslo supports OpenAI-compatible serving. See the [documentation](https://docs.vllm.ai/en/latest/) for more details. + + +## Creation + +This model was created by using the [llm-compressor](https://github.com/vllm-project/llm-compressor) library as presented in the code snipet below. + +```python +from transformers import AutoTokenizer +from datasets import Dataset +from llmcompressor.transformers import SparseAutoModelForCausalLM, oneshot +from llmcompressor.modifiers.quantization import GPTQModifier +import random + +model_id = "bigcode/starcoder2-3b" + +num_samples = 256 +max_seq_len = 8192 + +tokenizer = AutoTokenizer.from_pretrained(model_id) + +max_token_id = len(tokenizer.get_vocab()) - 1 +input_ids = [[random.randint(0, max_token_id) for _ in range(max_seq_len)] for _ in range(num_samples)] +attention_mask = num_samples * [max_seq_len * [1]] +ds = Dataset.from_dict({"input_ids": input_ids, "attention_mask": attention_mask}) + +recipe = GPTQModifier( + targets="Linear", + scheme="W8A8", + ignore=["lm_head"], + dampening_frac=0.01, +) + +model = SparseAutoModelForCausalLM.from_pretrained( + model_id, + device_map="auto", + trust_remote_code=True, +) + +oneshot( + model=model, + dataset=ds, + recipe=recipe, + max_seq_length=max_seq_len, + num_calibration_samples=num_samples, +) +model.save_pretrained("starcoder2-3b-quantized.w8a8") +``` + + +## Evaluation + +The model was evaluated on the [HumanEval](https://arxiv.org/abs/2107.03374) and [HumanEval+](https://arxiv.org/abs/2305.01210) benchmarks, using the generation configuration from [Big Code Models Leaderboard](https://huggingface.co/spaces/bigcode/bigcode-models-leaderboard). +We used Neural Magic's fork of [evalplus](https://github.com/neuralmagic/evalplus) and the [vLLM](https://docs.vllm.ai/en/stable/) engine, using the following commands: + +``` +python codegen/generate.py \ + --model neuralmagic/starcoder2-3b-quantized.w8a8 \ + --bs 16 \ + --temperature 0.2 \ + --n_samples 50 \ + --dataset humaneval \ + -- root "." + +python3 evalplus/sanitize.py humaneval/neuralmagic--starcoder2-3b-quantized.w8a8_vllm_temp_0.2 + +evalplus.evaluate --dataset humaneval --samples humaneval/neuralmagic--starcoder2-3b-quantized.w8a8_vllm_temp_0.2-sanitized +``` + +### Accuracy + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + + +
Benchmark + starcoder2-3b + starcoder2-3b-quantized.w8a8 (this model) + Recovery +
HumanEval pass@1 + 30.7 + 31.4 + 102.3% +
HumanEval pass@10 + 44.9 + 44.7 + 99.6% +
HumanEval+ pass@1 + 26.6 + 26.8 + 100.8% +
HumanEval+ pass@10 + 39.2 + 38.7 + 98.7% +
\ No newline at end of file diff --git a/config.json b/config.json new file mode 100644 index 0000000..48730b6 --- /dev/null +++ b/config.json @@ -0,0 +1,71 @@ +{ + "_name_or_path": "/root/.cache/huggingface/hub/models--bigcode--starcoder2-3b/snapshots/733247c55e3f73af49ce8e9c7949bf14af205928", + "architectures": [ + "Starcoder2ForCausalLM" + ], + "attention_dropout": 0.1, + "bos_token_id": 0, + "embedding_dropout": 0.1, + "eos_token_id": 0, + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 3072, + "initializer_range": 0.018042, + "intermediate_size": 12288, + "max_position_embeddings": 16384, + "mlp_type": "default", + "model_type": "starcoder2", + "norm_epsilon": 1e-05, + "norm_type": "layer_norm", + "num_attention_heads": 24, + "num_hidden_layers": 30, + "num_key_value_heads": 2, + "residual_dropout": 0.1, + "rope_theta": 999999.4420358813, + "sliding_window": 4096, + "torch_dtype": "float32", + "transformers_version": "4.43.3", + "use_bias": true, + "use_cache": true, + "vocab_size": 49152, + "quantization_config": { + "config_groups": { + "group_0": { + "input_activations": { + "block_structure": null, + "dynamic": true, + "group_size": null, + "num_bits": 8, + "observer": "memoryless", + "observer_kwargs": {}, + "strategy": "token", + "symmetric": true, + "type": "int" + }, + "output_activations": null, + "targets": [ + "Linear" + ], + "weights": { + "block_structure": null, + "dynamic": false, + "group_size": null, + "num_bits": 8, + "observer": "minmax", + "observer_kwargs": {}, + "strategy": "channel", + "symmetric": true, + "type": "int" + } + } + }, + "format": "int-quantized", + "global_compression_ratio": 1.3879017053810396, + "ignore": [ + "lm_head" + ], + "kv_cache_scheme": null, + "quant_method": "compressed-tensors", + "quantization_status": "compressed", + "sparsity_config": {} + } +} diff --git a/configuration.json b/configuration.json new file mode 100644 index 0000000..bbeeda1 --- /dev/null +++ b/configuration.json @@ -0,0 +1 @@ +{"framework": "pytorch", "task": "text-generation", "allow_remote": true} \ No newline at end of file diff --git a/generation_config.json b/generation_config.json new file mode 100644 index 0000000..f46fcf7 --- /dev/null +++ b/generation_config.json @@ -0,0 +1,6 @@ +{ + "_from_model_config": true, + "bos_token_id": 0, + "eos_token_id": 0, + "transformers_version": "4.43.3" +} diff --git a/merges.txt b/merges.txt new file mode 100644 index 0000000..966b9fd --- /dev/null +++ b/merges.txt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac9dfb2c5bd3b10bd46aa7fdaa8c54e8e1bcfe1618d9288320ebc82c518442fc +size 441705 diff --git a/model.safetensors b/model.safetensors new file mode 100644 index 0000000..738e606 --- /dev/null +++ b/model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:32d9b0f950dfbafcf0e0a5b30958b07f99b7e958dcf9f3f09c5a2947dc73d816 +size 4093158536 diff --git a/recipe.yaml b/recipe.yaml new file mode 100644 index 0000000..a9d4aa2 --- /dev/null +++ b/recipe.yaml @@ -0,0 +1,8 @@ +quant_stage: + quant_modifiers: + GPTQModifier: + sequential_update: false + dampening_frac: 0.01 + ignore: [lm_head] + scheme: W8A8 + targets: Linear diff --git a/special_tokens_map.json b/special_tokens_map.json new file mode 100644 index 0000000..8d8f8dd --- /dev/null +++ b/special_tokens_map.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b8d9e1e1efd775a22a19ee45b85a31b524766c0db89de00ed249a7433794762 +size 1332 diff --git a/tokenizer.json b/tokenizer.json new file mode 100644 index 0000000..16c4611 --- /dev/null +++ b/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:17fa145258b20c18287f1e3bd804e074cc13333f11984a2f5a2f11c5110437aa +size 2060947 diff --git a/tokenizer_config.json b/tokenizer_config.json new file mode 100644 index 0000000..a33707a --- /dev/null +++ b/tokenizer_config.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:967231a56db74a08ced2afea94492b831e581da94a411436cce9ec40a0541f00 +size 7909 diff --git a/vocab.json b/vocab.json new file mode 100644 index 0000000..56972d7 --- /dev/null +++ b/vocab.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:542ee3f8b27b3a825754728b4dc59eff31065fecce843a93f09b98f13dbd7de9 +size 777202