初始化项目,由ModelHub XC社区提供模型
Model: JongYeop/Qwen2.5-14B-Instruct-FP8-W8A8 Source: Original Platform
This commit is contained in:
32
recipe.yaml
Normal file
32
recipe.yaml
Normal file
@@ -0,0 +1,32 @@
|
||||
quant_stage:
|
||||
quant_modifiers:
|
||||
QuantizationModifier:
|
||||
config_groups:
|
||||
group_0:
|
||||
targets: [Linear]
|
||||
weights:
|
||||
num_bits: 8
|
||||
type: float
|
||||
symmetric: true
|
||||
group_size: null
|
||||
strategy: tensor
|
||||
block_structure: null
|
||||
dynamic: false
|
||||
actorder: null
|
||||
observer: minmax
|
||||
observer_kwargs: {}
|
||||
input_activations:
|
||||
num_bits: 8
|
||||
type: float
|
||||
symmetric: true
|
||||
group_size: null
|
||||
strategy: tensor
|
||||
block_structure: null
|
||||
dynamic: false
|
||||
actorder: null
|
||||
observer: minmax
|
||||
observer_kwargs: {}
|
||||
output_activations: null
|
||||
format: null
|
||||
targets: [Linear]
|
||||
ignore: [lm_head]
|
||||
Reference in New Issue
Block a user