init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,21 @@
model_name: "PaddlePaddle/ERNIE-4.5-21B-A3B-PT"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,flexible-extract"
value: 0.71
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "Tencent-Hunyuan/Hunyuan-A13B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 32768
gpu_memory_utilization: 0.90
enforce_eager: true
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.37
- name: "exact_match,flexible-extract"
value: 0.28
num_fewshot: 5
limit: 1000
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "OpenGVLab/InternVL3_5-8B-hf"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 40960
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.58
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,23 @@
model_name: "LLM-Research/Llama-3.2-3B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.71
- name: "exact_match,flexible-extract"
value: 0.76
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,24 @@
model_name: "nv-community/Minitron-8B-Base"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.9
enforce_eager: true
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.5436
- name: "exact_match,flexible-extract"
value: 0.5451
limit: 1000
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,32 @@
model_name: "mistralai/Mixtral-8x7B-Instruct-v0.1"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: bfloat16
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: true
enforce_eager: true
block_size: 128
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "200"
VLLM_ASCEND_ENABLE_MLAPO: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.45
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -0,0 +1,21 @@
model_name: "LLM-Research/Molmo-7B-D-0924"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.71
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,23 @@
model_name: "Qwen/Qwen2-Audio-7B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.44
- name: "exact_match,flexible-extract"
value: 0.45
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "Qwen/Qwen2.5-Math-RM-72B"
model_type: "vllm-rm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.9
trust_remote_code: false
# system_prompt controls the <|im_start|>system block passed to the reward model.
system_prompt: "Please reason step by step, and put your final answer within \\boxed{}."
tasks:
- name: "gsm8k_correctness"
dataset: "AI-ModelScope/gsm8k"
split: "test"
dataset_config: "main"
metrics:
- name: "accuracy"
value: 0.80
limit: 200
batch_size: 4

View File

@@ -0,0 +1,25 @@
model_name: "vllm-ascend/Qwen3-30B-A3B-W8A8"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
quantization: ascend
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.9
- name: "exact_match,flexible-extract"
value: 0.8
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -1,6 +1,15 @@
model_name: "Qwen/Qwen3-30B-A3B"
runner: "linux-aarch64-a2-2"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.6
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "gsm8k"
metrics:
@@ -12,9 +21,8 @@ tasks:
metrics:
- name: "acc,none"
value: 0.84
num_fewshot: 5
gpu_memory_utilization: 0.6
enable_expert_parallel: True
tensor_parallel_size: 2
apply_chat_template: False
fewshot_as_multiturn: False
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "vllm-ascend/Qwen3-8B-W8A8"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
quantization: ascend
enable_thinking: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.80
- name: "exact_match,flexible-extract"
value: 0.82
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,24 @@
model_name: "Qwen/Qwen3-8B"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
enable_thinking: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.765
- name: "exact_match,flexible-extract"
value: 0.81
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "Qwen/Qwen3-ASR-1.7B"
model_type: "vllm-asr"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "librispeech_test_clean"
dataset: "openslr/librispeech_asr"
split: "test"
dataset_config: "clean"
metrics:
- name: "wer"
value: 0.035
limit: 500

View File

@@ -0,0 +1,23 @@
model_name: "Qwen/Qwen3-Next-80B-A3B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
enforce_eager: true
tasks:
- name: "ceval-valid_accountant"
metrics:
- name: "acc,none"
value: 0.98
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: 1

View File

@@ -0,0 +1,22 @@
model_name: "Qwen/Qwen3-Omni-30B-A3B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.60
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,22 @@
model_name: "Qwen/Qwen3-VL-30B-A3B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 128000
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.58
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,22 @@
model_name: "vllm-ascend/Qwen3-VL-8B-Instruct-W8A8"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.8
trust_remote_code: false
quantization: ascend
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.52
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -0,0 +1,21 @@
model_name: "Qwen/Qwen3-VL-8B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.7
trust_remote_code: false
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.55
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -1,4 +1,14 @@
DeepSeek-V2-Lite.yaml
Qwen3-8B-Base.yaml
Qwen2.5-VL-7B-Instruct.yaml
Qwen3-30B-A3B.yaml
Qwen3-30B-A3B.yaml
Qwen3-8B.yaml
Qwen2-Audio-7B-Instruct.yaml
Qwen3-VL-30B-A3B-Instruct.yaml
Qwen3-VL-8B-Instruct.yaml
Qwen3-Omni-30B-A3B-Instruct.yaml
InternVL3_5-8B-hf.yaml
ERNIE-4.5-21B-A3B-PT.yaml
gemma-3-4b-it.yaml
internlm3-8b-instruct.yaml
Molmo-7B-D-0924.yaml
llava-onevision-qwen2-0.5b-ov-hf.yaml
Llama-3.2-3B-Instruct.yaml
Qwen3-ASR-1.7B.yaml

View File

@@ -0,0 +1,24 @@
model_name: "LLM-Research/gemma-3-4b-it"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enforce_eager: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.59
- name: "exact_match,flexible-extract"
value: 0.59
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "Shanghai_AI_Laboratory/internlm3-8b-instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: "bfloat16"
max_model_len: 2048
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.42
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "llava-hf/llava-onevision-qwen2-0.5b-ov-hf"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.42
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"