@@ -0,0 +1,84 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "10"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "36864"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1, "method": "mtp"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
|
||||
_benchmarks_acc: &benchmarks_acc
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
_benchmarks_perf: &benchmarks_perf
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 400
|
||||
max_out_len: 1500
|
||||
batch_size: 1000
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-R1-0528-W8A8-single"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--enforce-eager"
|
||||
benchmarks:
|
||||
|
||||
- name: "DeepSeek-R1-0528-W8A8-aclgraph"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks_acc
|
||||
<<: *benchmarks_perf
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V3.2-W8A8-DCP-replicated-indexer"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
envs:
|
||||
VLLM_ASCEND_ENABLE_NZ: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "20"
|
||||
HCCL_BUFFSIZE: "768"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_SERVER_DEV_MODE: "1"
|
||||
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
|
||||
ASCEND_LAUNCH_BLOCKING: "0"
|
||||
ASCEND_ENABLE_USE_FABRIC_MEM: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "0"
|
||||
PYTHONHASHSEED: "0"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "10000"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
CPU_AFFINITY_CONF: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "1024"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--pipeline-parallel-size"
|
||||
- "1"
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--prefill-context-parallel-size"
|
||||
- "1"
|
||||
- "--decode-context-parallel-size"
|
||||
- "16"
|
||||
- "--cp-kv-cache-interleave-size"
|
||||
- "1"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.95"
|
||||
- "--api-server-count"
|
||||
- "1"
|
||||
- "--safetensors-load-strategy"
|
||||
- "prefetch"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 16, 64, 128]}'
|
||||
- "--additional-config"
|
||||
- '{"enable_dsa_cp": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}, "multistream_overlap_shared_expert": true, "enable_mc2_hierarchy_comm": false, "enable_sparse_sfa_c8": true, "enable_sparse_li_c8": true, "enable_cpu_binding": true, "recompute_scheduler_enable": false}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
test_content: []
|
||||
benchmarks:
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 8192
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,80 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V3.2-W8A8-TP8-DP2"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_USE_V1: "1"
|
||||
HCCL_BUFFSIZE: "256"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "67000"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "8"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--async-scheduling"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.95"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,2,4,8,16,24,32,40,48], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
- "--reasoning-parser"
|
||||
- "deepseek_v3"
|
||||
- "--tokenizer_mode"
|
||||
- "deepseek_v32"
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 86.67
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
thinking: true
|
||||
threshold: 10
|
||||
|
||||
perf_2:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 16
|
||||
max_out_len: 1500
|
||||
batch_size: 4
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,80 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V4-Flash-W8A8-A3"
|
||||
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
|
||||
special_dependencies:
|
||||
transformers: "5.9.0"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
ASCEND_LAUNCH_BLOCKING: "0"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
server_cmd:
|
||||
- "--enable-prefix-caching"
|
||||
- "--max-model-len"
|
||||
- "1048576"
|
||||
- "--max-num-batched-tokens"
|
||||
- "10240"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--max-num-seqs"
|
||||
- "64"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--tokenizer-mode"
|
||||
- "deepseek_v4"
|
||||
- "--tool-call-parser"
|
||||
- "deepseek_v4"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--reasoning-parser"
|
||||
- "deepseek_v4"
|
||||
- "--safetensors-load-strategy"
|
||||
- "prefetch"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--api-server-count"
|
||||
- "1"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- "--async-scheduling"
|
||||
- "--additional-config"
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":"true","enable_shared_expert_dp":true,"multistream_overlap_shared_expert":true}'
|
||||
benchmarks:
|
||||
acc-gpqa:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 86.36
|
||||
threshold: 5
|
||||
thinking: true
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 64
|
||||
max_out_len: 1024
|
||||
batch_size: 16
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
66
tests/e2e/nightly/single_node/models/configs/GLM-4.7.yaml
Normal file
66
tests/e2e/nightly/single_node/models/configs/GLM-4.7.yaml
Normal file
@@ -0,0 +1,66 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
HCCL_BUFFSIZE: "512"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--enable-expert-parallel"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_shared_expert_dp": true, "ascend_fusion_config": {"fusion_ops_gmmswigluquant": false}}'
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 8
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "GLM-4.7-TP8-DP2-decodegraph"
|
||||
model: "Eco-Tech/GLM-4.7-W8A8-floatmtp"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes": [1,2,4,8,16,32,64,128,256,512], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,82 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "GLM-5.1-W8A8-PrefillMC2"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8" #need update
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_USE_V1: "1"
|
||||
HCCL_BUFFSIZE: "1800"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "10240"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--async-scheduling"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.94"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp", "enforce_eager": true}'
|
||||
- "--additional_config"
|
||||
- '{"enable_prefill_mc2": true}'
|
||||
- "--reasoning-parser"
|
||||
- "glm45"
|
||||
- "--tool-call-parser"
|
||||
- "glm47"
|
||||
|
||||
benchmarks:
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 8192
|
||||
batch_size: 32
|
||||
baseline: 96.88
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
thinking: true
|
||||
threshold: 5
|
||||
|
||||
perf_2:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 64
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,58 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-ep-weight-filter"
|
||||
- "--tool-call-parser"
|
||||
- "hy_v3"
|
||||
- "--reasoning-parser"
|
||||
- "hy_v3"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--max-model-len"
|
||||
- "32768"
|
||||
- "--max-num-seqs"
|
||||
- "8"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--speculative-config"
|
||||
- '{"method": "mtp", "num_speculative_tokens": 1}'
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_4_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 8
|
||||
baseline: 93.07
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Hy3-preview-TP16-EP-MTP"
|
||||
model: "Tencent-Hunyuan/Hy3-preview"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,52 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Kimi-K2-Thinking-TP16-Case"
|
||||
model: "moonshotai/Kimi-K2-Thinking"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "12"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--trust-remote-code"
|
||||
- "--enable-expert-parallel"
|
||||
- "--no-enable-prefix-caching"
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 256
|
||||
batch_size: 64
|
||||
trust_remote_code: true
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
91
tests/e2e/nightly/single_node/models/configs/Kimi-K2.5.yaml
Normal file
91
tests/e2e/nightly/single_node/models/configs/Kimi-K2.5.yaml
Normal file
@@ -0,0 +1,91 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
HCCL_BUFFSIZE: "512"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_NZ: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-prefix-caching"
|
||||
- "--enable-chunked-prefill"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "133120"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--seed"
|
||||
- "42"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
- "--additional-config"
|
||||
- '{"enable_shared_expert_dp":true}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--mm-encoder-tp-mode"
|
||||
- "data"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
temperature: 0.0
|
||||
top_p: 1
|
||||
top_k: -1
|
||||
repetition_penalty: 1.0
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 8
|
||||
max_out_len: 1024
|
||||
batch_size: 2
|
||||
trust_remote_code: true
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Kimi-K2.5-W4A8-Case"
|
||||
model: "Eco-Tech/Kimi-K2.5-W4A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,70 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
test_cases:
|
||||
- name: "Kimi-K2.6-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
|
||||
model: "Eco-Tech/Kimi-K2.6-w4a8"
|
||||
envs:
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_BUFFSIZE: "800"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
DYNAMIC_EPLB: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--trust-remote-code"
|
||||
- "--safetensors-load-strategy"
|
||||
- 'prefetch'
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--max-num-seqs"
|
||||
- "24"
|
||||
- "--max-model-len"
|
||||
- "6144"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.85"
|
||||
- "--seed"
|
||||
- "42"
|
||||
- "--async-scheduling"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--additional-config"
|
||||
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--mm-encoder-tp-mode"
|
||||
- "data"
|
||||
- "--speculative-config"
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 128
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1433.4454
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,91 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "100"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "3600000"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-seqs"
|
||||
- "14"
|
||||
- "--trust-remote-code"
|
||||
|
||||
_benchmarks_gsm8k: &benchmarks_gsm8k
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
_benchmarks_aime: &benchmarks_aime
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2024
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 86.67
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp2"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 2, "method": "mtp"}'
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.92"
|
||||
benchmarks:
|
||||
<<: *benchmarks_gsm8k
|
||||
|
||||
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp3"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-num-batched-tokens"
|
||||
- "2048"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method": "mtp"}'
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes": [56], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks_aime
|
||||
@@ -0,0 +1,88 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.5-w8a8"
|
||||
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "512"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
|
||||
HCCL_INTRA_PCIE_ENABLE: "1"
|
||||
HCCL_INTRA_ROCE_ENABLE: "0"
|
||||
OMP_PROC_BIND: "false"
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: "0"
|
||||
VLLM_TORCH_PROFILER_DIR: "./profile"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--model-loader-extra-config"
|
||||
- '{"enable_multithread_load":true,"num_threads":16}'
|
||||
- "--speculative_config"
|
||||
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-chunked-prefill"
|
||||
- "--enable-prefix-caching"
|
||||
- "--max-num-seqs"
|
||||
- "100"
|
||||
- "--max-model-len"
|
||||
- "196608"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-num-batched-tokens"
|
||||
- "6144"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--tool-call-parser"
|
||||
- "minimax_m2"
|
||||
- "--reasoning-parser"
|
||||
- "minimax_m2_append_think"
|
||||
- "--enable-force-include-usage"
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,16,40,80,160,256,400]}'
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gpqa_gen_0_shot_str
|
||||
max_out_len: 131072
|
||||
batch_size: 64
|
||||
baseline: 83
|
||||
threshold: 5
|
||||
bos_token_id: 200019
|
||||
do_sample: true
|
||||
eos_token_id: 200020
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 40
|
||||
transformers_version: 4.46.1
|
||||
ignore_eos: false
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 360
|
||||
max_out_len: 1500
|
||||
batch_size: 120
|
||||
request_rate: 0
|
||||
baseline: 2042
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,73 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.5-w8a8"
|
||||
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "512"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM-ASCEND_ENABLE_NZ: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.85"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
|
||||
- "--enable-expert-parallel"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--max-model-len"
|
||||
- "196608"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 131072
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
bos_token_id: 200019
|
||||
do_sample: true
|
||||
eos_token_id: 200020
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 40
|
||||
transformers_version: 4.46.1
|
||||
ignore_eos: false
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 1500
|
||||
batch_size: 128
|
||||
request_rate: 0
|
||||
baseline: 1116
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,85 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1200"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--host"
|
||||
- "0.0.0.0"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--enable-expert-parallel"
|
||||
- "--async-scheduling"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--safetensors-load-strategy"
|
||||
- 'prefetch'
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--tool-call-parser"
|
||||
- "minimax_m2"
|
||||
- "--speculative-config"
|
||||
- '{"method":"eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding": true, "enable_npugraph_ex": true, "enable_static_kernel": true,"enable_fused_mc2":true,"weight_nz_mode":true,"enable_flashcomm1":true}'
|
||||
|
||||
_benchmarks_3500: &benchmarks_3500
|
||||
perf_50:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 760
|
||||
max_out_len: 1500
|
||||
batch_size: 190
|
||||
request_rate: 0
|
||||
baseline: 4573.02
|
||||
threshold: 0.97
|
||||
perf_20:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 192
|
||||
max_out_len: 1500
|
||||
batch_size: 48
|
||||
request_rate: 0
|
||||
baseline: 2229.147
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.7-3500"
|
||||
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-model-len"
|
||||
- "70000"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.8"
|
||||
- "--no-enable-prefix-caching"
|
||||
benchmarks:
|
||||
<<: *benchmarks_3500
|
||||
@@ -0,0 +1,78 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "prefix-cache-deepseek-r1-0528-w8a8"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
OMP_NUM_THREADS: "10"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "5200"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1, "method": "mtp"}'
|
||||
test_content:
|
||||
- "benchmark_comparisons"
|
||||
benchmark_comparisons_args:
|
||||
- metric: "TTFT"
|
||||
baseline: "prefix0"
|
||||
target: "prefix75"
|
||||
ratio: 0.5
|
||||
operator: "<"
|
||||
benchmarks:
|
||||
warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1024-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 1000
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
prefix0:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix0-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 18
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
prefix75:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix75-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 18
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,70 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "prefix-cache-qwen3-32b-w8a8"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--reasoning-parser"
|
||||
- "qwen3"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "256"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
test_content:
|
||||
- "benchmark_comparisons"
|
||||
benchmark_comparisons_args:
|
||||
- metric: "TTFT"
|
||||
baseline: "prefix0"
|
||||
target: "prefix75"
|
||||
ratio: 0.4
|
||||
operator: "<"
|
||||
benchmarks:
|
||||
warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1024-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 1000
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
prefix0:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix0-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 48
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
prefix75:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix75-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 48
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,84 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "10"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "12"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
top_k: 20
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-235B-A22B-W8A8-full_graph"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
|
||||
- name: "Qwen3-235B-A22B-W8A8-piecewise"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "PIECEWISE"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
|
||||
- name: "Qwen3-235B-A22B-W8A8-EPLB"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
DYNAMIC_EPLB: "true"
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--additional-config"
|
||||
- '{"eplb_config": {"dynamic_eplb": "true", "expert_heat_collection_interval": 600, "algorithm_execution_interval": 50, "num_redundant_experts": 16, "eplb_policy_type": 2}}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,41 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-30B-A3B-W4A8-llm-compressor"
|
||||
model: "vllm-ascend/Qwen3-30B-A3B-Instruct-2507-quantized.w4a8"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.8"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
@@ -0,0 +1,43 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-30B-A3B-W8A8-TP1"
|
||||
model: "vllm-ascend/Qwen3-30B-A3B-W8A8"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--tensor-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "5600"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--max-num-seqs"
|
||||
- "100"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 180
|
||||
max_out_len: 1500
|
||||
batch_size: 45
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,40 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-30B-QuaRot"
|
||||
model: "vllm-ascend/Qwen3-30B-A3B-W8A8-QuaRot"
|
||||
envs:
|
||||
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
HCCL_BUFFSIZE: "768"
|
||||
server_cmd:
|
||||
- "--enforce-eager"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--trust-remote-code"
|
||||
- "--distributed-executor-backend"
|
||||
- "mp"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"method": "eagle3", "model": "AngelSlim/Qwen3-a3B_eagle3", "num_speculative_tokens": 3}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 80
|
||||
max_out_len: 1500
|
||||
batch_size: 20
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,78 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-batched-tokens"
|
||||
- "40960"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--trust-remote-code"
|
||||
- "--reasoning-parser"
|
||||
- "qwen3"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"weight_prefetch_config":{"enabled":true}}'
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2024
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 83.33
|
||||
threshold: 10
|
||||
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 288
|
||||
max_out_len: 1500
|
||||
batch_size: 72
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-32B-W8A8-aclgraph-a2"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1,12,16,20,24,32,48,60,64,68,72,76,80]}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
|
||||
- name: "Qwen3-32B-W8A8-single-a2"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--enforce-eager"
|
||||
benchmarks:
|
||||
@@ -0,0 +1,78 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--max-num-seqs"
|
||||
- "80"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-batched-tokens"
|
||||
- "40960"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"weight_prefetch_config":{"enabled":true}}'
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_noncot_chat_prompt
|
||||
max_out_len: 10240
|
||||
batch_size: 32
|
||||
baseline: 96
|
||||
threshold: 10
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 304
|
||||
max_out_len: 1500
|
||||
batch_size: 76
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-32B-W8A8-aclgraph-a3"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1,12,16,20,24,32,48,60,64,68,72,76,80]}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
|
||||
- name: "Qwen3-32B-W8A8-single-a3"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--enforce-eager"
|
||||
benchmarks:
|
||||
@@ -0,0 +1,38 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-32B-QuaRot"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8-QuaRot"
|
||||
envs:
|
||||
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--enforce-eager"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--trust-remote-code"
|
||||
- "--distributed-executor-backend"
|
||||
- "mp"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"method": "eagle3", "model": "RedHatAI/Qwen3-32B-speculator.eagle3", "num_speculative_tokens": 3}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 80
|
||||
max_out_len: 1500
|
||||
batch_size: 20
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,70 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "1"
|
||||
OMP_PROC_BIND: "false"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1536"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ASCEND_ENABLE_NZ: "2"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "32768"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.92"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/textvqa-lite
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
max_out_len: 2048
|
||||
batch_size: 128
|
||||
baseline: 83
|
||||
temperature: 0
|
||||
top_k: -1
|
||||
top_p: 1
|
||||
repetition_penalty: 1
|
||||
threshold: 5
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-VL-235B-A22B-Instruct-W8A8"
|
||||
model: "Eco-Tech/Qwen3-VL-235B-A22B-Instruct-w8a8-QuaRot"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation_config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,2,4,8,16,24,32]}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,67 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "1"
|
||||
OMP_PROC_BIND: "false"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "20000"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/textvqa-lite
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
max_out_len: 2048
|
||||
batch_size: 128
|
||||
baseline: 80
|
||||
temperature: 0
|
||||
top_k: -1
|
||||
top_p: 1
|
||||
repetition_penalty: 1
|
||||
threshold: 5
|
||||
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3-VL-32B-Instruct-W8A8"
|
||||
model: "Eco-Tech/Qwen3-VL-32B-Instruct-w8a8-QuaRot"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation_config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,12,16,20,24,32,48,64,68,72,76,80,128]}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,79 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "131072"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
|
||||
- "--trust-remote-code"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding": true, "enable_shared_expert_dp": true}'
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-122B-A10B-W8A8-A3"
|
||||
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
thinking: true
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
presence_penalty: 1.5
|
||||
repetition_penalty: 1.0
|
||||
ignore_eos: false
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 320
|
||||
max_out_len: 1500
|
||||
batch_size: 80
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,57 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-27B-w8a8"
|
||||
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
|
||||
envs:
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
|
||||
VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE: "1"
|
||||
VLLM_ASCEND_ENABLE_NZ: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--max-model-len"
|
||||
- "262144"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.95"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true, "enable_weight_nz_layout":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3,"enforce_eager": true}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144]}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--mm_processor_cache_type"
|
||||
- "shm"
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 128
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 604
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,73 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-27B-w8a8"
|
||||
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
|
||||
envs:
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "2"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--max-model-len"
|
||||
- "196608"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--async-scheduling"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3,"enforce_eager": true}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
ignore_eos: false
|
||||
thinking: true
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
presence_penalty: 1.5
|
||||
repetition_penalty: 1.0
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 140
|
||||
max_out_len: 1500
|
||||
batch_size: 35
|
||||
request_rate: 0
|
||||
baseline: 610.22
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,76 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-397B-A17B-w8a8-mtp"
|
||||
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
|
||||
envs:
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "6000"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--max-model-len"
|
||||
- "133120"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--safetensors-load-strategy"
|
||||
- "lazy"
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
temperature: 0.6
|
||||
top_p: 0.95
|
||||
top_k: 20
|
||||
min_p: 0.0
|
||||
presence_penalty: 0.0
|
||||
repetition_penalty: 1.0
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in131072-bs100-qwen3
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 8
|
||||
max_out_len: 1024
|
||||
batch_size: 2
|
||||
request_rate: 0
|
||||
baseline: 56.357
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,74 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-397B-A17B-w4a8-mtp"
|
||||
model: "Eco-Tech/Qwen3.5-397B-A17B-w4a8-mtp"
|
||||
envs:
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--max-model-len"
|
||||
- "133120"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true,"multistream_overlap_shared_expert":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,108,112,128,160,172,196,200,212,232,256,260,288,320,360,400], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
benchmarks:
|
||||
acc_GPQA:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 88.38
|
||||
threshold: 5
|
||||
# acc_aime2025:
|
||||
# case_type: accuracy
|
||||
# dataset_path: vllm-ascend/aime2025
|
||||
# request_conf: vllm_api_general_chat
|
||||
# dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
# max_out_len: 72348
|
||||
# batch_size: 32
|
||||
# baseline: 93.33
|
||||
# threshold: 5
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 256
|
||||
max_out_len: 1500
|
||||
batch_size: 64
|
||||
request_rate: 0
|
||||
baseline: 1040
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,62 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Qwen3.5-397B-A17B-w8a8-mtp-longseq"
|
||||
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
|
||||
envs:
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "6000"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--prefill-context-parallel-size"
|
||||
- "2"
|
||||
- "--decode-context-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--max-model-len"
|
||||
- "133120"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative_config"
|
||||
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,4,8,16,32,64], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--safetensors-load-strategy"
|
||||
- "lazy"
|
||||
benchmarks:
|
||||
acc_GPQA:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 50
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 84
|
||||
threshold: 8
|
||||
@@ -0,0 +1,312 @@
|
||||
# vLLM-Ascend Single-Node E2E Test Developer Guide
|
||||
|
||||
This document is intended to help developers understand the architecture of the single-node E2E (End-to-End) testing framework in `vllm-ascend`, how to run test scripts, and how to add custom testing functionality by writing YAML configuration files and extending the code.
|
||||
|
||||
## 1. Test Architecture Overview
|
||||
|
||||
To achieve high readability, extensibility, and decoupling of configuration from code, the single-node E2E test adopts a **"YAML-driven + Dispatcher"** architectural structure.
|
||||
|
||||
It consists of the following core components:
|
||||
|
||||
* **Configuration Parser (`single_node_config.py`)**: Responsible for reading `models/configs/*.yaml` files and parsing them into a strongly-typed `@dataclass` (`SingleNodeConfig`) via `SingleNodeConfigLoader`, while handling regex replacement for environment variables.
|
||||
* **Service Manager Framework (`test_single_node.py` and `conftest.py`)**: Based on the `service_mode` (`openai` or `epd`), it utilizes context managers to safely start/stop server processes.
|
||||
* **Test Function Dispatcher (`TEST_HANDLERS` Registry)**: Specific test logic is encapsulated into independent functions and registered in the global `TEST_HANDLERS` dictionary.
|
||||
* **Performance Benchmarking (`_run_benchmarks`)**: Calls `aisbench` for performance and TTFT testing based on the `benchmarks` parameters in the YAML.
|
||||
|
||||
### 1.1 Key Files and Responsibilities
|
||||
|
||||
* `tests/e2e/nightly/single_node/models/scripts/single_node_config.py`
|
||||
* Defines `SingleNodeConfig` and `SingleNodeConfigLoader`
|
||||
* Loads YAML from `tests/e2e/nightly/single_node/models/configs/<CONFIG_YAML_PATH>`
|
||||
* Auto-assigns ports when `envs` contains `DEFAULT_PORT` / missing values
|
||||
* Expands `$VAR` / `${VAR}` placeholders inside commands via `_expand_values`
|
||||
|
||||
* `tests/e2e/nightly/single_node/models/scripts/test_single_node.py`
|
||||
* Declares `configs = SingleNodeConfigLoader.from_yaml_cases()` (loaded at import time)
|
||||
* `pytest.mark.parametrize("config", configs, ids=[config.name for config in configs])` runs one test per YAML case
|
||||
* Controls server lifecycle via context managers
|
||||
* Dispatches `test_content` to functions registered in `TEST_HANDLERS`
|
||||
* Runs `aisbench` and optional benchmark assertions
|
||||
|
||||
### 1.2 End-to-End Flow (High Level)
|
||||
|
||||
```txt
|
||||
pytest starts
|
||||
|
|
||||
v
|
||||
import tests/e2e/nightly/single_node/models/scripts/test_single_node.py
|
||||
|
|
||||
v
|
||||
configs = SingleNodeConfigLoader.from_yaml_cases()
|
||||
|
|
||||
v
|
||||
pytest parametrize("config", configs) # one config == one test case
|
||||
|
|
||||
v
|
||||
test_single_node(config)
|
||||
|
|
||||
+-----------------------------------------------+
|
||||
| Start service (depends on service_mode) |
|
||||
| |
|
||||
| openai: start one vLLM OpenAI-compatible |
|
||||
| service process |
|
||||
| epd: start (encode service + decode/PD |
|
||||
| service) + start proxy process |
|
||||
+-----------------------------------------------+
|
||||
|
|
||||
v
|
||||
Run test phases (test_content)
|
||||
|
|
||||
v
|
||||
Optional benchmarks (if benchmarks is configured)
|
||||
|
|
||||
v
|
||||
Shutdown all started processes
|
||||
|
||||
Notes:
|
||||
- One YAML file may contain multiple test_cases; pytest will run them one by one.
|
||||
- The framework is "YAML-driven": changes are typically done by editing YAML rather than editing Python code.
|
||||
```
|
||||
|
||||
### 1.3 Function Call Relationships (Dispatcher)
|
||||
|
||||
`test_content` is a list of “phases”. Each phase maps to one handler function.
|
||||
|
||||
```txt
|
||||
For each test_case:
|
||||
|
||||
test_content (list of phases)
|
||||
|
|
||||
v
|
||||
[Dispatcher]
|
||||
|
|
||||
+--> phase "completion" -> send completion request(s)
|
||||
|
|
||||
+--> phase "chat_completion" -> send chat completion request(s)
|
||||
|
|
||||
+--> phase "image" -> send multimodal image request(s)
|
||||
|
|
||||
\--> (extendable) add your own phase by registering a new handler
|
||||
|
||||
After phases:
|
||||
if benchmarks is configured -> run aisbench
|
||||
|
||||
Notes:
|
||||
- The dispatcher only controls "what to run"; service lifecycle is controlled by the service manager.
|
||||
- Phases are intentionally small & composable so you can reuse them across YAML cases.
|
||||
```
|
||||
|
||||
## 2. Running and Debugging Steps
|
||||
|
||||
### 2.1 Dependencies
|
||||
|
||||
Ensure you are in an NPU environment and have installed `pytest`, `pyyaml`, `openai`, and `aisbench`.
|
||||
|
||||
### 2.2 Local Execution
|
||||
|
||||
The framework uses the `CONFIG_YAML_PATH` environment variable to specify the configuration file.
|
||||
|
||||
```bash
|
||||
# Switch to the project root directory
|
||||
cd /vllm-workspace/vllm-ascend
|
||||
|
||||
# Run a specific yaml test
|
||||
export CONFIG_YAML_PATH="Qwen3-32B.yaml"
|
||||
pytest -sv tests/e2e/nightly/single_node/models/scripts/test_single_node.py
|
||||
```
|
||||
|
||||
### 2.3 Tips for Debugging
|
||||
|
||||
* Only run a subset of cases: `pytest -sv ... -k <keyword>` (matches case names in the report output)
|
||||
* Stop on first failure: `pytest -sv ... -x`
|
||||
* Keep server logs visible: use `-s` (already included in `-sv`) and increase log verbosity via standard Python logging configuration if needed.
|
||||
|
||||
## 3. How to Write YAML Configuration Files
|
||||
|
||||
### 3.1 File Location and Selection Rules
|
||||
|
||||
* YAML files live under: `tests/e2e/nightly/single_node/models/configs/`
|
||||
* Selected by env var: `CONFIG_YAML_PATH=<YourConfig>.yaml`
|
||||
* If not set, the loader uses `SingleNodeConfigLoader.DEFAULT_CONFIG_NAME`
|
||||
|
||||
### 3.2 Field Descriptions
|
||||
|
||||
| Field Name | Type | Required | Default Value | Description |
|
||||
| :--------------- | :--------- | :------- | :-------------- | :------------------------------------------------------------------ |
|
||||
| `test_cases` | list | **Yes** | - | List of test case objects |
|
||||
| `name` | string | **Yes** | - | Human-readable case ID shown in pytest output and logs |
|
||||
| `model` | string | **Yes** | - | Model name or local path |
|
||||
| `service_mode` | string | No | `openai` | Service mode: `openai` or `epd` (disaggregated) |
|
||||
| `envs` | map | **Yes** | `{}` | Environment variables for the server process |
|
||||
| `server_cmd` | list | Cond. | `[]` | vLLM startup arguments (Required for non-EPD) |
|
||||
| `server_cmd_extra` | list | No | `[]` | Extra vLLM startup arguments appended after `server_cmd` |
|
||||
| `prompts` | list | No | built-in default | Prompts for completion/chat tests |
|
||||
| `api_keyword_args` | map | No | built-in default | OpenAI API keyword args (e.g., `max_tokens`, sampling params) |
|
||||
| `test_content` | list | No | `["completion"]` | Test phases: `completion`, `chat_completion`, `image` etc. |
|
||||
| `benchmarks` | map | No | `{}` | Configuration for `aisbench` performance verification |
|
||||
| `epd_server_cmds`| list[list] | Cond. | `[]` | (EPD Only) Command arrays for starting dual Encode/Decode processes |
|
||||
| `epd_proxy_args` | list | Cond. | `[]` | (EPD Only) Startup arguments for the EPD routing gateway |
|
||||
|
||||
**Notes / Behaviors**
|
||||
|
||||
* `name` is mandatory and must be a non-empty string.
|
||||
* It is used directly as pytest case id (e.g., `test_single_node[DeepSeek-R1-0528-W8A8-single]`).
|
||||
* It is also printed in `[single-node][START]` marker for log navigation.
|
||||
|
||||
* `envs` (ports): the config object recognizes these keys: `SERVER_PORT`, `ENCODE_PORT`, `PD_PORT`, `PROXY_PORT`.
|
||||
* If a port key is missing or set to `DEFAULT_PORT`, it will be automatically filled with an available open port.
|
||||
* `$SERVER_PORT` / `${SERVER_PORT}` placeholders in commands will be expanded using `envs`.
|
||||
|
||||
* `server_cmd` vs `server_cmd_extra`:
|
||||
* YAML can define `server_cmd_extra` to append additional args after `server_cmd`.
|
||||
* The loader merges them into a single `server_cmd` list.
|
||||
|
||||
* Extra fields:
|
||||
* Any non-standard fields in a case are stored in `config.extra_config`.
|
||||
* This is how extension configs are passed through without changing the dataclass.
|
||||
|
||||
### 3.3 YAML Examples
|
||||
|
||||
#### Single-Case (similar to DeepSeek-R1-W8A8-HBM)
|
||||
|
||||
```yaml
|
||||
test_cases:
|
||||
- name: "<your-case-name>"
|
||||
model: "<model-repo-or-local-path>"
|
||||
|
||||
# Optional: The default values are as follows
|
||||
prompts:
|
||||
- "San Francisco is a"
|
||||
api_keyword_args:
|
||||
max_tokens: 10
|
||||
|
||||
envs:
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
# Add only what you need.
|
||||
|
||||
server_cmd:
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
# plus your vLLM serve args...
|
||||
|
||||
# Optional: omit -> defaults to ["completion"]
|
||||
test_content:
|
||||
- "chat_completion"
|
||||
|
||||
# Optional: leave empty if you don't run aisbench
|
||||
benchmarks:
|
||||
```
|
||||
|
||||
#### Multi-Case + Shared Anchors
|
||||
|
||||
```yaml
|
||||
_envs: &envs
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
# shared envs...
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
# shared vLLM serve args...
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 400
|
||||
max_out_len: 1500
|
||||
batch_size: 1000
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
test_cases:
|
||||
- name: "case-a"
|
||||
model: "<model>"
|
||||
envs:
|
||||
<<: *envs
|
||||
DYNAMIC_EPLB: "true"
|
||||
# private envs...
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--enforce-eager"
|
||||
benchmarks:
|
||||
|
||||
- name: "case-b"
|
||||
model: "<model>"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
```
|
||||
|
||||
#### EPD / Disaggregated Case
|
||||
|
||||
```yaml
|
||||
test_cases:
|
||||
- name: "<your-epd-case>"
|
||||
model: "<model>"
|
||||
service_mode: "epd"
|
||||
envs:
|
||||
ENCODE_PORT: "DEFAULT_PORT"
|
||||
PD_PORT: "DEFAULT_PORT"
|
||||
PROXY_PORT: "DEFAULT_PORT"
|
||||
|
||||
epd_server_cmds:
|
||||
- ["--port", "$ENCODE_PORT", "--model", "<encode-model>"]
|
||||
- ["--port", "$PD_PORT", "--model", "<decode-model>"]
|
||||
|
||||
epd_proxy_args:
|
||||
- "--host"
|
||||
- "127.0.0.1"
|
||||
- "--port"
|
||||
- "$PROXY_PORT"
|
||||
- "--encode-servers-urls"
|
||||
- "http://localhost:$ENCODE_PORT"
|
||||
- "--decode-servers-urls"
|
||||
- "http://localhost:$PD_PORT"
|
||||
- "--prefill-servers-urls"
|
||||
- "disable"
|
||||
|
||||
test_content:
|
||||
- "chat_completion"
|
||||
```
|
||||
|
||||
## 4. How to Add Custom Tests (Extension)
|
||||
|
||||
### Step 1: Write your test logic in `test_single_node.py`
|
||||
|
||||
```python
|
||||
async def run_video_test(config: SingleNodeConfig, server: 'RemoteOpenAIServer | DisaggEpdProxy') -> None:
|
||||
client = server.get_async_client()
|
||||
# Your custom logic here...
|
||||
```
|
||||
|
||||
### Step 2: Register your function in `TEST_HANDLERS`
|
||||
|
||||
```python
|
||||
TEST_HANDLERS = {
|
||||
"completion": run_completion_test,
|
||||
"video": run_video_test, # Registered!
|
||||
}
|
||||
```
|
||||
|
||||
### Step 3: Enable in YAML
|
||||
|
||||
```yaml
|
||||
test_content:
|
||||
- "completion"
|
||||
- "video"
|
||||
```
|
||||
|
||||
## 5. Checklist (Before Submitting a New YAML)
|
||||
|
||||
* `test_cases` exists and is a list
|
||||
* Each case contains required fields for its `service_mode`
|
||||
* Common required: `name`, `model`, `envs`
|
||||
* `openai`: `server_cmd`
|
||||
* `epd`: `epd_server_cmds`, `epd_proxy_args`
|
||||
* Port envs are set to `DEFAULT_PORT` (or to explicit free ports)
|
||||
* If using `benchmarks`, ensure each benchmark case includes required aisbench fields (e.g., `case_type`, `dataset_path`, `request_conf`, `dataset_conf`, `max_out_len`, `batch_size`)
|
||||
16
tests/e2e/nightly/single_node/models/scripts/__init__.py
Normal file
16
tests/e2e/nightly/single_node/models/scripts/__init__.py
Normal file
@@ -0,0 +1,16 @@
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
@@ -0,0 +1,188 @@
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
import yaml
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or "tests/e2e/nightly/single_node/models/configs"
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
# Default prompts and API args fallback
|
||||
PROMPTS = [
|
||||
"San Francisco is a",
|
||||
]
|
||||
|
||||
API_KEYWORD_ARGS = {
|
||||
"max_tokens": 10,
|
||||
}
|
||||
|
||||
|
||||
@dataclass
|
||||
class SingleNodeConfig:
|
||||
name: str
|
||||
model: str
|
||||
envs: dict[str, Any] = field(default_factory=dict)
|
||||
special_dependencies: dict[str, Any] = field(default_factory=dict)
|
||||
prompts: list[str] = field(default_factory=lambda: PROMPTS)
|
||||
api_keyword_args: dict[str, Any] = field(default_factory=lambda: API_KEYWORD_ARGS)
|
||||
benchmarks: dict[str, Any] = field(default_factory=dict)
|
||||
server_cmd: list[str] = field(default_factory=list)
|
||||
test_content: list[str] = field(default_factory=lambda: ["completion"])
|
||||
service_mode: str = "openai"
|
||||
epd_server_cmds: list[list[str]] = field(default_factory=list)
|
||||
epd_proxy_args: list[str] = field(default_factory=list)
|
||||
extra_config: dict[str, Any] = field(default_factory=dict)
|
||||
|
||||
def __post_init__(self) -> None:
|
||||
port_keys = ["SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"]
|
||||
for env_key in port_keys:
|
||||
if self.envs.get(env_key) in ["DEFAULT_PORT", None]:
|
||||
self.envs[env_key] = str(get_open_port())
|
||||
|
||||
if self.prompts is None:
|
||||
self.prompts = PROMPTS
|
||||
if self.api_keyword_args is None:
|
||||
self.api_keyword_args = API_KEYWORD_ARGS
|
||||
if self.benchmarks is None:
|
||||
self.benchmarks = {}
|
||||
if self.special_dependencies is None:
|
||||
self.special_dependencies = {}
|
||||
if self.test_content is None:
|
||||
self.test_content = []
|
||||
|
||||
self.server_cmd = self._expand_values(self.server_cmd or [], self.envs)
|
||||
self.epd_server_cmds = [self._expand_values(cmd, self.envs) for cmd in self.epd_server_cmds]
|
||||
self.epd_proxy_args = self._expand_values(self.epd_proxy_args or [], self.envs)
|
||||
|
||||
for key, value in self.extra_config.items():
|
||||
setattr(self, key, value)
|
||||
|
||||
@staticmethod
|
||||
def _expand_values(values: list[str], envs: dict[str, Any]) -> list[str]:
|
||||
"""Interpolate $VAR/${VAR} placeholders with provided env values."""
|
||||
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
|
||||
|
||||
def repl(m: re.Match[str]) -> str:
|
||||
key = m.group(1) or m.group(2)
|
||||
return str(envs.get(key, m.group(0)))
|
||||
|
||||
return [pattern.sub(repl, str(arg)) for arg in values]
|
||||
|
||||
def _get_required_port(self, key: str) -> int:
|
||||
value = self.envs.get(key)
|
||||
if value is None:
|
||||
raise ValueError(f"Missing required port env: {key}")
|
||||
return int(value)
|
||||
|
||||
@property
|
||||
def server_port(self) -> int:
|
||||
return self._get_required_port("SERVER_PORT")
|
||||
|
||||
@property
|
||||
def encode_port(self) -> int:
|
||||
return self._get_required_port("ENCODE_PORT")
|
||||
|
||||
@property
|
||||
def pd_port(self) -> int:
|
||||
return self._get_required_port("PD_PORT")
|
||||
|
||||
@property
|
||||
def proxy_port(self) -> int:
|
||||
return self._get_required_port("PROXY_PORT")
|
||||
|
||||
|
||||
class SingleNodeConfigLoader:
|
||||
"""Load SingleNodeConfig from yaml file."""
|
||||
|
||||
DEFAULT_CONFIG_NAME = "Kimi-K2-Thinking.yaml"
|
||||
STANDARD_CASE_FIELDS = {
|
||||
"name",
|
||||
"model",
|
||||
"envs",
|
||||
"special_dependencies",
|
||||
"prompts",
|
||||
"api_keyword_args",
|
||||
"benchmarks",
|
||||
"service_mode",
|
||||
"server_cmd",
|
||||
"server_cmd_extra",
|
||||
"test_content",
|
||||
"epd_server_cmds",
|
||||
"epd_proxy_args",
|
||||
}
|
||||
|
||||
@classmethod
|
||||
def from_yaml_cases(cls, yaml_path: str | None = None) -> list[SingleNodeConfig]:
|
||||
config = cls._load_yaml(yaml_path)
|
||||
|
||||
if "test_cases" not in config:
|
||||
raise KeyError("test_cases field is required in config yaml")
|
||||
|
||||
cases = config.get("test_cases")
|
||||
if not isinstance(cases, list):
|
||||
raise TypeError("test_cases must be a list")
|
||||
cls._validate_para(cases)
|
||||
|
||||
return cls._parse_test_cases(cases)
|
||||
|
||||
@classmethod
|
||||
def _load_yaml(cls, yaml_path: str | None) -> dict[str, Any]:
|
||||
if not yaml_path:
|
||||
yaml_path = os.getenv("CONFIG_YAML_PATH", cls.DEFAULT_CONFIG_NAME)
|
||||
|
||||
full_path = os.path.join(CONFIG_BASE_PATH, yaml_path)
|
||||
logger.info("Loading config yaml: %s", full_path)
|
||||
|
||||
with open(full_path) as f:
|
||||
return yaml.safe_load(f)
|
||||
|
||||
@staticmethod
|
||||
def _validate_para(cases: list[dict[str, Any]]) -> None:
|
||||
if not cases:
|
||||
raise ValueError("test_cases is empty")
|
||||
for case in cases:
|
||||
mode = case.get("service_mode", "openai")
|
||||
required = ["name", "model", "envs"]
|
||||
if mode == "epd":
|
||||
required.extend(["epd_server_cmds", "epd_proxy_args"])
|
||||
else:
|
||||
required.append("server_cmd")
|
||||
missing = [k for k in required if k not in case]
|
||||
if missing:
|
||||
raise KeyError(f"Missing required config fields: {missing}")
|
||||
|
||||
if not isinstance(case["name"], str) or not case["name"].strip():
|
||||
raise ValueError("test case field 'name' must be a non-empty string")
|
||||
|
||||
@classmethod
|
||||
def _parse_test_cases(cls, cases: list[dict[str, Any]]) -> list[SingleNodeConfig]:
|
||||
result: list[SingleNodeConfig] = []
|
||||
for case in cases:
|
||||
server_cmd = case.get("server_cmd", [])
|
||||
server_cmd_extra = case.get("server_cmd_extra", [])
|
||||
full_cmd = list(server_cmd) + list(server_cmd_extra)
|
||||
extra_case_fields = {key: value for key, value in case.items() if key not in cls.STANDARD_CASE_FIELDS}
|
||||
|
||||
# Safe parsing mapping
|
||||
result.append(
|
||||
SingleNodeConfig(
|
||||
name=case["name"],
|
||||
model=case["model"],
|
||||
envs=case.get("envs", {}),
|
||||
special_dependencies=case.get("special_dependencies", {}),
|
||||
server_cmd=full_cmd,
|
||||
epd_server_cmds=case.get("epd_server_cmds", []),
|
||||
epd_proxy_args=case.get("epd_proxy_args", []),
|
||||
benchmarks=case.get("benchmarks", {}),
|
||||
prompts=case.get("prompts", PROMPTS),
|
||||
api_keyword_args=case.get("api_keyword_args", API_KEYWORD_ARGS),
|
||||
test_content=case.get("test_content", ["completion"]),
|
||||
service_mode=case.get("service_mode", "openai"),
|
||||
extra_config=extra_case_fields,
|
||||
)
|
||||
)
|
||||
return result
|
||||
443
tests/e2e/nightly/single_node/models/scripts/test_single_node.py
Normal file
443
tests/e2e/nightly/single_node/models/scripts/test_single_node.py
Normal file
@@ -0,0 +1,443 @@
|
||||
import asyncio
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
import openai
|
||||
import psutil
|
||||
import pytest
|
||||
import vllm
|
||||
|
||||
from tests.e2e.conftest import DisaggEpdProxy, RemoteEPDServer, RemoteOpenAIServer
|
||||
from tests.e2e.nightly.single_node.models.scripts.single_node_config import (
|
||||
SingleNodeConfig,
|
||||
SingleNodeConfigLoader,
|
||||
)
|
||||
from tools.aisbench import run_aisbench_cases
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
configs = SingleNodeConfigLoader.from_yaml_cases()
|
||||
|
||||
|
||||
async def run_completion_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
|
||||
client = server.get_async_client()
|
||||
batch = await client.completions.create(
|
||||
model=config.model,
|
||||
prompt=config.prompts,
|
||||
**config.api_keyword_args,
|
||||
)
|
||||
choices: list[openai.types.CompletionChoice] = batch.choices
|
||||
assert choices[0].text, "empty response"
|
||||
print(choices)
|
||||
|
||||
|
||||
async def run_image_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
|
||||
from tools.send_mm_request import send_image_request
|
||||
|
||||
send_image_request(config.model, server)
|
||||
|
||||
|
||||
async def run_chat_completion_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
|
||||
from tools.send_request import send_v1_chat_completions
|
||||
|
||||
send_v1_chat_completions(
|
||||
config.prompts[0],
|
||||
model=config.model,
|
||||
server=server,
|
||||
request_args=config.api_keyword_args,
|
||||
)
|
||||
|
||||
|
||||
def run_benchmark_comparisons(config: SingleNodeConfig, results: Any) -> None:
|
||||
"""General assertion engine for aisbench outcomes mapped directly from YAML."""
|
||||
|
||||
comparisons = config.extra_config.get("benchmark_comparisons_args", [])
|
||||
|
||||
if not comparisons:
|
||||
return
|
||||
|
||||
# Valid task keys defined in benchmarks mapping
|
||||
valid_keys = [k for k, v in config.benchmarks.items() if v]
|
||||
|
||||
metrics_cache = {}
|
||||
|
||||
for comp in comparisons:
|
||||
metric = comp.get("metric", "TTFT")
|
||||
baseline_key = comp.get("baseline")
|
||||
target_key = comp.get("target")
|
||||
ratio = comp.get("ratio", 1.0)
|
||||
op = comp.get("operator", "<")
|
||||
|
||||
if not baseline_key or not target_key:
|
||||
logger.warning("Invalid comparison config: missing baseline or target. %s", comp)
|
||||
continue
|
||||
|
||||
if metric not in metrics_cache:
|
||||
if metric == "TTFT":
|
||||
from tools.aisbench import get_TTFT
|
||||
|
||||
# map TTFT outputs directly to their corresponding benchmark test case names
|
||||
metrics_cache[metric] = dict(zip(valid_keys, get_TTFT(results)))
|
||||
else:
|
||||
logger.warning("Unsupported metric for comparison: %s", metric)
|
||||
continue
|
||||
|
||||
metric_dict = metrics_cache[metric]
|
||||
baseline_val = metric_dict.get(baseline_key)
|
||||
target_val = metric_dict.get(target_key)
|
||||
|
||||
if baseline_val is None or target_val is None:
|
||||
logger.warning("Missing data to compare %s and %s in metrics: %s", baseline_key, target_key, metric_dict)
|
||||
continue
|
||||
|
||||
expected_threshold = baseline_val * ratio
|
||||
|
||||
eval_str = f"metric {metric}: {target_key}({target_val}) {op} {baseline_key}({baseline_val}) * {ratio}"
|
||||
|
||||
if op == "<":
|
||||
assert target_val < expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
|
||||
elif op == ">":
|
||||
assert target_val > expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
|
||||
elif op == "<=":
|
||||
assert target_val <= expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
|
||||
elif op == ">=":
|
||||
assert target_val >= expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
|
||||
else:
|
||||
logger.warning("Unsupported comparison operator: %s", op)
|
||||
continue
|
||||
|
||||
print(f"✅ Comparison passed: {eval_str} [threshold: {expected_threshold}]")
|
||||
|
||||
|
||||
async def run_check_rank0_process_count(
|
||||
config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy"
|
||||
) -> None:
|
||||
proc = await asyncio.create_subprocess_exec(
|
||||
"npu-smi",
|
||||
"info",
|
||||
stdout=asyncio.subprocess.PIPE,
|
||||
stderr=asyncio.subprocess.PIPE,
|
||||
)
|
||||
stdout_bytes, stderr_bytes = await proc.communicate()
|
||||
if proc.returncode == 0:
|
||||
logger.info("npu-smi info:\n%s", stdout_bytes.decode(errors="ignore"))
|
||||
else:
|
||||
logger.warning("npu-smi info failed: %s", stderr_bytes.decode(errors="ignore"))
|
||||
|
||||
vllm_serve_procs = [
|
||||
p
|
||||
for p in psutil.process_iter(attrs=["pid", "cmdline"], ad_value=None)
|
||||
if p.info["cmdline"]
|
||||
and any("vllm" in arg for arg in p.info["cmdline"])
|
||||
and any("serve" in arg for arg in p.info["cmdline"])
|
||||
]
|
||||
count = len(vllm_serve_procs)
|
||||
assert count == 1, (
|
||||
f"rank0 process count check failed: expected exactly 1 vllm serve process on rank0, found {count}"
|
||||
)
|
||||
|
||||
|
||||
# Extend this dictionary to add new test capabilities
|
||||
TEST_HANDLERS = {
|
||||
"completion": run_completion_test,
|
||||
"image": run_image_test,
|
||||
"chat_completion": run_chat_completion_test,
|
||||
"check_rank0_process_count": run_check_rank0_process_count,
|
||||
}
|
||||
|
||||
|
||||
async def _dispatch_tests(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
|
||||
"""Dispatches requested tests defined in yaml."""
|
||||
for test_name in config.test_content:
|
||||
if test_name == "benchmark_comparisons":
|
||||
continue
|
||||
|
||||
handler = TEST_HANDLERS.get(test_name)
|
||||
if handler:
|
||||
await handler(config, server)
|
||||
else:
|
||||
logger.warning("No handler registered for test content type: %s", test_name)
|
||||
|
||||
|
||||
def _extract_server_cmd_value(server_cmd: list[str], flag: str) -> str | None:
|
||||
"""Return the value following `flag` in a server_cmd list, or None."""
|
||||
try:
|
||||
idx = server_cmd.index(flag)
|
||||
return server_cmd[idx + 1]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
|
||||
|
||||
def _extract_hardware(runner: str) -> str:
|
||||
"""Derive hardware label (e.g. 'A2', 'A3') from runner name."""
|
||||
runner_lower = runner.lower()
|
||||
for label in ("a3", "a2"):
|
||||
if label in runner_lower:
|
||||
return label.upper()
|
||||
return runner
|
||||
|
||||
|
||||
_PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
|
||||
|
||||
_FEATURE_ENVS: dict[str, str] = {
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
|
||||
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
|
||||
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
|
||||
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
|
||||
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
|
||||
}
|
||||
|
||||
_PERF_METRIC_RENAME: dict[str, str] = {
|
||||
"Benchmark Duration": "Benchmark_Duration(BD)",
|
||||
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
|
||||
"Input Token Throughput": "Input_Token_Throughput(ITT)",
|
||||
"Output Token Throughput": "Output_Token_Throughput(OTT)",
|
||||
"Total Token Throughput": "Total_Token_Throughput(TTT)",
|
||||
}
|
||||
|
||||
|
||||
def _extract_dtype(config: SingleNodeConfig) -> str:
|
||||
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and --quantization ascend is set, else bf16."""
|
||||
has_w8a8 = "w8a8" in config.model.lower()
|
||||
has_quant_ascend = _extract_server_cmd_value(config.server_cmd, "--quantization") == "ascend"
|
||||
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
|
||||
|
||||
|
||||
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
|
||||
"""Extract and JSON-parse the value following `flag` in a command list."""
|
||||
val = _extract_server_cmd_value(cmd_list, flag)
|
||||
if not val:
|
||||
return {}
|
||||
try:
|
||||
return json.loads(val)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return {}
|
||||
|
||||
|
||||
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
|
||||
"""Extract enabled feature names from server_cmd and environment variables."""
|
||||
if isinstance(server_cmd, str):
|
||||
try:
|
||||
cmd_list = shlex.split(server_cmd)
|
||||
except ValueError:
|
||||
cmd_list = server_cmd.split()
|
||||
else:
|
||||
cmd_list = list(server_cmd)
|
||||
|
||||
features: list[str] = []
|
||||
|
||||
# Features from --additional-config JSON
|
||||
additional = _parse_json_flag(cmd_list, "--additional-config")
|
||||
if additional.get("enable_weight_nz_layout"):
|
||||
features.append("weight_nz_layout")
|
||||
wp = additional.get("weight_prefetch_config") or {}
|
||||
if isinstance(wp, dict) and wp.get("enabled"):
|
||||
features.append("weight_prefetch")
|
||||
tc = additional.get("torchair_graph_config") or {}
|
||||
if isinstance(tc, dict) and tc.get("enabled"):
|
||||
features.append("torchair_graph")
|
||||
asc = additional.get("ascend_scheduler_config") or {}
|
||||
if isinstance(asc, dict) and asc.get("enabled"):
|
||||
features.append("ascend_scheduler")
|
||||
|
||||
# Features from --compilation-config JSON
|
||||
compilation = _parse_json_flag(cmd_list, "--compilation-config")
|
||||
if compilation.get("cudagraph_mode"):
|
||||
features.append("aclgraph")
|
||||
|
||||
# Features from --speculative-config JSON
|
||||
speculative = _parse_json_flag(cmd_list, "--speculative-config")
|
||||
if speculative:
|
||||
features.append(speculative.get("method", "speculative"))
|
||||
|
||||
# Features from direct flags
|
||||
if "--enable-expert-parallel" in cmd_list:
|
||||
features.append("expert_parallel")
|
||||
|
||||
# Features from environment variables
|
||||
for env_key, feature_name in _FEATURE_ENVS.items():
|
||||
val = str(envs.get(env_key, "0"))
|
||||
if val not in ("0", "", "false", "False"):
|
||||
features.append(feature_name)
|
||||
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
|
||||
features.append("flashcomm2")
|
||||
|
||||
return features
|
||||
|
||||
|
||||
def _build_serve_cmd(config: SingleNodeConfig) -> dict[str, str]:
|
||||
"""Build serve_cmd dict with mix key for single-node deployments."""
|
||||
args = " ".join(config.server_cmd)
|
||||
return {"mix": f"vllm serve {config.model} {args}".strip()}
|
||||
|
||||
|
||||
def _filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
|
||||
"""Return env vars with internal port keys removed."""
|
||||
return {k: v for k, v in envs.items() if k not in _PORT_ENV_KEYS}
|
||||
|
||||
|
||||
def _task_passed(case_config: dict[str, Any], result: Any) -> bool:
|
||||
"""Return True if a single benchmark result meets its baseline/threshold."""
|
||||
if result == "":
|
||||
return False
|
||||
case_type = case_config.get("case_type")
|
||||
baseline = case_config.get("baseline")
|
||||
threshold = case_config.get("threshold")
|
||||
if baseline is None or threshold is None:
|
||||
return True
|
||||
if case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
return abs(float(result) - float(baseline)) <= float(threshold)
|
||||
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
|
||||
try:
|
||||
throughput_val = float(throughput_str.replace("token/s", "").strip())
|
||||
return throughput_val >= float(threshold) * float(baseline)
|
||||
except (ValueError, AttributeError):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def _build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
|
||||
"""Build a single task dict in the required format."""
|
||||
dataset_path = case_config.get("dataset_path", "")
|
||||
dataset_conf = case_config.get("dataset_conf", "")
|
||||
if dataset_path:
|
||||
task_name = dataset_path.split("/", 1)[-1]
|
||||
elif dataset_conf:
|
||||
task_name = dataset_conf.split("/")[0]
|
||||
else:
|
||||
task_name = case_key
|
||||
case_type = case_config.get("case_type", "unknown")
|
||||
metrics: dict[str, float] = {}
|
||||
|
||||
if result == "":
|
||||
# benchmark run failed — no metrics available
|
||||
pass
|
||||
elif case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
metrics["accuracy"] = round(float(result), 4)
|
||||
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
for metric_name, metric_data in result_json.items():
|
||||
if not isinstance(metric_data, dict):
|
||||
continue
|
||||
total_str = metric_data.get("total", "")
|
||||
try:
|
||||
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
|
||||
metrics[_PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
|
||||
test_input = {k: case_config[k] for k in test_input_keys if k in case_config}
|
||||
|
||||
target: dict[str, Any] = {}
|
||||
if case_config.get("baseline") is not None:
|
||||
target["baseline"] = case_config["baseline"]
|
||||
if case_config.get("threshold") is not None:
|
||||
target["threshold"] = case_config["threshold"]
|
||||
|
||||
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
|
||||
if target:
|
||||
entry["target"] = target
|
||||
entry["pass_fail"] = "pass" if _task_passed(case_config, result) else "fail"
|
||||
return entry
|
||||
|
||||
|
||||
def _all_passed(case_configs: list[dict[str, Any]], results: list[Any]) -> bool:
|
||||
"""Return True only when every benchmark result meets its baseline/threshold."""
|
||||
return all(_task_passed(cfg, res) for cfg, res in zip(case_configs, results))
|
||||
|
||||
|
||||
def _save_benchmark_results_json(config: SingleNodeConfig, benchmark_keys: list[str], results: list[Any]) -> None:
|
||||
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
|
||||
runner = os.environ.get("VLLM_CI_RUNNER", "")
|
||||
case_configs = [config.benchmarks[k] for k in benchmark_keys]
|
||||
|
||||
tasks = [
|
||||
_build_task_entry(key, case_cfg, result) for key, case_cfg, result in zip(benchmark_keys, case_configs, results)
|
||||
]
|
||||
|
||||
passed = _all_passed(case_configs, results)
|
||||
|
||||
output: dict[str, Any] = {
|
||||
"model_name": config.model,
|
||||
"hardware": _extract_hardware(runner),
|
||||
"dtype": _extract_dtype(config),
|
||||
"feature": _extract_features(config.server_cmd, config.envs),
|
||||
"vllm_version": vllm.__version__,
|
||||
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_VERSION", ""),
|
||||
"tasks": tasks,
|
||||
"serve_cmd": _build_serve_cmd(config),
|
||||
"environment": _filter_environment(config.envs),
|
||||
"pass_fail": "pass" if passed else "fail",
|
||||
}
|
||||
|
||||
os.makedirs("benchmark_results", exist_ok=True)
|
||||
job_name = os.environ.get("BENCHMARK_JOB_NAME") or config.name
|
||||
safe_name = job_name.replace("/", "_").replace(" ", "_")
|
||||
output_path = os.path.join("benchmark_results", f"{safe_name}.json")
|
||||
with open(output_path, "w", encoding="utf-8") as f:
|
||||
json.dump(output, f, indent=2, ensure_ascii=False)
|
||||
logger.info("Benchmark results saved to %s", output_path)
|
||||
print(f"Benchmark results saved to {output_path}")
|
||||
|
||||
|
||||
def _run_benchmarks(config: SingleNodeConfig, port: int) -> None:
|
||||
"""Run Aisbench benchmarks and process benchmark-dependent custom assertions."""
|
||||
benchmark_keys = [k for k, v in config.benchmarks.items() if v]
|
||||
aisbench_cases = [config.benchmarks[k] for k in benchmark_keys]
|
||||
if not aisbench_cases:
|
||||
return
|
||||
|
||||
result = run_aisbench_cases(
|
||||
model=config.model,
|
||||
port=port,
|
||||
aisbench_cases=aisbench_cases,
|
||||
)
|
||||
|
||||
_save_benchmark_results_json(config, benchmark_keys, result)
|
||||
|
||||
if "benchmark_comparisons" in config.test_content:
|
||||
run_benchmark_comparisons(config, result)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
@pytest.mark.parametrize("config", configs, ids=[config.name for config in configs])
|
||||
async def test_single_node(config: SingleNodeConfig) -> None:
|
||||
# TODO: remove this part after the transformers version upgraded
|
||||
if config.special_dependencies:
|
||||
for k, v in config.special_dependencies.items():
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"pip",
|
||||
"install",
|
||||
f"{k}=={v}",
|
||||
]
|
||||
subprocess.call(command)
|
||||
if config.service_mode == "epd":
|
||||
with (
|
||||
RemoteEPDServer(vllm_serve_args=config.epd_server_cmds, env_dict=config.envs) as _,
|
||||
DisaggEpdProxy(proxy_args=config.epd_proxy_args, env_dict=config.envs) as proxy,
|
||||
):
|
||||
await _dispatch_tests(config, proxy)
|
||||
_run_benchmarks(config, proxy.port)
|
||||
return
|
||||
|
||||
# Standard OpenAI service mode
|
||||
with RemoteOpenAIServer(
|
||||
model=config.model,
|
||||
vllm_serve_args=config.server_cmd,
|
||||
server_port=config.server_port,
|
||||
env_dict=config.envs,
|
||||
auto_port=False,
|
||||
) as server:
|
||||
await _dispatch_tests(config, server)
|
||||
_run_benchmarks(config, config.server_port)
|
||||
Reference in New Issue
Block a user