init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,84 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "10"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "36864"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "16"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"num_speculative_tokens": 1, "method": "mtp"}'
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
_benchmarks_acc: &benchmarks_acc
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10
_benchmarks_perf: &benchmarks_perf
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 400
max_out_len: 1500
batch_size: 1000
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-R1-0528-W8A8-single"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enforce-eager"
benchmarks:
- name: "DeepSeek-R1-0528-W8A8-aclgraph"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks_acc
<<: *benchmarks_perf

View File

@@ -0,0 +1,81 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V3.2-W8A8-DCP-replicated-indexer"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
VLLM_ASCEND_ENABLE_NZ: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "20"
HCCL_BUFFSIZE: "768"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_SERVER_DEV_MODE: "1"
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
ASCEND_LAUNCH_BLOCKING: "0"
ASCEND_ENABLE_USE_FABRIC_MEM: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_MLAPO: "0"
PYTHONHASHSEED: "0"
ASCEND_A3_ENABLE: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "10000"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
TASK_QUEUE_ENABLE: "1"
CPU_AFFINITY_CONF: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "1024"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "1"
- "--pipeline-parallel-size"
- "1"
- "--tensor-parallel-size"
- "16"
- "--prefill-context-parallel-size"
- "1"
- "--decode-context-parallel-size"
- "16"
- "--cp-kv-cache-interleave-size"
- "1"
- "--block-size"
- "128"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.95"
- "--api-server-count"
- "1"
- "--safetensors-load-strategy"
- "prefetch"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 16, 64, 128]}'
- "--additional-config"
- '{"enable_dsa_cp": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}, "multistream_overlap_shared_expert": true, "enable_mc2_hierarchy_comm": false, "enable_sparse_sfa_c8": true, "enable_sparse_li_c8": true, "enable_cpu_binding": true, "recompute_scheduler_enable": false}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
test_content: []
benchmarks:
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 8192
batch_size: 32
baseline: 95
threshold: 5

View File

@@ -0,0 +1,80 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V3.2-W8A8-TP8-DP2"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "256"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "8"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--async-scheduling"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.95"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,2,4,8,16,24,32,40,48], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
- "--reasoning-parser"
- "deepseek_v3"
- "--tokenizer_mode"
- "deepseek_v32"
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 86.67
temperature: 1.0
top_p: 0.95
thinking: true
threshold: 10
perf_2:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1500
batch_size: 4
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,80 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V4-Flash-W8A8-A3"
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
special_dependencies:
transformers: "5.9.0"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
ASCEND_LAUNCH_BLOCKING: "0"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
server_cmd:
- "--enable-prefix-caching"
- "--max-model-len"
- "1048576"
- "--max-num-batched-tokens"
- "10240"
- "--gpu-memory-utilization"
- "0.9"
- "--max-num-seqs"
- "64"
- "--data-parallel-size"
- "4"
- "--tensor-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--tokenizer-mode"
- "deepseek_v4"
- "--tool-call-parser"
- "deepseek_v4"
- "--enable-auto-tool-choice"
- "--reasoning-parser"
- "deepseek_v4"
- "--safetensors-load-strategy"
- "prefetch"
- "--quantization"
- "ascend"
- "--api-server-count"
- "1"
- "--speculative-config"
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- "--port"
- "$SERVER_PORT"
- "--block-size"
- "128"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--async-scheduling"
- "--additional-config"
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":"true","enable_shared_expert_dp":true,"multistream_overlap_shared_expert":true}'
benchmarks:
acc-gpqa:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 86.36
threshold: 5
thinking: true
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 64
max_out_len: 1024
batch_size: 16
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,66 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "16"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"mtp"}'
- "--additional-config"
- '{"enable_shared_expert_dp": true, "ascend_fusion_config": {"fusion_ops_gmmswigluquant": false}}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 8
baseline: 95
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "GLM-4.7-TP8-DP2-decodegraph"
model: "Eco-Tech/GLM-4.7-W8A8-floatmtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_capture_sizes": [1,2,4,8,16,32,64,128,256,512], "cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,82 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "GLM-5.1-W8A8-PrefillMC2"
model: "Eco-Tech/GLM-5.1-w8a8" #need update
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "1800"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "10240"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--async-scheduling"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.94"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp", "enforce_eager": true}'
- "--additional_config"
- '{"enable_prefill_mc2": true}'
- "--reasoning-parser"
- "glm45"
- "--tool-call-parser"
- "glm47"
benchmarks:
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 8192
batch_size: 32
baseline: 96.88
temperature: 1.0
top_p: 0.95
thinking: true
threshold: 5
perf_2:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 64
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,58 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
VLLM_USE_MODELSCOPE: "true"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--tensor-parallel-size"
- "16"
- "--enable-expert-parallel"
- "--enable-ep-weight-filter"
- "--tool-call-parser"
- "hy_v3"
- "--reasoning-parser"
- "hy_v3"
- "--enable-auto-tool-choice"
- "--max-model-len"
- "32768"
- "--max-num-seqs"
- "8"
- "--port"
- "$SERVER_PORT"
- "--speculative-config"
- '{"method": "mtp", "num_speculative_tokens": 1}'
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks: &benchmarks
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_4_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 8
baseline: 93.07
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Hy3-preview-TP16-EP-MTP"
model: "Tencent-Hunyuan/Hy3-preview"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,52 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2-Thinking-TP16-Case"
model: "moonshotai/Kimi-K2-Thinking"
envs:
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
OMP_PROC_BIND: "false"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "12"
- "--gpu-memory-utilization"
- "0.9"
- "--trust-remote-code"
- "--enable-expert-parallel"
- "--no-enable-prefix-caching"
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 32
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 256
batch_size: 64
trust_remote_code: true
request_rate: 11.2
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,91 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_NZ: "1"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--enable-prefix-caching"
- "--enable-chunked-prefill"
- "--allowed-local-media-path"
- "/"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "16"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "42"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
_benchmarks: &benchmarks
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
temperature: 0.0
top_p: 1
top_k: -1
repetition_penalty: 1.0
batch_size: 32
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.5-W4A8-Case"
model: "Eco-Tech/Kimi-K2.5-W4A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.6-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "800"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "24"
- "--max-model-len"
- "6144"
- "--max-num-batched-tokens"
- "4096"
- "--gpu-memory-utilization"
- "0.85"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1433.4454
threshold: 0.97

View File

@@ -0,0 +1,91 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "100"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "3600000"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--seed"
- "1024"
- "--no-enable-prefix-caching"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-seqs"
- "14"
- "--trust-remote-code"
_benchmarks_gsm8k: &benchmarks_gsm8k
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10
_benchmarks_aime: &benchmarks_aime
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2024
request_conf: vllm_api_general_chat
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 86.67
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp2"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "4096"
- "--speculative-config"
- '{"num_speculative_tokens": 2, "method": "mtp"}'
- "--gpu-memory-utilization"
- "0.92"
benchmarks:
<<: *benchmarks_gsm8k
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp3"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
HCCL_OP_EXPANSION_MODE: "AIV"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "2048"
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "mtp"}'
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes": [56], "cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks_aime

View File

@@ -0,0 +1,88 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.5-w8a8"
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
envs:
HCCL_BUFFSIZE: "512"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
HCCL_INTRA_PCIE_ENABLE: "1"
HCCL_INTRA_ROCE_ENABLE: "0"
OMP_PROC_BIND: "false"
VLLM_TORCH_PROFILER_WITH_STACK: "0"
VLLM_TORCH_PROFILER_DIR: "./profile"
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--port"
- "$SERVER_PORT"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--model-loader-extra-config"
- '{"enable_multithread_load":true,"num_threads":16}'
- "--speculative_config"
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
- "--enable-expert-parallel"
- "--enable-chunked-prefill"
- "--enable-prefix-caching"
- "--max-num-seqs"
- "100"
- "--max-model-len"
- "196608"
- "--seed"
- "1024"
- "--max-num-batched-tokens"
- "6144"
- "--enable-auto-tool-choice"
- "--tool-call-parser"
- "minimax_m2"
- "--reasoning-parser"
- "minimax_m2_append_think"
- "--enable-force-include-usage"
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,16,40,80,160,256,400]}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gpqa_gen_0_shot_str
max_out_len: 131072
batch_size: 64
baseline: 83
threshold: 5
bos_token_id: 200019
do_sample: true
eos_token_id: 200020
temperature: 1.0
top_p: 0.95
top_k: 40
transformers_version: 4.46.1
ignore_eos: false
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 360
max_out_len: 1500
batch_size: 120
request_rate: 0
baseline: 2042
threshold: 0.97

View File

@@ -0,0 +1,73 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.5-w8a8"
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
envs:
HCCL_BUFFSIZE: "512"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM-ASCEND_ENABLE_NZ: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.85"
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
- "--enable-expert-parallel"
- "--max-num-seqs"
- "128"
- "--max-num-batched-tokens"
- "16384"
- "--max-model-len"
- "196608"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 131072
batch_size: 32
baseline: 90
threshold: 10
bos_token_id: 200019
do_sample: true
eos_token_id: 200020
temperature: 1.0
top_p: 0.95
top_k: 40
transformers_version: 4.46.1
ignore_eos: false
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 1500
batch_size: 128
request_rate: 0
baseline: 1116
threshold: 0.97

View File

@@ -0,0 +1,85 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1200"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
TASK_QUEUE_ENABLE: "1"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--host"
- "0.0.0.0"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--async-scheduling"
- "--max-num-seqs"
- "128"
- "--safetensors-load-strategy"
- 'prefetch'
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--enable-auto-tool-choice"
- "--tool-call-parser"
- "minimax_m2"
- "--speculative-config"
- '{"method":"eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"enable_cpu_binding": true, "enable_npugraph_ex": true, "enable_static_kernel": true,"enable_fused_mc2":true,"weight_nz_mode":true,"enable_flashcomm1":true}'
_benchmarks_3500: &benchmarks_3500
perf_50:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 760
max_out_len: 1500
batch_size: 190
request_rate: 0
baseline: 4573.02
threshold: 0.97
perf_20:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1500
batch_size: 48
request_rate: 0
baseline: 2229.147
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.7-3500"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "70000"
- "--gpu-memory-utilization"
- "0.8"
- "--no-enable-prefix-caching"
benchmarks:
<<: *benchmarks_3500

View File

@@ -0,0 +1,78 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "prefix-cache-deepseek-r1-0528-w8a8"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
OMP_NUM_THREADS: "10"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
server_cmd:
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "5200"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "16"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
- "--speculative-config"
- '{"num_speculative_tokens": 1, "method": "mtp"}'
test_content:
- "benchmark_comparisons"
benchmark_comparisons_args:
- metric: "TTFT"
baseline: "prefix0"
target: "prefix75"
ratio: 0.5
operator: "<"
benchmarks:
warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1024-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 1000
baseline: 0
threshold: 0.97
prefix0:
case_type: performance
dataset_path: vllm-ascend/prefix0-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 18
baseline: 1
threshold: 0.97
prefix75:
case_type: performance
dataset_path: vllm-ascend/prefix75-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 18
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "prefix-cache-qwen3-32b-w8a8"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--reasoning-parser"
- "qwen3"
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "256"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
test_content:
- "benchmark_comparisons"
benchmark_comparisons_args:
- metric: "TTFT"
baseline: "prefix0"
target: "prefix75"
ratio: 0.4
operator: "<"
benchmarks:
warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1024-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 1000
baseline: 0
threshold: 0.97
prefix0:
case_type: performance
dataset_path: vllm-ascend/prefix0-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 48
baseline: 1
threshold: 0.97
prefix75:
case_type: performance
dataset_path: vllm-ascend/prefix75-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 48
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,84 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "10"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "4"
- "--tensor-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "12"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
top_k: 20
baseline: 95
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-235B-A22B-W8A8-full_graph"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks
- name: "Qwen3-235B-A22B-W8A8-piecewise"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "PIECEWISE"}'
benchmarks:
<<: *benchmarks
- name: "Qwen3-235B-A22B-W8A8-EPLB"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
envs:
<<: *envs
DYNAMIC_EPLB: "true"
server_cmd: *server_cmd
server_cmd_extra:
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": "true", "expert_heat_collection_interval": 600, "algorithm_execution_interval": 50, "num_redundant_experts": 16, "eplb_policy_type": 2}}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,41 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-30B-A3B-W4A8-llm-compressor"
model: "vllm-ascend/Qwen3-30B-A3B-Instruct-2507-quantized.w4a8"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_BUFFSIZE: "1024"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.8"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10

View File

@@ -0,0 +1,43 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-30B-A3B-W8A8-TP1"
model: "vllm-ascend/Qwen3-30B-A3B-W8A8"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
HCCL_BUFFSIZE: "1024"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "5600"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "100"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 180
max_out_len: 1500
batch_size: 45
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,40 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-30B-QuaRot"
model: "vllm-ascend/Qwen3-30B-A3B-W8A8-QuaRot"
envs:
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
SERVER_PORT: "DEFAULT_PORT"
HCCL_BUFFSIZE: "768"
server_cmd:
- "--enforce-eager"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--trust-remote-code"
- "--distributed-executor-backend"
- "mp"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"method": "eagle3", "model": "AngelSlim/Qwen3-a3B_eagle3", "num_speculative_tokens": 3}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1500
batch_size: 20
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,78 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FLASHCOMM: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "40960"
- "--block-size"
- "128"
- "--trust-remote-code"
- "--reasoning-parser"
- "qwen3"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"weight_prefetch_config":{"enabled":true}}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2024
request_conf: vllm_api_general_chat
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 83.33
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 288
max_out_len: 1500
batch_size: 72
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-32B-W8A8-aclgraph-a2"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1,12,16,20,24,32,48,60,64,68,72,76,80]}'
benchmarks:
<<: *benchmarks
- name: "Qwen3-32B-W8A8-single-a2"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enforce-eager"
benchmarks:

View File

@@ -0,0 +1,78 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FLASHCOMM: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "4"
- "--max-num-seqs"
- "80"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "40960"
- "--block-size"
- "128"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"weight_prefetch_config":{"enabled":true}}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_noncot_chat_prompt
max_out_len: 10240
batch_size: 32
baseline: 96
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 304
max_out_len: 1500
batch_size: 76
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-32B-W8A8-aclgraph-a3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[1,12,16,20,24,32,48,60,64,68,72,76,80]}'
benchmarks:
<<: *benchmarks
- name: "Qwen3-32B-W8A8-single-a3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enforce-eager"
benchmarks:

View File

@@ -0,0 +1,38 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-32B-QuaRot"
model: "vllm-ascend/Qwen3-32B-W8A8-QuaRot"
envs:
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--enforce-eager"
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--trust-remote-code"
- "--distributed-executor-backend"
- "mp"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"method": "eagle3", "model": "RedHatAI/Qwen3-32B-speculator.eagle3", "num_speculative_tokens": 3}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1500
batch_size: 20
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,70 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "1"
OMP_PROC_BIND: "false"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1536"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_ENABLE_NZ: "2"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--mm-processor-cache-gb"
- "0"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "32768"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.92"
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/textvqa-lite
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
max_out_len: 2048
batch_size: 128
baseline: 83
temperature: 0
top_k: -1
top_p: 1
repetition_penalty: 1
threshold: 5
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-VL-235B-A22B-Instruct-W8A8"
model: "Eco-Tech/Qwen3-VL-235B-A22B-Instruct-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation_config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,2,4,8,16,24,32]}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,67 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "1"
OMP_PROC_BIND: "false"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--mm-processor-cache-gb"
- "0"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "20000"
- "--max-num-batched-tokens"
- "8192"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/textvqa-lite
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
max_out_len: 2048
batch_size: 128
baseline: 80
temperature: 0
top_k: -1
top_p: 1
repetition_penalty: 1
threshold: 5
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3-VL-32B-Instruct-W8A8"
model: "Eco-Tech/Qwen3-VL-32B-Instruct-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation_config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,12,16,20,24,32,48,64,68,72,76,80,128]}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,79 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
_envs: &envs
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "131072"
- "--max-num-batched-tokens"
- "16384"
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding": true, "enable_shared_expert_dp": true}'
test_cases:
- name: "Qwen3.5-122B-A10B-W8A8-A3"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
thinking: true
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 1.5
repetition_penalty: 1.0
ignore_eos: false
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 320
max_out_len: 1500
batch_size: 80
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,57 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-27B-w8a8"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_NZ: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "262144"
- "--max-num-batched-tokens"
- "8192"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.95"
- "--additional-config"
- '{"enable_cpu_binding":true, "enable_weight_nz_layout":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3,"enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144]}'
- "--mm-processor-cache-gb"
- "0"
- "--mm_processor_cache_type"
- "shm"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 604
threshold: 0.97

View File

@@ -0,0 +1,73 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-27B-w8a8"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "2"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "196608"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--async-scheduling"
- "--allowed-local-media-path"
- "/"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3,"enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--mm-processor-cache-gb"
- "0"
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
ignore_eos: false
thinking: true
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 1.5
repetition_penalty: 1.0
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 140
max_out_len: 1500
batch_size: 35
request_rate: 0
baseline: 610.22
threshold: 0.97

View File

@@ -0,0 +1,76 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w8a8-mtp"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "6000"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--safetensors-load-strategy"
- "lazy"
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 90
threshold: 10
temperature: 0.6
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 0.0
repetition_penalty: 1.0
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 56.357
threshold: 0.97

View File

@@ -0,0 +1,74 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w4a8-mtp"
model: "Eco-Tech/Qwen3.5-397B-A17B-w4a8-mtp"
envs:
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "1"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--no-enable-prefix-caching"
- "--additional-config"
- '{"enable_cpu_binding":true,"multistream_overlap_shared_expert":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,108,112,128,160,172,196,200,212,232,256,260,288,320,360,400], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
benchmarks:
acc_GPQA:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 88.38
threshold: 5
# acc_aime2025:
# case_type: accuracy
# dataset_path: vllm-ascend/aime2025
# request_conf: vllm_api_general_chat
# dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
# max_out_len: 72348
# batch_size: 32
# baseline: 93.33
# threshold: 5
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 256
max_out_len: 1500
batch_size: 64
request_rate: 0
baseline: 1040
threshold: 0.97

View File

@@ -0,0 +1,62 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w8a8-mtp-longseq"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "6000"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "1"
- "--prefill-context-parallel-size"
- "2"
- "--decode-context-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "32"
- "--quantization"
- "ascend"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,16,32,64], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--safetensors-load-strategy"
- "lazy"
benchmarks:
acc_GPQA:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
num_prompts: 50
max_out_len: 32768
batch_size: 32
baseline: 84
threshold: 8

View File

@@ -0,0 +1,312 @@
# vLLM-Ascend Single-Node E2E Test Developer Guide
This document is intended to help developers understand the architecture of the single-node E2E (End-to-End) testing framework in `vllm-ascend`, how to run test scripts, and how to add custom testing functionality by writing YAML configuration files and extending the code.
## 1. Test Architecture Overview
To achieve high readability, extensibility, and decoupling of configuration from code, the single-node E2E test adopts a **"YAML-driven + Dispatcher"** architectural structure.
It consists of the following core components:
* **Configuration Parser (`single_node_config.py`)**: Responsible for reading `models/configs/*.yaml` files and parsing them into a strongly-typed `@dataclass` (`SingleNodeConfig`) via `SingleNodeConfigLoader`, while handling regex replacement for environment variables.
* **Service Manager Framework (`test_single_node.py` and `conftest.py`)**: Based on the `service_mode` (`openai` or `epd`), it utilizes context managers to safely start/stop server processes.
* **Test Function Dispatcher (`TEST_HANDLERS` Registry)**: Specific test logic is encapsulated into independent functions and registered in the global `TEST_HANDLERS` dictionary.
* **Performance Benchmarking (`_run_benchmarks`)**: Calls `aisbench` for performance and TTFT testing based on the `benchmarks` parameters in the YAML.
### 1.1 Key Files and Responsibilities
* `tests/e2e/nightly/single_node/models/scripts/single_node_config.py`
* Defines `SingleNodeConfig` and `SingleNodeConfigLoader`
* Loads YAML from `tests/e2e/nightly/single_node/models/configs/<CONFIG_YAML_PATH>`
* Auto-assigns ports when `envs` contains `DEFAULT_PORT` / missing values
* Expands `$VAR` / `${VAR}` placeholders inside commands via `_expand_values`
* `tests/e2e/nightly/single_node/models/scripts/test_single_node.py`
* Declares `configs = SingleNodeConfigLoader.from_yaml_cases()` (loaded at import time)
* `pytest.mark.parametrize("config", configs, ids=[config.name for config in configs])` runs one test per YAML case
* Controls server lifecycle via context managers
* Dispatches `test_content` to functions registered in `TEST_HANDLERS`
* Runs `aisbench` and optional benchmark assertions
### 1.2 End-to-End Flow (High Level)
```txt
pytest starts
|
v
import tests/e2e/nightly/single_node/models/scripts/test_single_node.py
|
v
configs = SingleNodeConfigLoader.from_yaml_cases()
|
v
pytest parametrize("config", configs) # one config == one test case
|
v
test_single_node(config)
|
+-----------------------------------------------+
| Start service (depends on service_mode) |
| |
| openai: start one vLLM OpenAI-compatible |
| service process |
| epd: start (encode service + decode/PD |
| service) + start proxy process |
+-----------------------------------------------+
|
v
Run test phases (test_content)
|
v
Optional benchmarks (if benchmarks is configured)
|
v
Shutdown all started processes
Notes:
- One YAML file may contain multiple test_cases; pytest will run them one by one.
- The framework is "YAML-driven": changes are typically done by editing YAML rather than editing Python code.
```
### 1.3 Function Call Relationships (Dispatcher)
`test_content` is a list of “phases”. Each phase maps to one handler function.
```txt
For each test_case:
test_content (list of phases)
|
v
[Dispatcher]
|
+--> phase "completion" -> send completion request(s)
|
+--> phase "chat_completion" -> send chat completion request(s)
|
+--> phase "image" -> send multimodal image request(s)
|
\--> (extendable) add your own phase by registering a new handler
After phases:
if benchmarks is configured -> run aisbench
Notes:
- The dispatcher only controls "what to run"; service lifecycle is controlled by the service manager.
- Phases are intentionally small & composable so you can reuse them across YAML cases.
```
## 2. Running and Debugging Steps
### 2.1 Dependencies
Ensure you are in an NPU environment and have installed `pytest`, `pyyaml`, `openai`, and `aisbench`.
### 2.2 Local Execution
The framework uses the `CONFIG_YAML_PATH` environment variable to specify the configuration file.
```bash
# Switch to the project root directory
cd /vllm-workspace/vllm-ascend
# Run a specific yaml test
export CONFIG_YAML_PATH="Qwen3-32B.yaml"
pytest -sv tests/e2e/nightly/single_node/models/scripts/test_single_node.py
```
### 2.3 Tips for Debugging
* Only run a subset of cases: `pytest -sv ... -k <keyword>` (matches case names in the report output)
* Stop on first failure: `pytest -sv ... -x`
* Keep server logs visible: use `-s` (already included in `-sv`) and increase log verbosity via standard Python logging configuration if needed.
## 3. How to Write YAML Configuration Files
### 3.1 File Location and Selection Rules
* YAML files live under: `tests/e2e/nightly/single_node/models/configs/`
* Selected by env var: `CONFIG_YAML_PATH=<YourConfig>.yaml`
* If not set, the loader uses `SingleNodeConfigLoader.DEFAULT_CONFIG_NAME`
### 3.2 Field Descriptions
| Field Name | Type | Required | Default Value | Description |
| :--------------- | :--------- | :------- | :-------------- | :------------------------------------------------------------------ |
| `test_cases` | list | **Yes** | - | List of test case objects |
| `name` | string | **Yes** | - | Human-readable case ID shown in pytest output and logs |
| `model` | string | **Yes** | - | Model name or local path |
| `service_mode` | string | No | `openai` | Service mode: `openai` or `epd` (disaggregated) |
| `envs` | map | **Yes** | `{}` | Environment variables for the server process |
| `server_cmd` | list | Cond. | `[]` | vLLM startup arguments (Required for non-EPD) |
| `server_cmd_extra` | list | No | `[]` | Extra vLLM startup arguments appended after `server_cmd` |
| `prompts` | list | No | built-in default | Prompts for completion/chat tests |
| `api_keyword_args` | map | No | built-in default | OpenAI API keyword args (e.g., `max_tokens`, sampling params) |
| `test_content` | list | No | `["completion"]` | Test phases: `completion`, `chat_completion`, `image` etc. |
| `benchmarks` | map | No | `{}` | Configuration for `aisbench` performance verification |
| `epd_server_cmds`| list[list] | Cond. | `[]` | (EPD Only) Command arrays for starting dual Encode/Decode processes |
| `epd_proxy_args` | list | Cond. | `[]` | (EPD Only) Startup arguments for the EPD routing gateway |
**Notes / Behaviors**
* `name` is mandatory and must be a non-empty string.
* It is used directly as pytest case id (e.g., `test_single_node[DeepSeek-R1-0528-W8A8-single]`).
* It is also printed in `[single-node][START]` marker for log navigation.
* `envs` (ports): the config object recognizes these keys: `SERVER_PORT`, `ENCODE_PORT`, `PD_PORT`, `PROXY_PORT`.
* If a port key is missing or set to `DEFAULT_PORT`, it will be automatically filled with an available open port.
* `$SERVER_PORT` / `${SERVER_PORT}` placeholders in commands will be expanded using `envs`.
* `server_cmd` vs `server_cmd_extra`:
* YAML can define `server_cmd_extra` to append additional args after `server_cmd`.
* The loader merges them into a single `server_cmd` list.
* Extra fields:
* Any non-standard fields in a case are stored in `config.extra_config`.
* This is how extension configs are passed through without changing the dataclass.
### 3.3 YAML Examples
#### Single-Case (similar to DeepSeek-R1-W8A8-HBM)
```yaml
test_cases:
- name: "<your-case-name>"
model: "<model-repo-or-local-path>"
# Optional: The default values are as follows
prompts:
- "San Francisco is a"
api_keyword_args:
max_tokens: 10
envs:
SERVER_PORT: "DEFAULT_PORT"
# Add only what you need.
server_cmd:
- "--port"
- "$SERVER_PORT"
# plus your vLLM serve args...
# Optional: omit -> defaults to ["completion"]
test_content:
- "chat_completion"
# Optional: leave empty if you don't run aisbench
benchmarks:
```
#### Multi-Case + Shared Anchors
```yaml
_envs: &envs
SERVER_PORT: "DEFAULT_PORT"
# shared envs...
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
# shared vLLM serve args...
_benchmarks: &benchmarks
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 400
max_out_len: 1500
batch_size: 1000
baseline: 1
threshold: 0.97
test_cases:
- name: "case-a"
model: "<model>"
envs:
<<: *envs
DYNAMIC_EPLB: "true"
# private envs...
server_cmd: *server_cmd
server_cmd_extra:
- "--enforce-eager"
benchmarks:
- name: "case-b"
model: "<model>"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks
```
#### EPD / Disaggregated Case
```yaml
test_cases:
- name: "<your-epd-case>"
model: "<model>"
service_mode: "epd"
envs:
ENCODE_PORT: "DEFAULT_PORT"
PD_PORT: "DEFAULT_PORT"
PROXY_PORT: "DEFAULT_PORT"
epd_server_cmds:
- ["--port", "$ENCODE_PORT", "--model", "<encode-model>"]
- ["--port", "$PD_PORT", "--model", "<decode-model>"]
epd_proxy_args:
- "--host"
- "127.0.0.1"
- "--port"
- "$PROXY_PORT"
- "--encode-servers-urls"
- "http://localhost:$ENCODE_PORT"
- "--decode-servers-urls"
- "http://localhost:$PD_PORT"
- "--prefill-servers-urls"
- "disable"
test_content:
- "chat_completion"
```
## 4. How to Add Custom Tests (Extension)
### Step 1: Write your test logic in `test_single_node.py`
```python
async def run_video_test(config: SingleNodeConfig, server: 'RemoteOpenAIServer | DisaggEpdProxy') -> None:
client = server.get_async_client()
# Your custom logic here...
```
### Step 2: Register your function in `TEST_HANDLERS`
```python
TEST_HANDLERS = {
"completion": run_completion_test,
"video": run_video_test, # Registered!
}
```
### Step 3: Enable in YAML
```yaml
test_content:
- "completion"
- "video"
```
## 5. Checklist (Before Submitting a New YAML)
* `test_cases` exists and is a list
* Each case contains required fields for its `service_mode`
* Common required: `name`, `model`, `envs`
* `openai`: `server_cmd`
* `epd`: `epd_server_cmds`, `epd_proxy_args`
* Port envs are set to `DEFAULT_PORT` (or to explicit free ports)
* If using `benchmarks`, ensure each benchmark case includes required aisbench fields (e.g., `case_type`, `dataset_path`, `request_conf`, `dataset_conf`, `max_out_len`, `batch_size`)

View File

@@ -0,0 +1,16 @@
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#

View File

@@ -0,0 +1,188 @@
import logging
import os
from dataclasses import dataclass, field
from typing import Any
import regex as re
import yaml
from vllm.utils.network_utils import get_open_port
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or "tests/e2e/nightly/single_node/models/configs"
logger = logging.getLogger(__name__)
# Default prompts and API args fallback
PROMPTS = [
"San Francisco is a",
]
API_KEYWORD_ARGS = {
"max_tokens": 10,
}
@dataclass
class SingleNodeConfig:
name: str
model: str
envs: dict[str, Any] = field(default_factory=dict)
special_dependencies: dict[str, Any] = field(default_factory=dict)
prompts: list[str] = field(default_factory=lambda: PROMPTS)
api_keyword_args: dict[str, Any] = field(default_factory=lambda: API_KEYWORD_ARGS)
benchmarks: dict[str, Any] = field(default_factory=dict)
server_cmd: list[str] = field(default_factory=list)
test_content: list[str] = field(default_factory=lambda: ["completion"])
service_mode: str = "openai"
epd_server_cmds: list[list[str]] = field(default_factory=list)
epd_proxy_args: list[str] = field(default_factory=list)
extra_config: dict[str, Any] = field(default_factory=dict)
def __post_init__(self) -> None:
port_keys = ["SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"]
for env_key in port_keys:
if self.envs.get(env_key) in ["DEFAULT_PORT", None]:
self.envs[env_key] = str(get_open_port())
if self.prompts is None:
self.prompts = PROMPTS
if self.api_keyword_args is None:
self.api_keyword_args = API_KEYWORD_ARGS
if self.benchmarks is None:
self.benchmarks = {}
if self.special_dependencies is None:
self.special_dependencies = {}
if self.test_content is None:
self.test_content = []
self.server_cmd = self._expand_values(self.server_cmd or [], self.envs)
self.epd_server_cmds = [self._expand_values(cmd, self.envs) for cmd in self.epd_server_cmds]
self.epd_proxy_args = self._expand_values(self.epd_proxy_args or [], self.envs)
for key, value in self.extra_config.items():
setattr(self, key, value)
@staticmethod
def _expand_values(values: list[str], envs: dict[str, Any]) -> list[str]:
"""Interpolate $VAR/${VAR} placeholders with provided env values."""
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
def repl(m: re.Match[str]) -> str:
key = m.group(1) or m.group(2)
return str(envs.get(key, m.group(0)))
return [pattern.sub(repl, str(arg)) for arg in values]
def _get_required_port(self, key: str) -> int:
value = self.envs.get(key)
if value is None:
raise ValueError(f"Missing required port env: {key}")
return int(value)
@property
def server_port(self) -> int:
return self._get_required_port("SERVER_PORT")
@property
def encode_port(self) -> int:
return self._get_required_port("ENCODE_PORT")
@property
def pd_port(self) -> int:
return self._get_required_port("PD_PORT")
@property
def proxy_port(self) -> int:
return self._get_required_port("PROXY_PORT")
class SingleNodeConfigLoader:
"""Load SingleNodeConfig from yaml file."""
DEFAULT_CONFIG_NAME = "Kimi-K2-Thinking.yaml"
STANDARD_CASE_FIELDS = {
"name",
"model",
"envs",
"special_dependencies",
"prompts",
"api_keyword_args",
"benchmarks",
"service_mode",
"server_cmd",
"server_cmd_extra",
"test_content",
"epd_server_cmds",
"epd_proxy_args",
}
@classmethod
def from_yaml_cases(cls, yaml_path: str | None = None) -> list[SingleNodeConfig]:
config = cls._load_yaml(yaml_path)
if "test_cases" not in config:
raise KeyError("test_cases field is required in config yaml")
cases = config.get("test_cases")
if not isinstance(cases, list):
raise TypeError("test_cases must be a list")
cls._validate_para(cases)
return cls._parse_test_cases(cases)
@classmethod
def _load_yaml(cls, yaml_path: str | None) -> dict[str, Any]:
if not yaml_path:
yaml_path = os.getenv("CONFIG_YAML_PATH", cls.DEFAULT_CONFIG_NAME)
full_path = os.path.join(CONFIG_BASE_PATH, yaml_path)
logger.info("Loading config yaml: %s", full_path)
with open(full_path) as f:
return yaml.safe_load(f)
@staticmethod
def _validate_para(cases: list[dict[str, Any]]) -> None:
if not cases:
raise ValueError("test_cases is empty")
for case in cases:
mode = case.get("service_mode", "openai")
required = ["name", "model", "envs"]
if mode == "epd":
required.extend(["epd_server_cmds", "epd_proxy_args"])
else:
required.append("server_cmd")
missing = [k for k in required if k not in case]
if missing:
raise KeyError(f"Missing required config fields: {missing}")
if not isinstance(case["name"], str) or not case["name"].strip():
raise ValueError("test case field 'name' must be a non-empty string")
@classmethod
def _parse_test_cases(cls, cases: list[dict[str, Any]]) -> list[SingleNodeConfig]:
result: list[SingleNodeConfig] = []
for case in cases:
server_cmd = case.get("server_cmd", [])
server_cmd_extra = case.get("server_cmd_extra", [])
full_cmd = list(server_cmd) + list(server_cmd_extra)
extra_case_fields = {key: value for key, value in case.items() if key not in cls.STANDARD_CASE_FIELDS}
# Safe parsing mapping
result.append(
SingleNodeConfig(
name=case["name"],
model=case["model"],
envs=case.get("envs", {}),
special_dependencies=case.get("special_dependencies", {}),
server_cmd=full_cmd,
epd_server_cmds=case.get("epd_server_cmds", []),
epd_proxy_args=case.get("epd_proxy_args", []),
benchmarks=case.get("benchmarks", {}),
prompts=case.get("prompts", PROMPTS),
api_keyword_args=case.get("api_keyword_args", API_KEYWORD_ARGS),
test_content=case.get("test_content", ["completion"]),
service_mode=case.get("service_mode", "openai"),
extra_config=extra_case_fields,
)
)
return result

View File

@@ -0,0 +1,443 @@
import asyncio
import json
import logging
import os
import shlex
import subprocess
import sys
from typing import Any
import openai
import psutil
import pytest
import vllm
from tests.e2e.conftest import DisaggEpdProxy, RemoteEPDServer, RemoteOpenAIServer
from tests.e2e.nightly.single_node.models.scripts.single_node_config import (
SingleNodeConfig,
SingleNodeConfigLoader,
)
from tools.aisbench import run_aisbench_cases
logger = logging.getLogger(__name__)
configs = SingleNodeConfigLoader.from_yaml_cases()
async def run_completion_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
client = server.get_async_client()
batch = await client.completions.create(
model=config.model,
prompt=config.prompts,
**config.api_keyword_args,
)
choices: list[openai.types.CompletionChoice] = batch.choices
assert choices[0].text, "empty response"
print(choices)
async def run_image_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
from tools.send_mm_request import send_image_request
send_image_request(config.model, server)
async def run_chat_completion_test(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
from tools.send_request import send_v1_chat_completions
send_v1_chat_completions(
config.prompts[0],
model=config.model,
server=server,
request_args=config.api_keyword_args,
)
def run_benchmark_comparisons(config: SingleNodeConfig, results: Any) -> None:
"""General assertion engine for aisbench outcomes mapped directly from YAML."""
comparisons = config.extra_config.get("benchmark_comparisons_args", [])
if not comparisons:
return
# Valid task keys defined in benchmarks mapping
valid_keys = [k for k, v in config.benchmarks.items() if v]
metrics_cache = {}
for comp in comparisons:
metric = comp.get("metric", "TTFT")
baseline_key = comp.get("baseline")
target_key = comp.get("target")
ratio = comp.get("ratio", 1.0)
op = comp.get("operator", "<")
if not baseline_key or not target_key:
logger.warning("Invalid comparison config: missing baseline or target. %s", comp)
continue
if metric not in metrics_cache:
if metric == "TTFT":
from tools.aisbench import get_TTFT
# map TTFT outputs directly to their corresponding benchmark test case names
metrics_cache[metric] = dict(zip(valid_keys, get_TTFT(results)))
else:
logger.warning("Unsupported metric for comparison: %s", metric)
continue
metric_dict = metrics_cache[metric]
baseline_val = metric_dict.get(baseline_key)
target_val = metric_dict.get(target_key)
if baseline_val is None or target_val is None:
logger.warning("Missing data to compare %s and %s in metrics: %s", baseline_key, target_key, metric_dict)
continue
expected_threshold = baseline_val * ratio
eval_str = f"metric {metric}: {target_key}({target_val}) {op} {baseline_key}({baseline_val}) * {ratio}"
if op == "<":
assert target_val < expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
elif op == ">":
assert target_val > expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
elif op == "<=":
assert target_val <= expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
elif op == ">=":
assert target_val >= expected_threshold, f"Assertion Failed: {eval_str} [threshold: {expected_threshold}]"
else:
logger.warning("Unsupported comparison operator: %s", op)
continue
print(f"✅ Comparison passed: {eval_str} [threshold: {expected_threshold}]")
async def run_check_rank0_process_count(
config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy"
) -> None:
proc = await asyncio.create_subprocess_exec(
"npu-smi",
"info",
stdout=asyncio.subprocess.PIPE,
stderr=asyncio.subprocess.PIPE,
)
stdout_bytes, stderr_bytes = await proc.communicate()
if proc.returncode == 0:
logger.info("npu-smi info:\n%s", stdout_bytes.decode(errors="ignore"))
else:
logger.warning("npu-smi info failed: %s", stderr_bytes.decode(errors="ignore"))
vllm_serve_procs = [
p
for p in psutil.process_iter(attrs=["pid", "cmdline"], ad_value=None)
if p.info["cmdline"]
and any("vllm" in arg for arg in p.info["cmdline"])
and any("serve" in arg for arg in p.info["cmdline"])
]
count = len(vllm_serve_procs)
assert count == 1, (
f"rank0 process count check failed: expected exactly 1 vllm serve process on rank0, found {count}"
)
# Extend this dictionary to add new test capabilities
TEST_HANDLERS = {
"completion": run_completion_test,
"image": run_image_test,
"chat_completion": run_chat_completion_test,
"check_rank0_process_count": run_check_rank0_process_count,
}
async def _dispatch_tests(config: SingleNodeConfig, server: "RemoteOpenAIServer | DisaggEpdProxy") -> None:
"""Dispatches requested tests defined in yaml."""
for test_name in config.test_content:
if test_name == "benchmark_comparisons":
continue
handler = TEST_HANDLERS.get(test_name)
if handler:
await handler(config, server)
else:
logger.warning("No handler registered for test content type: %s", test_name)
def _extract_server_cmd_value(server_cmd: list[str], flag: str) -> str | None:
"""Return the value following `flag` in a server_cmd list, or None."""
try:
idx = server_cmd.index(flag)
return server_cmd[idx + 1]
except (ValueError, IndexError):
return None
def _extract_hardware(runner: str) -> str:
"""Derive hardware label (e.g. 'A2', 'A3') from runner name."""
runner_lower = runner.lower()
for label in ("a3", "a2"):
if label in runner_lower:
return label.upper()
return runner
_PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
_FEATURE_ENVS: dict[str, str] = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
_PERF_METRIC_RENAME: dict[str, str] = {
"Benchmark Duration": "Benchmark_Duration(BD)",
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
"Input Token Throughput": "Input_Token_Throughput(ITT)",
"Output Token Throughput": "Output_Token_Throughput(OTT)",
"Total Token Throughput": "Total_Token_Throughput(TTT)",
}
def _extract_dtype(config: SingleNodeConfig) -> str:
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and --quantization ascend is set, else bf16."""
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = _extract_server_cmd_value(config.server_cmd, "--quantization") == "ascend"
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
"""Extract and JSON-parse the value following `flag` in a command list."""
val = _extract_server_cmd_value(cmd_list, flag)
if not val:
return {}
try:
return json.loads(val)
except (json.JSONDecodeError, ValueError):
return {}
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
"""Extract enabled feature names from server_cmd and environment variables."""
if isinstance(server_cmd, str):
try:
cmd_list = shlex.split(server_cmd)
except ValueError:
cmd_list = server_cmd.split()
else:
cmd_list = list(server_cmd)
features: list[str] = []
# Features from --additional-config JSON
additional = _parse_json_flag(cmd_list, "--additional-config")
if additional.get("enable_weight_nz_layout"):
features.append("weight_nz_layout")
wp = additional.get("weight_prefetch_config") or {}
if isinstance(wp, dict) and wp.get("enabled"):
features.append("weight_prefetch")
tc = additional.get("torchair_graph_config") or {}
if isinstance(tc, dict) and tc.get("enabled"):
features.append("torchair_graph")
asc = additional.get("ascend_scheduler_config") or {}
if isinstance(asc, dict) and asc.get("enabled"):
features.append("ascend_scheduler")
# Features from --compilation-config JSON
compilation = _parse_json_flag(cmd_list, "--compilation-config")
if compilation.get("cudagraph_mode"):
features.append("aclgraph")
# Features from --speculative-config JSON
speculative = _parse_json_flag(cmd_list, "--speculative-config")
if speculative:
features.append(speculative.get("method", "speculative"))
# Features from direct flags
if "--enable-expert-parallel" in cmd_list:
features.append("expert_parallel")
# Features from environment variables
for env_key, feature_name in _FEATURE_ENVS.items():
val = str(envs.get(env_key, "0"))
if val not in ("0", "", "false", "False"):
features.append(feature_name)
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
features.append("flashcomm2")
return features
def _build_serve_cmd(config: SingleNodeConfig) -> dict[str, str]:
"""Build serve_cmd dict with mix key for single-node deployments."""
args = " ".join(config.server_cmd)
return {"mix": f"vllm serve {config.model} {args}".strip()}
def _filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
"""Return env vars with internal port keys removed."""
return {k: v for k, v in envs.items() if k not in _PORT_ENV_KEYS}
def _task_passed(case_config: dict[str, Any], result: Any) -> bool:
"""Return True if a single benchmark result meets its baseline/threshold."""
if result == "":
return False
case_type = case_config.get("case_type")
baseline = case_config.get("baseline")
threshold = case_config.get("threshold")
if baseline is None or threshold is None:
return True
if case_type == "accuracy" and isinstance(result, (int, float)):
return abs(float(result) - float(baseline)) <= float(threshold)
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
try:
throughput_val = float(throughput_str.replace("token/s", "").strip())
return throughput_val >= float(threshold) * float(baseline)
except (ValueError, AttributeError):
return False
return True
def _build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
"""Build a single task dict in the required format."""
dataset_path = case_config.get("dataset_path", "")
dataset_conf = case_config.get("dataset_conf", "")
if dataset_path:
task_name = dataset_path.split("/", 1)[-1]
elif dataset_conf:
task_name = dataset_conf.split("/")[0]
else:
task_name = case_key
case_type = case_config.get("case_type", "unknown")
metrics: dict[str, float] = {}
if result == "":
# benchmark run failed — no metrics available
pass
elif case_type == "accuracy" and isinstance(result, (int, float)):
metrics["accuracy"] = round(float(result), 4)
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
for metric_name, metric_data in result_json.items():
if not isinstance(metric_data, dict):
continue
total_str = metric_data.get("total", "")
try:
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
metrics[_PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
except (ValueError, AttributeError):
pass
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
test_input = {k: case_config[k] for k in test_input_keys if k in case_config}
target: dict[str, Any] = {}
if case_config.get("baseline") is not None:
target["baseline"] = case_config["baseline"]
if case_config.get("threshold") is not None:
target["threshold"] = case_config["threshold"]
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
if target:
entry["target"] = target
entry["pass_fail"] = "pass" if _task_passed(case_config, result) else "fail"
return entry
def _all_passed(case_configs: list[dict[str, Any]], results: list[Any]) -> bool:
"""Return True only when every benchmark result meets its baseline/threshold."""
return all(_task_passed(cfg, res) for cfg, res in zip(case_configs, results))
def _save_benchmark_results_json(config: SingleNodeConfig, benchmark_keys: list[str], results: list[Any]) -> None:
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
runner = os.environ.get("VLLM_CI_RUNNER", "")
case_configs = [config.benchmarks[k] for k in benchmark_keys]
tasks = [
_build_task_entry(key, case_cfg, result) for key, case_cfg, result in zip(benchmark_keys, case_configs, results)
]
passed = _all_passed(case_configs, results)
output: dict[str, Any] = {
"model_name": config.model,
"hardware": _extract_hardware(runner),
"dtype": _extract_dtype(config),
"feature": _extract_features(config.server_cmd, config.envs),
"vllm_version": vllm.__version__,
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_VERSION", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config),
"environment": _filter_environment(config.envs),
"pass_fail": "pass" if passed else "fail",
}
os.makedirs("benchmark_results", exist_ok=True)
job_name = os.environ.get("BENCHMARK_JOB_NAME") or config.name
safe_name = job_name.replace("/", "_").replace(" ", "_")
output_path = os.path.join("benchmark_results", f"{safe_name}.json")
with open(output_path, "w", encoding="utf-8") as f:
json.dump(output, f, indent=2, ensure_ascii=False)
logger.info("Benchmark results saved to %s", output_path)
print(f"Benchmark results saved to {output_path}")
def _run_benchmarks(config: SingleNodeConfig, port: int) -> None:
"""Run Aisbench benchmarks and process benchmark-dependent custom assertions."""
benchmark_keys = [k for k, v in config.benchmarks.items() if v]
aisbench_cases = [config.benchmarks[k] for k in benchmark_keys]
if not aisbench_cases:
return
result = run_aisbench_cases(
model=config.model,
port=port,
aisbench_cases=aisbench_cases,
)
_save_benchmark_results_json(config, benchmark_keys, result)
if "benchmark_comparisons" in config.test_content:
run_benchmark_comparisons(config, result)
@pytest.mark.asyncio
@pytest.mark.parametrize("config", configs, ids=[config.name for config in configs])
async def test_single_node(config: SingleNodeConfig) -> None:
# TODO: remove this part after the transformers version upgraded
if config.special_dependencies:
for k, v in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{k}=={v}",
]
subprocess.call(command)
if config.service_mode == "epd":
with (
RemoteEPDServer(vllm_serve_args=config.epd_server_cmds, env_dict=config.envs) as _,
DisaggEpdProxy(proxy_args=config.epd_proxy_args, env_dict=config.envs) as proxy,
):
await _dispatch_tests(config, proxy)
_run_benchmarks(config, proxy.port)
return
# Standard OpenAI service mode
with RemoteOpenAIServer(
model=config.model,
vllm_serve_args=config.server_cmd,
server_port=config.server_port,
env_dict=config.envs,
auto_port=False,
) as server:
await _dispatch_tests(config, server)
_run_benchmarks(config, config.server_port)