init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

View File

@@ -0,0 +1 @@
"""External DP nightly test package."""

View File

@@ -0,0 +1,308 @@
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
benchmarks:
perf_1M_1k_prefix99_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1
batch_size: 1
request_rate: 0
baseline: 0
threshold: 0.95
perf_1M_1k_prefix99:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 11.93
threshold: 0.95

View File

@@ -0,0 +1,324 @@
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 0
tp_size: 2
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 8
tp_size: 2
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_BUFFSIZE: "1800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
perf_TPOT50_128k_1_prefix_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1
batch_size: 4
request_rate: 0
baseline: 0
threshold: 0.95
perf_TPOT50_128k_1_prefix90:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1024
batch_size: 48
request_rate: 1
baseline: 869.13
threshold: 0.95
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,176 @@
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 16
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "1500"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --trust-remote-code
- --block-size
- "32"
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.9"
- --quantization
- "ascend"
- --enforce-eager
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "240"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,335 @@
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 4
npu_per_node: 16
special_dependencies:
transformers: "5.2.0"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0, 1]
decoder: [2, 3]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 4
tp_size: 4
dp_address: "${NODE_2_IP}"
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
VLLM_TORCH_PROFILER_WITH_STACK: "0"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_INTRA_PCIE_ENABLE: "1"
HCCL_INTRA_ROCE_ENABLE: "0"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
templates:
- node_index: 0
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 1
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 2
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 3
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1500
batch_size: 40
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,161 @@
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
model: "Eco-Tech/Kimi-K2.6-w4a8"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_1_IP}"
env_common: &env_common
SERVER_PORT: "${PORT}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_SERVER_DEV_MODE: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_USE_MODELSCOPE: "true"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "512"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --safetensors-load-strategy
- 'prefetch'
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --enforce-eager
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --safetensors-load-strategy
- 'prefetch'
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "256"
- --max-num-seqs
- "16"
- --trust-remote-code
- --gpu-memory-utilization
- "0.92"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 60
max_out_len: 1024
batch_size: 15
request_rate: 0.4
baseline: 347.4475
threshold: 0.97
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 93.33
threshold: 10
temperature: 1.0
top_p: 1

View File

@@ -0,0 +1,197 @@
test_name: "Minimax_m2.7_in3_5_tpot50"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
PYTHONHASHSEED: "0"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "2048"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --enforce-eager
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "55880",
"engine_id": "0",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}} }'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --async-scheduling
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "56900",
"engine_id": "1",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}}'
benchmarks:
perf_warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2
max_out_len: 1
batch_size: 2
request_rate: 0
baseline: 0
threshold: 0.97
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1024
batch_size: 40
request_rate: 0
baseline: 717.5332
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
temperature: 1
top_p: 1
top_k: 40
ignore_eos: false

View File

@@ -0,0 +1,360 @@
# External DP Config Template
This document shows how to write YAML configs consumed by
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
`server_cmd_template` contains only the arguments after
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
`model` automatically.
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
and roles are selected by `routing.groups`.
## Generic DP Template
Use this template for generic external data parallel serving. This mode uses
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
independent vLLM instances instead of external DP rank arguments.
```yaml
test_name: "test Qwen3-30B-A3B generic external dp"
model: "Qwen/Qwen3-30B-A3B"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "generic_dp"
groups:
worker: [0, 1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 1
dp_address: "${NODE_0_IP}"
templates:
- node_index: 0
envs: &generic_env
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "1024"
SERVER_PORT: "${PORT}"
server_cmd_template: &generic_server_cmd
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --max-model-len
- "4096"
- --trust-remote-code
- --enable-expert-parallel
- node_index: 1
envs:
<<: *generic_env
server_cmd_template: *generic_server_cmd
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 16
batch_size: 1
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
num_prompts: 4
max_out_len: 16
batch_size: 1
baseline: 0
threshold: 100
```
## Disaggregated Prefill Template
Use this template for PD disaggregation. `routing.groups` decides which config
entries run as prefillers or decoders. The framework derives the PD proxy script
from `routing.type`, so do not write `proxy_*` fields in YAML.
```yaml
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0]
decoder: [1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
HCCL_OP_
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "256"
SERVER_PORT: "${PORT}"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
- node_index: 1
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
max_out_len: 128
batch_size: 4
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 48
batch_size: 4
baseline: 0
threshold: 100
```
## Field Notes
- `test_name`: Human-readable test name. It is also used when writing benchmark
result metadata.
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
- `num_nodes`: Number of config entries and templates expected.
- `npu_per_node`: Device capacity validation for each node.
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
needs fixed hosts.
- `routing.type`: Supported values are `generic_dp` and
`disaggregated_prefill`.
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
`kv_consumer` for decoder templates.
- `config[].dp_size`: Global DP size for this DP group.
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
- `config[].dp_rank_start`: First global DP rank owned by this node.
- `config[].dp_address`: DP master address. For one global DP group, use
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
address for prefiller nodes and the decoder master address for decoder nodes.
- `templates`: One template per config entry. The framework expands one command
per local DP rank.
The framework injects distributed network envs at startup:
```text
HCCL_IF_IP
HCCL_SOCKET_IFNAME
GLOO_SOCKET_IFNAME
TP_SOCKET_IFNAME
LOCAL_IP
NIC_NAME
MASTER_IP
```
The framework also derives proxy metadata from `routing.type`:
```text
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
```
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
for benchmark requests.
## Template Variables
The following variables are available in `envs` and `server_cmd_template`:
```text
${MODEL}
${PORT_START}
${PORT}
${DP_SIZE}
${DP_SIZE_LOCAL}
${DP_RANK_START}
${DP_RANK}
${LOCAL_RANK}
${TP_SIZE}
${CP_SIZE}
${SP_SIZE}
${PP_SIZE}
${DP_ADDRESS}
${DP_RPC_PORT}
${VISIBLE_DEVICES}
${NODE_INDEX}
${CONFIG_INDEX}
${NODE_0_IP}, ${NODE_1_IP}, ...
${LOCAL_IP}
${MASTER_IP}
${LWS_WORKER_INDEX}
```
Command arguments can also reference rendered environment variables with
shell-style `$VARNAME`, for example:
```yaml
envs:
SERVER_PORT: "${PORT}"
server_cmd_template:
- --port
- $SERVER_PORT
```
## Checks Before Running
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
- Make sure each config index is assigned to exactly one routing group.
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
`--enable-expert-parallel`.
- Set `--max-model-len` large enough for benchmark input tokens plus
`max_out_len`.

View File

@@ -0,0 +1 @@
"""External DP nightly test helpers."""

View File

@@ -0,0 +1,449 @@
import logging
import os
from dataclasses import dataclass, field
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.scripts.utils import (
load_yaml_mapping,
resolve_cluster_ips,
)
from tests.e2e.nightly.multi_node.scripts.utils import (
resolve_current_node_index as resolve_node_index,
)
logger = logging.getLogger(__name__)
ROUTING_GENERIC_DP = "generic_dp"
ROUTING_DISAGGREGATED_PREFILL = "disaggregated_prefill"
PROXY_SCRIPT_BY_ROUTING_TYPE = {
ROUTING_GENERIC_DP: "examples/external_online_dp/dp_load_balance_proxy_server.py",
ROUTING_DISAGGREGATED_PREFILL: "examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
}
CLUSTER_PLACEHOLDER_RE = re.compile(r"\$\{(NODE_(\d+)_IP|LOCAL_IP|MASTER_IP|LWS_WORKER_INDEX)\}")
@dataclass(frozen=True)
class RoutingConfig:
"""Proxy routing metadata shared by all external DP ranks."""
type: str
proxy_node_index: int
proxy_host: str
proxy_port: int
proxy_script: str
groups: dict[str, list[int]]
@dataclass(frozen=True)
class NodeInfo:
"""Per-node external DP server topology loaded from one config entry."""
ip: str
port_start: int
dp_rpc_port: int
dp_size: int
dp_size_local: int
dp_rank_start: int
tp_size: int
dp_address: str
cp_size: int = 1
sp_size: int = 1
pp_size: int = 1
@property
def devices_per_rank(self) -> int:
return self.tp_size * self.cp_size * self.sp_size * self.pp_size
@property
def devices_per_node(self) -> int:
return self.dp_size_local * self.devices_per_rank
@dataclass(frozen=True)
class NodeTemplate:
"""Per-node env and argument template for launching vLLM servers."""
envs: dict[str, Any]
server_cmd_template: list[str]
@dataclass(frozen=True)
class RankInfo:
"""One concrete vLLM server rank expanded from a node config."""
node_index: int
role: str
local_rank: int
dp_rank: int
host: str
port: int
visible_devices: str
dp_size: int
dp_size_local: int
tp_size: int
cp_size: int
sp_size: int
pp_size: int
dp_address: str
dp_rpc_port: int
port_start: int
@dataclass(frozen=True)
class ExternalDPConfig:
"""Top-level external DP test config after YAML anchors are merged."""
test_name: str
model: str
num_nodes: int
npu_per_node: int
cluster_hosts: list[str] | None
cluster_ips: list[str]
routing: RoutingConfig
nodes: list[NodeInfo]
launch_templates: list[NodeTemplate]
benchmark_cases: list[dict[str, Any]] = field(default_factory=list)
special_dependencies: dict[str, str] = field(default_factory=dict)
@property
def is_disaggregated_prefill(self) -> bool:
return self.routing.type == ROUTING_DISAGGREGATED_PREFILL
def replace_cluster_placeholders(
value: Any,
*,
cluster_ips: list[str],
local_ip: str | None = None,
current_node_index: int | None = None,
) -> Any:
if isinstance(value, dict):
return {
key: replace_cluster_placeholders(
val,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for key, val in value.items()
}
if isinstance(value, list):
return [
replace_cluster_placeholders(
item,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for item in value
]
if not isinstance(value, str):
return value
def repl(match: re.Match[str]) -> str:
token = match.group(1)
node_index = match.group(2)
if node_index is not None:
idx = int(node_index)
if idx >= len(cluster_ips):
raise ValueError(f"Cluster placeholder ${{{token}}} is out of range")
return cluster_ips[idx]
if token == "MASTER_IP":
return cluster_ips[0]
if token == "LOCAL_IP":
if local_ip is None:
return match.group(0)
return local_ip
if token == "LWS_WORKER_INDEX":
if current_node_index is None:
return os.environ.get("LWS_WORKER_INDEX", match.group(0))
return str(current_node_index)
return match.group(0)
return CLUSTER_PLACEHOLDER_RE.sub(repl, value)
def resolve_current_node_index(config: ExternalDPConfig) -> int:
return resolve_node_index(config.cluster_ips)
class ExternalDPConfigLoader:
"""Load, normalize, and validate external DP YAML files."""
@classmethod
def from_yaml(
cls,
yaml_path: str | None = None,
*,
cluster_ips: list[str] | None = None,
) -> ExternalDPConfig:
raw_config = cls._load_yaml(yaml_path)
cls._validate_root(raw_config)
num_nodes = int(raw_config["num_nodes"])
resolved_cluster_ips = cls._resolve_cluster_ips(raw_config, num_nodes, cluster_ips)
model = str(raw_config["model"])
routing = cls._parse_routing(raw_config["routing"], resolved_cluster_ips)
nodes = cls._parse_nodes(raw_config, resolved_cluster_ips)
launch_templates = cls._parse_templates(raw_config)
benchmark_cases = cls._parse_benchmarks(raw_config)
config = ExternalDPConfig(
test_name=str(raw_config.get("test_name", "external_dp_test")),
model=model,
num_nodes=num_nodes,
npu_per_node=int(raw_config["npu_per_node"]),
cluster_hosts=raw_config.get("cluster_hosts"),
cluster_ips=resolved_cluster_ips,
routing=routing,
nodes=nodes,
launch_templates=launch_templates,
benchmark_cases=benchmark_cases,
special_dependencies=dict(raw_config.get("special_dependencies", {})),
)
cls._validate_config(config)
return config
@staticmethod
def _load_yaml(yaml_path: str | None) -> dict[str, Any]:
default_config_name = "GLM5_1-W8A8-EP-external.yaml"
default_config_base_path = "tests/e2e/nightly/multi_node/external_dp/config/"
return load_yaml_mapping(
yaml_path,
default_name=default_config_name,
default_base_path=default_config_base_path,
description="external DP config",
)
@staticmethod
def _validate_root(config: dict[str, Any]) -> None:
required = ["model", "num_nodes", "npu_per_node", "routing", "config", "templates", "benchmarks"]
missing = [key for key in required if key not in config]
if missing:
raise KeyError(f"Missing required external DP config fields: {missing}")
if int(config["num_nodes"]) <= 0:
raise ValueError("num_nodes must be greater than 0")
@staticmethod
def _resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
cluster_ips: list[str] | None,
) -> list[str]:
return resolve_cluster_ips(
raw_config,
num_nodes,
cluster_ips,
dns_log_message="Resolving external DP cluster IPs via LWS DNS",
)
@staticmethod
def _parse_routing(raw_routing: dict[str, Any], cluster_ips: list[str]) -> RoutingConfig:
routing_type = str(raw_routing["type"])
if routing_type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {routing_type}")
proxy_node_index = 0
proxy_port = 1999
if proxy_node_index >= len(cluster_ips) or proxy_node_index < 0:
raise ValueError("routing.proxy_node_index out of range")
local_ip = cluster_ips[proxy_node_index]
routing = replace_cluster_placeholders(
raw_routing,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=proxy_node_index,
)
return RoutingConfig(
type=routing_type,
proxy_node_index=proxy_node_index,
proxy_host=local_ip,
proxy_port=proxy_port,
proxy_script=PROXY_SCRIPT_BY_ROUTING_TYPE[routing_type],
groups={
str(name): [int(index) for index in indices] for name, indices in routing.get("groups", {}).items()
},
)
@staticmethod
def _parse_nodes(raw_config: dict[str, Any], cluster_ips: list[str]) -> list[NodeInfo]:
nodes: list[NodeInfo] = []
for index, raw_node in enumerate(raw_config["config"]):
raw_node_index = raw_node.get("node_index")
if raw_node_index is not None and int(raw_node_index) != index:
raise ValueError(f"config[{index}].node_index must equal {index}")
node = replace_cluster_placeholders(
raw_node,
cluster_ips=cluster_ips,
local_ip=cluster_ips[index],
current_node_index=index,
)
nodes.append(
NodeInfo(
ip=cluster_ips[index],
port_start=int(node["port_start"]),
dp_rpc_port=int(node["dp_rpc_port"]),
dp_size=int(node.get("dp_size", 1)),
dp_size_local=int(node.get("dp_size_local", 1)),
dp_rank_start=int(node.get("dp_rank_start", 0)),
tp_size=int(node.get("tp_size", 1)),
cp_size=int(node.get("cp_size", 1)),
sp_size=int(node.get("sp_size", 1)),
dp_address=str(node["dp_address"]),
pp_size=int(node.get("pp_size", 1)),
)
)
return nodes
@staticmethod
def _parse_templates(raw_config: dict[str, Any]) -> list[NodeTemplate]:
templates: list[NodeTemplate] = []
for index, raw_template in enumerate(raw_config["templates"]):
envs = raw_template.get("envs")
server_cmd_template = raw_template.get("server_cmd_template")
if envs is None or server_cmd_template is None:
raise KeyError(f"templates[{index}] must contain envs and server_cmd_template")
if not isinstance(server_cmd_template, list):
raise TypeError(f"templates[{index}].server_cmd_template must be a list")
templates.append(
NodeTemplate(
envs=dict(envs),
server_cmd_template=[str(arg) for arg in server_cmd_template],
)
)
return templates
@staticmethod
def _parse_benchmarks(raw_config: dict[str, Any]) -> list[dict[str, Any]]:
benchmark_cases: list[dict[str, Any]] = []
for name, case in (raw_config.get("benchmarks") or {}).items():
case_with_name = dict(case)
case_with_name["case_name"] = name
benchmark_cases.append(case_with_name)
return benchmark_cases
@classmethod
def _validate_config(cls, config: ExternalDPConfig) -> None:
cls._validate_config_sizes(config)
cls._validate_routing(config)
cls._validate_node_parallel_config(config)
@staticmethod
def _validate_config_sizes(config: ExternalDPConfig) -> None:
if len(config.nodes) != config.num_nodes:
raise AssertionError(f"config size ({len(config.nodes)}) != num_nodes ({config.num_nodes})")
if len(config.launch_templates) != config.num_nodes:
raise AssertionError(f"templates size ({len(config.launch_templates)}) != num_nodes ({config.num_nodes})")
if config.cluster_hosts and len(config.cluster_hosts) != config.num_nodes:
raise AssertionError("cluster_hosts size mismatch")
@staticmethod
def _validate_routing(config: ExternalDPConfig) -> None:
if config.routing.type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {config.routing.type}")
groups = config.routing.groups
if config.routing.type == ROUTING_GENERIC_DP and not groups.get("worker"):
raise ValueError("generic_dp routing requires routing.groups.worker")
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL and (
not groups.get("prefiller") or not groups.get("decoder")
):
raise ValueError("disaggregated_prefill routing requires prefiller and decoder groups")
seen_group_indices: dict[int, str] = {}
for group_name, indices in groups.items():
for index in indices:
if index < 0 or index >= config.num_nodes:
raise ValueError(f"routing.groups.{group_name} index out of range: {index}")
if index in seen_group_indices:
raise ValueError(f"node index {index} appears in both {seen_group_indices[index]} and {group_name}")
seen_group_indices[index] = group_name
if config.routing.proxy_node_index < 0 or config.routing.proxy_node_index >= config.num_nodes:
raise ValueError("routing.proxy_node_index out of range")
@staticmethod
def _validate_node_parallel_config(config: ExternalDPConfig) -> None:
for node_index, node in enumerate(config.nodes):
parallel_sizes = {
"dp_size": node.dp_size,
"dp_size_local": node.dp_size_local,
"tp_size": node.tp_size,
"cp_size": node.cp_size,
"sp_size": node.sp_size,
"pp_size": node.pp_size,
}
invalid_sizes = {name: value for name, value in parallel_sizes.items() if value < 1}
if invalid_sizes:
raise ValueError(f"node {node_index} parallel sizes must be >= 1: {invalid_sizes}")
if node.dp_rank_start < 0:
raise ValueError(f"node {node_index} dp_rank_start must be >= 0")
if node.devices_per_node > config.npu_per_node:
raise ValueError(
f"node {node_index} uses {node.devices_per_node} NPUs, but npu_per_node is {config.npu_per_node}"
)
if node.dp_rank_start + node.dp_size_local > node.dp_size:
raise ValueError(f"node {node_index} dp rank range exceeds dp_size")
class RankResolver:
"""Expand node-level configs into concrete vLLM server ranks."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def resolve(self) -> list[RankInfo]:
role_by_node_index = self._role_by_node_index()
ranks: list[RankInfo] = []
for node_index, node_info in enumerate(self.config.nodes):
role = role_by_node_index[node_index]
ranks.extend(self._expand_node(node_index, role, node_info))
return ranks
def _role_by_node_index(self) -> dict[int, str]:
role_by_index: dict[int, str] = {}
for role, node_indices in self.config.routing.groups.items():
for index in node_indices:
role_by_index[index] = role
missing = [index for index in range(self.config.num_nodes) if index not in role_by_index]
if missing:
raise ValueError(f"routing.groups does not assign role for node indices: {missing}")
return role_by_index
@staticmethod
def _expand_node(node_index: int, role: str, node_info: NodeInfo) -> list[RankInfo]:
ranks: list[RankInfo] = []
for local_rank in range(node_info.dp_size_local):
dp_rank = node_info.dp_rank_start + local_rank
port = node_info.port_start + local_rank
device_range = range(
local_rank * node_info.devices_per_rank,
(local_rank + 1) * node_info.devices_per_rank,
)
visible_devices = ",".join(str(device) for device in device_range)
ranks.append(
RankInfo(
node_index=node_index,
role=role,
local_rank=local_rank,
dp_rank=dp_rank,
host=node_info.ip,
port=port,
visible_devices=visible_devices,
dp_size=node_info.dp_size,
dp_size_local=node_info.dp_size_local,
tp_size=node_info.tp_size,
cp_size=node_info.cp_size,
sp_size=node_info.sp_size,
pp_size=node_info.pp_size,
dp_address=node_info.dp_address,
dp_rpc_port=node_info.dp_rpc_port,
port_start=node_info.port_start,
)
)
return ranks

View File

@@ -0,0 +1,435 @@
import logging
import os
import subprocess
import sys
import time
from collections.abc import Iterable
from dataclasses import dataclass
from pathlib import Path
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ROUTING_GENERIC_DP,
ExternalDPConfig,
NodeTemplate,
RankInfo,
replace_cluster_placeholders,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
format_server_cmd,
is_http_ready,
start_logged_process,
terminate_process_tree,
wait_http_ready,
wait_http_unready,
)
from tests.e2e.nightly.multi_node.scripts.utils import get_net_interface
logger = logging.getLogger(__name__)
SERVER_READY_TIMEOUT_SECONDS = 3600
TEMPLATE_VAR_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
ENV_VAR_RE = re.compile(r"(?<!\$)\$([A-Za-z_][A-Za-z0-9_]*)")
@dataclass(frozen=True)
class ServerCommand:
"""Rendered command, env, and printable command line."""
cmd: list[str]
env: dict[str, str]
display_cmd: str
RankProcess = tuple[subprocess.Popen, RankInfo, Path]
class ServerCommandBuilder:
"""Render rank templates into vLLM serve commands."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def build(self, rank: RankInfo, template: NodeTemplate) -> ServerCommand:
variables = self._build_variables(rank)
rendered_env = self._render_envs(template.envs, rank, variables)
rendered_args = [
self._render_string(
arg,
rank=rank,
braced_variables=variables,
unbraced_variables=rendered_env,
allow_missing_unbraced=False,
)
for arg in template.server_cmd_template
]
cmd = ["vllm", "serve", self.config.model, *rendered_args]
env = {key: str(value) for key, value in rendered_env.items()}
display_cmd = format_server_cmd(cmd, env)
logger.info(
"External DP server command node=%s rank=%s: %s",
rank.node_index,
rank.local_rank,
display_cmd,
)
return ServerCommand(cmd=cmd, env=env, display_cmd=display_cmd)
def build_all(self, ranks: list[RankInfo]) -> list[ServerCommand]:
return [self.build(rank, self.config.launch_templates[rank.node_index]) for rank in ranks]
def _build_variables(self, rank: RankInfo) -> dict[str, str]:
return {
"MODEL": self.config.model,
"PORT_START": str(rank.port_start),
"PORT": str(rank.port),
"DP_SIZE": str(rank.dp_size),
"DP_SIZE_LOCAL": str(rank.dp_size_local),
"DP_RANK_START": str(rank.dp_rank - rank.local_rank),
"DP_RANK": str(rank.dp_rank),
"LOCAL_RANK": str(rank.local_rank),
"TP_SIZE": str(rank.tp_size),
"CP_SIZE": str(rank.cp_size),
"SP_SIZE": str(rank.sp_size),
"PP_SIZE": str(rank.pp_size),
"DP_ADDRESS": rank.dp_address,
"DP_RPC_PORT": str(rank.dp_rpc_port),
"VISIBLE_DEVICES": rank.visible_devices,
"NODE_INDEX": str(rank.node_index),
"CONFIG_INDEX": str(rank.node_index),
}
def _render_envs(
self,
envs: dict[str, Any],
rank: RankInfo,
variables: dict[str, str],
) -> dict[str, str]:
rendered_envs: dict[str, str] = {}
for key, value in envs.items():
if isinstance(value, str):
value = self._render_string(
value,
rank=rank,
braced_variables=variables,
unbraced_variables={**os.environ, **rendered_envs},
allow_missing_unbraced=True,
)
rendered_envs[str(key)] = str(value)
return rendered_envs
def _render_string(
self,
value: str,
*,
rank: RankInfo,
braced_variables: dict[str, str],
unbraced_variables: dict[str, str],
allow_missing_unbraced: bool,
) -> str:
value = replace_cluster_placeholders(
value,
cluster_ips=self.config.cluster_ips,
local_ip=rank.host,
current_node_index=rank.node_index,
)
value = self._render_variables(
value,
braced_variables,
pattern=TEMPLATE_VAR_RE,
allow_missing=False,
)
return self._render_variables(
value,
unbraced_variables,
pattern=ENV_VAR_RE,
allow_missing=allow_missing_unbraced,
)
@staticmethod
def _render_variables(
value: str,
variables: dict[str, str],
*,
pattern: re.Pattern[str],
allow_missing: bool,
) -> str:
def repl(match: re.Match[str]) -> str:
key = match.group(1)
if key not in variables:
if allow_missing:
return ""
raise KeyError(f"Unknown external DP template variable: {key}")
return variables[key]
return pattern.sub(repl, value)
class ExternalDPServerManager:
"""Start and stop the external DP ranks owned by the current node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.command_builder = ServerCommandBuilder(config)
self.dist_envs = build_dist_envs(
config.cluster_ips[current_node_index],
config.cluster_ips[0],
)
self.rank_processes: list[RankProcess] = []
def start_current_node(self) -> None:
local_ranks = [rank for rank in self.ranks if rank.node_index == self.current_node_index]
logger.info("Starting %d external DP ranks on node %d", len(local_ranks), self.current_node_index)
try:
for rank in local_ranks:
template = self.config.launch_templates[rank.node_index]
template = type(template)(
envs={**template.envs, **self.dist_envs},
server_cmd_template=template.server_cmd_template,
)
server_cmd = self.command_builder.build(rank, template)
log_file = self._rank_log_file(rank)
process = start_logged_process(server_cmd.cmd, server_cmd.env, log_file)
self.rank_processes.append((process, rank, log_file))
wait_ranks_ready(
local_ranks,
timeout=SERVER_READY_TIMEOUT_SECONDS,
rank_processes=self.rank_processes,
)
except Exception:
self.cleanup()
raise
def __enter__(self):
self.start_current_node()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
for process, rank, _log_file in reversed(self.rank_processes):
logger.info(
"Stopping external DP rank node=%d rank=%d pid=%d",
rank.node_index,
rank.local_rank,
process.pid,
)
terminate_process_tree(process.pid)
self.rank_processes.clear()
def _rank_log_file(self, rank: RankInfo) -> Path:
return self.log_root / f"node-{rank.node_index}" / f"rank-{rank.local_rank}.log"
class ExternalDPProxyLauncher:
"""Launch the external DP proxy on the configured proxy node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.pid: int | None = None
def start(self) -> None:
if self.current_node_index != self.config.routing.proxy_node_index:
logger.info("Current node is not proxy node, skip launching external DP proxy")
return
cmd = build_proxy_server_cmd(self.config, self.ranks)
log_file = self.log_root / f"node-{self.current_node_index}" / "proxy.log"
process = start_logged_process(cmd, {}, log_file)
self.pid = process.pid
logger.info("External DP proxy launched: %s", proxy_server_health_url(self.config))
def wait_ready(self, timeout: int = 300) -> None:
wait_http_ready(proxy_server_health_url(self.config), timeout=timeout)
logger.info("External DP proxy ready: %s", proxy_server_health_url(self.config))
def __enter__(self):
self.start()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
if self.pid is None:
return
logger.info("Stopping external DP proxy pid=%d", self.pid)
terminate_process_tree(self.pid)
self.pid = None
def build_all_server_commands(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[ServerCommand]:
return ServerCommandBuilder(config).build_all(ranks)
def build_dist_envs(cur_ip: str, master_ip: str) -> dict[str, str]:
nic_name = get_net_interface(cur_ip)
return {
"HCCL_IF_IP": cur_ip,
"HCCL_SOCKET_IFNAME": nic_name,
"GLOO_SOCKET_IFNAME": nic_name,
"TP_SOCKET_IFNAME": nic_name,
"LOCAL_IP": cur_ip,
"NIC_NAME": nic_name,
"MASTER_IP": master_ip,
}
def build_proxy_server_cmd(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[str]:
routing = config.routing
cmd = [sys.executable, routing.proxy_script, "--host", routing.proxy_host, "--port", str(routing.proxy_port)]
if routing.type == ROUTING_GENERIC_DP:
worker_ranks = [rank for rank in ranks if rank.role == "worker"]
if not worker_ranks:
raise ValueError("generic_dp proxy requires worker ranks")
cmd.extend(["--dp-hosts", *[rank.host for rank in worker_ranks]])
cmd.extend(["--dp-ports", *[str(rank.port) for rank in worker_ranks]])
return cmd
if routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefiller_ranks = [rank for rank in ranks if rank.role == "prefiller"]
decoder_ranks = [rank for rank in ranks if rank.role == "decoder"]
if not prefiller_ranks or not decoder_ranks:
raise ValueError("disaggregated_prefill proxy requires prefiller and decoder ranks")
cmd.extend(["--prefiller-hosts", *[rank.host for rank in prefiller_ranks]])
cmd.extend(["--prefiller-ports", *[str(rank.port) for rank in prefiller_ranks]])
cmd.extend(["--decoder-hosts", *[rank.host for rank in decoder_ranks]])
cmd.extend(["--decoder-ports", *[str(rank.port) for rank in decoder_ranks]])
return cmd
raise ValueError(f"Unsupported routing.type: {routing.type}")
def proxy_server_health_url(config: ExternalDPConfig) -> str:
return f"http://{config.routing.proxy_host}:{config.routing.proxy_port}/healthcheck"
def rank_health_url(rank: RankInfo) -> str:
return f"http://{rank.host}:{rank.port}/health"
def master_rank_health_url(ranks: list[RankInfo]) -> str:
for rank in ranks:
if rank.node_index == 0 and rank.local_rank == 0:
return rank_health_url(rank)
raise RuntimeError("External DP master rank was not found")
def rank_label(rank: RankInfo) -> str:
return f"node={rank.node_index} rank={rank.local_rank} role={rank.role} url={rank_health_url(rank)}"
def format_http_status(label: str, url: str) -> str:
status = "ready" if is_http_ready(url, timeout=1.0) else "waiting"
return f"{label}={status} url={url}"
def _format_rank_statuses(
ranks: list[RankInfo],
rank_ready: dict[RankInfo, bool],
) -> str:
parts = []
for rank in ranks:
status = "ready" if rank_ready[rank] else "waiting"
parts.append(f" {rank_label(rank)} status={status}")
return "\n".join(parts)
def _raise_if_rank_process_exited(rank_processes: list[RankProcess] | None) -> None:
if not rank_processes:
return
exited = []
for process, rank, log_file in rank_processes:
returncode = process.poll()
if returncode is not None:
exited.append(f"{rank_label(rank)} pid={process.pid} returncode={returncode} log={log_file}")
if exited:
raise RuntimeError("External DP rank process exited before ready: " + "; ".join(exited))
def wait_ranks_ready(
ranks: Iterable[RankInfo],
timeout: int,
rank_processes: list[RankProcess] | None = None,
) -> None:
ranks = list(ranks)
rank_ready = {rank: False for rank in ranks}
deadline = time.monotonic() + timeout
last_log_time = 0.0
while True:
_raise_if_rank_process_exited(rank_processes)
all_ready = True
unhealthy_after_ready = []
for rank in ranks:
is_ready = is_http_ready(rank_health_url(rank), timeout=1.0)
if is_ready:
if not rank_ready[rank]:
logger.info("[READY] External DP rank %s", rank_label(rank))
rank_ready[rank] = True
continue
all_ready = False
if rank_ready[rank]:
unhealthy_after_ready.append(rank)
if unhealthy_after_ready:
failed = "; ".join(rank_label(rank) for rank in unhealthy_after_ready)
raise RuntimeError(f"External DP rank became unhealthy after ready: {failed}")
if all_ready:
return
now = time.monotonic()
if now - last_log_time >= 30:
logger.info(
"Polling external DP ranks: ready=%d/%d\n%s",
sum(rank_ready.values()),
len(ranks),
_format_rank_statuses(ranks, rank_ready),
)
last_log_time = now
if now >= deadline:
pending = [rank for rank in ranks if not rank_ready[rank]]
pending_labels = "; ".join(rank_label(rank) for rank in pending)
raise TimeoutError(f"Timed out waiting for external DP ranks ready: {pending_labels}")
time.sleep(5)
def wait_master_rank_stopped(ranks: list[RankInfo], timeout: int) -> None:
url = master_rank_health_url(ranks)
wait_http_ready(url, timeout=SERVER_READY_TIMEOUT_SECONDS)
logger.info("Hanging until master external DP rank stops: %s", url)
wait_http_unready(url, timeout=timeout)

View File

@@ -0,0 +1,163 @@
import logging
import os
import subprocess
import sys
import threading
import time
from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ExternalDPConfig,
ExternalDPConfigLoader,
RankResolver,
resolve_current_node_index,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import (
ExternalDPProxyLauncher,
ExternalDPServerManager,
build_all_server_commands,
format_http_status,
master_rank_health_url,
proxy_server_health_url,
wait_master_rank_stopped,
wait_ranks_ready,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
collect_logs,
write_benchmark_results_json,
)
from tools.aisbench import run_aisbench_cases
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
logger = logging.getLogger(__name__)
DEFAULT_LOG_ROOT = Path("/tmp/external_dp_logs")
def _install_special_dependencies(config: ExternalDPConfig) -> None:
for package, version in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{package}=={version}",
]
subprocess.call(command)
@contextmanager
def _heartbeat(
task_name: str,
*,
interval: int = 30,
status_fn: Callable[[], str] | None = None,
):
start_time = time.monotonic()
stop_event = threading.Event()
def report_progress() -> None:
while not stop_event.wait(interval):
elapsed = int(time.monotonic() - start_time)
status = ""
if status_fn is not None:
try:
status = f" {status_fn()}"
except Exception as exc: # pragma: no cover - diagnostic only
status = f" status_error={exc!r}"
logger.info("%s still running: elapsed=%ds%s", task_name, elapsed, status)
logger.info("%s started", task_name)
thread = threading.Thread(target=report_progress, daemon=True)
thread.start()
try:
yield
finally:
stop_event.set()
thread.join(timeout=1)
elapsed = int(time.monotonic() - start_time)
logger.info("%s finished: elapsed=%ds", task_name, elapsed)
def _format_benchmark_cases(config: ExternalDPConfig) -> str:
names = [str(case.get("case_name", "<unnamed>")) for case in config.benchmark_cases]
return ", ".join(names) if names else "<none>"
def _archive_rank_logs(log_root: Path, current_node_index: int) -> None:
log_prefix = os.environ.get("LOG_PREFIX")
if not log_prefix:
return
node_log_dir = log_root / f"node-{current_node_index}"
output_tar = Path(log_prefix) / f"node_{current_node_index}_external_dp_logs.tar.gz"
collect_logs(node_log_dir, output_tar)
def test_external_dp() -> None:
config = ExternalDPConfigLoader.from_yaml()
_install_special_dependencies(config)
ranks = RankResolver(config).resolve()
current_node_index = resolve_current_node_index(config)
log_root = Path(os.environ.get("EXTERNAL_DP_LOG_DIR", str(DEFAULT_LOG_ROOT)))
max_wait_seconds = int(os.environ.get("EXTERNAL_DP_MAX_WAIT_SECONDS", "3600"))
is_master = current_node_index == 0
server_manager = ExternalDPServerManager(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
proxy_launcher = ExternalDPProxyLauncher(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
try:
with server_manager, proxy_launcher:
if is_master:
wait_ranks_ready(ranks, timeout=max_wait_seconds)
proxy_launcher.wait_ready()
target = f"http://{config.routing.proxy_host}:{config.routing.proxy_port}"
logger.info(
"Running AISBench cases: model=%s target=%s cases=[%s]",
config.model,
target,
_format_benchmark_cases(config),
)
with _heartbeat(
"Running AISBench",
status_fn=lambda: format_http_status("proxy", proxy_server_health_url(config)),
):
results = run_aisbench_cases(
model=config.model,
port=config.routing.proxy_port,
aisbench_cases=config.benchmark_cases,
host_ip=config.routing.proxy_host,
)
logger.info("AISBench completed: results=%d", len(results or []))
all_commands = build_all_server_commands(config, ranks)
write_benchmark_results_json(
config=config,
ranks=ranks,
commands=all_commands,
results=results,
)
wait_ranks_ready(ranks, timeout=30)
else:
master_url = master_rank_health_url(ranks)
with _heartbeat(
"Waiting for master external DP rank to stop",
status_fn=lambda: format_http_status("master", master_url),
):
wait_master_rank_stopped(ranks, timeout=max_wait_seconds)
finally:
_archive_rank_logs(log_root, current_node_index)

View File

@@ -0,0 +1,236 @@
import logging
import os
import shlex
import signal
import subprocess
import tarfile
import time
import urllib.error
import urllib.request
from pathlib import Path
from typing import TYPE_CHECKING, Any
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ExternalDPConfig,
RankInfo,
)
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
build_task_entry,
extract_hardware,
filter_environment,
get_vllm_version,
write_results_json,
)
logger = logging.getLogger(__name__)
if TYPE_CHECKING:
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import ServerCommand
SENSITIVE_ENV_TOKENS = ("TOKEN", "SECRET", "PASSWORD", "ACCESS_KEY")
def format_server_cmd(cmd: list[str], env: dict[str, str] | None = None) -> str:
env_parts: list[str] = []
for key, value in sorted((env or {}).items()):
display_value = "***" if any(token in key.upper() for token in SENSITIVE_ENV_TOKENS) else str(value)
env_parts.append(f"{key}={shlex.quote(display_value)}")
return " ".join([*env_parts, shlex.join(cmd)])
def start_logged_process(cmd: list[str], env: dict[str, str], log_file: Path) -> subprocess.Popen:
log_file.parent.mkdir(parents=True, exist_ok=True)
merged_env = {**os.environ, **env}
with log_file.open("ab") as f:
f.write(f"Starting command: {format_server_cmd(cmd, env)}\n".encode())
f.flush()
return subprocess.Popen(
cmd,
stdout=f,
stderr=subprocess.STDOUT,
env=merged_env,
start_new_session=True,
)
def terminate_process_tree(pid: int, timeout: int = 30) -> None:
try:
import psutil
except ModuleNotFoundError:
try:
os.killpg(pid, signal.SIGTERM)
except ProcessLookupError:
return
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
os.kill(pid, 0)
except ProcessLookupError:
return
time.sleep(0.2)
try:
os.killpg(pid, signal.SIGKILL)
except ProcessLookupError:
return
return
try:
parent = psutil.Process(pid)
except psutil.NoSuchProcess:
return
children = parent.children(recursive=True)
for process in children:
process.terminate()
parent.terminate()
gone, alive = psutil.wait_procs([parent, *children], timeout=timeout)
del gone
for process in alive:
process.kill()
def is_http_ready(url: str, timeout: float = 5.0) -> bool:
try:
with urllib.request.urlopen(url, timeout=timeout) as response:
return 200 <= response.status < 300
except (urllib.error.URLError, TimeoutError, OSError):
return False
def wait_http_ready(url: str, timeout: int, interval: float = 2.0) -> None:
deadline = time.monotonic() + timeout
last_error: Exception | None = None
while time.monotonic() < deadline:
try:
with urllib.request.urlopen(url, timeout=5) as response:
if 200 <= response.status < 300:
return
except (urllib.error.URLError, TimeoutError, OSError) as exc:
last_error = exc
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP ready: {url}; last_error={last_error}")
def wait_http_unready(url: str, timeout: int, interval: float = 5.0) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if not is_http_ready(url):
return
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP unready: {url}")
def collect_logs(src_dir: Path, output_tar: Path) -> None:
if not src_dir.exists():
return
output_tar.parent.mkdir(parents=True, exist_ok=True)
with tarfile.open(output_tar, "w:gz") as tar:
tar.add(src_dir, arcname=src_dir.name)
def _common_command_envs(commands: list["ServerCommand"]) -> dict[str, str]:
if not commands:
return {}
common_keys = set(commands[0].env)
for command in commands[1:]:
common_keys.intersection_update(command.env)
common_envs: dict[str, str] = {}
for key in sorted(common_keys):
values = {command.env[key] for command in commands}
if len(values) == 1:
common_envs[key] = next(iter(values))
return common_envs
def _extract_dtype(config: ExternalDPConfig, commands: list["ServerCommand"]) -> str:
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = any("--quantization ascend" in command.display_cmd for command in commands)
return "w8a8" if has_w8a8 and has_quant_ascend else "bf16"
def _extract_features(commands: list["ServerCommand"]) -> list[str]:
if not commands:
return []
features: list[str] = []
command_args = [command.cmd for command in commands]
command_displays = [" ".join(shlex.quote(arg) for arg in command.cmd) for command in commands]
if any("--async-scheduling" in cmd for cmd in command_args):
features.append("async_scheduling")
if any("--enable-expert-parallel" in cmd for cmd in command_args):
features.append("expert_parallel")
if any("--speculative-config" in cmd for cmd in command_args):
features.append("speculative")
if any("cudagraph_mode" in display for display in command_displays):
features.append("aclgraph")
feature_envs = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
for env_key, feature_name in feature_envs.items():
values = [str(command.env.get(env_key, "0")) for command in commands]
if any(value not in ("0", "", "false", "False") for value in values):
features.append(feature_name)
return features
def _build_serve_cmd(
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
) -> dict[str, Any]:
entries: dict[str, str] = {}
for rank, command in zip(ranks, commands):
prefix = rank.role
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefix = "prefill" if rank.role == "prefiller" else "decode"
entries[f"{prefix}-node{rank.node_index}-rank{rank.local_rank}"] = command.display_cmd
key = "external_dp_pd" if config.routing.type == ROUTING_DISAGGREGATED_PREFILL else "external_dp"
return {key: entries}
def build_benchmark_results(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
) -> dict[str, Any]:
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
tasks = [build_task_entry(key, case, result) for (key, case), result in zip(valid_items, results)]
runner = os.environ.get("VLLM_CI_RUNNER", "")
common_envs = _common_command_envs(commands)
return {
"model_name": config.model,
"hardware": extract_hardware(runner),
"dtype": _extract_dtype(config, commands),
"feature": _extract_features(commands),
"vllm_version": get_vllm_version(),
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config, ranks, commands),
"environment": filter_environment(common_envs),
}
def write_benchmark_results_json(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
output_dir: Path | None = None,
) -> Path:
output = build_benchmark_results(config=config, ranks=ranks, commands=commands, results=results)
job_name = os.environ.get("BENCHMARK_JOB_NAME", "") or config.test_name.replace(" ", "-")
return write_results_json(output, job_name=job_name, output_dir=output_dir)

View File

@@ -0,0 +1,196 @@
test_name: "test DeepSeek-R1-W8A8 disaggregated_prefill"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 10
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_DETERMINISTIC: True
TASK_QUEUE_ENABLE: 1
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0, L2:0"
DYNAMIC_EPLB: true
VLLM_ENGINE_READY_TIMEOUT_S: 3000
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 4
--max-model-len 36864
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 4
--max-model-len 36864
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 32
--data-parallel-size-local 16
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 1
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 28
--max-model-len 36864
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"multistream_overlap_shared_expert":true,"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--headless
--data-parallel-size 32
--data-parallel-size-local 16
--data-parallel-start-rank 16
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 1
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 28
--max-model-len 36864
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"multistream_overlap_shared_expert":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 512
baseline: 95
threshold: 5

View File

@@ -0,0 +1,114 @@
test_name: "test DeepSeek-R1-W8A8-longseq disaggregated_prefill"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 768
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_DETERMINISTIC: True
TASK_QUEUE_ENABLE: 1
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0"
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
VLLM_ENGINE_READY_TIMEOUT_S: 3000
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 1
--decode-context-parallel-size 8
--prefill-context-parallel-size 2
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 32
--max-model-len 32768
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.85
--enable-chunked-prefill
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--decode-context-parallel-size 2
--prefill-context-parallel-size 1
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 32
--max-model-len 32768
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.85
--compilation_config '{"cudagraph_capture_sizes":[4,8,16,32],"cudagraph_mode": "FULL_DECODE_ONLY"}'
--enable-chunked-prefill
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
--additional-config '{"recompute_scheduler_enable":true}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
num_prompts: 360
max_out_len: 4096
batch_size: 32
baseline: 95
threshold: 5

View File

@@ -0,0 +1,85 @@
test_name: "test DeepSeek-V3.1-BF16 on A3"
model: "unsloth/DeepSeek-V3.1-BF16"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 2048
SERVER_PORT: 8080
OMP_PROC_BIND: false
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
OMP_NUM_THREADS: 1
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_BALANCE_SCHEDULING: 1
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve unsloth/DeepSeek-V3.1-BF16
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--tensor-parallel-size 8
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13399
--no-enable-prefix-caching
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 4096
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.95
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
--additional_config '{"enable_multistream_moe": true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve unsloth/DeepSeek-V3.1-BF16
--headless
--data-parallel-size 4
--tensor-parallel-size 8
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13399
--no-enable-prefix-caching
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 4096
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.95
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
--additional_config '{"enable_multistream_moe": true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 512
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 512
baseline: 95
threshold: 10

View File

@@ -0,0 +1,127 @@
test_name: "test DeepSeek-V3.2-W8A8 on A3"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
ASCEND_A3_EBA_ENABLE: 1
VLLM_ENGINE_READY_TIMEOUT_S: 3000
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13399
--tensor-parallel-size 8
--quantization ascend
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 90000
--max-num-batched-tokens 4096
--no-enable-prefix-caching
--gpu-memory-utilization 0.85
--trust-remote-code
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--headless
--data-parallel-size 4
--data-parallel-rpc-port 13399
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--tensor-parallel-size 8
--quantization ascend
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 90000
--max-num-batched-tokens 4096
--no-enable-prefix-caching
--gpu-memory-utilization 0.85
--trust-remote-code
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
benchmarks:
perf_short_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 3000
batch_size: 512
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_long_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 3000
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_short:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 3000
batch_size: 256
request_rate: 11.2
baseline: 305.2903
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 128
baseline: 95
threshold: 10
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 80000
batch_size: 32
baseline: 57
threshold: 10

View File

@@ -0,0 +1,268 @@
test_name: "test DeepSeek-V3.2-W8A8-EP disaggregated_prefill"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 10
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: 360
VLLM_TORCH_PROFILER_WITH_STACK: 0
ASCEND_AGGREGATE_ENABLE: 1
ASCEND_TRANSPORT_PRINT: 1
ACL_OP_INIT_MODE: 1
ASCEND_A3_ENABLE: 1
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
VLLM_ENGINE_READY_TIMEOUT_S: 3000
HCCL_CONNECT_TIMEOUT: 1200
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
deployment:
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-start-rank 0
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 16
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 133000
--max-num-batched-tokens 8192
--trust-remote-code
--gpu-memory-utilization 0.90
--enforce-eager
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-start-rank 1
--data-parallel-size-local 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 16
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 133000
--max-num-batched-tokens 8192
--trust-remote-code
--gpu-memory-utilization 0.90
--enforce-eager
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 1100
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-model-len 133000
--max-num-batched-tokens 42
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
--trust-remote-code
--max-num-seqs 14
--gpu-memory-utilization 0.90
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 1100
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-model-len 133000
--max-num-batched-tokens 42
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
--trust-remote-code
--max-num-seqs 14
--gpu-memory-utilization 0.90
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
benchmarks:
perf_short_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1500
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_long_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1024
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_short:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1
threshold: 0.97
perf_long:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 1
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 64
baseline: 96.88
threshold: 10

View File

@@ -0,0 +1,102 @@
test_name: "multi-node-GLM-5.1-W8A8C8-MTP-A3_64k/128k"
model: "Eco-Tech/GLM-5.1-w8a8c8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
VLLM_USE_MODELSCOPE: "true"
OMP_NUM_THREADS: "1"
HCCL_BUFFSIZE: "400"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: 8077
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8c8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--data-parallel-rpc-port 12981
--tensor-parallel-size 4
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
--seed 1024
--tool-call-parser glm47
--reasoning-parser glm45
--enable-auto-tool-choice
--max-num-seqs 6
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.92
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8c8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--data-parallel-rpc-port 12981
--tensor-parallel-size 4
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
--seed 1024
--tool-call-parser glm47
--reasoning-parser glm45
--enable-auto-tool-choice
--max-num-seqs 6
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.9
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
benchmarks:
perf_128k_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1
batch_size: 1
request_rate: 0
baseline: 0
threshold: 0.97
perf_128k:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 256
max_out_len: 1024
batch_size: 64
request_rate: 0
baseline: 290.8308
threshold: 0.97

View File

@@ -0,0 +1,84 @@
test_name: "multi-node-GLM-5.1-w8a8-A2"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: 0
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_USE_MODELSCOPE: true
VLLM_ENGINE_READY_TIMEOUT_S: 3000
VLLM_RPC_TIMEOUT: 600
SERVER_PORT: 8078
special_dependencies:
transformers: "5.2.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 8
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 64
--max-model-len 38000
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--no-enable-prefix-caching
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 8
--data-parallel-size-local 1
--data-parallel-start-rank 1
--headless
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--enable-expert-parallel
--seed 1024
--max-num-seqs 64
--max-model-len 38000
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--no-enable-prefix-caching
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 72
max_out_len: 1500
batch_size: 18
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,98 @@
test_name: "multi-node-GLM-5.1-w8a8-A3"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
VLLM_USE_MODELSCOPE: true
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: 0
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "600"
SERVER_PORT: 8080
special_dependencies:
transformers: "5.2.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-rpc-port 13389
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 72348
batch_size: 32
baseline: 90
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 12
max_out_len: 1024
batch_size: 3
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,240 @@
test_name: "multi-node-GLM-5.1-w8a8-EP"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: 1024
VLLM_TORCH_PROFILER_WITH_STACK: 0
ASCEND_AGGREGATE_ENABLE: 1
ASCEND_TRANSPORT_PRINT: 1
ACL_OP_INIT_MODE: 1
ASCEND_A3_ENABLE: 1
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
VLLM_ASCEND_ENABLE_FUSED_MC2: 1
special_dependencies:
transformers: "5.2.0"
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--tensor-parallel-size 8
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 131072
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
--max-num-batched-tokens 4096
--trust-remote-code
--max-num-seqs 64
--quantization ascend
--gpu-memory-utilization 0.95
--enforce-eager
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--tensor-parallel-size 8
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 131072
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
--max-num-batched-tokens 4096
--trust-remote-code
--max-num-seqs 64
--quantization ascend
--gpu-memory-utilization 0.95
--enforce-eager
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 10543
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 202752
--max-num-batched-tokens 32
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
--trust-remote-code
--max-num-seqs 8
--gpu-memory-utilization 0.92
--quantization ascend
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--headless
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 10543
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 202752
--max-num-batched-tokens 32
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
--trust-remote-code
--max-num-seqs 8
--gpu-memory-utilization 0.92
--quantization ascend
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1500
batch_size: 40
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,91 @@
test_name: "multi-node-GLM-5.2-w8a8-A3"
model: "Eco-Tech/GLM-5.2-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
VLLM_USE_MODELSCOPE: true
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "600"
SERVER_PORT: 8080
special_dependencies:
transformers: "5.12.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.2-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.2-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-rpc-port 13389
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 72348
batch_size: 32
baseline: 90
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 12
max_out_len: 1024
batch_size: 3
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,90 @@
test_name: "test Kimi-K2.5-W4A8 A2 dual nodes"
model: "Eco-Tech/Kimi-K2.5-W4A8"
num_nodes: 2
npu_per_node: 8
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 512
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/Kimi-K2.5-W4A8
--host 0.0.0.0
--port $SERVER_PORT
--quantization ascend
--allowed-local-media-path /
--trust-remote-code
--no-enable-prefix-caching
--seed 42
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--enable-expert-parallel
--max-num-seqs 64
--max-model-len 51200
--max-num-batched-tokens 8192
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
--mm-processor-cache-gb 0
--mm-encoder-tp-mode data
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/Kimi-K2.5-W4A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--quantization ascend
--allowed-local-media-path /
--trust-remote-code
--no-enable-prefix-caching
--seed 42
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--enable-expert-parallel
--max-num-seqs 64
--max-model-len 51200
--max-num-batched-tokens 8192
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
--mm-processor-cache-gb 0
--mm-encoder-tp-mode data
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 320
max_out_len: 1500
batch_size: 80
trust_remote_code: True
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,74 @@
test_name: "test Qwen3-235B-A22B multi-dp on A2"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 8
env_common: &env_common
VLLM_USE_MODELSCOPE: true
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
TASK_QUEUE_ENABLE: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 40960
--max-num-batched-tokens 2048
--trust-remote-code
--gpu-memory-utilization 0.9
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--headless
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 128
--max-model-len 40960
--max-num-batched-tokens 2048
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 256
request_rate: 4.8
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 256
baseline: 96
threshold: 10

View File

@@ -0,0 +1,77 @@
test_name: "test Qwen3-235B-A22B multi-dp"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
TASK_QUEUE_ENABLE: 1
VLLM_USE_MODELSCOPE: true
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--headless
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 512
baseline: 95
threshold: 3

View File

@@ -0,0 +1,93 @@
test_name: "test Qwen3-235B-A22B-W8A8 EPLB"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
DYNAMIC_EPLB: true
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":50,"algorithm_execution_interval":5}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":600,"algorithm_execution_interval":50}}'
benchmarks:

View File

@@ -0,0 +1,100 @@
test_name: "test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
DYNAMIC_EPLB: true
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 1
--decode-context-parallel-size 2
--prefill-context-parallel-size 2
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--seed 1024
--enforce-eager
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"dynamic_eplb":true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--decode-context-parallel-size 2
--prefill-context-parallel-size 1
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--compilation_config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"dynamic_eplb":true}'
benchmarks:

View File

@@ -0,0 +1,89 @@
test_name: "test Qwen3-235B-A22B-W8A8 disaggregated_prefill"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
benchmarks:

View File

@@ -0,0 +1,118 @@
test_name: "test Qwen3-235B-A22B disaggregated_prefill"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_BUFFSIZE: 1024
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
VLLM_ASCEND_ENABLE_FUSED_MC2: 2
TASK_QUEUE_ENABLE: 1
SERVER_PORT: 8080
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
--no-enable-prefix-caching
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
--no-enable-prefix-caching
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 512
baseline: 97
threshold: 10

View File

@@ -0,0 +1,108 @@
test_name: "test Qwen3-VL-235B-A22B disaggregated_prefill"
model: "Qwen/Qwen3-VL-235B-A22B-Instruct"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_OP_EXPANSION_MODE: "AIV"
TASK_QUEUE_ENABLE: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 4
--tensor-parallel-size 4
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/textvqa-perf-1080p
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
num_prompts: 2800
max_out_len: 1500
batch_size: 64
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/textvqa-lite
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
max_out_len: 7680
batch_size: 64
baseline: 85
threshold: 5

View File

@@ -0,0 +1,331 @@
import logging
import os
import subprocess
from dataclasses import dataclass
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.scripts.utils import (
get_available_port,
get_net_interface,
load_yaml_mapping,
resolve_cluster_ips,
resolve_current_node_index,
setup_logger,
)
setup_logger()
logger = logging.getLogger(__name__)
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
DEFAULT_SERVER_PORT = 8080
@dataclass(frozen=True)
class NodeInfo:
index: int
ip: str
server_cmd: str
envs: dict[str, Any] | None = None
headless: bool = False
def __post_init__(self):
if not self.ip:
raise ValueError("NodeInfo.ip must not be empty")
def __str__(self) -> str:
return f"NodeInfo(\n index={self.index},\n ip={self.ip},\n headless={self.headless},\n)"
class DisaggregatedPrefillCfg:
def __init__(self, raw_cfg: dict, num_nodes: int):
self.prefiller_indices: list[int] = raw_cfg.get("prefiller_host_index", [])
self.decoder_indices: list[int] = raw_cfg.get("decoder_host_index", [])
if not self.decoder_indices:
raise RuntimeError("decoder_host_index must be provided")
self._validate(num_nodes)
self.decode_start_index = self.decoder_indices[0]
self.num_prefillers = len(self.prefiller_indices)
self.num_decoders = len(self.decoder_indices)
def _validate(self, num_nodes: int):
overlap = set(self.prefiller_indices) & set(self.decoder_indices)
if overlap:
raise AssertionError(f"Prefiller and decoder overlap: {overlap}")
all_indices = self.prefiller_indices + self.decoder_indices
if any(i >= num_nodes for i in all_indices):
raise ValueError("Disaggregated prefill index out of range")
def is_prefiller(self, index: int) -> bool:
return index in self.prefiller_indices
def is_decoder(self, index: int) -> bool:
return index in self.decoder_indices
def master_ip_for_node(self, index: int, nodes: list[NodeInfo]) -> str:
if self.is_prefiller(index):
return nodes[0].ip
return nodes[self.decode_start_index].ip
class DistEnvBuilder:
def __init__(
self,
*,
cur_node: NodeInfo,
master_ip: str,
):
self.cur_ip = cur_node.ip
self.nic_name = get_net_interface(self.cur_ip)
self.master_ip = master_ip
self.base_envs = dict(cur_node.envs or {})
def build(self) -> dict:
envs = dict(self.base_envs)
envs.update(
{
"HCCL_IF_IP": self.cur_ip,
"HCCL_SOCKET_IFNAME": self.nic_name,
"GLOO_SOCKET_IFNAME": self.nic_name,
"TP_SOCKET_IFNAME": self.nic_name,
"LOCAL_IP": self.cur_ip,
"NIC_NAME": self.nic_name,
"MASTER_IP": self.master_ip,
}
)
return {k: str(v) for k, v in envs.items()}
class ProxyLauncher:
def __init__(
self,
*,
nodes: list[NodeInfo],
envs: dict,
proxy_port: int,
cur_index: int,
disagg_cfg: DisaggregatedPrefillCfg | None = None,
):
self.nodes = nodes
self.cfg = disagg_cfg
self.server_port = envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
self.proxy_port = proxy_port
self.proxy_script = envs.get(
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
"examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
)
self.envs = envs
self.is_master = cur_index == 0
self.cur_ip = nodes[cur_index].ip
self.process: subprocess.Popen[bytes] | None = None
def __enter__(self):
if not self.is_master or self.cfg is None:
logger.info("Not launching proxy on non-master node")
return self
prefiller_ips = [self.nodes[i].ip for i in self.cfg.prefiller_indices if not self.nodes[i].headless]
decoder_ips = [self.nodes[i].ip for i in self.cfg.decoder_indices if not self.nodes[i].headless]
cmd = [
"python",
self.proxy_script,
"--host",
self.cur_ip,
"--port",
str(self.proxy_port),
"--prefiller-hosts",
*prefiller_ips,
"--prefiller-ports",
*[str(self.server_port)] * len(prefiller_ips),
"--decoder-hosts",
*decoder_ips,
"--decoder-ports",
*[str(self.server_port)] * len(decoder_ips),
]
logger.info("Launching proxy: %s", " ".join(cmd))
self.process = subprocess.Popen(cmd, env={**os.environ, **self.envs})
return self
def __exit__(self, exc_type, exc, tb):
if not self.process:
return
logger.info("Stopping proxy server...")
self.process.terminate()
try:
self.process.wait(timeout=5)
except subprocess.TimeoutExpired:
self.process.kill()
class MultiNodeConfig:
def __init__(
self,
*,
model: str,
test_name: str,
nodes: list[NodeInfo],
npu_per_node: int,
disaggregated_prefill: dict | None,
benchmark_cases: list[dict],
special_dependencies: dict,
):
self.model = model
self.test_name = test_name
self.nodes = nodes
self.npu_per_node = npu_per_node
self.benchmark_cases = benchmark_cases
self.cur_index = self._resolve_cur_index()
self.cur_node = self.nodes[self.cur_index]
self.special_dependencies = special_dependencies
self.disagg_cfg = DisaggregatedPrefillCfg(disaggregated_prefill, len(nodes)) if disaggregated_prefill else None
master_ip = (
self.disagg_cfg.master_ip_for_node(self.cur_index, self.nodes) if self.disagg_cfg else self.nodes[0].ip
)
self.proxy_port = get_available_port()
self.envs = DistEnvBuilder(
cur_node=self.cur_node,
master_ip=master_ip,
).build()
logger.info("Node %d envs: %s", self.cur_index, self.envs)
self.server_cmd = self._expand_env(self.cur_node.server_cmd)
def _resolve_cur_index(self) -> int:
return resolve_current_node_index([node.ip for node in self.nodes])
def _expand_env(self, cmd: str) -> str:
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
def repl(m):
key = m.group(1) or m.group(2)
return self.envs.get(key, m.group(0))
return pattern.sub(repl, cmd)
@property
def world_size(self) -> int:
return len(self.nodes) * self.npu_per_node
@property
def is_master(self) -> bool:
return self.cur_index == 0
@property
def server_port(self) -> int:
return self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
@property
def master_ip(self) -> str:
return self.nodes[0].ip
@property
def benchmark_endpoint(self) -> tuple[str, int]:
"""
Endpoint used by benchmark clients.
"""
master_ip = self.nodes[0].ip
server_port = self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
if self.disagg_cfg:
return master_ip, self.proxy_port
return master_ip, server_port
class MultiNodeConfigLoader:
"""Load MultiNodeConfig from yaml file."""
DEFAULT_CONFIG_NAME = "DeepSeek-V3.yaml"
@classmethod
def from_yaml(cls, yaml_path: str | None = None) -> MultiNodeConfig:
config = cls._load_yaml(yaml_path)
cls._validate_root(config)
nodes = cls._parse_nodes(config)
benchmarks = cls._parse_benchmarks(config)
return MultiNodeConfig(
model=config["model"],
test_name=config.get("test_name", "untitled_test"),
nodes=nodes,
npu_per_node=config.get("npu_per_node", 16),
disaggregated_prefill=config.get("disaggregated_prefill"),
special_dependencies=config.get("special_dependencies", {}),
benchmark_cases=list(benchmarks.values()),
)
@classmethod
def _load_yaml(cls, yaml_path: str | None) -> dict:
return load_yaml_mapping(
yaml_path,
default_name=cls.DEFAULT_CONFIG_NAME,
default_base_path=DEFAULT_CONFIG_BASE_PATH,
description="config",
)
@staticmethod
def _validate_root(cfg: dict):
required = ["model", "deployment", "num_nodes", "npu_per_node", "benchmarks"]
missing = [k for k in required if k not in cfg]
if missing:
raise KeyError(f"Missing required config fields: {missing}")
@classmethod
def _parse_nodes(cls, cfg: dict) -> list[NodeInfo]:
num_nodes = cfg["num_nodes"]
deployments = cfg["deployment"]
if len(deployments) != num_nodes:
raise AssertionError(f"deployment size ({len(deployments)}) != num_nodes ({num_nodes})")
for idx, deploy in enumerate(deployments):
if deploy.get("envs") is None:
raise KeyError(f"deployment[{idx}].envs is required for multi-node configs")
cluster_ips = cls._resolve_cluster_ips(cfg, num_nodes)
nodes: list[NodeInfo] = []
for idx, deploy in enumerate(deployments):
cmd = deploy.get("server_cmd", "")
envs = deploy["envs"]
nodes.append(
NodeInfo(
index=idx,
ip=cluster_ips[idx],
server_cmd=cmd,
envs=envs,
headless="--headless" in cmd,
)
)
return nodes
@staticmethod
def _parse_benchmarks(cfg: dict) -> dict:
benchmarks = cfg.get("benchmarks") or {}
for name, case in benchmarks.items():
case["case_name"] = name
return benchmarks
@staticmethod
def _resolve_cluster_ips(cfg: dict, num_nodes: int) -> list[str]:
return resolve_cluster_ips(
cfg,
num_nodes,
cluster_hosts_log_message=(
"Using cluster_hosts from config. This typically indicates that your current environment is a "
"non-Kubernetes environment."
),
dns_log_message="Resolving cluster IPs via DNS...",
)

View File

@@ -0,0 +1,207 @@
import json
import logging
import os
import shlex
import subprocess
import sys
from typing import Any
import pytest
import vllm
from tests.e2e.conftest import RemoteOpenAIServer
from tests.e2e.nightly.multi_node.internal_dp.scripts.multi_node_config import (
MultiNodeConfig,
MultiNodeConfigLoader,
ProxyLauncher,
)
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
build_task_entry,
extract_hardware,
filter_environment,
write_results_json,
)
from tools.aisbench import run_aisbench_cases
logger = logging.getLogger(__name__)
_FEATURE_ENVS: dict[str, str] = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
def _extract_dtype(config: MultiNodeConfig) -> str:
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and any node uses --quantization ascend."""
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = any("--quantization ascend" in node.server_cmd for node in config.nodes)
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
def _cmd_to_list(server_cmd: list[str] | str) -> list[str]:
"""Normalize server_cmd to a list of argument strings."""
if isinstance(server_cmd, str):
try:
return shlex.split(server_cmd)
except ValueError:
return server_cmd.split()
return list(server_cmd)
def _extract_server_cmd_value(cmd_list: list[str], flag: str) -> str | None:
"""Return the value following `flag` in a command list, or None."""
try:
idx = cmd_list.index(flag)
return cmd_list[idx + 1]
except (ValueError, IndexError):
return None
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
"""Extract and JSON-parse the value following `flag` in a command list."""
val = _extract_server_cmd_value(cmd_list, flag)
if not val:
return {}
try:
return json.loads(val)
except (json.JSONDecodeError, ValueError):
return {}
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
"""Extract enabled feature names from server_cmd and environment variables."""
cmd_list = _cmd_to_list(server_cmd)
features: list[str] = []
# Features from --additional-config JSON
additional = _parse_json_flag(cmd_list, "--additional-config")
if additional.get("enable_weight_nz_layout"):
features.append("weight_nz_layout")
wp = additional.get("weight_prefetch_config") or {}
if isinstance(wp, dict) and wp.get("enabled"):
features.append("weight_prefetch")
tc = additional.get("torchair_graph_config") or {}
if isinstance(tc, dict) and tc.get("enabled"):
features.append("torchair_graph")
asc = additional.get("ascend_scheduler_config") or {}
if isinstance(asc, dict) and asc.get("enabled"):
features.append("ascend_scheduler")
# Features from --compilation-config JSON
compilation = _parse_json_flag(cmd_list, "--compilation-config")
if compilation.get("cudagraph_mode"):
features.append("aclgraph")
# Features from --speculative-config JSON
speculative = _parse_json_flag(cmd_list, "--speculative-config")
if speculative:
features.append(speculative.get("method", "speculative"))
# Features from direct flags
if "--enable-expert-parallel" in cmd_list:
features.append("expert_parallel")
# Features from environment variables
for env_key, feature_name in _FEATURE_ENVS.items():
val = str(envs.get(env_key, "0"))
if val not in ("0", "", "false", "False"):
features.append(feature_name)
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
features.append("flashcomm2")
return features
def _build_serve_cmd(config: MultiNodeConfig) -> dict[str, Any]:
"""Build serve_cmd dict: pd format for disaggregated, dp format for multi-node."""
if config.disagg_cfg:
pd: dict[str, str] = {}
for node in config.nodes:
idx = node.index
if config.disagg_cfg.is_prefiller(idx):
n = config.disagg_cfg.prefiller_indices.index(idx)
pd[f"prefill-{n}"] = node.server_cmd
elif config.disagg_cfg.is_decoder(idx):
n = config.disagg_cfg.decoder_indices.index(idx)
pd[f"decode-{n}"] = node.server_cmd
return {"pd": pd}
return {"dp": {f"node{node.index}": node.server_cmd for node in config.nodes}}
def _save_benchmark_results_json(config: MultiNodeConfig, results: list[Any]) -> None:
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
runner = os.environ.get("VLLM_CI_RUNNER", "")
# Filter out None benchmark cases; results align with the non-None ones in order
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
tasks = [build_task_entry(key, case_cfg, result) for (key, case_cfg), result in zip(valid_items, results)]
output: dict[str, Any] = {
"model_name": config.model,
"hardware": extract_hardware(runner),
"dtype": _extract_dtype(config),
"feature": _extract_features(config.nodes[0].server_cmd, config.envs),
"vllm_version": vllm.__version__,
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config),
"environment": filter_environment(config.envs),
}
job_name = os.environ.get("BENCHMARK_JOB_NAME", "")
write_results_json(output, job_name=job_name)
@pytest.mark.asyncio
async def test_multi_node() -> None:
config = MultiNodeConfigLoader.from_yaml()
if config.special_dependencies:
for k, v in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{k}=={v}",
]
subprocess.call(command)
with (
ProxyLauncher(
nodes=config.nodes,
disagg_cfg=config.disagg_cfg,
envs=config.envs,
proxy_port=config.proxy_port,
cur_index=config.cur_index,
) as proxy,
RemoteOpenAIServer(
model=config.model,
vllm_serve_args=config.server_cmd,
server_port=config.server_port,
server_host=config.master_ip,
env_dict=config.envs,
auto_port=False,
proxy_port=proxy.proxy_port,
disaggregated_prefill=config.disagg_cfg,
nodes_info=config.nodes,
max_wait_seconds=2800,
) as server,
):
host, port = config.benchmark_endpoint
if config.is_master:
results = run_aisbench_cases(
model=config.model,
port=port,
aisbench_cases=config.benchmark_cases,
host_ip=host,
)
_save_benchmark_results_json(config, results)
else:
# We should keep listening on the master node's server url determining when to exit.
server.hang_until_terminated(f"http://{host}:{config.server_port}/health")

View File

@@ -0,0 +1,28 @@
import os
from tests.e2e.nightly.multi_node.scripts.utils import (
get_all_ipv4,
get_available_port,
get_cluster_ips,
get_net_interface,
setup_logger,
temp_env,
)
DISAGGEGATED_PREFILL_PORT = 5333
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or DEFAULT_CONFIG_BASE_PATH
DEFAULT_SERVER_PORT = 8080
__all__ = [
"CONFIG_BASE_PATH",
"DEFAULT_CONFIG_BASE_PATH",
"DEFAULT_SERVER_PORT",
"DISAGGEGATED_PREFILL_PORT",
"get_all_ipv4",
"get_available_port",
"get_cluster_ips",
"get_net_interface",
"setup_logger",
"temp_env",
]

View File

@@ -0,0 +1 @@

View File

@@ -0,0 +1,128 @@
import json
import logging
from pathlib import Path
from typing import Any
logger = logging.getLogger(__name__)
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
INFRA_ENV_KEYS = {
"HCCL_IF_IP",
"HCCL_SOCKET_IFNAME",
"GLOO_SOCKET_IFNAME",
"TP_SOCKET_IFNAME",
"LOCAL_IP",
"NIC_NAME",
"MASTER_IP",
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
}
PERF_METRIC_RENAME: dict[str, str] = {
"Benchmark Duration": "Benchmark_Duration(BD)",
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
"Input Token Throughput": "Input_Token_Throughput(ITT)",
"Output Token Throughput": "Output_Token_Throughput(OTT)",
"Total Token Throughput": "Total_Token_Throughput(TTT)",
}
def extract_hardware(runner: str) -> str:
runner_lower = runner.lower()
for label in ("a3", "a2"):
if label in runner_lower:
return label.upper()
return runner
def get_vllm_version() -> str:
try:
import vllm
return vllm.__version__
except Exception:
return ""
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
if result == "":
return False
case_type = case_config.get("case_type")
baseline = case_config.get("baseline")
threshold = case_config.get("threshold")
if baseline is None or threshold is None:
return True
if case_type == "accuracy" and isinstance(result, (int, float)):
return abs(float(result) - float(baseline)) <= float(threshold)
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
try:
throughput_val = float(throughput_str.replace("token/s", "").strip())
return throughput_val >= float(threshold) * float(baseline)
except (ValueError, AttributeError):
return False
return True
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
dataset_path = case_config.get("dataset_path", "")
dataset_conf = case_config.get("dataset_conf", "")
if dataset_path:
task_name = dataset_path.split("/", 1)[-1]
elif dataset_conf:
task_name = dataset_conf.split("/")[0]
else:
task_name = case_key
case_type = case_config.get("case_type", "unknown")
metrics: dict[str, float] = {}
if result == "":
pass
elif case_type == "accuracy" and isinstance(result, (int, float)):
metrics["accuracy"] = round(float(result), 4)
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
for metric_name, metric_data in result_json.items():
if not isinstance(metric_data, dict):
continue
total_str = metric_data.get("total", "")
try:
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
except (ValueError, AttributeError):
pass
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
target: dict[str, Any] = {}
if case_config.get("baseline") is not None:
target["baseline"] = case_config["baseline"]
if case_config.get("threshold") is not None:
target["threshold"] = case_config["threshold"]
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
if target:
entry["target"] = target
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
return entry
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
return {key: value for key, value in envs.items() if key not in exclude}
def write_results_json(
output: dict[str, Any],
*,
job_name: str,
output_dir: Path | None = None,
) -> Path:
if output_dir is None:
output_dir = Path("/root/.cache/benchmark_results") / job_name
output_dir.mkdir(parents=True, exist_ok=True)
output_path = output_dir / f"{job_name}.json"
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
logger.info("Benchmark results saved to PVC at %s", output_path)
print(f"Benchmark results saved to PVC at {output_path}")
return output_path

View File

@@ -0,0 +1,174 @@
apiVersion: leaderworkerset.x-k8s.io/v1
kind: LeaderWorkerSet
metadata:
name: {{ lws_name | default("vllm") }}
namespace: vllm-project
spec:
replicas: {{ replicas | default(1) }}
leaderWorkerTemplate:
size: {{ size | default(2) }}
restartPolicy: None
leaderTemplate:
metadata:
labels:
role: leader
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-leader
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: VLLM_ASCEND_VERSION
value: {{ vllm_ascend_ref | default("main") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
ports:
- containerPort: 8080
# readinessProbe:
# tcpSocket:
# port: 8080
# initialDelaySeconds: 15
# periodSeconds: 10
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
workerTemplate:
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-worker
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
---
apiVersion: v1
kind: Service
metadata:
name: {{ lws_name | default("vllm") }}-leader
namespace: vllm-project
spec:
ports:
- name: http
port: 8080
protocol: TCP
targetPort: 8080
selector:
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
role: leader
type: ClusterIP

View File

@@ -0,0 +1,466 @@
#!/bin/bash
set -euo pipefail
# Color definitions
GREEN="\033[0;32m"
BLUE="\033[0;34m"
YELLOW="\033[0;33m"
RED="\033[0;31m"
NC="\033[0m" # No Color
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
else
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
fi
fi
# Configuration
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
# cann and atb environment setup
source /usr/local/Ascend/ascend-toolkit/set_env.sh
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
set +eu
source /usr/local/Ascend/nnal/atb/set_env.sh
set -eu
# Home path for aisbench
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
# Logging configurations
export VLLM_LOGGING_LEVEL="INFO"
# Reduce glog verbosity for mooncake
export GLOG_minloglevel=1
# Set transformers to offline mode to avoid downloading models during tests
export HF_HUB_OFFLINE="1"
# Default is 600s
export VLLM_ENGINE_READY_TIMEOUT_S=1800
# Function to print section headers
print_section() {
echo -e "\n${BLUE}=== $1 ===${NC}"
}
print_failure() {
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
exit 1
}
# Function to print success messages
print_success() {
echo -e "${GREEN}✓ $1${NC}"
}
# Function to print error messages and exit
print_error() {
echo -e "${RED}✗ ERROR: $1${NC}"
exit 1
}
show_vllm_info() {
cd "$WORKSPACE"
echo "Installed vLLM-related Python packages:"
pip list | grep vllm || echo "No vllm packages found."
echo ""
echo "============================"
echo "vLLM Git information"
echo "============================"
cd vllm
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm"
fi
cd ..
echo ""
echo "============================"
echo "vLLM-Ascend Git information"
echo "============================"
cd vllm-ascend
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm-ascend"
fi
cd ..
}
check_npu_info() {
echo "====> Check NPU info"
npu-smi info
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
}
check_and_config() {
echo "====> Configure mirrors and git proxy"
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
}
install_extra_components() {
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
return 1
fi
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
echo "Failed to download custom_ops wheel"
return 1
fi
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
source /usr/local/Ascend/ascend-toolkit/set_env.sh
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
echo "====> Extra components installation completed"
}
checkout_src() {
echo "====> Checkout source code"
mkdir -p "$WORKSPACE"
cd "$WORKSPACE"
pip uninstall -y vllm-ascend || true
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
rm -rf "$WORKSPACE/vllm-ascend"
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
cd "$WORKSPACE/vllm-ascend"
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
if [ -n "$PR_REF" ]; then
git fetch --depth 1 origin "$PR_REF"
git checkout FETCH_HEAD
else
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
git checkout "$VLLM_ASCEND_REF"
fi
git submodule update --init --recursive
fi
}
install_vllm_ascend() {
echo "====> Install vllm-ascend"
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
pip install -e "$WORKSPACE/vllm-ascend"
}
install_aisbench() {
echo "====> Install AISBench benchmark"
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
cp -r /tmp/aisbench-backup "$BENCH_DIR"
cd "$BENCH_DIR"
pip install -e . \
-r requirements/api.txt \
-r requirements/extra.txt
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
}
show_triton_ascend_info() {
echo "====> Check triton ascend info"
clang -v
which bishengir-compile
pip show triton-ascend
}
kill_npu_processes() {
pgrep python3 | xargs -r kill -9
pgrep VLLM | xargs -r kill -9
sleep 4
}
run_tests_with_log() {
set +e
kill_npu_processes
mkdir -p "${LOG_PREFIX}"
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
ret=$?
echo "pytest exit code: ret=${ret}"
set -e
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
if [ $ret -eq 0 ]; then
print_success "All tests passed!"
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
else
echo "Leader: waiting 10s for worker logs..."
sleep 10
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
set +e; aop_pipeline; set -e
fi
local done_file="${LOG_PREFIX}/aop_done"
touch "$done_file"
echo "Leader: notifying workers (${done_file})"
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
exit 1
fi
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
if [ $ret -eq 0 ]; then
echo "Worker: test passed, waiting for leader..."
local wait_timeout=30
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
sleep 1
wait_timeout=$((wait_timeout - 1))
done
fi
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
local release="${LOG_PREFIX}/aop_done"
mkdir -p "$coord"
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
cd "$WORKSPACE/vllm-ascend"
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--coord-dir "${coord}" \
--release-file "${release}"
while [ ! -f "$release" ]; do sleep 5; done
echo "Worker: release signal received, exiting"
exit 1
else
echo "Worker: leader finished successfully, exiting"
fi
fi
}
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
aop_pipeline() {
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
local table="${GOOD_TABLE:-}"
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
local case_name="${BENCHMARK_JOB_NAME#*-}"
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
case_name="${CONFIG_YAML_PATH%.yaml}"
fi
echo "============================================"
echo " AOP Pipeline (Pod) - START"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Case name : ${case_name}"
echo " Rules file : ${rules}"
echo " Table file : ${table}"
echo " Log prefix : ${LOG_PREFIX}"
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
echo "============================================"
# ---- Step 1: Classify ----
echo ""
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
echo " Rules content:"
if [ -f "$rules" ]; then
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
else
echo " (rules file not found)"
fi
echo ""
echo " Pod logs found:"
local found_any=0
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
echo " - ${f} ($(wc -l < "$f") lines)"
found_any=1
fi
done
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
local env_count=0
if [ -f "$rules" ]; then
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
local n
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
n=${n%%[!0-9]*}
echo " Scan ${f}: ${n} matches"
env_count=$((env_count + n))
if [ "$n" -gt 0 ]; then
echo " Matched lines:"
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
fi
fi
done
fi
echo " Classify result: env_count=${env_count}"
if [ "$found_any" -eq 0 ]; then
echo " Decision: no pod logs → SKIP"
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
return 1
fi
if [ "$env_count" -gt 0 ]; then
echo " Decision: env_failure → SKIP"
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
return 1
fi
# ---- Step 2: Check age ----
echo ""
echo "--- [2/3] Check commit age ---"
echo " Looking up: ${case_name}"
local skip_age=0
if [ ! -f "$table" ]; then
echo " Table file not found: ${table}"
echo " Decision: no table → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Only consider success rows
local success_rows
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
if [ -z "$success_rows" ]; then
echo " No success row found for '${case_name}'"
echo " Decision: no success entry → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Pick most recent success row
local best_date=""
while IFS= read -r row; do
local d
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
[ -z "$d" ] && continue
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
best_date="$d"
fi
done <<< "$success_rows"
if [ -z "$best_date" ]; then
echo " No valid date in success rows"
echo " Decision: no date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
local last_ts now_ts age_days
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
echo " Date parse failed: ${best_date}"
echo " Decision: invalid date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
now_ts=$(date +%s)
age_days=$(( (now_ts - last_ts) / 86400 ))
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
if [ "$age_days" -gt 3 ]; then
echo " Decision: old commit (> 3 days) → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# ---- Step 3: Bisect ----
echo ""
echo "--- [3/3] Run bisect ---"
echo " Scene : multi_node"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Bad commit : HEAD"
echo " Name : ${case_name}"
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
echo " Coord dir : ${coord}"
# Wait for all workers to signal ready
echo " Waiting for workers..."
for i in $(seq 1 30); do
local ready_count=0
for f in "${coord}"/worker_ready_*; do
[ -e "$f" ] && ready_count=$((ready_count + 1))
done
echo " [${i}/30] ready workers: ${ready_count}"
if [ "$ready_count" -ge 1 ]; then break; fi
sleep 2
done
cd "$WORKSPACE/vllm-ascend"
local bisect_rc=0
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--good-table "${table}" \
--name "${case_name}" \
--coord-dir "${coord}" || bisect_rc=$?
echo " bisect completed (exit code: ${bisect_rc})"
echo "=== AOP Pipeline (Pod) - END ==="
return 1
}
clear_logs() {
print_section "Clearing logs from previous runs"
rm -fr "$HOME/ascend/log" || true
}
backup_ascend_logs() {
if [ -n "${LOG_PREFIX:-}" ]; then
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
mkdir -p "$dest"
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
echo "Ascend logs backed up to $dest"
fi
}
main() {
trap backup_ascend_logs EXIT
check_npu_info
clear_logs
check_and_config
if [[ "$IS_PR_TEST" == "true" ]]; then
checkout_src
install_vllm_ascend
install_aisbench
fi
show_vllm_info
show_triton_ascend_info
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
install_extra_components
fi
cd "$WORKSPACE/vllm-ascend"
run_tests_with_log
}
main "$@"

View File

@@ -0,0 +1,183 @@
import logging
import os
import socket
import time
from contextlib import contextmanager
from pathlib import Path
from typing import Any
import yaml
logger = logging.getLogger(__name__)
@contextmanager
def temp_env(env_dict: dict[str, Any]):
old_env = {}
for key, value in env_dict.items():
old_env[key] = os.environ.get(key)
os.environ[key] = str(value)
try:
yield
finally:
for key, value in old_env.items():
if value is None:
os.environ.pop(key, None)
else:
os.environ[key] = value
def setup_logger() -> None:
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
def load_yaml_mapping(
yaml_path: str | None,
*,
default_name: str,
default_base_path: str,
description: str,
) -> dict[str, Any]:
if not yaml_path:
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
path = Path(yaml_path)
if not path.is_absolute() and not path.exists():
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
path = Path(base_path) / yaml_path
logger.info("Loading %s yaml: %s", description, path)
with path.open(encoding="utf-8") as f:
data = yaml.safe_load(f)
if not isinstance(data, dict):
raise TypeError(f"{description} must be a mapping: {path}")
return data
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
def resolve(dns: str) -> str:
delay = base_delay
for attempt in range(retries):
try:
return socket.gethostbyname(dns)
except socket.gaierror:
if attempt == retries - 1:
raise
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError(f"Unable to resolve DNS: {dns}")
return resolve
def get_cluster_dns_list(world_size: int) -> list[str]:
if world_size < 1:
raise ValueError(f"world_size must be >= 1, got {world_size}")
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
if not leader_dns:
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
parts = leader_dns.split(".")
if len(parts) < 3:
raise ValueError(f"invalid leader DNS format: {leader_dns}")
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
return [leader_dns, *worker_dns_list]
def get_cluster_ips(world_size: int = 2) -> list[str]:
resolver = dns_resolver()
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
def resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
explicit_cluster_ips: list[str] | None = None,
*,
cluster_hosts_log_message: str | None = None,
dns_log_message: str = "Resolving cluster IPs via DNS...",
) -> list[str]:
if explicit_cluster_ips is not None:
if len(explicit_cluster_ips) != num_nodes:
raise AssertionError("cluster_ips size mismatch")
return explicit_cluster_ips
cluster_hosts = raw_config.get("cluster_hosts")
if cluster_hosts:
if cluster_hosts_log_message:
logger.info(cluster_hosts_log_message)
if len(cluster_hosts) != num_nodes:
raise AssertionError("cluster_hosts size mismatch")
return list(cluster_hosts)
logger.info(dns_log_message)
return get_cluster_ips(num_nodes)
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
for port in range(start_port, end_port):
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
try:
s.bind(("", port))
return port
except OSError:
continue
raise RuntimeError("No available port found")
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
delay = base_delay
for attempt in range(retries):
try:
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
s.connect(("8.8.8.8", 80))
return s.getsockname()[0]
except Exception:
try:
return socket.gethostbyname(socket.gethostname())
except Exception:
if attempt == retries - 1:
raise RuntimeError("Failed to determine local IP address")
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError("Failed to determine local IP address")
def get_net_interface(ip: str | None = None) -> str:
import psutil
if ip is None:
ip = get_cur_ip()
for iface, addrs in psutil.net_if_addrs().items():
for addr in addrs:
if addr.family == socket.AF_INET and addr.address == ip:
return iface
raise RuntimeError(f"No network interface found for IP {ip}")
def get_all_ipv4() -> list[str]:
ipv4s = {"127.0.0.1"}
hostname = socket.gethostname()
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
ipv4s.add(info[4][0])
return list(ipv4s)
def resolve_current_node_index(cluster_ips: list[str]) -> int:
worker_index = os.environ.get("LWS_WORKER_INDEX")
if worker_index:
return int(worker_index)
local_ips = set(get_all_ipv4())
for index, ip in enumerate(cluster_ips):
if ip in local_ips:
return index
raise RuntimeError("Unable to determine current node index")