init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1 @@
"""External DP nightly test package."""

View File

@@ -0,0 +1,308 @@
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
benchmarks:
perf_1M_1k_prefix99_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1
batch_size: 1
request_rate: 0
baseline: 0
threshold: 0.95
perf_1M_1k_prefix99:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 11.93
threshold: 0.95

View File

@@ -0,0 +1,324 @@
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 0
tp_size: 2
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 8
tp_size: 2
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_BUFFSIZE: "1800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
perf_TPOT50_128k_1_prefix_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1
batch_size: 4
request_rate: 0
baseline: 0
threshold: 0.95
perf_TPOT50_128k_1_prefix90:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1024
batch_size: 48
request_rate: 1
baseline: 869.13
threshold: 0.95
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,176 @@
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 16
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "1500"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --trust-remote-code
- --block-size
- "32"
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.9"
- --quantization
- "ascend"
- --enforce-eager
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "240"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,335 @@
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 4
npu_per_node: 16
special_dependencies:
transformers: "5.2.0"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0, 1]
decoder: [2, 3]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 4
tp_size: 4
dp_address: "${NODE_2_IP}"
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
VLLM_TORCH_PROFILER_WITH_STACK: "0"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_INTRA_PCIE_ENABLE: "1"
HCCL_INTRA_ROCE_ENABLE: "0"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
templates:
- node_index: 0
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 1
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 2
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 3
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1500
batch_size: 40
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,161 @@
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
model: "Eco-Tech/Kimi-K2.6-w4a8"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_1_IP}"
env_common: &env_common
SERVER_PORT: "${PORT}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_SERVER_DEV_MODE: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_USE_MODELSCOPE: "true"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "512"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --safetensors-load-strategy
- 'prefetch'
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --enforce-eager
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --safetensors-load-strategy
- 'prefetch'
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "256"
- --max-num-seqs
- "16"
- --trust-remote-code
- --gpu-memory-utilization
- "0.92"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 60
max_out_len: 1024
batch_size: 15
request_rate: 0.4
baseline: 347.4475
threshold: 0.97
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 93.33
threshold: 10
temperature: 1.0
top_p: 1

View File

@@ -0,0 +1,197 @@
test_name: "Minimax_m2.7_in3_5_tpot50"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
PYTHONHASHSEED: "0"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "2048"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --enforce-eager
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "55880",
"engine_id": "0",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}} }'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --async-scheduling
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "56900",
"engine_id": "1",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}}'
benchmarks:
perf_warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2
max_out_len: 1
batch_size: 2
request_rate: 0
baseline: 0
threshold: 0.97
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1024
batch_size: 40
request_rate: 0
baseline: 717.5332
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
temperature: 1
top_p: 1
top_k: 40
ignore_eos: false

View File

@@ -0,0 +1,360 @@
# External DP Config Template
This document shows how to write YAML configs consumed by
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
`server_cmd_template` contains only the arguments after
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
`model` automatically.
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
and roles are selected by `routing.groups`.
## Generic DP Template
Use this template for generic external data parallel serving. This mode uses
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
independent vLLM instances instead of external DP rank arguments.
```yaml
test_name: "test Qwen3-30B-A3B generic external dp"
model: "Qwen/Qwen3-30B-A3B"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "generic_dp"
groups:
worker: [0, 1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 1
dp_address: "${NODE_0_IP}"
templates:
- node_index: 0
envs: &generic_env
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "1024"
SERVER_PORT: "${PORT}"
server_cmd_template: &generic_server_cmd
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --max-model-len
- "4096"
- --trust-remote-code
- --enable-expert-parallel
- node_index: 1
envs:
<<: *generic_env
server_cmd_template: *generic_server_cmd
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 16
batch_size: 1
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
num_prompts: 4
max_out_len: 16
batch_size: 1
baseline: 0
threshold: 100
```
## Disaggregated Prefill Template
Use this template for PD disaggregation. `routing.groups` decides which config
entries run as prefillers or decoders. The framework derives the PD proxy script
from `routing.type`, so do not write `proxy_*` fields in YAML.
```yaml
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0]
decoder: [1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
HCCL_OP_
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "256"
SERVER_PORT: "${PORT}"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
- node_index: 1
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
max_out_len: 128
batch_size: 4
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 48
batch_size: 4
baseline: 0
threshold: 100
```
## Field Notes
- `test_name`: Human-readable test name. It is also used when writing benchmark
result metadata.
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
- `num_nodes`: Number of config entries and templates expected.
- `npu_per_node`: Device capacity validation for each node.
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
needs fixed hosts.
- `routing.type`: Supported values are `generic_dp` and
`disaggregated_prefill`.
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
`kv_consumer` for decoder templates.
- `config[].dp_size`: Global DP size for this DP group.
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
- `config[].dp_rank_start`: First global DP rank owned by this node.
- `config[].dp_address`: DP master address. For one global DP group, use
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
address for prefiller nodes and the decoder master address for decoder nodes.
- `templates`: One template per config entry. The framework expands one command
per local DP rank.
The framework injects distributed network envs at startup:
```text
HCCL_IF_IP
HCCL_SOCKET_IFNAME
GLOO_SOCKET_IFNAME
TP_SOCKET_IFNAME
LOCAL_IP
NIC_NAME
MASTER_IP
```
The framework also derives proxy metadata from `routing.type`:
```text
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
```
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
for benchmark requests.
## Template Variables
The following variables are available in `envs` and `server_cmd_template`:
```text
${MODEL}
${PORT_START}
${PORT}
${DP_SIZE}
${DP_SIZE_LOCAL}
${DP_RANK_START}
${DP_RANK}
${LOCAL_RANK}
${TP_SIZE}
${CP_SIZE}
${SP_SIZE}
${PP_SIZE}
${DP_ADDRESS}
${DP_RPC_PORT}
${VISIBLE_DEVICES}
${NODE_INDEX}
${CONFIG_INDEX}
${NODE_0_IP}, ${NODE_1_IP}, ...
${LOCAL_IP}
${MASTER_IP}
${LWS_WORKER_INDEX}
```
Command arguments can also reference rendered environment variables with
shell-style `$VARNAME`, for example:
```yaml
envs:
SERVER_PORT: "${PORT}"
server_cmd_template:
- --port
- $SERVER_PORT
```
## Checks Before Running
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
- Make sure each config index is assigned to exactly one routing group.
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
`--enable-expert-parallel`.
- Set `--max-model-len` large enough for benchmark input tokens plus
`max_out_len`.

View File

@@ -0,0 +1 @@
"""External DP nightly test helpers."""

View File

@@ -0,0 +1,449 @@
import logging
import os
from dataclasses import dataclass, field
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.scripts.utils import (
load_yaml_mapping,
resolve_cluster_ips,
)
from tests.e2e.nightly.multi_node.scripts.utils import (
resolve_current_node_index as resolve_node_index,
)
logger = logging.getLogger(__name__)
ROUTING_GENERIC_DP = "generic_dp"
ROUTING_DISAGGREGATED_PREFILL = "disaggregated_prefill"
PROXY_SCRIPT_BY_ROUTING_TYPE = {
ROUTING_GENERIC_DP: "examples/external_online_dp/dp_load_balance_proxy_server.py",
ROUTING_DISAGGREGATED_PREFILL: "examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
}
CLUSTER_PLACEHOLDER_RE = re.compile(r"\$\{(NODE_(\d+)_IP|LOCAL_IP|MASTER_IP|LWS_WORKER_INDEX)\}")
@dataclass(frozen=True)
class RoutingConfig:
"""Proxy routing metadata shared by all external DP ranks."""
type: str
proxy_node_index: int
proxy_host: str
proxy_port: int
proxy_script: str
groups: dict[str, list[int]]
@dataclass(frozen=True)
class NodeInfo:
"""Per-node external DP server topology loaded from one config entry."""
ip: str
port_start: int
dp_rpc_port: int
dp_size: int
dp_size_local: int
dp_rank_start: int
tp_size: int
dp_address: str
cp_size: int = 1
sp_size: int = 1
pp_size: int = 1
@property
def devices_per_rank(self) -> int:
return self.tp_size * self.cp_size * self.sp_size * self.pp_size
@property
def devices_per_node(self) -> int:
return self.dp_size_local * self.devices_per_rank
@dataclass(frozen=True)
class NodeTemplate:
"""Per-node env and argument template for launching vLLM servers."""
envs: dict[str, Any]
server_cmd_template: list[str]
@dataclass(frozen=True)
class RankInfo:
"""One concrete vLLM server rank expanded from a node config."""
node_index: int
role: str
local_rank: int
dp_rank: int
host: str
port: int
visible_devices: str
dp_size: int
dp_size_local: int
tp_size: int
cp_size: int
sp_size: int
pp_size: int
dp_address: str
dp_rpc_port: int
port_start: int
@dataclass(frozen=True)
class ExternalDPConfig:
"""Top-level external DP test config after YAML anchors are merged."""
test_name: str
model: str
num_nodes: int
npu_per_node: int
cluster_hosts: list[str] | None
cluster_ips: list[str]
routing: RoutingConfig
nodes: list[NodeInfo]
launch_templates: list[NodeTemplate]
benchmark_cases: list[dict[str, Any]] = field(default_factory=list)
special_dependencies: dict[str, str] = field(default_factory=dict)
@property
def is_disaggregated_prefill(self) -> bool:
return self.routing.type == ROUTING_DISAGGREGATED_PREFILL
def replace_cluster_placeholders(
value: Any,
*,
cluster_ips: list[str],
local_ip: str | None = None,
current_node_index: int | None = None,
) -> Any:
if isinstance(value, dict):
return {
key: replace_cluster_placeholders(
val,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for key, val in value.items()
}
if isinstance(value, list):
return [
replace_cluster_placeholders(
item,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for item in value
]
if not isinstance(value, str):
return value
def repl(match: re.Match[str]) -> str:
token = match.group(1)
node_index = match.group(2)
if node_index is not None:
idx = int(node_index)
if idx >= len(cluster_ips):
raise ValueError(f"Cluster placeholder ${{{token}}} is out of range")
return cluster_ips[idx]
if token == "MASTER_IP":
return cluster_ips[0]
if token == "LOCAL_IP":
if local_ip is None:
return match.group(0)
return local_ip
if token == "LWS_WORKER_INDEX":
if current_node_index is None:
return os.environ.get("LWS_WORKER_INDEX", match.group(0))
return str(current_node_index)
return match.group(0)
return CLUSTER_PLACEHOLDER_RE.sub(repl, value)
def resolve_current_node_index(config: ExternalDPConfig) -> int:
return resolve_node_index(config.cluster_ips)
class ExternalDPConfigLoader:
"""Load, normalize, and validate external DP YAML files."""
@classmethod
def from_yaml(
cls,
yaml_path: str | None = None,
*,
cluster_ips: list[str] | None = None,
) -> ExternalDPConfig:
raw_config = cls._load_yaml(yaml_path)
cls._validate_root(raw_config)
num_nodes = int(raw_config["num_nodes"])
resolved_cluster_ips = cls._resolve_cluster_ips(raw_config, num_nodes, cluster_ips)
model = str(raw_config["model"])
routing = cls._parse_routing(raw_config["routing"], resolved_cluster_ips)
nodes = cls._parse_nodes(raw_config, resolved_cluster_ips)
launch_templates = cls._parse_templates(raw_config)
benchmark_cases = cls._parse_benchmarks(raw_config)
config = ExternalDPConfig(
test_name=str(raw_config.get("test_name", "external_dp_test")),
model=model,
num_nodes=num_nodes,
npu_per_node=int(raw_config["npu_per_node"]),
cluster_hosts=raw_config.get("cluster_hosts"),
cluster_ips=resolved_cluster_ips,
routing=routing,
nodes=nodes,
launch_templates=launch_templates,
benchmark_cases=benchmark_cases,
special_dependencies=dict(raw_config.get("special_dependencies", {})),
)
cls._validate_config(config)
return config
@staticmethod
def _load_yaml(yaml_path: str | None) -> dict[str, Any]:
default_config_name = "GLM5_1-W8A8-EP-external.yaml"
default_config_base_path = "tests/e2e/nightly/multi_node/external_dp/config/"
return load_yaml_mapping(
yaml_path,
default_name=default_config_name,
default_base_path=default_config_base_path,
description="external DP config",
)
@staticmethod
def _validate_root(config: dict[str, Any]) -> None:
required = ["model", "num_nodes", "npu_per_node", "routing", "config", "templates", "benchmarks"]
missing = [key for key in required if key not in config]
if missing:
raise KeyError(f"Missing required external DP config fields: {missing}")
if int(config["num_nodes"]) <= 0:
raise ValueError("num_nodes must be greater than 0")
@staticmethod
def _resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
cluster_ips: list[str] | None,
) -> list[str]:
return resolve_cluster_ips(
raw_config,
num_nodes,
cluster_ips,
dns_log_message="Resolving external DP cluster IPs via LWS DNS",
)
@staticmethod
def _parse_routing(raw_routing: dict[str, Any], cluster_ips: list[str]) -> RoutingConfig:
routing_type = str(raw_routing["type"])
if routing_type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {routing_type}")
proxy_node_index = 0
proxy_port = 1999
if proxy_node_index >= len(cluster_ips) or proxy_node_index < 0:
raise ValueError("routing.proxy_node_index out of range")
local_ip = cluster_ips[proxy_node_index]
routing = replace_cluster_placeholders(
raw_routing,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=proxy_node_index,
)
return RoutingConfig(
type=routing_type,
proxy_node_index=proxy_node_index,
proxy_host=local_ip,
proxy_port=proxy_port,
proxy_script=PROXY_SCRIPT_BY_ROUTING_TYPE[routing_type],
groups={
str(name): [int(index) for index in indices] for name, indices in routing.get("groups", {}).items()
},
)
@staticmethod
def _parse_nodes(raw_config: dict[str, Any], cluster_ips: list[str]) -> list[NodeInfo]:
nodes: list[NodeInfo] = []
for index, raw_node in enumerate(raw_config["config"]):
raw_node_index = raw_node.get("node_index")
if raw_node_index is not None and int(raw_node_index) != index:
raise ValueError(f"config[{index}].node_index must equal {index}")
node = replace_cluster_placeholders(
raw_node,
cluster_ips=cluster_ips,
local_ip=cluster_ips[index],
current_node_index=index,
)
nodes.append(
NodeInfo(
ip=cluster_ips[index],
port_start=int(node["port_start"]),
dp_rpc_port=int(node["dp_rpc_port"]),
dp_size=int(node.get("dp_size", 1)),
dp_size_local=int(node.get("dp_size_local", 1)),
dp_rank_start=int(node.get("dp_rank_start", 0)),
tp_size=int(node.get("tp_size", 1)),
cp_size=int(node.get("cp_size", 1)),
sp_size=int(node.get("sp_size", 1)),
dp_address=str(node["dp_address"]),
pp_size=int(node.get("pp_size", 1)),
)
)
return nodes
@staticmethod
def _parse_templates(raw_config: dict[str, Any]) -> list[NodeTemplate]:
templates: list[NodeTemplate] = []
for index, raw_template in enumerate(raw_config["templates"]):
envs = raw_template.get("envs")
server_cmd_template = raw_template.get("server_cmd_template")
if envs is None or server_cmd_template is None:
raise KeyError(f"templates[{index}] must contain envs and server_cmd_template")
if not isinstance(server_cmd_template, list):
raise TypeError(f"templates[{index}].server_cmd_template must be a list")
templates.append(
NodeTemplate(
envs=dict(envs),
server_cmd_template=[str(arg) for arg in server_cmd_template],
)
)
return templates
@staticmethod
def _parse_benchmarks(raw_config: dict[str, Any]) -> list[dict[str, Any]]:
benchmark_cases: list[dict[str, Any]] = []
for name, case in (raw_config.get("benchmarks") or {}).items():
case_with_name = dict(case)
case_with_name["case_name"] = name
benchmark_cases.append(case_with_name)
return benchmark_cases
@classmethod
def _validate_config(cls, config: ExternalDPConfig) -> None:
cls._validate_config_sizes(config)
cls._validate_routing(config)
cls._validate_node_parallel_config(config)
@staticmethod
def _validate_config_sizes(config: ExternalDPConfig) -> None:
if len(config.nodes) != config.num_nodes:
raise AssertionError(f"config size ({len(config.nodes)}) != num_nodes ({config.num_nodes})")
if len(config.launch_templates) != config.num_nodes:
raise AssertionError(f"templates size ({len(config.launch_templates)}) != num_nodes ({config.num_nodes})")
if config.cluster_hosts and len(config.cluster_hosts) != config.num_nodes:
raise AssertionError("cluster_hosts size mismatch")
@staticmethod
def _validate_routing(config: ExternalDPConfig) -> None:
if config.routing.type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {config.routing.type}")
groups = config.routing.groups
if config.routing.type == ROUTING_GENERIC_DP and not groups.get("worker"):
raise ValueError("generic_dp routing requires routing.groups.worker")
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL and (
not groups.get("prefiller") or not groups.get("decoder")
):
raise ValueError("disaggregated_prefill routing requires prefiller and decoder groups")
seen_group_indices: dict[int, str] = {}
for group_name, indices in groups.items():
for index in indices:
if index < 0 or index >= config.num_nodes:
raise ValueError(f"routing.groups.{group_name} index out of range: {index}")
if index in seen_group_indices:
raise ValueError(f"node index {index} appears in both {seen_group_indices[index]} and {group_name}")
seen_group_indices[index] = group_name
if config.routing.proxy_node_index < 0 or config.routing.proxy_node_index >= config.num_nodes:
raise ValueError("routing.proxy_node_index out of range")
@staticmethod
def _validate_node_parallel_config(config: ExternalDPConfig) -> None:
for node_index, node in enumerate(config.nodes):
parallel_sizes = {
"dp_size": node.dp_size,
"dp_size_local": node.dp_size_local,
"tp_size": node.tp_size,
"cp_size": node.cp_size,
"sp_size": node.sp_size,
"pp_size": node.pp_size,
}
invalid_sizes = {name: value for name, value in parallel_sizes.items() if value < 1}
if invalid_sizes:
raise ValueError(f"node {node_index} parallel sizes must be >= 1: {invalid_sizes}")
if node.dp_rank_start < 0:
raise ValueError(f"node {node_index} dp_rank_start must be >= 0")
if node.devices_per_node > config.npu_per_node:
raise ValueError(
f"node {node_index} uses {node.devices_per_node} NPUs, but npu_per_node is {config.npu_per_node}"
)
if node.dp_rank_start + node.dp_size_local > node.dp_size:
raise ValueError(f"node {node_index} dp rank range exceeds dp_size")
class RankResolver:
"""Expand node-level configs into concrete vLLM server ranks."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def resolve(self) -> list[RankInfo]:
role_by_node_index = self._role_by_node_index()
ranks: list[RankInfo] = []
for node_index, node_info in enumerate(self.config.nodes):
role = role_by_node_index[node_index]
ranks.extend(self._expand_node(node_index, role, node_info))
return ranks
def _role_by_node_index(self) -> dict[int, str]:
role_by_index: dict[int, str] = {}
for role, node_indices in self.config.routing.groups.items():
for index in node_indices:
role_by_index[index] = role
missing = [index for index in range(self.config.num_nodes) if index not in role_by_index]
if missing:
raise ValueError(f"routing.groups does not assign role for node indices: {missing}")
return role_by_index
@staticmethod
def _expand_node(node_index: int, role: str, node_info: NodeInfo) -> list[RankInfo]:
ranks: list[RankInfo] = []
for local_rank in range(node_info.dp_size_local):
dp_rank = node_info.dp_rank_start + local_rank
port = node_info.port_start + local_rank
device_range = range(
local_rank * node_info.devices_per_rank,
(local_rank + 1) * node_info.devices_per_rank,
)
visible_devices = ",".join(str(device) for device in device_range)
ranks.append(
RankInfo(
node_index=node_index,
role=role,
local_rank=local_rank,
dp_rank=dp_rank,
host=node_info.ip,
port=port,
visible_devices=visible_devices,
dp_size=node_info.dp_size,
dp_size_local=node_info.dp_size_local,
tp_size=node_info.tp_size,
cp_size=node_info.cp_size,
sp_size=node_info.sp_size,
pp_size=node_info.pp_size,
dp_address=node_info.dp_address,
dp_rpc_port=node_info.dp_rpc_port,
port_start=node_info.port_start,
)
)
return ranks

View File

@@ -0,0 +1,435 @@
import logging
import os
import subprocess
import sys
import time
from collections.abc import Iterable
from dataclasses import dataclass
from pathlib import Path
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ROUTING_GENERIC_DP,
ExternalDPConfig,
NodeTemplate,
RankInfo,
replace_cluster_placeholders,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
format_server_cmd,
is_http_ready,
start_logged_process,
terminate_process_tree,
wait_http_ready,
wait_http_unready,
)
from tests.e2e.nightly.multi_node.scripts.utils import get_net_interface
logger = logging.getLogger(__name__)
SERVER_READY_TIMEOUT_SECONDS = 3600
TEMPLATE_VAR_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
ENV_VAR_RE = re.compile(r"(?<!\$)\$([A-Za-z_][A-Za-z0-9_]*)")
@dataclass(frozen=True)
class ServerCommand:
"""Rendered command, env, and printable command line."""
cmd: list[str]
env: dict[str, str]
display_cmd: str
RankProcess = tuple[subprocess.Popen, RankInfo, Path]
class ServerCommandBuilder:
"""Render rank templates into vLLM serve commands."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def build(self, rank: RankInfo, template: NodeTemplate) -> ServerCommand:
variables = self._build_variables(rank)
rendered_env = self._render_envs(template.envs, rank, variables)
rendered_args = [
self._render_string(
arg,
rank=rank,
braced_variables=variables,
unbraced_variables=rendered_env,
allow_missing_unbraced=False,
)
for arg in template.server_cmd_template
]
cmd = ["vllm", "serve", self.config.model, *rendered_args]
env = {key: str(value) for key, value in rendered_env.items()}
display_cmd = format_server_cmd(cmd, env)
logger.info(
"External DP server command node=%s rank=%s: %s",
rank.node_index,
rank.local_rank,
display_cmd,
)
return ServerCommand(cmd=cmd, env=env, display_cmd=display_cmd)
def build_all(self, ranks: list[RankInfo]) -> list[ServerCommand]:
return [self.build(rank, self.config.launch_templates[rank.node_index]) for rank in ranks]
def _build_variables(self, rank: RankInfo) -> dict[str, str]:
return {
"MODEL": self.config.model,
"PORT_START": str(rank.port_start),
"PORT": str(rank.port),
"DP_SIZE": str(rank.dp_size),
"DP_SIZE_LOCAL": str(rank.dp_size_local),
"DP_RANK_START": str(rank.dp_rank - rank.local_rank),
"DP_RANK": str(rank.dp_rank),
"LOCAL_RANK": str(rank.local_rank),
"TP_SIZE": str(rank.tp_size),
"CP_SIZE": str(rank.cp_size),
"SP_SIZE": str(rank.sp_size),
"PP_SIZE": str(rank.pp_size),
"DP_ADDRESS": rank.dp_address,
"DP_RPC_PORT": str(rank.dp_rpc_port),
"VISIBLE_DEVICES": rank.visible_devices,
"NODE_INDEX": str(rank.node_index),
"CONFIG_INDEX": str(rank.node_index),
}
def _render_envs(
self,
envs: dict[str, Any],
rank: RankInfo,
variables: dict[str, str],
) -> dict[str, str]:
rendered_envs: dict[str, str] = {}
for key, value in envs.items():
if isinstance(value, str):
value = self._render_string(
value,
rank=rank,
braced_variables=variables,
unbraced_variables={**os.environ, **rendered_envs},
allow_missing_unbraced=True,
)
rendered_envs[str(key)] = str(value)
return rendered_envs
def _render_string(
self,
value: str,
*,
rank: RankInfo,
braced_variables: dict[str, str],
unbraced_variables: dict[str, str],
allow_missing_unbraced: bool,
) -> str:
value = replace_cluster_placeholders(
value,
cluster_ips=self.config.cluster_ips,
local_ip=rank.host,
current_node_index=rank.node_index,
)
value = self._render_variables(
value,
braced_variables,
pattern=TEMPLATE_VAR_RE,
allow_missing=False,
)
return self._render_variables(
value,
unbraced_variables,
pattern=ENV_VAR_RE,
allow_missing=allow_missing_unbraced,
)
@staticmethod
def _render_variables(
value: str,
variables: dict[str, str],
*,
pattern: re.Pattern[str],
allow_missing: bool,
) -> str:
def repl(match: re.Match[str]) -> str:
key = match.group(1)
if key not in variables:
if allow_missing:
return ""
raise KeyError(f"Unknown external DP template variable: {key}")
return variables[key]
return pattern.sub(repl, value)
class ExternalDPServerManager:
"""Start and stop the external DP ranks owned by the current node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.command_builder = ServerCommandBuilder(config)
self.dist_envs = build_dist_envs(
config.cluster_ips[current_node_index],
config.cluster_ips[0],
)
self.rank_processes: list[RankProcess] = []
def start_current_node(self) -> None:
local_ranks = [rank for rank in self.ranks if rank.node_index == self.current_node_index]
logger.info("Starting %d external DP ranks on node %d", len(local_ranks), self.current_node_index)
try:
for rank in local_ranks:
template = self.config.launch_templates[rank.node_index]
template = type(template)(
envs={**template.envs, **self.dist_envs},
server_cmd_template=template.server_cmd_template,
)
server_cmd = self.command_builder.build(rank, template)
log_file = self._rank_log_file(rank)
process = start_logged_process(server_cmd.cmd, server_cmd.env, log_file)
self.rank_processes.append((process, rank, log_file))
wait_ranks_ready(
local_ranks,
timeout=SERVER_READY_TIMEOUT_SECONDS,
rank_processes=self.rank_processes,
)
except Exception:
self.cleanup()
raise
def __enter__(self):
self.start_current_node()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
for process, rank, _log_file in reversed(self.rank_processes):
logger.info(
"Stopping external DP rank node=%d rank=%d pid=%d",
rank.node_index,
rank.local_rank,
process.pid,
)
terminate_process_tree(process.pid)
self.rank_processes.clear()
def _rank_log_file(self, rank: RankInfo) -> Path:
return self.log_root / f"node-{rank.node_index}" / f"rank-{rank.local_rank}.log"
class ExternalDPProxyLauncher:
"""Launch the external DP proxy on the configured proxy node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.pid: int | None = None
def start(self) -> None:
if self.current_node_index != self.config.routing.proxy_node_index:
logger.info("Current node is not proxy node, skip launching external DP proxy")
return
cmd = build_proxy_server_cmd(self.config, self.ranks)
log_file = self.log_root / f"node-{self.current_node_index}" / "proxy.log"
process = start_logged_process(cmd, {}, log_file)
self.pid = process.pid
logger.info("External DP proxy launched: %s", proxy_server_health_url(self.config))
def wait_ready(self, timeout: int = 300) -> None:
wait_http_ready(proxy_server_health_url(self.config), timeout=timeout)
logger.info("External DP proxy ready: %s", proxy_server_health_url(self.config))
def __enter__(self):
self.start()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
if self.pid is None:
return
logger.info("Stopping external DP proxy pid=%d", self.pid)
terminate_process_tree(self.pid)
self.pid = None
def build_all_server_commands(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[ServerCommand]:
return ServerCommandBuilder(config).build_all(ranks)
def build_dist_envs(cur_ip: str, master_ip: str) -> dict[str, str]:
nic_name = get_net_interface(cur_ip)
return {
"HCCL_IF_IP": cur_ip,
"HCCL_SOCKET_IFNAME": nic_name,
"GLOO_SOCKET_IFNAME": nic_name,
"TP_SOCKET_IFNAME": nic_name,
"LOCAL_IP": cur_ip,
"NIC_NAME": nic_name,
"MASTER_IP": master_ip,
}
def build_proxy_server_cmd(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[str]:
routing = config.routing
cmd = [sys.executable, routing.proxy_script, "--host", routing.proxy_host, "--port", str(routing.proxy_port)]
if routing.type == ROUTING_GENERIC_DP:
worker_ranks = [rank for rank in ranks if rank.role == "worker"]
if not worker_ranks:
raise ValueError("generic_dp proxy requires worker ranks")
cmd.extend(["--dp-hosts", *[rank.host for rank in worker_ranks]])
cmd.extend(["--dp-ports", *[str(rank.port) for rank in worker_ranks]])
return cmd
if routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefiller_ranks = [rank for rank in ranks if rank.role == "prefiller"]
decoder_ranks = [rank for rank in ranks if rank.role == "decoder"]
if not prefiller_ranks or not decoder_ranks:
raise ValueError("disaggregated_prefill proxy requires prefiller and decoder ranks")
cmd.extend(["--prefiller-hosts", *[rank.host for rank in prefiller_ranks]])
cmd.extend(["--prefiller-ports", *[str(rank.port) for rank in prefiller_ranks]])
cmd.extend(["--decoder-hosts", *[rank.host for rank in decoder_ranks]])
cmd.extend(["--decoder-ports", *[str(rank.port) for rank in decoder_ranks]])
return cmd
raise ValueError(f"Unsupported routing.type: {routing.type}")
def proxy_server_health_url(config: ExternalDPConfig) -> str:
return f"http://{config.routing.proxy_host}:{config.routing.proxy_port}/healthcheck"
def rank_health_url(rank: RankInfo) -> str:
return f"http://{rank.host}:{rank.port}/health"
def master_rank_health_url(ranks: list[RankInfo]) -> str:
for rank in ranks:
if rank.node_index == 0 and rank.local_rank == 0:
return rank_health_url(rank)
raise RuntimeError("External DP master rank was not found")
def rank_label(rank: RankInfo) -> str:
return f"node={rank.node_index} rank={rank.local_rank} role={rank.role} url={rank_health_url(rank)}"
def format_http_status(label: str, url: str) -> str:
status = "ready" if is_http_ready(url, timeout=1.0) else "waiting"
return f"{label}={status} url={url}"
def _format_rank_statuses(
ranks: list[RankInfo],
rank_ready: dict[RankInfo, bool],
) -> str:
parts = []
for rank in ranks:
status = "ready" if rank_ready[rank] else "waiting"
parts.append(f" {rank_label(rank)} status={status}")
return "\n".join(parts)
def _raise_if_rank_process_exited(rank_processes: list[RankProcess] | None) -> None:
if not rank_processes:
return
exited = []
for process, rank, log_file in rank_processes:
returncode = process.poll()
if returncode is not None:
exited.append(f"{rank_label(rank)} pid={process.pid} returncode={returncode} log={log_file}")
if exited:
raise RuntimeError("External DP rank process exited before ready: " + "; ".join(exited))
def wait_ranks_ready(
ranks: Iterable[RankInfo],
timeout: int,
rank_processes: list[RankProcess] | None = None,
) -> None:
ranks = list(ranks)
rank_ready = {rank: False for rank in ranks}
deadline = time.monotonic() + timeout
last_log_time = 0.0
while True:
_raise_if_rank_process_exited(rank_processes)
all_ready = True
unhealthy_after_ready = []
for rank in ranks:
is_ready = is_http_ready(rank_health_url(rank), timeout=1.0)
if is_ready:
if not rank_ready[rank]:
logger.info("[READY] External DP rank %s", rank_label(rank))
rank_ready[rank] = True
continue
all_ready = False
if rank_ready[rank]:
unhealthy_after_ready.append(rank)
if unhealthy_after_ready:
failed = "; ".join(rank_label(rank) for rank in unhealthy_after_ready)
raise RuntimeError(f"External DP rank became unhealthy after ready: {failed}")
if all_ready:
return
now = time.monotonic()
if now - last_log_time >= 30:
logger.info(
"Polling external DP ranks: ready=%d/%d\n%s",
sum(rank_ready.values()),
len(ranks),
_format_rank_statuses(ranks, rank_ready),
)
last_log_time = now
if now >= deadline:
pending = [rank for rank in ranks if not rank_ready[rank]]
pending_labels = "; ".join(rank_label(rank) for rank in pending)
raise TimeoutError(f"Timed out waiting for external DP ranks ready: {pending_labels}")
time.sleep(5)
def wait_master_rank_stopped(ranks: list[RankInfo], timeout: int) -> None:
url = master_rank_health_url(ranks)
wait_http_ready(url, timeout=SERVER_READY_TIMEOUT_SECONDS)
logger.info("Hanging until master external DP rank stops: %s", url)
wait_http_unready(url, timeout=timeout)

View File

@@ -0,0 +1,163 @@
import logging
import os
import subprocess
import sys
import threading
import time
from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ExternalDPConfig,
ExternalDPConfigLoader,
RankResolver,
resolve_current_node_index,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import (
ExternalDPProxyLauncher,
ExternalDPServerManager,
build_all_server_commands,
format_http_status,
master_rank_health_url,
proxy_server_health_url,
wait_master_rank_stopped,
wait_ranks_ready,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
collect_logs,
write_benchmark_results_json,
)
from tools.aisbench import run_aisbench_cases
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
logger = logging.getLogger(__name__)
DEFAULT_LOG_ROOT = Path("/tmp/external_dp_logs")
def _install_special_dependencies(config: ExternalDPConfig) -> None:
for package, version in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{package}=={version}",
]
subprocess.call(command)
@contextmanager
def _heartbeat(
task_name: str,
*,
interval: int = 30,
status_fn: Callable[[], str] | None = None,
):
start_time = time.monotonic()
stop_event = threading.Event()
def report_progress() -> None:
while not stop_event.wait(interval):
elapsed = int(time.monotonic() - start_time)
status = ""
if status_fn is not None:
try:
status = f" {status_fn()}"
except Exception as exc: # pragma: no cover - diagnostic only
status = f" status_error={exc!r}"
logger.info("%s still running: elapsed=%ds%s", task_name, elapsed, status)
logger.info("%s started", task_name)
thread = threading.Thread(target=report_progress, daemon=True)
thread.start()
try:
yield
finally:
stop_event.set()
thread.join(timeout=1)
elapsed = int(time.monotonic() - start_time)
logger.info("%s finished: elapsed=%ds", task_name, elapsed)
def _format_benchmark_cases(config: ExternalDPConfig) -> str:
names = [str(case.get("case_name", "<unnamed>")) for case in config.benchmark_cases]
return ", ".join(names) if names else "<none>"
def _archive_rank_logs(log_root: Path, current_node_index: int) -> None:
log_prefix = os.environ.get("LOG_PREFIX")
if not log_prefix:
return
node_log_dir = log_root / f"node-{current_node_index}"
output_tar = Path(log_prefix) / f"node_{current_node_index}_external_dp_logs.tar.gz"
collect_logs(node_log_dir, output_tar)
def test_external_dp() -> None:
config = ExternalDPConfigLoader.from_yaml()
_install_special_dependencies(config)
ranks = RankResolver(config).resolve()
current_node_index = resolve_current_node_index(config)
log_root = Path(os.environ.get("EXTERNAL_DP_LOG_DIR", str(DEFAULT_LOG_ROOT)))
max_wait_seconds = int(os.environ.get("EXTERNAL_DP_MAX_WAIT_SECONDS", "3600"))
is_master = current_node_index == 0
server_manager = ExternalDPServerManager(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
proxy_launcher = ExternalDPProxyLauncher(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
try:
with server_manager, proxy_launcher:
if is_master:
wait_ranks_ready(ranks, timeout=max_wait_seconds)
proxy_launcher.wait_ready()
target = f"http://{config.routing.proxy_host}:{config.routing.proxy_port}"
logger.info(
"Running AISBench cases: model=%s target=%s cases=[%s]",
config.model,
target,
_format_benchmark_cases(config),
)
with _heartbeat(
"Running AISBench",
status_fn=lambda: format_http_status("proxy", proxy_server_health_url(config)),
):
results = run_aisbench_cases(
model=config.model,
port=config.routing.proxy_port,
aisbench_cases=config.benchmark_cases,
host_ip=config.routing.proxy_host,
)
logger.info("AISBench completed: results=%d", len(results or []))
all_commands = build_all_server_commands(config, ranks)
write_benchmark_results_json(
config=config,
ranks=ranks,
commands=all_commands,
results=results,
)
wait_ranks_ready(ranks, timeout=30)
else:
master_url = master_rank_health_url(ranks)
with _heartbeat(
"Waiting for master external DP rank to stop",
status_fn=lambda: format_http_status("master", master_url),
):
wait_master_rank_stopped(ranks, timeout=max_wait_seconds)
finally:
_archive_rank_logs(log_root, current_node_index)

View File

@@ -0,0 +1,236 @@
import logging
import os
import shlex
import signal
import subprocess
import tarfile
import time
import urllib.error
import urllib.request
from pathlib import Path
from typing import TYPE_CHECKING, Any
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ExternalDPConfig,
RankInfo,
)
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
build_task_entry,
extract_hardware,
filter_environment,
get_vllm_version,
write_results_json,
)
logger = logging.getLogger(__name__)
if TYPE_CHECKING:
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import ServerCommand
SENSITIVE_ENV_TOKENS = ("TOKEN", "SECRET", "PASSWORD", "ACCESS_KEY")
def format_server_cmd(cmd: list[str], env: dict[str, str] | None = None) -> str:
env_parts: list[str] = []
for key, value in sorted((env or {}).items()):
display_value = "***" if any(token in key.upper() for token in SENSITIVE_ENV_TOKENS) else str(value)
env_parts.append(f"{key}={shlex.quote(display_value)}")
return " ".join([*env_parts, shlex.join(cmd)])
def start_logged_process(cmd: list[str], env: dict[str, str], log_file: Path) -> subprocess.Popen:
log_file.parent.mkdir(parents=True, exist_ok=True)
merged_env = {**os.environ, **env}
with log_file.open("ab") as f:
f.write(f"Starting command: {format_server_cmd(cmd, env)}\n".encode())
f.flush()
return subprocess.Popen(
cmd,
stdout=f,
stderr=subprocess.STDOUT,
env=merged_env,
start_new_session=True,
)
def terminate_process_tree(pid: int, timeout: int = 30) -> None:
try:
import psutil
except ModuleNotFoundError:
try:
os.killpg(pid, signal.SIGTERM)
except ProcessLookupError:
return
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
os.kill(pid, 0)
except ProcessLookupError:
return
time.sleep(0.2)
try:
os.killpg(pid, signal.SIGKILL)
except ProcessLookupError:
return
return
try:
parent = psutil.Process(pid)
except psutil.NoSuchProcess:
return
children = parent.children(recursive=True)
for process in children:
process.terminate()
parent.terminate()
gone, alive = psutil.wait_procs([parent, *children], timeout=timeout)
del gone
for process in alive:
process.kill()
def is_http_ready(url: str, timeout: float = 5.0) -> bool:
try:
with urllib.request.urlopen(url, timeout=timeout) as response:
return 200 <= response.status < 300
except (urllib.error.URLError, TimeoutError, OSError):
return False
def wait_http_ready(url: str, timeout: int, interval: float = 2.0) -> None:
deadline = time.monotonic() + timeout
last_error: Exception | None = None
while time.monotonic() < deadline:
try:
with urllib.request.urlopen(url, timeout=5) as response:
if 200 <= response.status < 300:
return
except (urllib.error.URLError, TimeoutError, OSError) as exc:
last_error = exc
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP ready: {url}; last_error={last_error}")
def wait_http_unready(url: str, timeout: int, interval: float = 5.0) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if not is_http_ready(url):
return
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP unready: {url}")
def collect_logs(src_dir: Path, output_tar: Path) -> None:
if not src_dir.exists():
return
output_tar.parent.mkdir(parents=True, exist_ok=True)
with tarfile.open(output_tar, "w:gz") as tar:
tar.add(src_dir, arcname=src_dir.name)
def _common_command_envs(commands: list["ServerCommand"]) -> dict[str, str]:
if not commands:
return {}
common_keys = set(commands[0].env)
for command in commands[1:]:
common_keys.intersection_update(command.env)
common_envs: dict[str, str] = {}
for key in sorted(common_keys):
values = {command.env[key] for command in commands}
if len(values) == 1:
common_envs[key] = next(iter(values))
return common_envs
def _extract_dtype(config: ExternalDPConfig, commands: list["ServerCommand"]) -> str:
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = any("--quantization ascend" in command.display_cmd for command in commands)
return "w8a8" if has_w8a8 and has_quant_ascend else "bf16"
def _extract_features(commands: list["ServerCommand"]) -> list[str]:
if not commands:
return []
features: list[str] = []
command_args = [command.cmd for command in commands]
command_displays = [" ".join(shlex.quote(arg) for arg in command.cmd) for command in commands]
if any("--async-scheduling" in cmd for cmd in command_args):
features.append("async_scheduling")
if any("--enable-expert-parallel" in cmd for cmd in command_args):
features.append("expert_parallel")
if any("--speculative-config" in cmd for cmd in command_args):
features.append("speculative")
if any("cudagraph_mode" in display for display in command_displays):
features.append("aclgraph")
feature_envs = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
for env_key, feature_name in feature_envs.items():
values = [str(command.env.get(env_key, "0")) for command in commands]
if any(value not in ("0", "", "false", "False") for value in values):
features.append(feature_name)
return features
def _build_serve_cmd(
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
) -> dict[str, Any]:
entries: dict[str, str] = {}
for rank, command in zip(ranks, commands):
prefix = rank.role
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefix = "prefill" if rank.role == "prefiller" else "decode"
entries[f"{prefix}-node{rank.node_index}-rank{rank.local_rank}"] = command.display_cmd
key = "external_dp_pd" if config.routing.type == ROUTING_DISAGGREGATED_PREFILL else "external_dp"
return {key: entries}
def build_benchmark_results(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
) -> dict[str, Any]:
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
tasks = [build_task_entry(key, case, result) for (key, case), result in zip(valid_items, results)]
runner = os.environ.get("VLLM_CI_RUNNER", "")
common_envs = _common_command_envs(commands)
return {
"model_name": config.model,
"hardware": extract_hardware(runner),
"dtype": _extract_dtype(config, commands),
"feature": _extract_features(commands),
"vllm_version": get_vllm_version(),
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config, ranks, commands),
"environment": filter_environment(common_envs),
}
def write_benchmark_results_json(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
output_dir: Path | None = None,
) -> Path:
output = build_benchmark_results(config=config, ranks=ranks, commands=commands, results=results)
job_name = os.environ.get("BENCHMARK_JOB_NAME", "") or config.test_name.replace(" ", "-")
return write_results_json(output, job_name=job_name, output_dir=output_dir)