0
tests/e2e/nightly/multi_node/__init__.py
Normal file
0
tests/e2e/nightly/multi_node/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/external_dp/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/external_dp/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""External DP nightly test package."""
|
||||
@@ -0,0 +1,308 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_1M_1k_prefix99_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_1M_1k_prefix99:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1024
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 11.93
|
||||
threshold: 0.95
|
||||
@@ -0,0 +1,324 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 0
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 8
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_BUFFSIZE: "1800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_TPOT50_128k_1_prefix_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1
|
||||
batch_size: 4
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_TPOT50_128k_1_prefix90:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 192
|
||||
max_out_len: 1024
|
||||
batch_size: 48
|
||||
request_rate: 1
|
||||
baseline: 869.13
|
||||
threshold: 0.95
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,176 @@
|
||||
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
|
||||
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 16
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1500"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --trust-remote-code
|
||||
- --block-size
|
||||
- "32"
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.9"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "240"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,335 @@
|
||||
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0, 1]
|
||||
decoder: [2, 3]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 4
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: "0"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_INTRA_PCIE_ENABLE: "1"
|
||||
HCCL_INTRA_ROCE_ENABLE: "0"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1500
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,161 @@
|
||||
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
|
||||
model: "Eco-Tech/Kimi-K2.6-w4a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_SERVER_DEV_MODE: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "512"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --enforce-eager
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "256"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 60
|
||||
max_out_len: 1024
|
||||
batch_size: 15
|
||||
request_rate: 0.4
|
||||
baseline: 347.4475
|
||||
threshold: 0.97
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 93.33
|
||||
threshold: 10
|
||||
temperature: 1.0
|
||||
top_p: 1
|
||||
@@ -0,0 +1,197 @@
|
||||
test_name: "Minimax_m2.7_in3_5_tpot50"
|
||||
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
|
||||
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
PYTHONHASHSEED: "0"
|
||||
|
||||
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "2048"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "55880",
|
||||
"engine_id": "0",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}} }'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --async-scheduling
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "56900",
|
||||
"engine_id": "1",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf_warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2
|
||||
max_out_len: 1
|
||||
batch_size: 2
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1024
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 717.5332
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
temperature: 1
|
||||
top_p: 1
|
||||
top_k: 40
|
||||
ignore_eos: false
|
||||
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
@@ -0,0 +1,360 @@
|
||||
# External DP Config Template
|
||||
|
||||
This document shows how to write YAML configs consumed by
|
||||
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
|
||||
|
||||
`server_cmd_template` contains only the arguments after
|
||||
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
|
||||
`model` automatically.
|
||||
|
||||
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
|
||||
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
|
||||
and roles are selected by `routing.groups`.
|
||||
|
||||
## Generic DP Template
|
||||
|
||||
Use this template for generic external data parallel serving. This mode uses
|
||||
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
|
||||
independent vLLM instances instead of external DP rank arguments.
|
||||
|
||||
```yaml
|
||||
test_name: "test Qwen3-30B-A3B generic external dp"
|
||||
model: "Qwen/Qwen3-30B-A3B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "generic_dp"
|
||||
groups:
|
||||
worker: [0, 1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs: &generic_env
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template: &generic_server_cmd
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --max-model-len
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --enable-expert-parallel
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *generic_env
|
||||
server_cmd_template: *generic_server_cmd
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Disaggregated Prefill Template
|
||||
|
||||
Use this template for PD disaggregation. `routing.groups` decides which config
|
||||
entries run as prefillers or decoders. The framework derives the PD proxy script
|
||||
from `routing.type`, so do not write `proxy_*` fields in YAML.
|
||||
|
||||
```yaml
|
||||
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0]
|
||||
decoder: [1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "256"
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
max_out_len: 128
|
||||
batch_size: 4
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 48
|
||||
batch_size: 4
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Field Notes
|
||||
|
||||
- `test_name`: Human-readable test name. It is also used when writing benchmark
|
||||
result metadata.
|
||||
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
|
||||
- `num_nodes`: Number of config entries and templates expected.
|
||||
- `npu_per_node`: Device capacity validation for each node.
|
||||
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
|
||||
needs fixed hosts.
|
||||
- `routing.type`: Supported values are `generic_dp` and
|
||||
`disaggregated_prefill`.
|
||||
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
|
||||
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
|
||||
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
|
||||
`kv_consumer` for decoder templates.
|
||||
- `config[].dp_size`: Global DP size for this DP group.
|
||||
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
|
||||
- `config[].dp_rank_start`: First global DP rank owned by this node.
|
||||
- `config[].dp_address`: DP master address. For one global DP group, use
|
||||
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
|
||||
address for prefiller nodes and the decoder master address for decoder nodes.
|
||||
- `templates`: One template per config entry. The framework expands one command
|
||||
per local DP rank.
|
||||
|
||||
The framework injects distributed network envs at startup:
|
||||
|
||||
```text
|
||||
HCCL_IF_IP
|
||||
HCCL_SOCKET_IFNAME
|
||||
GLOO_SOCKET_IFNAME
|
||||
TP_SOCKET_IFNAME
|
||||
LOCAL_IP
|
||||
NIC_NAME
|
||||
MASTER_IP
|
||||
```
|
||||
|
||||
The framework also derives proxy metadata from `routing.type`:
|
||||
|
||||
```text
|
||||
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
|
||||
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
|
||||
```
|
||||
|
||||
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
|
||||
for benchmark requests.
|
||||
|
||||
## Template Variables
|
||||
|
||||
The following variables are available in `envs` and `server_cmd_template`:
|
||||
|
||||
```text
|
||||
${MODEL}
|
||||
${PORT_START}
|
||||
${PORT}
|
||||
${DP_SIZE}
|
||||
${DP_SIZE_LOCAL}
|
||||
${DP_RANK_START}
|
||||
${DP_RANK}
|
||||
${LOCAL_RANK}
|
||||
${TP_SIZE}
|
||||
${CP_SIZE}
|
||||
${SP_SIZE}
|
||||
${PP_SIZE}
|
||||
${DP_ADDRESS}
|
||||
${DP_RPC_PORT}
|
||||
${VISIBLE_DEVICES}
|
||||
${NODE_INDEX}
|
||||
${CONFIG_INDEX}
|
||||
${NODE_0_IP}, ${NODE_1_IP}, ...
|
||||
${LOCAL_IP}
|
||||
${MASTER_IP}
|
||||
${LWS_WORKER_INDEX}
|
||||
```
|
||||
|
||||
Command arguments can also reference rendered environment variables with
|
||||
shell-style `$VARNAME`, for example:
|
||||
|
||||
```yaml
|
||||
envs:
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template:
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
```
|
||||
|
||||
## Checks Before Running
|
||||
|
||||
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
|
||||
- Make sure each config index is assigned to exactly one routing group.
|
||||
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
|
||||
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
|
||||
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
|
||||
`--enable-expert-parallel`.
|
||||
- Set `--max-model-len` large enough for benchmark input tokens plus
|
||||
`max_out_len`.
|
||||
@@ -0,0 +1 @@
|
||||
"""External DP nightly test helpers."""
|
||||
@@ -0,0 +1,449 @@
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
load_yaml_mapping,
|
||||
resolve_cluster_ips,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
resolve_current_node_index as resolve_node_index,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ROUTING_GENERIC_DP = "generic_dp"
|
||||
ROUTING_DISAGGREGATED_PREFILL = "disaggregated_prefill"
|
||||
PROXY_SCRIPT_BY_ROUTING_TYPE = {
|
||||
ROUTING_GENERIC_DP: "examples/external_online_dp/dp_load_balance_proxy_server.py",
|
||||
ROUTING_DISAGGREGATED_PREFILL: "examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
|
||||
}
|
||||
|
||||
CLUSTER_PLACEHOLDER_RE = re.compile(r"\$\{(NODE_(\d+)_IP|LOCAL_IP|MASTER_IP|LWS_WORKER_INDEX)\}")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RoutingConfig:
|
||||
"""Proxy routing metadata shared by all external DP ranks."""
|
||||
|
||||
type: str
|
||||
proxy_node_index: int
|
||||
proxy_host: str
|
||||
proxy_port: int
|
||||
proxy_script: str
|
||||
groups: dict[str, list[int]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeInfo:
|
||||
"""Per-node external DP server topology loaded from one config entry."""
|
||||
|
||||
ip: str
|
||||
port_start: int
|
||||
dp_rpc_port: int
|
||||
dp_size: int
|
||||
dp_size_local: int
|
||||
dp_rank_start: int
|
||||
tp_size: int
|
||||
dp_address: str
|
||||
cp_size: int = 1
|
||||
sp_size: int = 1
|
||||
pp_size: int = 1
|
||||
|
||||
@property
|
||||
def devices_per_rank(self) -> int:
|
||||
return self.tp_size * self.cp_size * self.sp_size * self.pp_size
|
||||
|
||||
@property
|
||||
def devices_per_node(self) -> int:
|
||||
return self.dp_size_local * self.devices_per_rank
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeTemplate:
|
||||
"""Per-node env and argument template for launching vLLM servers."""
|
||||
|
||||
envs: dict[str, Any]
|
||||
server_cmd_template: list[str]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RankInfo:
|
||||
"""One concrete vLLM server rank expanded from a node config."""
|
||||
|
||||
node_index: int
|
||||
role: str
|
||||
local_rank: int
|
||||
dp_rank: int
|
||||
host: str
|
||||
port: int
|
||||
visible_devices: str
|
||||
dp_size: int
|
||||
dp_size_local: int
|
||||
tp_size: int
|
||||
cp_size: int
|
||||
sp_size: int
|
||||
pp_size: int
|
||||
dp_address: str
|
||||
dp_rpc_port: int
|
||||
port_start: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExternalDPConfig:
|
||||
"""Top-level external DP test config after YAML anchors are merged."""
|
||||
|
||||
test_name: str
|
||||
model: str
|
||||
num_nodes: int
|
||||
npu_per_node: int
|
||||
cluster_hosts: list[str] | None
|
||||
cluster_ips: list[str]
|
||||
routing: RoutingConfig
|
||||
nodes: list[NodeInfo]
|
||||
launch_templates: list[NodeTemplate]
|
||||
benchmark_cases: list[dict[str, Any]] = field(default_factory=list)
|
||||
special_dependencies: dict[str, str] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def is_disaggregated_prefill(self) -> bool:
|
||||
return self.routing.type == ROUTING_DISAGGREGATED_PREFILL
|
||||
|
||||
|
||||
def replace_cluster_placeholders(
|
||||
value: Any,
|
||||
*,
|
||||
cluster_ips: list[str],
|
||||
local_ip: str | None = None,
|
||||
current_node_index: int | None = None,
|
||||
) -> Any:
|
||||
if isinstance(value, dict):
|
||||
return {
|
||||
key: replace_cluster_placeholders(
|
||||
val,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=current_node_index,
|
||||
)
|
||||
for key, val in value.items()
|
||||
}
|
||||
if isinstance(value, list):
|
||||
return [
|
||||
replace_cluster_placeholders(
|
||||
item,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=current_node_index,
|
||||
)
|
||||
for item in value
|
||||
]
|
||||
if not isinstance(value, str):
|
||||
return value
|
||||
|
||||
def repl(match: re.Match[str]) -> str:
|
||||
token = match.group(1)
|
||||
node_index = match.group(2)
|
||||
if node_index is not None:
|
||||
idx = int(node_index)
|
||||
if idx >= len(cluster_ips):
|
||||
raise ValueError(f"Cluster placeholder ${{{token}}} is out of range")
|
||||
return cluster_ips[idx]
|
||||
if token == "MASTER_IP":
|
||||
return cluster_ips[0]
|
||||
if token == "LOCAL_IP":
|
||||
if local_ip is None:
|
||||
return match.group(0)
|
||||
return local_ip
|
||||
if token == "LWS_WORKER_INDEX":
|
||||
if current_node_index is None:
|
||||
return os.environ.get("LWS_WORKER_INDEX", match.group(0))
|
||||
return str(current_node_index)
|
||||
return match.group(0)
|
||||
|
||||
return CLUSTER_PLACEHOLDER_RE.sub(repl, value)
|
||||
|
||||
|
||||
def resolve_current_node_index(config: ExternalDPConfig) -> int:
|
||||
return resolve_node_index(config.cluster_ips)
|
||||
|
||||
|
||||
class ExternalDPConfigLoader:
|
||||
"""Load, normalize, and validate external DP YAML files."""
|
||||
|
||||
@classmethod
|
||||
def from_yaml(
|
||||
cls,
|
||||
yaml_path: str | None = None,
|
||||
*,
|
||||
cluster_ips: list[str] | None = None,
|
||||
) -> ExternalDPConfig:
|
||||
raw_config = cls._load_yaml(yaml_path)
|
||||
cls._validate_root(raw_config)
|
||||
|
||||
num_nodes = int(raw_config["num_nodes"])
|
||||
resolved_cluster_ips = cls._resolve_cluster_ips(raw_config, num_nodes, cluster_ips)
|
||||
|
||||
model = str(raw_config["model"])
|
||||
routing = cls._parse_routing(raw_config["routing"], resolved_cluster_ips)
|
||||
nodes = cls._parse_nodes(raw_config, resolved_cluster_ips)
|
||||
launch_templates = cls._parse_templates(raw_config)
|
||||
benchmark_cases = cls._parse_benchmarks(raw_config)
|
||||
|
||||
config = ExternalDPConfig(
|
||||
test_name=str(raw_config.get("test_name", "external_dp_test")),
|
||||
model=model,
|
||||
num_nodes=num_nodes,
|
||||
npu_per_node=int(raw_config["npu_per_node"]),
|
||||
cluster_hosts=raw_config.get("cluster_hosts"),
|
||||
cluster_ips=resolved_cluster_ips,
|
||||
routing=routing,
|
||||
nodes=nodes,
|
||||
launch_templates=launch_templates,
|
||||
benchmark_cases=benchmark_cases,
|
||||
special_dependencies=dict(raw_config.get("special_dependencies", {})),
|
||||
)
|
||||
cls._validate_config(config)
|
||||
return config
|
||||
|
||||
@staticmethod
|
||||
def _load_yaml(yaml_path: str | None) -> dict[str, Any]:
|
||||
default_config_name = "GLM5_1-W8A8-EP-external.yaml"
|
||||
default_config_base_path = "tests/e2e/nightly/multi_node/external_dp/config/"
|
||||
return load_yaml_mapping(
|
||||
yaml_path,
|
||||
default_name=default_config_name,
|
||||
default_base_path=default_config_base_path,
|
||||
description="external DP config",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_root(config: dict[str, Any]) -> None:
|
||||
required = ["model", "num_nodes", "npu_per_node", "routing", "config", "templates", "benchmarks"]
|
||||
missing = [key for key in required if key not in config]
|
||||
if missing:
|
||||
raise KeyError(f"Missing required external DP config fields: {missing}")
|
||||
if int(config["num_nodes"]) <= 0:
|
||||
raise ValueError("num_nodes must be greater than 0")
|
||||
|
||||
@staticmethod
|
||||
def _resolve_cluster_ips(
|
||||
raw_config: dict[str, Any],
|
||||
num_nodes: int,
|
||||
cluster_ips: list[str] | None,
|
||||
) -> list[str]:
|
||||
return resolve_cluster_ips(
|
||||
raw_config,
|
||||
num_nodes,
|
||||
cluster_ips,
|
||||
dns_log_message="Resolving external DP cluster IPs via LWS DNS",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _parse_routing(raw_routing: dict[str, Any], cluster_ips: list[str]) -> RoutingConfig:
|
||||
routing_type = str(raw_routing["type"])
|
||||
if routing_type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
|
||||
raise ValueError(f"Unsupported routing.type: {routing_type}")
|
||||
|
||||
proxy_node_index = 0
|
||||
proxy_port = 1999
|
||||
if proxy_node_index >= len(cluster_ips) or proxy_node_index < 0:
|
||||
raise ValueError("routing.proxy_node_index out of range")
|
||||
local_ip = cluster_ips[proxy_node_index]
|
||||
routing = replace_cluster_placeholders(
|
||||
raw_routing,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=proxy_node_index,
|
||||
)
|
||||
return RoutingConfig(
|
||||
type=routing_type,
|
||||
proxy_node_index=proxy_node_index,
|
||||
proxy_host=local_ip,
|
||||
proxy_port=proxy_port,
|
||||
proxy_script=PROXY_SCRIPT_BY_ROUTING_TYPE[routing_type],
|
||||
groups={
|
||||
str(name): [int(index) for index in indices] for name, indices in routing.get("groups", {}).items()
|
||||
},
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _parse_nodes(raw_config: dict[str, Any], cluster_ips: list[str]) -> list[NodeInfo]:
|
||||
nodes: list[NodeInfo] = []
|
||||
for index, raw_node in enumerate(raw_config["config"]):
|
||||
raw_node_index = raw_node.get("node_index")
|
||||
if raw_node_index is not None and int(raw_node_index) != index:
|
||||
raise ValueError(f"config[{index}].node_index must equal {index}")
|
||||
node = replace_cluster_placeholders(
|
||||
raw_node,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=cluster_ips[index],
|
||||
current_node_index=index,
|
||||
)
|
||||
nodes.append(
|
||||
NodeInfo(
|
||||
ip=cluster_ips[index],
|
||||
port_start=int(node["port_start"]),
|
||||
dp_rpc_port=int(node["dp_rpc_port"]),
|
||||
dp_size=int(node.get("dp_size", 1)),
|
||||
dp_size_local=int(node.get("dp_size_local", 1)),
|
||||
dp_rank_start=int(node.get("dp_rank_start", 0)),
|
||||
tp_size=int(node.get("tp_size", 1)),
|
||||
cp_size=int(node.get("cp_size", 1)),
|
||||
sp_size=int(node.get("sp_size", 1)),
|
||||
dp_address=str(node["dp_address"]),
|
||||
pp_size=int(node.get("pp_size", 1)),
|
||||
)
|
||||
)
|
||||
return nodes
|
||||
|
||||
@staticmethod
|
||||
def _parse_templates(raw_config: dict[str, Any]) -> list[NodeTemplate]:
|
||||
templates: list[NodeTemplate] = []
|
||||
for index, raw_template in enumerate(raw_config["templates"]):
|
||||
envs = raw_template.get("envs")
|
||||
server_cmd_template = raw_template.get("server_cmd_template")
|
||||
if envs is None or server_cmd_template is None:
|
||||
raise KeyError(f"templates[{index}] must contain envs and server_cmd_template")
|
||||
if not isinstance(server_cmd_template, list):
|
||||
raise TypeError(f"templates[{index}].server_cmd_template must be a list")
|
||||
templates.append(
|
||||
NodeTemplate(
|
||||
envs=dict(envs),
|
||||
server_cmd_template=[str(arg) for arg in server_cmd_template],
|
||||
)
|
||||
)
|
||||
return templates
|
||||
|
||||
@staticmethod
|
||||
def _parse_benchmarks(raw_config: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
benchmark_cases: list[dict[str, Any]] = []
|
||||
for name, case in (raw_config.get("benchmarks") or {}).items():
|
||||
case_with_name = dict(case)
|
||||
case_with_name["case_name"] = name
|
||||
benchmark_cases.append(case_with_name)
|
||||
return benchmark_cases
|
||||
|
||||
@classmethod
|
||||
def _validate_config(cls, config: ExternalDPConfig) -> None:
|
||||
cls._validate_config_sizes(config)
|
||||
cls._validate_routing(config)
|
||||
cls._validate_node_parallel_config(config)
|
||||
|
||||
@staticmethod
|
||||
def _validate_config_sizes(config: ExternalDPConfig) -> None:
|
||||
if len(config.nodes) != config.num_nodes:
|
||||
raise AssertionError(f"config size ({len(config.nodes)}) != num_nodes ({config.num_nodes})")
|
||||
if len(config.launch_templates) != config.num_nodes:
|
||||
raise AssertionError(f"templates size ({len(config.launch_templates)}) != num_nodes ({config.num_nodes})")
|
||||
if config.cluster_hosts and len(config.cluster_hosts) != config.num_nodes:
|
||||
raise AssertionError("cluster_hosts size mismatch")
|
||||
|
||||
@staticmethod
|
||||
def _validate_routing(config: ExternalDPConfig) -> None:
|
||||
if config.routing.type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
|
||||
raise ValueError(f"Unsupported routing.type: {config.routing.type}")
|
||||
|
||||
groups = config.routing.groups
|
||||
if config.routing.type == ROUTING_GENERIC_DP and not groups.get("worker"):
|
||||
raise ValueError("generic_dp routing requires routing.groups.worker")
|
||||
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL and (
|
||||
not groups.get("prefiller") or not groups.get("decoder")
|
||||
):
|
||||
raise ValueError("disaggregated_prefill routing requires prefiller and decoder groups")
|
||||
|
||||
seen_group_indices: dict[int, str] = {}
|
||||
for group_name, indices in groups.items():
|
||||
for index in indices:
|
||||
if index < 0 or index >= config.num_nodes:
|
||||
raise ValueError(f"routing.groups.{group_name} index out of range: {index}")
|
||||
if index in seen_group_indices:
|
||||
raise ValueError(f"node index {index} appears in both {seen_group_indices[index]} and {group_name}")
|
||||
seen_group_indices[index] = group_name
|
||||
|
||||
if config.routing.proxy_node_index < 0 or config.routing.proxy_node_index >= config.num_nodes:
|
||||
raise ValueError("routing.proxy_node_index out of range")
|
||||
|
||||
@staticmethod
|
||||
def _validate_node_parallel_config(config: ExternalDPConfig) -> None:
|
||||
for node_index, node in enumerate(config.nodes):
|
||||
parallel_sizes = {
|
||||
"dp_size": node.dp_size,
|
||||
"dp_size_local": node.dp_size_local,
|
||||
"tp_size": node.tp_size,
|
||||
"cp_size": node.cp_size,
|
||||
"sp_size": node.sp_size,
|
||||
"pp_size": node.pp_size,
|
||||
}
|
||||
invalid_sizes = {name: value for name, value in parallel_sizes.items() if value < 1}
|
||||
if invalid_sizes:
|
||||
raise ValueError(f"node {node_index} parallel sizes must be >= 1: {invalid_sizes}")
|
||||
if node.dp_rank_start < 0:
|
||||
raise ValueError(f"node {node_index} dp_rank_start must be >= 0")
|
||||
if node.devices_per_node > config.npu_per_node:
|
||||
raise ValueError(
|
||||
f"node {node_index} uses {node.devices_per_node} NPUs, but npu_per_node is {config.npu_per_node}"
|
||||
)
|
||||
if node.dp_rank_start + node.dp_size_local > node.dp_size:
|
||||
raise ValueError(f"node {node_index} dp rank range exceeds dp_size")
|
||||
|
||||
|
||||
class RankResolver:
|
||||
"""Expand node-level configs into concrete vLLM server ranks."""
|
||||
|
||||
def __init__(self, config: ExternalDPConfig):
|
||||
self.config = config
|
||||
|
||||
def resolve(self) -> list[RankInfo]:
|
||||
role_by_node_index = self._role_by_node_index()
|
||||
ranks: list[RankInfo] = []
|
||||
for node_index, node_info in enumerate(self.config.nodes):
|
||||
role = role_by_node_index[node_index]
|
||||
ranks.extend(self._expand_node(node_index, role, node_info))
|
||||
return ranks
|
||||
|
||||
def _role_by_node_index(self) -> dict[int, str]:
|
||||
role_by_index: dict[int, str] = {}
|
||||
for role, node_indices in self.config.routing.groups.items():
|
||||
for index in node_indices:
|
||||
role_by_index[index] = role
|
||||
|
||||
missing = [index for index in range(self.config.num_nodes) if index not in role_by_index]
|
||||
if missing:
|
||||
raise ValueError(f"routing.groups does not assign role for node indices: {missing}")
|
||||
return role_by_index
|
||||
|
||||
@staticmethod
|
||||
def _expand_node(node_index: int, role: str, node_info: NodeInfo) -> list[RankInfo]:
|
||||
ranks: list[RankInfo] = []
|
||||
for local_rank in range(node_info.dp_size_local):
|
||||
dp_rank = node_info.dp_rank_start + local_rank
|
||||
port = node_info.port_start + local_rank
|
||||
device_range = range(
|
||||
local_rank * node_info.devices_per_rank,
|
||||
(local_rank + 1) * node_info.devices_per_rank,
|
||||
)
|
||||
visible_devices = ",".join(str(device) for device in device_range)
|
||||
ranks.append(
|
||||
RankInfo(
|
||||
node_index=node_index,
|
||||
role=role,
|
||||
local_rank=local_rank,
|
||||
dp_rank=dp_rank,
|
||||
host=node_info.ip,
|
||||
port=port,
|
||||
visible_devices=visible_devices,
|
||||
dp_size=node_info.dp_size,
|
||||
dp_size_local=node_info.dp_size_local,
|
||||
tp_size=node_info.tp_size,
|
||||
cp_size=node_info.cp_size,
|
||||
sp_size=node_info.sp_size,
|
||||
pp_size=node_info.pp_size,
|
||||
dp_address=node_info.dp_address,
|
||||
dp_rpc_port=node_info.dp_rpc_port,
|
||||
port_start=node_info.port_start,
|
||||
)
|
||||
)
|
||||
return ranks
|
||||
435
tests/e2e/nightly/multi_node/external_dp/scripts/runtime.py
Normal file
435
tests/e2e/nightly/multi_node/external_dp/scripts/runtime.py
Normal file
@@ -0,0 +1,435 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Iterable
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ROUTING_DISAGGREGATED_PREFILL,
|
||||
ROUTING_GENERIC_DP,
|
||||
ExternalDPConfig,
|
||||
NodeTemplate,
|
||||
RankInfo,
|
||||
replace_cluster_placeholders,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
|
||||
format_server_cmd,
|
||||
is_http_ready,
|
||||
start_logged_process,
|
||||
terminate_process_tree,
|
||||
wait_http_ready,
|
||||
wait_http_unready,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import get_net_interface
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SERVER_READY_TIMEOUT_SECONDS = 3600
|
||||
TEMPLATE_VAR_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
|
||||
ENV_VAR_RE = re.compile(r"(?<!\$)\$([A-Za-z_][A-Za-z0-9_]*)")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ServerCommand:
|
||||
"""Rendered command, env, and printable command line."""
|
||||
|
||||
cmd: list[str]
|
||||
env: dict[str, str]
|
||||
display_cmd: str
|
||||
|
||||
|
||||
RankProcess = tuple[subprocess.Popen, RankInfo, Path]
|
||||
|
||||
|
||||
class ServerCommandBuilder:
|
||||
"""Render rank templates into vLLM serve commands."""
|
||||
|
||||
def __init__(self, config: ExternalDPConfig):
|
||||
self.config = config
|
||||
|
||||
def build(self, rank: RankInfo, template: NodeTemplate) -> ServerCommand:
|
||||
variables = self._build_variables(rank)
|
||||
rendered_env = self._render_envs(template.envs, rank, variables)
|
||||
rendered_args = [
|
||||
self._render_string(
|
||||
arg,
|
||||
rank=rank,
|
||||
braced_variables=variables,
|
||||
unbraced_variables=rendered_env,
|
||||
allow_missing_unbraced=False,
|
||||
)
|
||||
for arg in template.server_cmd_template
|
||||
]
|
||||
cmd = ["vllm", "serve", self.config.model, *rendered_args]
|
||||
|
||||
env = {key: str(value) for key, value in rendered_env.items()}
|
||||
display_cmd = format_server_cmd(cmd, env)
|
||||
logger.info(
|
||||
"External DP server command node=%s rank=%s: %s",
|
||||
rank.node_index,
|
||||
rank.local_rank,
|
||||
display_cmd,
|
||||
)
|
||||
return ServerCommand(cmd=cmd, env=env, display_cmd=display_cmd)
|
||||
|
||||
def build_all(self, ranks: list[RankInfo]) -> list[ServerCommand]:
|
||||
return [self.build(rank, self.config.launch_templates[rank.node_index]) for rank in ranks]
|
||||
|
||||
def _build_variables(self, rank: RankInfo) -> dict[str, str]:
|
||||
return {
|
||||
"MODEL": self.config.model,
|
||||
"PORT_START": str(rank.port_start),
|
||||
"PORT": str(rank.port),
|
||||
"DP_SIZE": str(rank.dp_size),
|
||||
"DP_SIZE_LOCAL": str(rank.dp_size_local),
|
||||
"DP_RANK_START": str(rank.dp_rank - rank.local_rank),
|
||||
"DP_RANK": str(rank.dp_rank),
|
||||
"LOCAL_RANK": str(rank.local_rank),
|
||||
"TP_SIZE": str(rank.tp_size),
|
||||
"CP_SIZE": str(rank.cp_size),
|
||||
"SP_SIZE": str(rank.sp_size),
|
||||
"PP_SIZE": str(rank.pp_size),
|
||||
"DP_ADDRESS": rank.dp_address,
|
||||
"DP_RPC_PORT": str(rank.dp_rpc_port),
|
||||
"VISIBLE_DEVICES": rank.visible_devices,
|
||||
"NODE_INDEX": str(rank.node_index),
|
||||
"CONFIG_INDEX": str(rank.node_index),
|
||||
}
|
||||
|
||||
def _render_envs(
|
||||
self,
|
||||
envs: dict[str, Any],
|
||||
rank: RankInfo,
|
||||
variables: dict[str, str],
|
||||
) -> dict[str, str]:
|
||||
rendered_envs: dict[str, str] = {}
|
||||
for key, value in envs.items():
|
||||
if isinstance(value, str):
|
||||
value = self._render_string(
|
||||
value,
|
||||
rank=rank,
|
||||
braced_variables=variables,
|
||||
unbraced_variables={**os.environ, **rendered_envs},
|
||||
allow_missing_unbraced=True,
|
||||
)
|
||||
rendered_envs[str(key)] = str(value)
|
||||
return rendered_envs
|
||||
|
||||
def _render_string(
|
||||
self,
|
||||
value: str,
|
||||
*,
|
||||
rank: RankInfo,
|
||||
braced_variables: dict[str, str],
|
||||
unbraced_variables: dict[str, str],
|
||||
allow_missing_unbraced: bool,
|
||||
) -> str:
|
||||
value = replace_cluster_placeholders(
|
||||
value,
|
||||
cluster_ips=self.config.cluster_ips,
|
||||
local_ip=rank.host,
|
||||
current_node_index=rank.node_index,
|
||||
)
|
||||
value = self._render_variables(
|
||||
value,
|
||||
braced_variables,
|
||||
pattern=TEMPLATE_VAR_RE,
|
||||
allow_missing=False,
|
||||
)
|
||||
return self._render_variables(
|
||||
value,
|
||||
unbraced_variables,
|
||||
pattern=ENV_VAR_RE,
|
||||
allow_missing=allow_missing_unbraced,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _render_variables(
|
||||
value: str,
|
||||
variables: dict[str, str],
|
||||
*,
|
||||
pattern: re.Pattern[str],
|
||||
allow_missing: bool,
|
||||
) -> str:
|
||||
def repl(match: re.Match[str]) -> str:
|
||||
key = match.group(1)
|
||||
if key not in variables:
|
||||
if allow_missing:
|
||||
return ""
|
||||
raise KeyError(f"Unknown external DP template variable: {key}")
|
||||
return variables[key]
|
||||
|
||||
return pattern.sub(repl, value)
|
||||
|
||||
|
||||
class ExternalDPServerManager:
|
||||
"""Start and stop the external DP ranks owned by the current node."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
current_node_index: int,
|
||||
log_root: Path,
|
||||
):
|
||||
self.config = config
|
||||
self.ranks = ranks
|
||||
self.current_node_index = current_node_index
|
||||
self.log_root = log_root
|
||||
self.command_builder = ServerCommandBuilder(config)
|
||||
self.dist_envs = build_dist_envs(
|
||||
config.cluster_ips[current_node_index],
|
||||
config.cluster_ips[0],
|
||||
)
|
||||
self.rank_processes: list[RankProcess] = []
|
||||
|
||||
def start_current_node(self) -> None:
|
||||
local_ranks = [rank for rank in self.ranks if rank.node_index == self.current_node_index]
|
||||
logger.info("Starting %d external DP ranks on node %d", len(local_ranks), self.current_node_index)
|
||||
try:
|
||||
for rank in local_ranks:
|
||||
template = self.config.launch_templates[rank.node_index]
|
||||
template = type(template)(
|
||||
envs={**template.envs, **self.dist_envs},
|
||||
server_cmd_template=template.server_cmd_template,
|
||||
)
|
||||
server_cmd = self.command_builder.build(rank, template)
|
||||
log_file = self._rank_log_file(rank)
|
||||
process = start_logged_process(server_cmd.cmd, server_cmd.env, log_file)
|
||||
self.rank_processes.append((process, rank, log_file))
|
||||
|
||||
wait_ranks_ready(
|
||||
local_ranks,
|
||||
timeout=SERVER_READY_TIMEOUT_SECONDS,
|
||||
rank_processes=self.rank_processes,
|
||||
)
|
||||
except Exception:
|
||||
self.cleanup()
|
||||
raise
|
||||
|
||||
def __enter__(self):
|
||||
self.start_current_node()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
self.cleanup()
|
||||
|
||||
def cleanup(self) -> None:
|
||||
for process, rank, _log_file in reversed(self.rank_processes):
|
||||
logger.info(
|
||||
"Stopping external DP rank node=%d rank=%d pid=%d",
|
||||
rank.node_index,
|
||||
rank.local_rank,
|
||||
process.pid,
|
||||
)
|
||||
terminate_process_tree(process.pid)
|
||||
self.rank_processes.clear()
|
||||
|
||||
def _rank_log_file(self, rank: RankInfo) -> Path:
|
||||
return self.log_root / f"node-{rank.node_index}" / f"rank-{rank.local_rank}.log"
|
||||
|
||||
|
||||
class ExternalDPProxyLauncher:
|
||||
"""Launch the external DP proxy on the configured proxy node."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
current_node_index: int,
|
||||
log_root: Path,
|
||||
):
|
||||
self.config = config
|
||||
self.ranks = ranks
|
||||
self.current_node_index = current_node_index
|
||||
self.log_root = log_root
|
||||
self.pid: int | None = None
|
||||
|
||||
def start(self) -> None:
|
||||
if self.current_node_index != self.config.routing.proxy_node_index:
|
||||
logger.info("Current node is not proxy node, skip launching external DP proxy")
|
||||
return
|
||||
|
||||
cmd = build_proxy_server_cmd(self.config, self.ranks)
|
||||
log_file = self.log_root / f"node-{self.current_node_index}" / "proxy.log"
|
||||
process = start_logged_process(cmd, {}, log_file)
|
||||
self.pid = process.pid
|
||||
logger.info("External DP proxy launched: %s", proxy_server_health_url(self.config))
|
||||
|
||||
def wait_ready(self, timeout: int = 300) -> None:
|
||||
wait_http_ready(proxy_server_health_url(self.config), timeout=timeout)
|
||||
logger.info("External DP proxy ready: %s", proxy_server_health_url(self.config))
|
||||
|
||||
def __enter__(self):
|
||||
self.start()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
self.cleanup()
|
||||
|
||||
def cleanup(self) -> None:
|
||||
if self.pid is None:
|
||||
return
|
||||
logger.info("Stopping external DP proxy pid=%d", self.pid)
|
||||
terminate_process_tree(self.pid)
|
||||
self.pid = None
|
||||
|
||||
|
||||
def build_all_server_commands(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[ServerCommand]:
|
||||
return ServerCommandBuilder(config).build_all(ranks)
|
||||
|
||||
|
||||
def build_dist_envs(cur_ip: str, master_ip: str) -> dict[str, str]:
|
||||
nic_name = get_net_interface(cur_ip)
|
||||
return {
|
||||
"HCCL_IF_IP": cur_ip,
|
||||
"HCCL_SOCKET_IFNAME": nic_name,
|
||||
"GLOO_SOCKET_IFNAME": nic_name,
|
||||
"TP_SOCKET_IFNAME": nic_name,
|
||||
"LOCAL_IP": cur_ip,
|
||||
"NIC_NAME": nic_name,
|
||||
"MASTER_IP": master_ip,
|
||||
}
|
||||
|
||||
|
||||
def build_proxy_server_cmd(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[str]:
|
||||
routing = config.routing
|
||||
cmd = [sys.executable, routing.proxy_script, "--host", routing.proxy_host, "--port", str(routing.proxy_port)]
|
||||
|
||||
if routing.type == ROUTING_GENERIC_DP:
|
||||
worker_ranks = [rank for rank in ranks if rank.role == "worker"]
|
||||
if not worker_ranks:
|
||||
raise ValueError("generic_dp proxy requires worker ranks")
|
||||
cmd.extend(["--dp-hosts", *[rank.host for rank in worker_ranks]])
|
||||
cmd.extend(["--dp-ports", *[str(rank.port) for rank in worker_ranks]])
|
||||
return cmd
|
||||
|
||||
if routing.type == ROUTING_DISAGGREGATED_PREFILL:
|
||||
prefiller_ranks = [rank for rank in ranks if rank.role == "prefiller"]
|
||||
decoder_ranks = [rank for rank in ranks if rank.role == "decoder"]
|
||||
if not prefiller_ranks or not decoder_ranks:
|
||||
raise ValueError("disaggregated_prefill proxy requires prefiller and decoder ranks")
|
||||
cmd.extend(["--prefiller-hosts", *[rank.host for rank in prefiller_ranks]])
|
||||
cmd.extend(["--prefiller-ports", *[str(rank.port) for rank in prefiller_ranks]])
|
||||
cmd.extend(["--decoder-hosts", *[rank.host for rank in decoder_ranks]])
|
||||
cmd.extend(["--decoder-ports", *[str(rank.port) for rank in decoder_ranks]])
|
||||
return cmd
|
||||
|
||||
raise ValueError(f"Unsupported routing.type: {routing.type}")
|
||||
|
||||
|
||||
def proxy_server_health_url(config: ExternalDPConfig) -> str:
|
||||
return f"http://{config.routing.proxy_host}:{config.routing.proxy_port}/healthcheck"
|
||||
|
||||
|
||||
def rank_health_url(rank: RankInfo) -> str:
|
||||
return f"http://{rank.host}:{rank.port}/health"
|
||||
|
||||
|
||||
def master_rank_health_url(ranks: list[RankInfo]) -> str:
|
||||
for rank in ranks:
|
||||
if rank.node_index == 0 and rank.local_rank == 0:
|
||||
return rank_health_url(rank)
|
||||
raise RuntimeError("External DP master rank was not found")
|
||||
|
||||
|
||||
def rank_label(rank: RankInfo) -> str:
|
||||
return f"node={rank.node_index} rank={rank.local_rank} role={rank.role} url={rank_health_url(rank)}"
|
||||
|
||||
|
||||
def format_http_status(label: str, url: str) -> str:
|
||||
status = "ready" if is_http_ready(url, timeout=1.0) else "waiting"
|
||||
return f"{label}={status} url={url}"
|
||||
|
||||
|
||||
def _format_rank_statuses(
|
||||
ranks: list[RankInfo],
|
||||
rank_ready: dict[RankInfo, bool],
|
||||
) -> str:
|
||||
parts = []
|
||||
for rank in ranks:
|
||||
status = "ready" if rank_ready[rank] else "waiting"
|
||||
parts.append(f" {rank_label(rank)} status={status}")
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _raise_if_rank_process_exited(rank_processes: list[RankProcess] | None) -> None:
|
||||
if not rank_processes:
|
||||
return
|
||||
|
||||
exited = []
|
||||
for process, rank, log_file in rank_processes:
|
||||
returncode = process.poll()
|
||||
if returncode is not None:
|
||||
exited.append(f"{rank_label(rank)} pid={process.pid} returncode={returncode} log={log_file}")
|
||||
|
||||
if exited:
|
||||
raise RuntimeError("External DP rank process exited before ready: " + "; ".join(exited))
|
||||
|
||||
|
||||
def wait_ranks_ready(
|
||||
ranks: Iterable[RankInfo],
|
||||
timeout: int,
|
||||
rank_processes: list[RankProcess] | None = None,
|
||||
) -> None:
|
||||
ranks = list(ranks)
|
||||
rank_ready = {rank: False for rank in ranks}
|
||||
deadline = time.monotonic() + timeout
|
||||
last_log_time = 0.0
|
||||
|
||||
while True:
|
||||
_raise_if_rank_process_exited(rank_processes)
|
||||
|
||||
all_ready = True
|
||||
unhealthy_after_ready = []
|
||||
|
||||
for rank in ranks:
|
||||
is_ready = is_http_ready(rank_health_url(rank), timeout=1.0)
|
||||
if is_ready:
|
||||
if not rank_ready[rank]:
|
||||
logger.info("[READY] External DP rank %s", rank_label(rank))
|
||||
rank_ready[rank] = True
|
||||
continue
|
||||
|
||||
all_ready = False
|
||||
if rank_ready[rank]:
|
||||
unhealthy_after_ready.append(rank)
|
||||
|
||||
if unhealthy_after_ready:
|
||||
failed = "; ".join(rank_label(rank) for rank in unhealthy_after_ready)
|
||||
raise RuntimeError(f"External DP rank became unhealthy after ready: {failed}")
|
||||
|
||||
if all_ready:
|
||||
return
|
||||
|
||||
now = time.monotonic()
|
||||
if now - last_log_time >= 30:
|
||||
logger.info(
|
||||
"Polling external DP ranks: ready=%d/%d\n%s",
|
||||
sum(rank_ready.values()),
|
||||
len(ranks),
|
||||
_format_rank_statuses(ranks, rank_ready),
|
||||
)
|
||||
last_log_time = now
|
||||
|
||||
if now >= deadline:
|
||||
pending = [rank for rank in ranks if not rank_ready[rank]]
|
||||
pending_labels = "; ".join(rank_label(rank) for rank in pending)
|
||||
raise TimeoutError(f"Timed out waiting for external DP ranks ready: {pending_labels}")
|
||||
|
||||
time.sleep(5)
|
||||
|
||||
|
||||
def wait_master_rank_stopped(ranks: list[RankInfo], timeout: int) -> None:
|
||||
url = master_rank_health_url(ranks)
|
||||
wait_http_ready(url, timeout=SERVER_READY_TIMEOUT_SECONDS)
|
||||
logger.info("Hanging until master external DP rank stops: %s", url)
|
||||
wait_http_unready(url, timeout=timeout)
|
||||
@@ -0,0 +1,163 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ExternalDPConfig,
|
||||
ExternalDPConfigLoader,
|
||||
RankResolver,
|
||||
resolve_current_node_index,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import (
|
||||
ExternalDPProxyLauncher,
|
||||
ExternalDPServerManager,
|
||||
build_all_server_commands,
|
||||
format_http_status,
|
||||
master_rank_health_url,
|
||||
proxy_server_health_url,
|
||||
wait_master_rank_stopped,
|
||||
wait_ranks_ready,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
|
||||
collect_logs,
|
||||
write_benchmark_results_json,
|
||||
)
|
||||
from tools.aisbench import run_aisbench_cases
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_LOG_ROOT = Path("/tmp/external_dp_logs")
|
||||
|
||||
|
||||
def _install_special_dependencies(config: ExternalDPConfig) -> None:
|
||||
for package, version in config.special_dependencies.items():
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"pip",
|
||||
"install",
|
||||
f"{package}=={version}",
|
||||
]
|
||||
subprocess.call(command)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _heartbeat(
|
||||
task_name: str,
|
||||
*,
|
||||
interval: int = 30,
|
||||
status_fn: Callable[[], str] | None = None,
|
||||
):
|
||||
start_time = time.monotonic()
|
||||
stop_event = threading.Event()
|
||||
|
||||
def report_progress() -> None:
|
||||
while not stop_event.wait(interval):
|
||||
elapsed = int(time.monotonic() - start_time)
|
||||
status = ""
|
||||
if status_fn is not None:
|
||||
try:
|
||||
status = f" {status_fn()}"
|
||||
except Exception as exc: # pragma: no cover - diagnostic only
|
||||
status = f" status_error={exc!r}"
|
||||
logger.info("%s still running: elapsed=%ds%s", task_name, elapsed, status)
|
||||
|
||||
logger.info("%s started", task_name)
|
||||
thread = threading.Thread(target=report_progress, daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
stop_event.set()
|
||||
thread.join(timeout=1)
|
||||
elapsed = int(time.monotonic() - start_time)
|
||||
logger.info("%s finished: elapsed=%ds", task_name, elapsed)
|
||||
|
||||
|
||||
def _format_benchmark_cases(config: ExternalDPConfig) -> str:
|
||||
names = [str(case.get("case_name", "<unnamed>")) for case in config.benchmark_cases]
|
||||
return ", ".join(names) if names else "<none>"
|
||||
|
||||
|
||||
def _archive_rank_logs(log_root: Path, current_node_index: int) -> None:
|
||||
log_prefix = os.environ.get("LOG_PREFIX")
|
||||
if not log_prefix:
|
||||
return
|
||||
node_log_dir = log_root / f"node-{current_node_index}"
|
||||
output_tar = Path(log_prefix) / f"node_{current_node_index}_external_dp_logs.tar.gz"
|
||||
collect_logs(node_log_dir, output_tar)
|
||||
|
||||
|
||||
def test_external_dp() -> None:
|
||||
config = ExternalDPConfigLoader.from_yaml()
|
||||
_install_special_dependencies(config)
|
||||
ranks = RankResolver(config).resolve()
|
||||
current_node_index = resolve_current_node_index(config)
|
||||
log_root = Path(os.environ.get("EXTERNAL_DP_LOG_DIR", str(DEFAULT_LOG_ROOT)))
|
||||
max_wait_seconds = int(os.environ.get("EXTERNAL_DP_MAX_WAIT_SECONDS", "3600"))
|
||||
is_master = current_node_index == 0
|
||||
|
||||
server_manager = ExternalDPServerManager(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
current_node_index=current_node_index,
|
||||
log_root=log_root,
|
||||
)
|
||||
proxy_launcher = ExternalDPProxyLauncher(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
current_node_index=current_node_index,
|
||||
log_root=log_root,
|
||||
)
|
||||
|
||||
try:
|
||||
with server_manager, proxy_launcher:
|
||||
if is_master:
|
||||
wait_ranks_ready(ranks, timeout=max_wait_seconds)
|
||||
proxy_launcher.wait_ready()
|
||||
target = f"http://{config.routing.proxy_host}:{config.routing.proxy_port}"
|
||||
logger.info(
|
||||
"Running AISBench cases: model=%s target=%s cases=[%s]",
|
||||
config.model,
|
||||
target,
|
||||
_format_benchmark_cases(config),
|
||||
)
|
||||
with _heartbeat(
|
||||
"Running AISBench",
|
||||
status_fn=lambda: format_http_status("proxy", proxy_server_health_url(config)),
|
||||
):
|
||||
results = run_aisbench_cases(
|
||||
model=config.model,
|
||||
port=config.routing.proxy_port,
|
||||
aisbench_cases=config.benchmark_cases,
|
||||
host_ip=config.routing.proxy_host,
|
||||
)
|
||||
logger.info("AISBench completed: results=%d", len(results or []))
|
||||
all_commands = build_all_server_commands(config, ranks)
|
||||
write_benchmark_results_json(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
commands=all_commands,
|
||||
results=results,
|
||||
)
|
||||
wait_ranks_ready(ranks, timeout=30)
|
||||
else:
|
||||
master_url = master_rank_health_url(ranks)
|
||||
with _heartbeat(
|
||||
"Waiting for master external DP rank to stop",
|
||||
status_fn=lambda: format_http_status("master", master_url),
|
||||
):
|
||||
wait_master_rank_stopped(ranks, timeout=max_wait_seconds)
|
||||
finally:
|
||||
_archive_rank_logs(log_root, current_node_index)
|
||||
236
tests/e2e/nightly/multi_node/external_dp/scripts/utils.py
Normal file
236
tests/e2e/nightly/multi_node/external_dp/scripts/utils.py
Normal file
@@ -0,0 +1,236 @@
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import signal
|
||||
import subprocess
|
||||
import tarfile
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ROUTING_DISAGGREGATED_PREFILL,
|
||||
ExternalDPConfig,
|
||||
RankInfo,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
|
||||
build_task_entry,
|
||||
extract_hardware,
|
||||
filter_environment,
|
||||
get_vllm_version,
|
||||
write_results_json,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import ServerCommand
|
||||
|
||||
SENSITIVE_ENV_TOKENS = ("TOKEN", "SECRET", "PASSWORD", "ACCESS_KEY")
|
||||
|
||||
|
||||
def format_server_cmd(cmd: list[str], env: dict[str, str] | None = None) -> str:
|
||||
env_parts: list[str] = []
|
||||
for key, value in sorted((env or {}).items()):
|
||||
display_value = "***" if any(token in key.upper() for token in SENSITIVE_ENV_TOKENS) else str(value)
|
||||
env_parts.append(f"{key}={shlex.quote(display_value)}")
|
||||
return " ".join([*env_parts, shlex.join(cmd)])
|
||||
|
||||
|
||||
def start_logged_process(cmd: list[str], env: dict[str, str], log_file: Path) -> subprocess.Popen:
|
||||
log_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
merged_env = {**os.environ, **env}
|
||||
with log_file.open("ab") as f:
|
||||
f.write(f"Starting command: {format_server_cmd(cmd, env)}\n".encode())
|
||||
f.flush()
|
||||
return subprocess.Popen(
|
||||
cmd,
|
||||
stdout=f,
|
||||
stderr=subprocess.STDOUT,
|
||||
env=merged_env,
|
||||
start_new_session=True,
|
||||
)
|
||||
|
||||
|
||||
def terminate_process_tree(pid: int, timeout: int = 30) -> None:
|
||||
try:
|
||||
import psutil
|
||||
except ModuleNotFoundError:
|
||||
try:
|
||||
os.killpg(pid, signal.SIGTERM)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
time.sleep(0.2)
|
||||
try:
|
||||
os.killpg(pid, signal.SIGKILL)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
return
|
||||
|
||||
try:
|
||||
parent = psutil.Process(pid)
|
||||
except psutil.NoSuchProcess:
|
||||
return
|
||||
|
||||
children = parent.children(recursive=True)
|
||||
for process in children:
|
||||
process.terminate()
|
||||
parent.terminate()
|
||||
|
||||
gone, alive = psutil.wait_procs([parent, *children], timeout=timeout)
|
||||
del gone
|
||||
for process in alive:
|
||||
process.kill()
|
||||
|
||||
|
||||
def is_http_ready(url: str, timeout: float = 5.0) -> bool:
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=timeout) as response:
|
||||
return 200 <= response.status < 300
|
||||
except (urllib.error.URLError, TimeoutError, OSError):
|
||||
return False
|
||||
|
||||
|
||||
def wait_http_ready(url: str, timeout: int, interval: float = 2.0) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
last_error: Exception | None = None
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=5) as response:
|
||||
if 200 <= response.status < 300:
|
||||
return
|
||||
except (urllib.error.URLError, TimeoutError, OSError) as exc:
|
||||
last_error = exc
|
||||
time.sleep(interval)
|
||||
raise TimeoutError(f"Timed out waiting for HTTP ready: {url}; last_error={last_error}")
|
||||
|
||||
|
||||
def wait_http_unready(url: str, timeout: int, interval: float = 5.0) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if not is_http_ready(url):
|
||||
return
|
||||
time.sleep(interval)
|
||||
raise TimeoutError(f"Timed out waiting for HTTP unready: {url}")
|
||||
|
||||
|
||||
def collect_logs(src_dir: Path, output_tar: Path) -> None:
|
||||
if not src_dir.exists():
|
||||
return
|
||||
output_tar.parent.mkdir(parents=True, exist_ok=True)
|
||||
with tarfile.open(output_tar, "w:gz") as tar:
|
||||
tar.add(src_dir, arcname=src_dir.name)
|
||||
|
||||
|
||||
def _common_command_envs(commands: list["ServerCommand"]) -> dict[str, str]:
|
||||
if not commands:
|
||||
return {}
|
||||
|
||||
common_keys = set(commands[0].env)
|
||||
for command in commands[1:]:
|
||||
common_keys.intersection_update(command.env)
|
||||
|
||||
common_envs: dict[str, str] = {}
|
||||
for key in sorted(common_keys):
|
||||
values = {command.env[key] for command in commands}
|
||||
if len(values) == 1:
|
||||
common_envs[key] = next(iter(values))
|
||||
return common_envs
|
||||
|
||||
|
||||
def _extract_dtype(config: ExternalDPConfig, commands: list["ServerCommand"]) -> str:
|
||||
has_w8a8 = "w8a8" in config.model.lower()
|
||||
has_quant_ascend = any("--quantization ascend" in command.display_cmd for command in commands)
|
||||
return "w8a8" if has_w8a8 and has_quant_ascend else "bf16"
|
||||
|
||||
|
||||
def _extract_features(commands: list["ServerCommand"]) -> list[str]:
|
||||
if not commands:
|
||||
return []
|
||||
features: list[str] = []
|
||||
command_args = [command.cmd for command in commands]
|
||||
command_displays = [" ".join(shlex.quote(arg) for arg in command.cmd) for command in commands]
|
||||
|
||||
if any("--async-scheduling" in cmd for cmd in command_args):
|
||||
features.append("async_scheduling")
|
||||
if any("--enable-expert-parallel" in cmd for cmd in command_args):
|
||||
features.append("expert_parallel")
|
||||
if any("--speculative-config" in cmd for cmd in command_args):
|
||||
features.append("speculative")
|
||||
if any("cudagraph_mode" in display for display in command_displays):
|
||||
features.append("aclgraph")
|
||||
|
||||
feature_envs = {
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
|
||||
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
|
||||
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
|
||||
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
|
||||
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
|
||||
}
|
||||
for env_key, feature_name in feature_envs.items():
|
||||
values = [str(command.env.get(env_key, "0")) for command in commands]
|
||||
if any(value not in ("0", "", "false", "False") for value in values):
|
||||
features.append(feature_name)
|
||||
return features
|
||||
|
||||
|
||||
def _build_serve_cmd(
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
) -> dict[str, Any]:
|
||||
entries: dict[str, str] = {}
|
||||
for rank, command in zip(ranks, commands):
|
||||
prefix = rank.role
|
||||
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL:
|
||||
prefix = "prefill" if rank.role == "prefiller" else "decode"
|
||||
entries[f"{prefix}-node{rank.node_index}-rank{rank.local_rank}"] = command.display_cmd
|
||||
key = "external_dp_pd" if config.routing.type == ROUTING_DISAGGREGATED_PREFILL else "external_dp"
|
||||
return {key: entries}
|
||||
|
||||
|
||||
def build_benchmark_results(
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
results: list[Any],
|
||||
) -> dict[str, Any]:
|
||||
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
|
||||
tasks = [build_task_entry(key, case, result) for (key, case), result in zip(valid_items, results)]
|
||||
runner = os.environ.get("VLLM_CI_RUNNER", "")
|
||||
common_envs = _common_command_envs(commands)
|
||||
|
||||
return {
|
||||
"model_name": config.model,
|
||||
"hardware": extract_hardware(runner),
|
||||
"dtype": _extract_dtype(config, commands),
|
||||
"feature": _extract_features(commands),
|
||||
"vllm_version": get_vllm_version(),
|
||||
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
|
||||
"tasks": tasks,
|
||||
"serve_cmd": _build_serve_cmd(config, ranks, commands),
|
||||
"environment": filter_environment(common_envs),
|
||||
}
|
||||
|
||||
|
||||
def write_benchmark_results_json(
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
results: list[Any],
|
||||
output_dir: Path | None = None,
|
||||
) -> Path:
|
||||
output = build_benchmark_results(config=config, ranks=ranks, commands=commands, results=results)
|
||||
job_name = os.environ.get("BENCHMARK_JOB_NAME", "") or config.test_name.replace(" ", "-")
|
||||
return write_results_json(output, job_name=job_name, output_dir=output_dir)
|
||||
@@ -0,0 +1,196 @@
|
||||
test_name: "test DeepSeek-R1-W8A8 disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 10
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_DETERMINISTIC: True
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0, L2:0"
|
||||
DYNAMIC_EPLB: true
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 4
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 4
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 32
|
||||
--data-parallel-size-local 16
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 1
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 28
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"multistream_overlap_shared_expert":true,"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--headless
|
||||
--data-parallel-size 32
|
||||
--data-parallel-size-local 16
|
||||
--data-parallel-start-rank 16
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 1
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 28
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"multistream_overlap_shared_expert":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,114 @@
|
||||
test_name: "test DeepSeek-R1-W8A8-longseq disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 768
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_DETERMINISTIC: True
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0"
|
||||
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 1
|
||||
--decode-context-parallel-size 8
|
||||
--prefill-context-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 32
|
||||
--max-model-len 32768
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.85
|
||||
--enable-chunked-prefill
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 1
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 32
|
||||
--max-model-len 32768
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.85
|
||||
--compilation_config '{"cudagraph_capture_sizes":[4,8,16,32],"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--enable-chunked-prefill
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
--additional-config '{"recompute_scheduler_enable":true}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 360
|
||||
max_out_len: 4096
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,85 @@
|
||||
test_name: "test DeepSeek-V3.1-BF16 on A3"
|
||||
model: "unsloth/DeepSeek-V3.1-BF16"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 2048
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
OMP_NUM_THREADS: 1
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 1
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve unsloth/DeepSeek-V3.1-BF16
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--no-enable-prefix-caching
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 4096
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
|
||||
--additional_config '{"enable_multistream_moe": true}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve unsloth/DeepSeek-V3.1-BF16
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--no-enable-prefix-caching
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 4096
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
|
||||
--additional_config '{"enable_multistream_moe": true}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 512
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
@@ -0,0 +1,127 @@
|
||||
test_name: "test DeepSeek-V3.2-W8A8 on A3"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
ASCEND_A3_EBA_ENABLE: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--tensor-parallel-size 8
|
||||
--quantization ascend
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 90000
|
||||
--max-num-batched-tokens 4096
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.85
|
||||
--trust-remote-code
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--data-parallel-rpc-port 13399
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--tensor-parallel-size 8
|
||||
--quantization ascend
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 90000
|
||||
--max-num-batched-tokens 4096
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.85
|
||||
--trust-remote-code
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
benchmarks:
|
||||
perf_short_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 3000
|
||||
batch_size: 512
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 3000
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_short:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 3000
|
||||
batch_size: 256
|
||||
request_rate: 11.2
|
||||
baseline: 305.2903
|
||||
threshold: 0.97
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 128
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 80000
|
||||
batch_size: 32
|
||||
baseline: 57
|
||||
threshold: 10
|
||||
@@ -0,0 +1,268 @@
|
||||
test_name: "test DeepSeek-V3.2-W8A8-EP disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 10
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: 360
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: 0
|
||||
ASCEND_AGGREGATE_ENABLE: 1
|
||||
ASCEND_TRANSPORT_PRINT: 1
|
||||
ACL_OP_INIT_MODE: 1
|
||||
ASCEND_A3_ENABLE: 1
|
||||
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
HCCL_CONNECT_TIMEOUT: 1200
|
||||
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 16
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.90
|
||||
--enforce-eager
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 16
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.90
|
||||
--enforce-eager
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 1100
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 42
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 14
|
||||
--gpu-memory-utilization 0.90
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 1100
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 42
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 14
|
||||
--gpu-memory-utilization 0.90
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf_short_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1500
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1024
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_short:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 128
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 16
|
||||
max_out_len: 1024
|
||||
batch_size: 4
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 64
|
||||
baseline: 96.88
|
||||
threshold: 10
|
||||
@@ -0,0 +1,102 @@
|
||||
test_name: "multi-node-GLM-5.1-W8A8C8-MTP-A3_64k/128k"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8c8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_BUFFSIZE: "400"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
SERVER_PORT: 8077
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8c8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--data-parallel-rpc-port 12981
|
||||
--tensor-parallel-size 4
|
||||
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
|
||||
--seed 1024
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--enable-auto-tool-choice
|
||||
--max-num-seqs 6
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8c8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--data-parallel-rpc-port 12981
|
||||
--tensor-parallel-size 4
|
||||
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
|
||||
--seed 1024
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--enable-auto-tool-choice
|
||||
--max-num-seqs 6
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
|
||||
benchmarks:
|
||||
perf_128k_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
perf_128k:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 256
|
||||
max_out_len: 1024
|
||||
batch_size: 64
|
||||
request_rate: 0
|
||||
baseline: 290.8308
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,84 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-A2"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 0
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
VLLM_RPC_TIMEOUT: 600
|
||||
SERVER_PORT: 8078
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 64
|
||||
--max-model-len 38000
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 64
|
||||
--max-model-len 38000
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 72
|
||||
max_out_len: 1500
|
||||
batch_size: 18
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,98 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-A3"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 0
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "600"
|
||||
SERVER_PORT: 8080
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-rpc-port 13389
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 72348
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 12
|
||||
max_out_len: 1024
|
||||
batch_size: 3
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,240 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-EP"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: 1024
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: 0
|
||||
ASCEND_AGGREGATE_ENABLE: 1
|
||||
ASCEND_TRANSPORT_PRINT: 1
|
||||
ACL_OP_INIT_MODE: 1
|
||||
ASCEND_A3_ENABLE: 1
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: 1
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 131072
|
||||
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--max-num-seqs 64
|
||||
--quantization ascend
|
||||
--gpu-memory-utilization 0.95
|
||||
--enforce-eager
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 131072
|
||||
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--max-num-seqs 64
|
||||
--quantization ascend
|
||||
--gpu-memory-utilization 0.95
|
||||
--enforce-eager
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 10543
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 202752
|
||||
--max-num-batched-tokens 32
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
|
||||
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 8
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 10543
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 202752
|
||||
--max-num-batched-tokens 32
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
|
||||
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 8
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1500
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,91 @@
|
||||
test_name: "multi-node-GLM-5.2-w8a8-A3"
|
||||
model: "Eco-Tech/GLM-5.2-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "600"
|
||||
SERVER_PORT: 8080
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.12.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.2-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.2-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-rpc-port 13389
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 72348
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 12
|
||||
max_out_len: 1024
|
||||
batch_size: 3
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,90 @@
|
||||
test_name: "test Kimi-K2.5-W4A8 A2 dual nodes"
|
||||
model: "Eco-Tech/Kimi-K2.5-W4A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 8
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 512
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/Kimi-K2.5-W4A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--quantization ascend
|
||||
--allowed-local-media-path /
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--seed 42
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 64
|
||||
--max-model-len 51200
|
||||
--max-num-batched-tokens 8192
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
--mm-processor-cache-gb 0
|
||||
--mm-encoder-tp-mode data
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/Kimi-K2.5-W4A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--quantization ascend
|
||||
--allowed-local-media-path /
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--seed 42
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 64
|
||||
--max-model-len 51200
|
||||
--max-num-batched-tokens 8192
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
--mm-processor-cache-gb 0
|
||||
--mm-encoder-tp-mode data
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 320
|
||||
max_out_len: 1500
|
||||
batch_size: 80
|
||||
trust_remote_code: True
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,74 @@
|
||||
test_name: "test Qwen3-235B-A22B multi-dp on A2"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 8
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 40960
|
||||
--max-num-batched-tokens 2048
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--headless
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 128
|
||||
--max-model-len 40960
|
||||
--max-num-batched-tokens 2048
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 256
|
||||
request_rate: 4.8
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 256
|
||||
baseline: 96
|
||||
threshold: 10
|
||||
@@ -0,0 +1,77 @@
|
||||
test_name: "test Qwen3-235B-A22B multi-dp"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 3
|
||||
@@ -0,0 +1,93 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8 EPLB"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
DYNAMIC_EPLB: true
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":50,"algorithm_execution_interval":5}}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":600,"algorithm_execution_interval":50}}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,100 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
DYNAMIC_EPLB: true
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 1
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--seed 1024
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"dynamic_eplb":true}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 1
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation_config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"dynamic_eplb":true}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,89 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8 disaggregated_prefill"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,118 @@
|
||||
test_name: "test Qwen3-235B-A22B disaggregated_prefill"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_BUFFSIZE: 1024
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: 2
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
SERVER_PORT: 8080
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--no-enable-prefix-caching
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--no-enable-prefix-caching
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 512
|
||||
baseline: 97
|
||||
threshold: 10
|
||||
@@ -0,0 +1,108 @@
|
||||
test_name: "test Qwen3-VL-235B-A22B disaggregated_prefill"
|
||||
model: "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 4
|
||||
--tensor-parallel-size 4
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/textvqa-perf-1080p
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 64
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/textvqa-lite
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
max_out_len: 7680
|
||||
batch_size: 64
|
||||
baseline: 85
|
||||
threshold: 5
|
||||
@@ -0,0 +1,331 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
get_available_port,
|
||||
get_net_interface,
|
||||
load_yaml_mapping,
|
||||
resolve_cluster_ips,
|
||||
resolve_current_node_index,
|
||||
setup_logger,
|
||||
)
|
||||
|
||||
setup_logger()
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
|
||||
DEFAULT_SERVER_PORT = 8080
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeInfo:
|
||||
index: int
|
||||
ip: str
|
||||
server_cmd: str
|
||||
envs: dict[str, Any] | None = None
|
||||
headless: bool = False
|
||||
|
||||
def __post_init__(self):
|
||||
if not self.ip:
|
||||
raise ValueError("NodeInfo.ip must not be empty")
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"NodeInfo(\n index={self.index},\n ip={self.ip},\n headless={self.headless},\n)"
|
||||
|
||||
|
||||
class DisaggregatedPrefillCfg:
|
||||
def __init__(self, raw_cfg: dict, num_nodes: int):
|
||||
self.prefiller_indices: list[int] = raw_cfg.get("prefiller_host_index", [])
|
||||
self.decoder_indices: list[int] = raw_cfg.get("decoder_host_index", [])
|
||||
|
||||
if not self.decoder_indices:
|
||||
raise RuntimeError("decoder_host_index must be provided")
|
||||
|
||||
self._validate(num_nodes)
|
||||
|
||||
self.decode_start_index = self.decoder_indices[0]
|
||||
self.num_prefillers = len(self.prefiller_indices)
|
||||
self.num_decoders = len(self.decoder_indices)
|
||||
|
||||
def _validate(self, num_nodes: int):
|
||||
overlap = set(self.prefiller_indices) & set(self.decoder_indices)
|
||||
if overlap:
|
||||
raise AssertionError(f"Prefiller and decoder overlap: {overlap}")
|
||||
|
||||
all_indices = self.prefiller_indices + self.decoder_indices
|
||||
if any(i >= num_nodes for i in all_indices):
|
||||
raise ValueError("Disaggregated prefill index out of range")
|
||||
|
||||
def is_prefiller(self, index: int) -> bool:
|
||||
return index in self.prefiller_indices
|
||||
|
||||
def is_decoder(self, index: int) -> bool:
|
||||
return index in self.decoder_indices
|
||||
|
||||
def master_ip_for_node(self, index: int, nodes: list[NodeInfo]) -> str:
|
||||
if self.is_prefiller(index):
|
||||
return nodes[0].ip
|
||||
return nodes[self.decode_start_index].ip
|
||||
|
||||
|
||||
class DistEnvBuilder:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
cur_node: NodeInfo,
|
||||
master_ip: str,
|
||||
):
|
||||
self.cur_ip = cur_node.ip
|
||||
self.nic_name = get_net_interface(self.cur_ip)
|
||||
self.master_ip = master_ip
|
||||
|
||||
self.base_envs = dict(cur_node.envs or {})
|
||||
|
||||
def build(self) -> dict:
|
||||
envs = dict(self.base_envs)
|
||||
|
||||
envs.update(
|
||||
{
|
||||
"HCCL_IF_IP": self.cur_ip,
|
||||
"HCCL_SOCKET_IFNAME": self.nic_name,
|
||||
"GLOO_SOCKET_IFNAME": self.nic_name,
|
||||
"TP_SOCKET_IFNAME": self.nic_name,
|
||||
"LOCAL_IP": self.cur_ip,
|
||||
"NIC_NAME": self.nic_name,
|
||||
"MASTER_IP": self.master_ip,
|
||||
}
|
||||
)
|
||||
|
||||
return {k: str(v) for k, v in envs.items()}
|
||||
|
||||
|
||||
class ProxyLauncher:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
nodes: list[NodeInfo],
|
||||
envs: dict,
|
||||
proxy_port: int,
|
||||
cur_index: int,
|
||||
disagg_cfg: DisaggregatedPrefillCfg | None = None,
|
||||
):
|
||||
self.nodes = nodes
|
||||
self.cfg = disagg_cfg
|
||||
self.server_port = envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
self.proxy_port = proxy_port
|
||||
self.proxy_script = envs.get(
|
||||
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
|
||||
"examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
|
||||
)
|
||||
self.envs = envs
|
||||
self.is_master = cur_index == 0
|
||||
self.cur_ip = nodes[cur_index].ip
|
||||
self.process: subprocess.Popen[bytes] | None = None
|
||||
|
||||
def __enter__(self):
|
||||
if not self.is_master or self.cfg is None:
|
||||
logger.info("Not launching proxy on non-master node")
|
||||
return self
|
||||
prefiller_ips = [self.nodes[i].ip for i in self.cfg.prefiller_indices if not self.nodes[i].headless]
|
||||
decoder_ips = [self.nodes[i].ip for i in self.cfg.decoder_indices if not self.nodes[i].headless]
|
||||
|
||||
cmd = [
|
||||
"python",
|
||||
self.proxy_script,
|
||||
"--host",
|
||||
self.cur_ip,
|
||||
"--port",
|
||||
str(self.proxy_port),
|
||||
"--prefiller-hosts",
|
||||
*prefiller_ips,
|
||||
"--prefiller-ports",
|
||||
*[str(self.server_port)] * len(prefiller_ips),
|
||||
"--decoder-hosts",
|
||||
*decoder_ips,
|
||||
"--decoder-ports",
|
||||
*[str(self.server_port)] * len(decoder_ips),
|
||||
]
|
||||
|
||||
logger.info("Launching proxy: %s", " ".join(cmd))
|
||||
self.process = subprocess.Popen(cmd, env={**os.environ, **self.envs})
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc, tb):
|
||||
if not self.process:
|
||||
return
|
||||
logger.info("Stopping proxy server...")
|
||||
self.process.terminate()
|
||||
try:
|
||||
self.process.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
self.process.kill()
|
||||
|
||||
|
||||
class MultiNodeConfig:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
model: str,
|
||||
test_name: str,
|
||||
nodes: list[NodeInfo],
|
||||
npu_per_node: int,
|
||||
disaggregated_prefill: dict | None,
|
||||
benchmark_cases: list[dict],
|
||||
special_dependencies: dict,
|
||||
):
|
||||
self.model = model
|
||||
self.test_name = test_name
|
||||
self.nodes = nodes
|
||||
self.npu_per_node = npu_per_node
|
||||
self.benchmark_cases = benchmark_cases
|
||||
|
||||
self.cur_index = self._resolve_cur_index()
|
||||
self.cur_node = self.nodes[self.cur_index]
|
||||
self.special_dependencies = special_dependencies
|
||||
|
||||
self.disagg_cfg = DisaggregatedPrefillCfg(disaggregated_prefill, len(nodes)) if disaggregated_prefill else None
|
||||
|
||||
master_ip = (
|
||||
self.disagg_cfg.master_ip_for_node(self.cur_index, self.nodes) if self.disagg_cfg else self.nodes[0].ip
|
||||
)
|
||||
self.proxy_port = get_available_port()
|
||||
|
||||
self.envs = DistEnvBuilder(
|
||||
cur_node=self.cur_node,
|
||||
master_ip=master_ip,
|
||||
).build()
|
||||
logger.info("Node %d envs: %s", self.cur_index, self.envs)
|
||||
|
||||
self.server_cmd = self._expand_env(self.cur_node.server_cmd)
|
||||
|
||||
def _resolve_cur_index(self) -> int:
|
||||
return resolve_current_node_index([node.ip for node in self.nodes])
|
||||
|
||||
def _expand_env(self, cmd: str) -> str:
|
||||
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
|
||||
|
||||
def repl(m):
|
||||
key = m.group(1) or m.group(2)
|
||||
return self.envs.get(key, m.group(0))
|
||||
|
||||
return pattern.sub(repl, cmd)
|
||||
|
||||
@property
|
||||
def world_size(self) -> int:
|
||||
return len(self.nodes) * self.npu_per_node
|
||||
|
||||
@property
|
||||
def is_master(self) -> bool:
|
||||
return self.cur_index == 0
|
||||
|
||||
@property
|
||||
def server_port(self) -> int:
|
||||
return self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
|
||||
@property
|
||||
def master_ip(self) -> str:
|
||||
return self.nodes[0].ip
|
||||
|
||||
@property
|
||||
def benchmark_endpoint(self) -> tuple[str, int]:
|
||||
"""
|
||||
Endpoint used by benchmark clients.
|
||||
"""
|
||||
master_ip = self.nodes[0].ip
|
||||
server_port = self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
if self.disagg_cfg:
|
||||
return master_ip, self.proxy_port
|
||||
return master_ip, server_port
|
||||
|
||||
|
||||
class MultiNodeConfigLoader:
|
||||
"""Load MultiNodeConfig from yaml file."""
|
||||
|
||||
DEFAULT_CONFIG_NAME = "DeepSeek-V3.yaml"
|
||||
|
||||
@classmethod
|
||||
def from_yaml(cls, yaml_path: str | None = None) -> MultiNodeConfig:
|
||||
config = cls._load_yaml(yaml_path)
|
||||
cls._validate_root(config)
|
||||
|
||||
nodes = cls._parse_nodes(config)
|
||||
benchmarks = cls._parse_benchmarks(config)
|
||||
|
||||
return MultiNodeConfig(
|
||||
model=config["model"],
|
||||
test_name=config.get("test_name", "untitled_test"),
|
||||
nodes=nodes,
|
||||
npu_per_node=config.get("npu_per_node", 16),
|
||||
disaggregated_prefill=config.get("disaggregated_prefill"),
|
||||
special_dependencies=config.get("special_dependencies", {}),
|
||||
benchmark_cases=list(benchmarks.values()),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _load_yaml(cls, yaml_path: str | None) -> dict:
|
||||
return load_yaml_mapping(
|
||||
yaml_path,
|
||||
default_name=cls.DEFAULT_CONFIG_NAME,
|
||||
default_base_path=DEFAULT_CONFIG_BASE_PATH,
|
||||
description="config",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_root(cfg: dict):
|
||||
required = ["model", "deployment", "num_nodes", "npu_per_node", "benchmarks"]
|
||||
missing = [k for k in required if k not in cfg]
|
||||
if missing:
|
||||
raise KeyError(f"Missing required config fields: {missing}")
|
||||
|
||||
@classmethod
|
||||
def _parse_nodes(cls, cfg: dict) -> list[NodeInfo]:
|
||||
num_nodes = cfg["num_nodes"]
|
||||
deployments = cfg["deployment"]
|
||||
|
||||
if len(deployments) != num_nodes:
|
||||
raise AssertionError(f"deployment size ({len(deployments)}) != num_nodes ({num_nodes})")
|
||||
|
||||
for idx, deploy in enumerate(deployments):
|
||||
if deploy.get("envs") is None:
|
||||
raise KeyError(f"deployment[{idx}].envs is required for multi-node configs")
|
||||
|
||||
cluster_ips = cls._resolve_cluster_ips(cfg, num_nodes)
|
||||
|
||||
nodes: list[NodeInfo] = []
|
||||
for idx, deploy in enumerate(deployments):
|
||||
cmd = deploy.get("server_cmd", "")
|
||||
envs = deploy["envs"]
|
||||
nodes.append(
|
||||
NodeInfo(
|
||||
index=idx,
|
||||
ip=cluster_ips[idx],
|
||||
server_cmd=cmd,
|
||||
envs=envs,
|
||||
headless="--headless" in cmd,
|
||||
)
|
||||
)
|
||||
return nodes
|
||||
|
||||
@staticmethod
|
||||
def _parse_benchmarks(cfg: dict) -> dict:
|
||||
benchmarks = cfg.get("benchmarks") or {}
|
||||
for name, case in benchmarks.items():
|
||||
case["case_name"] = name
|
||||
return benchmarks
|
||||
|
||||
@staticmethod
|
||||
def _resolve_cluster_ips(cfg: dict, num_nodes: int) -> list[str]:
|
||||
return resolve_cluster_ips(
|
||||
cfg,
|
||||
num_nodes,
|
||||
cluster_hosts_log_message=(
|
||||
"Using cluster_hosts from config. This typically indicates that your current environment is a "
|
||||
"non-Kubernetes environment."
|
||||
),
|
||||
dns_log_message="Resolving cluster IPs via DNS...",
|
||||
)
|
||||
@@ -0,0 +1,207 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import vllm
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
from tests.e2e.nightly.multi_node.internal_dp.scripts.multi_node_config import (
|
||||
MultiNodeConfig,
|
||||
MultiNodeConfigLoader,
|
||||
ProxyLauncher,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
|
||||
build_task_entry,
|
||||
extract_hardware,
|
||||
filter_environment,
|
||||
write_results_json,
|
||||
)
|
||||
from tools.aisbench import run_aisbench_cases
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_FEATURE_ENVS: dict[str, str] = {
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
|
||||
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
|
||||
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
|
||||
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
|
||||
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
|
||||
}
|
||||
|
||||
|
||||
def _extract_dtype(config: MultiNodeConfig) -> str:
|
||||
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and any node uses --quantization ascend."""
|
||||
has_w8a8 = "w8a8" in config.model.lower()
|
||||
has_quant_ascend = any("--quantization ascend" in node.server_cmd for node in config.nodes)
|
||||
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
|
||||
|
||||
|
||||
def _cmd_to_list(server_cmd: list[str] | str) -> list[str]:
|
||||
"""Normalize server_cmd to a list of argument strings."""
|
||||
if isinstance(server_cmd, str):
|
||||
try:
|
||||
return shlex.split(server_cmd)
|
||||
except ValueError:
|
||||
return server_cmd.split()
|
||||
return list(server_cmd)
|
||||
|
||||
|
||||
def _extract_server_cmd_value(cmd_list: list[str], flag: str) -> str | None:
|
||||
"""Return the value following `flag` in a command list, or None."""
|
||||
try:
|
||||
idx = cmd_list.index(flag)
|
||||
return cmd_list[idx + 1]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
|
||||
|
||||
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
|
||||
"""Extract and JSON-parse the value following `flag` in a command list."""
|
||||
val = _extract_server_cmd_value(cmd_list, flag)
|
||||
if not val:
|
||||
return {}
|
||||
try:
|
||||
return json.loads(val)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return {}
|
||||
|
||||
|
||||
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
|
||||
"""Extract enabled feature names from server_cmd and environment variables."""
|
||||
cmd_list = _cmd_to_list(server_cmd)
|
||||
features: list[str] = []
|
||||
|
||||
# Features from --additional-config JSON
|
||||
additional = _parse_json_flag(cmd_list, "--additional-config")
|
||||
if additional.get("enable_weight_nz_layout"):
|
||||
features.append("weight_nz_layout")
|
||||
wp = additional.get("weight_prefetch_config") or {}
|
||||
if isinstance(wp, dict) and wp.get("enabled"):
|
||||
features.append("weight_prefetch")
|
||||
tc = additional.get("torchair_graph_config") or {}
|
||||
if isinstance(tc, dict) and tc.get("enabled"):
|
||||
features.append("torchair_graph")
|
||||
asc = additional.get("ascend_scheduler_config") or {}
|
||||
if isinstance(asc, dict) and asc.get("enabled"):
|
||||
features.append("ascend_scheduler")
|
||||
|
||||
# Features from --compilation-config JSON
|
||||
compilation = _parse_json_flag(cmd_list, "--compilation-config")
|
||||
if compilation.get("cudagraph_mode"):
|
||||
features.append("aclgraph")
|
||||
|
||||
# Features from --speculative-config JSON
|
||||
speculative = _parse_json_flag(cmd_list, "--speculative-config")
|
||||
if speculative:
|
||||
features.append(speculative.get("method", "speculative"))
|
||||
|
||||
# Features from direct flags
|
||||
if "--enable-expert-parallel" in cmd_list:
|
||||
features.append("expert_parallel")
|
||||
|
||||
# Features from environment variables
|
||||
for env_key, feature_name in _FEATURE_ENVS.items():
|
||||
val = str(envs.get(env_key, "0"))
|
||||
if val not in ("0", "", "false", "False"):
|
||||
features.append(feature_name)
|
||||
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
|
||||
features.append("flashcomm2")
|
||||
|
||||
return features
|
||||
|
||||
|
||||
def _build_serve_cmd(config: MultiNodeConfig) -> dict[str, Any]:
|
||||
"""Build serve_cmd dict: pd format for disaggregated, dp format for multi-node."""
|
||||
if config.disagg_cfg:
|
||||
pd: dict[str, str] = {}
|
||||
for node in config.nodes:
|
||||
idx = node.index
|
||||
if config.disagg_cfg.is_prefiller(idx):
|
||||
n = config.disagg_cfg.prefiller_indices.index(idx)
|
||||
pd[f"prefill-{n}"] = node.server_cmd
|
||||
elif config.disagg_cfg.is_decoder(idx):
|
||||
n = config.disagg_cfg.decoder_indices.index(idx)
|
||||
pd[f"decode-{n}"] = node.server_cmd
|
||||
return {"pd": pd}
|
||||
return {"dp": {f"node{node.index}": node.server_cmd for node in config.nodes}}
|
||||
|
||||
|
||||
def _save_benchmark_results_json(config: MultiNodeConfig, results: list[Any]) -> None:
|
||||
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
|
||||
runner = os.environ.get("VLLM_CI_RUNNER", "")
|
||||
|
||||
# Filter out None benchmark cases; results align with the non-None ones in order
|
||||
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
|
||||
|
||||
tasks = [build_task_entry(key, case_cfg, result) for (key, case_cfg), result in zip(valid_items, results)]
|
||||
|
||||
output: dict[str, Any] = {
|
||||
"model_name": config.model,
|
||||
"hardware": extract_hardware(runner),
|
||||
"dtype": _extract_dtype(config),
|
||||
"feature": _extract_features(config.nodes[0].server_cmd, config.envs),
|
||||
"vllm_version": vllm.__version__,
|
||||
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
|
||||
"tasks": tasks,
|
||||
"serve_cmd": _build_serve_cmd(config),
|
||||
"environment": filter_environment(config.envs),
|
||||
}
|
||||
|
||||
job_name = os.environ.get("BENCHMARK_JOB_NAME", "")
|
||||
write_results_json(output, job_name=job_name)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_multi_node() -> None:
|
||||
config = MultiNodeConfigLoader.from_yaml()
|
||||
if config.special_dependencies:
|
||||
for k, v in config.special_dependencies.items():
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"pip",
|
||||
"install",
|
||||
f"{k}=={v}",
|
||||
]
|
||||
subprocess.call(command)
|
||||
|
||||
with (
|
||||
ProxyLauncher(
|
||||
nodes=config.nodes,
|
||||
disagg_cfg=config.disagg_cfg,
|
||||
envs=config.envs,
|
||||
proxy_port=config.proxy_port,
|
||||
cur_index=config.cur_index,
|
||||
) as proxy,
|
||||
RemoteOpenAIServer(
|
||||
model=config.model,
|
||||
vllm_serve_args=config.server_cmd,
|
||||
server_port=config.server_port,
|
||||
server_host=config.master_ip,
|
||||
env_dict=config.envs,
|
||||
auto_port=False,
|
||||
proxy_port=proxy.proxy_port,
|
||||
disaggregated_prefill=config.disagg_cfg,
|
||||
nodes_info=config.nodes,
|
||||
max_wait_seconds=2800,
|
||||
) as server,
|
||||
):
|
||||
host, port = config.benchmark_endpoint
|
||||
|
||||
if config.is_master:
|
||||
results = run_aisbench_cases(
|
||||
model=config.model,
|
||||
port=port,
|
||||
aisbench_cases=config.benchmark_cases,
|
||||
host_ip=host,
|
||||
)
|
||||
_save_benchmark_results_json(config, results)
|
||||
else:
|
||||
# We should keep listening on the master node's server url determining when to exit.
|
||||
server.hang_until_terminated(f"http://{host}:{config.server_port}/health")
|
||||
28
tests/e2e/nightly/multi_node/internal_dp/scripts/utils.py
Normal file
28
tests/e2e/nightly/multi_node/internal_dp/scripts/utils.py
Normal file
@@ -0,0 +1,28 @@
|
||||
import os
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
get_all_ipv4,
|
||||
get_available_port,
|
||||
get_cluster_ips,
|
||||
get_net_interface,
|
||||
setup_logger,
|
||||
temp_env,
|
||||
)
|
||||
|
||||
DISAGGEGATED_PREFILL_PORT = 5333
|
||||
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
|
||||
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or DEFAULT_CONFIG_BASE_PATH
|
||||
DEFAULT_SERVER_PORT = 8080
|
||||
|
||||
__all__ = [
|
||||
"CONFIG_BASE_PATH",
|
||||
"DEFAULT_CONFIG_BASE_PATH",
|
||||
"DEFAULT_SERVER_PORT",
|
||||
"DISAGGEGATED_PREFILL_PORT",
|
||||
"get_all_ipv4",
|
||||
"get_available_port",
|
||||
"get_cluster_ips",
|
||||
"get_net_interface",
|
||||
"setup_logger",
|
||||
"temp_env",
|
||||
]
|
||||
1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
|
||||
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
@@ -0,0 +1,128 @@
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
|
||||
INFRA_ENV_KEYS = {
|
||||
"HCCL_IF_IP",
|
||||
"HCCL_SOCKET_IFNAME",
|
||||
"GLOO_SOCKET_IFNAME",
|
||||
"TP_SOCKET_IFNAME",
|
||||
"LOCAL_IP",
|
||||
"NIC_NAME",
|
||||
"MASTER_IP",
|
||||
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
|
||||
}
|
||||
PERF_METRIC_RENAME: dict[str, str] = {
|
||||
"Benchmark Duration": "Benchmark_Duration(BD)",
|
||||
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
|
||||
"Input Token Throughput": "Input_Token_Throughput(ITT)",
|
||||
"Output Token Throughput": "Output_Token_Throughput(OTT)",
|
||||
"Total Token Throughput": "Total_Token_Throughput(TTT)",
|
||||
}
|
||||
|
||||
|
||||
def extract_hardware(runner: str) -> str:
|
||||
runner_lower = runner.lower()
|
||||
for label in ("a3", "a2"):
|
||||
if label in runner_lower:
|
||||
return label.upper()
|
||||
return runner
|
||||
|
||||
|
||||
def get_vllm_version() -> str:
|
||||
try:
|
||||
import vllm
|
||||
|
||||
return vllm.__version__
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
|
||||
if result == "":
|
||||
return False
|
||||
case_type = case_config.get("case_type")
|
||||
baseline = case_config.get("baseline")
|
||||
threshold = case_config.get("threshold")
|
||||
if baseline is None or threshold is None:
|
||||
return True
|
||||
if case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
return abs(float(result) - float(baseline)) <= float(threshold)
|
||||
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
|
||||
try:
|
||||
throughput_val = float(throughput_str.replace("token/s", "").strip())
|
||||
return throughput_val >= float(threshold) * float(baseline)
|
||||
except (ValueError, AttributeError):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
|
||||
dataset_path = case_config.get("dataset_path", "")
|
||||
dataset_conf = case_config.get("dataset_conf", "")
|
||||
if dataset_path:
|
||||
task_name = dataset_path.split("/", 1)[-1]
|
||||
elif dataset_conf:
|
||||
task_name = dataset_conf.split("/")[0]
|
||||
else:
|
||||
task_name = case_key
|
||||
|
||||
case_type = case_config.get("case_type", "unknown")
|
||||
metrics: dict[str, float] = {}
|
||||
if result == "":
|
||||
pass
|
||||
elif case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
metrics["accuracy"] = round(float(result), 4)
|
||||
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
for metric_name, metric_data in result_json.items():
|
||||
if not isinstance(metric_data, dict):
|
||||
continue
|
||||
total_str = metric_data.get("total", "")
|
||||
try:
|
||||
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
|
||||
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
|
||||
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
|
||||
|
||||
target: dict[str, Any] = {}
|
||||
if case_config.get("baseline") is not None:
|
||||
target["baseline"] = case_config["baseline"]
|
||||
if case_config.get("threshold") is not None:
|
||||
target["threshold"] = case_config["threshold"]
|
||||
|
||||
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
|
||||
if target:
|
||||
entry["target"] = target
|
||||
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
|
||||
return entry
|
||||
|
||||
|
||||
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
|
||||
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
|
||||
return {key: value for key, value in envs.items() if key not in exclude}
|
||||
|
||||
|
||||
def write_results_json(
|
||||
output: dict[str, Any],
|
||||
*,
|
||||
job_name: str,
|
||||
output_dir: Path | None = None,
|
||||
) -> Path:
|
||||
if output_dir is None:
|
||||
output_dir = Path("/root/.cache/benchmark_results") / job_name
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
output_path = output_dir / f"{job_name}.json"
|
||||
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
logger.info("Benchmark results saved to PVC at %s", output_path)
|
||||
print(f"Benchmark results saved to PVC at {output_path}")
|
||||
return output_path
|
||||
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
@@ -0,0 +1,174 @@
|
||||
apiVersion: leaderworkerset.x-k8s.io/v1
|
||||
kind: LeaderWorkerSet
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
replicas: {{ replicas | default(1) }}
|
||||
leaderWorkerTemplate:
|
||||
size: {{ size | default(2) }}
|
||||
restartPolicy: None
|
||||
leaderTemplate:
|
||||
metadata:
|
||||
labels:
|
||||
role: leader
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-leader
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: VLLM_ASCEND_VERSION
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
# readinessProbe:
|
||||
# tcpSocket:
|
||||
# port: 8080
|
||||
# initialDelaySeconds: 15
|
||||
# periodSeconds: 10
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
workerTemplate:
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-worker
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}-leader
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
ports:
|
||||
- name: http
|
||||
port: 8080
|
||||
protocol: TCP
|
||||
targetPort: 8080
|
||||
selector:
|
||||
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
|
||||
role: leader
|
||||
type: ClusterIP
|
||||
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
@@ -0,0 +1,466 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Color definitions
|
||||
GREEN="\033[0;32m"
|
||||
BLUE="\033[0;34m"
|
||||
YELLOW="\033[0;33m"
|
||||
RED="\033[0;31m"
|
||||
NC="\033[0m" # No Color
|
||||
|
||||
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
|
||||
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
|
||||
|
||||
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
|
||||
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
|
||||
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
|
||||
else
|
||||
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Configuration
|
||||
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
|
||||
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
|
||||
# cann and atb environment setup
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
|
||||
|
||||
set +eu
|
||||
source /usr/local/Ascend/nnal/atb/set_env.sh
|
||||
set -eu
|
||||
|
||||
# Home path for aisbench
|
||||
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
|
||||
|
||||
# Logging configurations
|
||||
export VLLM_LOGGING_LEVEL="INFO"
|
||||
# Reduce glog verbosity for mooncake
|
||||
export GLOG_minloglevel=1
|
||||
# Set transformers to offline mode to avoid downloading models during tests
|
||||
export HF_HUB_OFFLINE="1"
|
||||
# Default is 600s
|
||||
export VLLM_ENGINE_READY_TIMEOUT_S=1800
|
||||
|
||||
# Function to print section headers
|
||||
print_section() {
|
||||
echo -e "\n${BLUE}=== $1 ===${NC}"
|
||||
}
|
||||
|
||||
print_failure() {
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Function to print success messages
|
||||
print_success() {
|
||||
echo -e "${GREEN}✓ $1${NC}"
|
||||
}
|
||||
|
||||
# Function to print error messages and exit
|
||||
print_error() {
|
||||
echo -e "${RED}✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
show_vllm_info() {
|
||||
cd "$WORKSPACE"
|
||||
echo "Installed vLLM-related Python packages:"
|
||||
pip list | grep vllm || echo "No vllm packages found."
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM Git information"
|
||||
echo "============================"
|
||||
cd vllm
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm"
|
||||
fi
|
||||
cd ..
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM-Ascend Git information"
|
||||
echo "============================"
|
||||
cd vllm-ascend
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm-ascend"
|
||||
fi
|
||||
cd ..
|
||||
}
|
||||
|
||||
check_npu_info() {
|
||||
echo "====> Check NPU info"
|
||||
npu-smi info
|
||||
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
|
||||
}
|
||||
|
||||
check_and_config() {
|
||||
echo "====> Configure mirrors and git proxy"
|
||||
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
|
||||
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
|
||||
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
|
||||
}
|
||||
|
||||
install_extra_components() {
|
||||
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
|
||||
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
|
||||
return 1
|
||||
fi
|
||||
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
|
||||
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
|
||||
echo "Failed to download custom_ops wheel"
|
||||
return 1
|
||||
fi
|
||||
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
|
||||
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
|
||||
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
|
||||
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
|
||||
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
echo "====> Extra components installation completed"
|
||||
}
|
||||
|
||||
checkout_src() {
|
||||
echo "====> Checkout source code"
|
||||
mkdir -p "$WORKSPACE"
|
||||
cd "$WORKSPACE"
|
||||
pip uninstall -y vllm-ascend || true
|
||||
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
|
||||
rm -rf "$WORKSPACE/vllm-ascend"
|
||||
|
||||
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
|
||||
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
|
||||
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
|
||||
if [ -n "$PR_REF" ]; then
|
||||
git fetch --depth 1 origin "$PR_REF"
|
||||
git checkout FETCH_HEAD
|
||||
else
|
||||
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
|
||||
git checkout "$VLLM_ASCEND_REF"
|
||||
fi
|
||||
git submodule update --init --recursive
|
||||
fi
|
||||
}
|
||||
|
||||
install_vllm_ascend() {
|
||||
echo "====> Install vllm-ascend"
|
||||
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
|
||||
pip install -e "$WORKSPACE/vllm-ascend"
|
||||
}
|
||||
|
||||
install_aisbench() {
|
||||
echo "====> Install AISBench benchmark"
|
||||
|
||||
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
|
||||
|
||||
cp -r /tmp/aisbench-backup "$BENCH_DIR"
|
||||
|
||||
cd "$BENCH_DIR"
|
||||
pip install -e . \
|
||||
-r requirements/api.txt \
|
||||
-r requirements/extra.txt
|
||||
|
||||
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
|
||||
|
||||
}
|
||||
|
||||
show_triton_ascend_info() {
|
||||
echo "====> Check triton ascend info"
|
||||
clang -v
|
||||
which bishengir-compile
|
||||
pip show triton-ascend
|
||||
}
|
||||
|
||||
kill_npu_processes() {
|
||||
pgrep python3 | xargs -r kill -9
|
||||
pgrep VLLM | xargs -r kill -9
|
||||
|
||||
sleep 4
|
||||
}
|
||||
|
||||
run_tests_with_log() {
|
||||
set +e
|
||||
kill_npu_processes
|
||||
mkdir -p "${LOG_PREFIX}"
|
||||
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
|
||||
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
|
||||
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
|
||||
ret=$?
|
||||
echo "pytest exit code: ret=${ret}"
|
||||
set -e
|
||||
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
print_success "All tests passed!"
|
||||
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
|
||||
else
|
||||
echo "Leader: waiting 10s for worker logs..."
|
||||
sleep 10
|
||||
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
set +e; aop_pipeline; set -e
|
||||
fi
|
||||
local done_file="${LOG_PREFIX}/aop_done"
|
||||
touch "$done_file"
|
||||
echo "Leader: notifying workers (${done_file})"
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
|
||||
exit 1
|
||||
fi
|
||||
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
echo "Worker: test passed, waiting for leader..."
|
||||
local wait_timeout=30
|
||||
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
|
||||
sleep 1
|
||||
wait_timeout=$((wait_timeout - 1))
|
||||
done
|
||||
fi
|
||||
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
local release="${LOG_PREFIX}/aop_done"
|
||||
mkdir -p "$coord"
|
||||
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--coord-dir "${coord}" \
|
||||
--release-file "${release}"
|
||||
while [ ! -f "$release" ]; do sleep 5; done
|
||||
echo "Worker: release signal received, exiting"
|
||||
exit 1
|
||||
else
|
||||
echo "Worker: leader finished successfully, exiting"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
|
||||
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
|
||||
aop_pipeline() {
|
||||
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
|
||||
local table="${GOOD_TABLE:-}"
|
||||
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
|
||||
local case_name="${BENCHMARK_JOB_NAME#*-}"
|
||||
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
|
||||
case_name="${CONFIG_YAML_PATH%.yaml}"
|
||||
fi
|
||||
|
||||
echo "============================================"
|
||||
echo " AOP Pipeline (Pod) - START"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Case name : ${case_name}"
|
||||
echo " Rules file : ${rules}"
|
||||
echo " Table file : ${table}"
|
||||
echo " Log prefix : ${LOG_PREFIX}"
|
||||
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
|
||||
echo "============================================"
|
||||
|
||||
# ---- Step 1: Classify ----
|
||||
echo ""
|
||||
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
|
||||
echo " Rules content:"
|
||||
if [ -f "$rules" ]; then
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
|
||||
else
|
||||
echo " (rules file not found)"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo " Pod logs found:"
|
||||
local found_any=0
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
echo " - ${f} ($(wc -l < "$f") lines)"
|
||||
found_any=1
|
||||
fi
|
||||
done
|
||||
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
|
||||
|
||||
local env_count=0
|
||||
if [ -f "$rules" ]; then
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
local n
|
||||
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
|
||||
n=${n%%[!0-9]*}
|
||||
echo " Scan ${f}: ${n} matches"
|
||||
env_count=$((env_count + n))
|
||||
if [ "$n" -gt 0 ]; then
|
||||
echo " Matched lines:"
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
|
||||
fi
|
||||
fi
|
||||
done
|
||||
fi
|
||||
echo " Classify result: env_count=${env_count}"
|
||||
|
||||
if [ "$found_any" -eq 0 ]; then
|
||||
echo " Decision: no pod logs → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [ "$env_count" -gt 0 ]; then
|
||||
echo " Decision: env_failure → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 2: Check age ----
|
||||
echo ""
|
||||
echo "--- [2/3] Check commit age ---"
|
||||
echo " Looking up: ${case_name}"
|
||||
local skip_age=0
|
||||
if [ ! -f "$table" ]; then
|
||||
echo " Table file not found: ${table}"
|
||||
echo " Decision: no table → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Only consider success rows
|
||||
local success_rows
|
||||
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
|
||||
if [ -z "$success_rows" ]; then
|
||||
echo " No success row found for '${case_name}'"
|
||||
echo " Decision: no success entry → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Pick most recent success row
|
||||
local best_date=""
|
||||
while IFS= read -r row; do
|
||||
local d
|
||||
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
|
||||
[ -z "$d" ] && continue
|
||||
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
|
||||
best_date="$d"
|
||||
fi
|
||||
done <<< "$success_rows"
|
||||
|
||||
if [ -z "$best_date" ]; then
|
||||
echo " No valid date in success rows"
|
||||
echo " Decision: no date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
|
||||
local last_ts now_ts age_days
|
||||
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
|
||||
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
|
||||
echo " Date parse failed: ${best_date}"
|
||||
echo " Decision: invalid date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
now_ts=$(date +%s)
|
||||
age_days=$(( (now_ts - last_ts) / 86400 ))
|
||||
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
|
||||
|
||||
if [ "$age_days" -gt 3 ]; then
|
||||
echo " Decision: old commit (> 3 days) → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 3: Bisect ----
|
||||
echo ""
|
||||
echo "--- [3/3] Run bisect ---"
|
||||
echo " Scene : multi_node"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Bad commit : HEAD"
|
||||
echo " Name : ${case_name}"
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
echo " Coord dir : ${coord}"
|
||||
|
||||
# Wait for all workers to signal ready
|
||||
echo " Waiting for workers..."
|
||||
for i in $(seq 1 30); do
|
||||
local ready_count=0
|
||||
for f in "${coord}"/worker_ready_*; do
|
||||
[ -e "$f" ] && ready_count=$((ready_count + 1))
|
||||
done
|
||||
echo " [${i}/30] ready workers: ${ready_count}"
|
||||
if [ "$ready_count" -ge 1 ]; then break; fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
local bisect_rc=0
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--good-table "${table}" \
|
||||
--name "${case_name}" \
|
||||
--coord-dir "${coord}" || bisect_rc=$?
|
||||
echo " bisect completed (exit code: ${bisect_rc})"
|
||||
echo "=== AOP Pipeline (Pod) - END ==="
|
||||
return 1
|
||||
}
|
||||
|
||||
clear_logs() {
|
||||
print_section "Clearing logs from previous runs"
|
||||
rm -fr "$HOME/ascend/log" || true
|
||||
}
|
||||
|
||||
backup_ascend_logs() {
|
||||
if [ -n "${LOG_PREFIX:-}" ]; then
|
||||
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
|
||||
mkdir -p "$dest"
|
||||
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
|
||||
echo "Ascend logs backed up to $dest"
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
trap backup_ascend_logs EXIT
|
||||
check_npu_info
|
||||
clear_logs
|
||||
check_and_config
|
||||
if [[ "$IS_PR_TEST" == "true" ]]; then
|
||||
checkout_src
|
||||
install_vllm_ascend
|
||||
install_aisbench
|
||||
fi
|
||||
show_vllm_info
|
||||
show_triton_ascend_info
|
||||
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
|
||||
install_extra_components
|
||||
fi
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
run_tests_with_log
|
||||
}
|
||||
|
||||
main "$@"
|
||||
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
@@ -0,0 +1,183 @@
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import time
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def temp_env(env_dict: dict[str, Any]):
|
||||
old_env = {}
|
||||
for key, value in env_dict.items():
|
||||
old_env[key] = os.environ.get(key)
|
||||
os.environ[key] = str(value)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
for key, value in old_env.items():
|
||||
if value is None:
|
||||
os.environ.pop(key, None)
|
||||
else:
|
||||
os.environ[key] = value
|
||||
|
||||
|
||||
def setup_logger() -> None:
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
|
||||
|
||||
def load_yaml_mapping(
|
||||
yaml_path: str | None,
|
||||
*,
|
||||
default_name: str,
|
||||
default_base_path: str,
|
||||
description: str,
|
||||
) -> dict[str, Any]:
|
||||
if not yaml_path:
|
||||
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
|
||||
|
||||
path = Path(yaml_path)
|
||||
if not path.is_absolute() and not path.exists():
|
||||
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
|
||||
path = Path(base_path) / yaml_path
|
||||
|
||||
logger.info("Loading %s yaml: %s", description, path)
|
||||
with path.open(encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f)
|
||||
if not isinstance(data, dict):
|
||||
raise TypeError(f"{description} must be a mapping: {path}")
|
||||
return data
|
||||
|
||||
|
||||
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
|
||||
def resolve(dns: str) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
return socket.gethostbyname(dns)
|
||||
except socket.gaierror:
|
||||
if attempt == retries - 1:
|
||||
raise
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError(f"Unable to resolve DNS: {dns}")
|
||||
|
||||
return resolve
|
||||
|
||||
|
||||
def get_cluster_dns_list(world_size: int) -> list[str]:
|
||||
if world_size < 1:
|
||||
raise ValueError(f"world_size must be >= 1, got {world_size}")
|
||||
|
||||
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
|
||||
if not leader_dns:
|
||||
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
|
||||
|
||||
parts = leader_dns.split(".")
|
||||
if len(parts) < 3:
|
||||
raise ValueError(f"invalid leader DNS format: {leader_dns}")
|
||||
|
||||
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
|
||||
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
|
||||
return [leader_dns, *worker_dns_list]
|
||||
|
||||
|
||||
def get_cluster_ips(world_size: int = 2) -> list[str]:
|
||||
resolver = dns_resolver()
|
||||
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
|
||||
|
||||
|
||||
def resolve_cluster_ips(
|
||||
raw_config: dict[str, Any],
|
||||
num_nodes: int,
|
||||
explicit_cluster_ips: list[str] | None = None,
|
||||
*,
|
||||
cluster_hosts_log_message: str | None = None,
|
||||
dns_log_message: str = "Resolving cluster IPs via DNS...",
|
||||
) -> list[str]:
|
||||
if explicit_cluster_ips is not None:
|
||||
if len(explicit_cluster_ips) != num_nodes:
|
||||
raise AssertionError("cluster_ips size mismatch")
|
||||
return explicit_cluster_ips
|
||||
|
||||
cluster_hosts = raw_config.get("cluster_hosts")
|
||||
if cluster_hosts:
|
||||
if cluster_hosts_log_message:
|
||||
logger.info(cluster_hosts_log_message)
|
||||
if len(cluster_hosts) != num_nodes:
|
||||
raise AssertionError("cluster_hosts size mismatch")
|
||||
return list(cluster_hosts)
|
||||
|
||||
logger.info(dns_log_message)
|
||||
return get_cluster_ips(num_nodes)
|
||||
|
||||
|
||||
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
|
||||
for port in range(start_port, end_port):
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
try:
|
||||
s.bind(("", port))
|
||||
return port
|
||||
except OSError:
|
||||
continue
|
||||
raise RuntimeError("No available port found")
|
||||
|
||||
|
||||
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
|
||||
s.connect(("8.8.8.8", 80))
|
||||
return s.getsockname()[0]
|
||||
except Exception:
|
||||
try:
|
||||
return socket.gethostbyname(socket.gethostname())
|
||||
except Exception:
|
||||
if attempt == retries - 1:
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
|
||||
|
||||
def get_net_interface(ip: str | None = None) -> str:
|
||||
import psutil
|
||||
|
||||
if ip is None:
|
||||
ip = get_cur_ip()
|
||||
|
||||
for iface, addrs in psutil.net_if_addrs().items():
|
||||
for addr in addrs:
|
||||
if addr.family == socket.AF_INET and addr.address == ip:
|
||||
return iface
|
||||
raise RuntimeError(f"No network interface found for IP {ip}")
|
||||
|
||||
|
||||
def get_all_ipv4() -> list[str]:
|
||||
ipv4s = {"127.0.0.1"}
|
||||
hostname = socket.gethostname()
|
||||
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
|
||||
ipv4s.add(info[4][0])
|
||||
return list(ipv4s)
|
||||
|
||||
|
||||
def resolve_current_node_index(cluster_ips: list[str]) -> int:
|
||||
worker_index = os.environ.get("LWS_WORKER_INDEX")
|
||||
if worker_index:
|
||||
return int(worker_index)
|
||||
|
||||
local_ips = set(get_all_ipv4())
|
||||
for index, ip in enumerate(cluster_ips):
|
||||
if ip in local_ips:
|
||||
return index
|
||||
raise RuntimeError("Unable to determine current node index")
|
||||
Reference in New Issue
Block a user