@@ -0,0 +1,308 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_1M_1k_prefix99_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_1M_1k_prefix99:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1024
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 11.93
|
||||
threshold: 0.95
|
||||
@@ -0,0 +1,324 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 0
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 8
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_BUFFSIZE: "1800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_TPOT50_128k_1_prefix_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1
|
||||
batch_size: 4
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_TPOT50_128k_1_prefix90:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 192
|
||||
max_out_len: 1024
|
||||
batch_size: 48
|
||||
request_rate: 1
|
||||
baseline: 869.13
|
||||
threshold: 0.95
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,176 @@
|
||||
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
|
||||
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 16
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1500"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --trust-remote-code
|
||||
- --block-size
|
||||
- "32"
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.9"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "240"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,335 @@
|
||||
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0, 1]
|
||||
decoder: [2, 3]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 4
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: "0"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_INTRA_PCIE_ENABLE: "1"
|
||||
HCCL_INTRA_ROCE_ENABLE: "0"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1500
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,161 @@
|
||||
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
|
||||
model: "Eco-Tech/Kimi-K2.6-w4a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_SERVER_DEV_MODE: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "512"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --enforce-eager
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "256"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 60
|
||||
max_out_len: 1024
|
||||
batch_size: 15
|
||||
request_rate: 0.4
|
||||
baseline: 347.4475
|
||||
threshold: 0.97
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 93.33
|
||||
threshold: 10
|
||||
temperature: 1.0
|
||||
top_p: 1
|
||||
@@ -0,0 +1,197 @@
|
||||
test_name: "Minimax_m2.7_in3_5_tpot50"
|
||||
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
|
||||
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
PYTHONHASHSEED: "0"
|
||||
|
||||
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "2048"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "55880",
|
||||
"engine_id": "0",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}} }'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --async-scheduling
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "56900",
|
||||
"engine_id": "1",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf_warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2
|
||||
max_out_len: 1
|
||||
batch_size: 2
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1024
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 717.5332
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
temperature: 1
|
||||
top_p: 1
|
||||
top_k: 40
|
||||
ignore_eos: false
|
||||
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
@@ -0,0 +1,360 @@
|
||||
# External DP Config Template
|
||||
|
||||
This document shows how to write YAML configs consumed by
|
||||
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
|
||||
|
||||
`server_cmd_template` contains only the arguments after
|
||||
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
|
||||
`model` automatically.
|
||||
|
||||
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
|
||||
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
|
||||
and roles are selected by `routing.groups`.
|
||||
|
||||
## Generic DP Template
|
||||
|
||||
Use this template for generic external data parallel serving. This mode uses
|
||||
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
|
||||
independent vLLM instances instead of external DP rank arguments.
|
||||
|
||||
```yaml
|
||||
test_name: "test Qwen3-30B-A3B generic external dp"
|
||||
model: "Qwen/Qwen3-30B-A3B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "generic_dp"
|
||||
groups:
|
||||
worker: [0, 1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs: &generic_env
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template: &generic_server_cmd
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --max-model-len
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --enable-expert-parallel
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *generic_env
|
||||
server_cmd_template: *generic_server_cmd
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Disaggregated Prefill Template
|
||||
|
||||
Use this template for PD disaggregation. `routing.groups` decides which config
|
||||
entries run as prefillers or decoders. The framework derives the PD proxy script
|
||||
from `routing.type`, so do not write `proxy_*` fields in YAML.
|
||||
|
||||
```yaml
|
||||
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0]
|
||||
decoder: [1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "256"
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
max_out_len: 128
|
||||
batch_size: 4
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 48
|
||||
batch_size: 4
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Field Notes
|
||||
|
||||
- `test_name`: Human-readable test name. It is also used when writing benchmark
|
||||
result metadata.
|
||||
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
|
||||
- `num_nodes`: Number of config entries and templates expected.
|
||||
- `npu_per_node`: Device capacity validation for each node.
|
||||
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
|
||||
needs fixed hosts.
|
||||
- `routing.type`: Supported values are `generic_dp` and
|
||||
`disaggregated_prefill`.
|
||||
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
|
||||
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
|
||||
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
|
||||
`kv_consumer` for decoder templates.
|
||||
- `config[].dp_size`: Global DP size for this DP group.
|
||||
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
|
||||
- `config[].dp_rank_start`: First global DP rank owned by this node.
|
||||
- `config[].dp_address`: DP master address. For one global DP group, use
|
||||
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
|
||||
address for prefiller nodes and the decoder master address for decoder nodes.
|
||||
- `templates`: One template per config entry. The framework expands one command
|
||||
per local DP rank.
|
||||
|
||||
The framework injects distributed network envs at startup:
|
||||
|
||||
```text
|
||||
HCCL_IF_IP
|
||||
HCCL_SOCKET_IFNAME
|
||||
GLOO_SOCKET_IFNAME
|
||||
TP_SOCKET_IFNAME
|
||||
LOCAL_IP
|
||||
NIC_NAME
|
||||
MASTER_IP
|
||||
```
|
||||
|
||||
The framework also derives proxy metadata from `routing.type`:
|
||||
|
||||
```text
|
||||
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
|
||||
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
|
||||
```
|
||||
|
||||
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
|
||||
for benchmark requests.
|
||||
|
||||
## Template Variables
|
||||
|
||||
The following variables are available in `envs` and `server_cmd_template`:
|
||||
|
||||
```text
|
||||
${MODEL}
|
||||
${PORT_START}
|
||||
${PORT}
|
||||
${DP_SIZE}
|
||||
${DP_SIZE_LOCAL}
|
||||
${DP_RANK_START}
|
||||
${DP_RANK}
|
||||
${LOCAL_RANK}
|
||||
${TP_SIZE}
|
||||
${CP_SIZE}
|
||||
${SP_SIZE}
|
||||
${PP_SIZE}
|
||||
${DP_ADDRESS}
|
||||
${DP_RPC_PORT}
|
||||
${VISIBLE_DEVICES}
|
||||
${NODE_INDEX}
|
||||
${CONFIG_INDEX}
|
||||
${NODE_0_IP}, ${NODE_1_IP}, ...
|
||||
${LOCAL_IP}
|
||||
${MASTER_IP}
|
||||
${LWS_WORKER_INDEX}
|
||||
```
|
||||
|
||||
Command arguments can also reference rendered environment variables with
|
||||
shell-style `$VARNAME`, for example:
|
||||
|
||||
```yaml
|
||||
envs:
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template:
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
```
|
||||
|
||||
## Checks Before Running
|
||||
|
||||
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
|
||||
- Make sure each config index is assigned to exactly one routing group.
|
||||
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
|
||||
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
|
||||
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
|
||||
`--enable-expert-parallel`.
|
||||
- Set `--max-model-len` large enough for benchmark input tokens plus
|
||||
`max_out_len`.
|
||||
Reference in New Issue
Block a user