init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,149 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
_envs: &envs
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "256"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "8"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--gpu-memory-utilization"
- "0.95"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,2,4,8,16,24,32,40,48], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--no-enable-prefix-caching"
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
- "--reasoning-parser"
- "deepseek_v3"
- "--tokenizer-mode"
- "deepseek_v32"
_benchmarks: &benchmarks
perf_16384_bs4:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs1200_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 26.10
threshold: 0.97
perf_16384_bs16:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs1200_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0.5
baseline: 60.53
threshold: 0.97
perf_32768_bs4:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 512
batch_size: 1
request_rate: 0
baseline: 22.68
threshold: 0.97
perf_32768_bs8:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 512
batch_size: 2
request_rate: 0.2
baseline: 32.73
threshold: 0.97
perf_65536_bs4:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs400_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 22.77
threshold: 0.97
perf_65536_bs8:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs400_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 1
baseline: 33.80
threshold: 0.97
_benchmarks_2048: &benchmarks_2048
perf_16384_bs4:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in2048-bs4000-ds3.2
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 2048
batch_size: 32
request_rate: 0
baseline: 220
threshold: 0.97
test_cases:
- name: "DeepSeek-V3.2-W8A8-weekly"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks
- name: "DeepSeek-V3.2-W8A8-mix-perf"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--async-scheduling"
benchmarks:
<<: *benchmarks_2048

View File

@@ -0,0 +1,81 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
LCCL_DETERMINISTIC: "1"
VLLM_USE_V1: "1"
HCCL_DETERMINISTIC: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
ATB_MATMUL_SHUFFLE_K_ENABLE: "0"
ATB_LLM_LCOC_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--data-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--seed"
- "1024"
- "--tensor-parallel-size"
- "8"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "16"
- "--quantization"
- "ascend"
- "--async-scheduling"
- "--speculative-config"
- '{"num_speculative_tokens": 1, "method":"deepseek_mtp"}'
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--disable-log-stats"
- "--compilation_config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
_benchmarks: &benchmarks
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 64
max_out_len: 1500
batch_size: 16
request_rate: 0
baseline: 411
threshold: 0.95
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Deepseek_R1_W8A8_Performance_0006"
model: "Eco-Tech/DeepSeek-R1-0528-w8a8-mtp-QuaRot"
envs:
<<: *envs
OMP_PROC_BIND: "1"
OMP_NUM_THREADS: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "5500"
benchmarks:
<<: *benchmarks
- name: "chunked_prefill_func_test_005"
model: "Eco-Tech/DeepSeek-R1-0528-w8a8-mtp-QuaRot"
envs:
<<: *envs
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "32768"

View File

@@ -0,0 +1,71 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
LCCL_DETERMINISTIC: "1"
VLLM_USE_V1: "1"
HCCL_DETERMINISTIC: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
ATB_MATMUL_SHUFFLE_K_ENABLE: "0"
ATB_LLM_LCOC_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "100"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--data-parallel-size"
- "8"
- "--data-parallel-size-local"
- "8"
- "--tensor-parallel-size"
- "2"
- "--seed"
- "1024"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "32"
- "--max-num-batched-tokens"
- "6000"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--enforce-eager"
- "--gpu-memory-utilization"
- "0.92"
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--reasoning-parser"
- "deepseek_r1"
- "--additional-config"
- '{"ascend_scheduler_config":{"enabled":false},"torchair_graph_config":{"enabled":false,"enable_multistream_shared_expert":false}}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k_lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
num_prompts: 32
batch_size: 16
baseline: 100
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Deepseek_R1_W8A8_reasoning_output_012"
model: "vllm-ascend/DeepSeek-R1-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,75 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "1024"
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: "0"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "16384"
- "--max-num-batched-tokens"
- "4096"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.95"
- "--max-num-seqs"
- "8"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"multistream_overlap_shared_expert": false, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 8192
batch_size: 8
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1500
batch_size: 8
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "GLM-5-TP16-DP1-decodegraph"
model: "Eco-Tech/GLM-5-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_capture_sizes": [4,8,16,32,64,128,256,512], "cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,158 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_BUFFSIZE: "200"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-num-batched-tokens"
- "4096"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--enable-chunked-prefill"
- "--additional-config"
- '{"multistream_overlap_shared_expert":true}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager": true}'
_benchmarks: &benchmarks
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 20
max_out_len: 1024
batch_size: 5
request_rate: 0
baseline: 1
threshold: 0.97
_benchmarks_bs8: &benchmarks_bs8
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 512
batch_size: 2
request_rate: 0
baseline: 1
threshold: 0.97
_benchmarks_bs32: &benchmarks_bs32
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 512
batch_size: 8
request_rate: 0
baseline: 1
threshold: 0.97
_benchmarks_bs10: &benchmarks_bs10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 20
max_out_len: 1024
batch_size: 5
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
# 16384 1024 0% 20
- name: "GLM-5_1-w8a8-High—Throughput-bs20"
model: "Eco-Tech/GLM-5.1-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--max-model-len"
- "20480"
- "--max-num-seqs"
- "48"
- "--gpu-memory-utilization"
- "0.95"
benchmarks:
<<: *benchmarks
#32768 512 0% 8
- name: "GLM-5_1-w8a8-High—Throughput-bs8"
model: "Eco-Tech/GLM-5.1-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--max-model-len"
- "33792"
- "--max-num-seqs"
- "48"
- "--gpu-memory-utilization"
- "0.95"
benchmarks:
<<: *benchmarks_bs8
#32768 512 90% 32
- name: "GLM-5_1-w8a8-High—Throughput-bs32"
model: "Eco-Tech/GLM-5.1-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "33792"
- "--max-num-seqs"
- "48"
- "--gpu-memory-utilization"
- "0.95"
benchmarks:
<<: *benchmarks_bs32
#65536 1024 90% 10
- name: "GLM-5_1-w8a8-High—Throughput-bs10"
model: "Eco-Tech/GLM-5.1-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "67584"
- "--max-num-seqs"
- "8"
- "--gpu-memory-utilization"
- "0.97"
- "--enable-prefix-caching"
benchmarks:
<<: *benchmarks_bs10

View File

@@ -0,0 +1,191 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_NZ: "1"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--allowed-local-media-path"
- "/"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "42"
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
_benchmark_TPOT50_32K_0_5K: &_benchmark_TPOT50_32K_0_5K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs200_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 512
batch_size: 4
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT50_32K_0_5K_pc90: &_benchmark_TPOT50_32K_0_5K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 40
max_out_len: 512
batch_size: 10
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT20_32K_0_5K: &_benchmark_TPOT20_32K_0_5K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs200_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 512
batch_size: 1
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT20_32K_0_5K_pc90: &_benchmark_TPOT20_32K_0_5K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 512
batch_size: 1
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.6-W4A8-TOPT50-32k-0.5k"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "450"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "32"
- "--max-model-len"
- "33792"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[16,32,48,64,80,96,112,128], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp": true}'
benchmarks:
perf: *_benchmark_TPOT50_32K_0_5K
- name: "Kimi-K2.6-W4A8-TOPT50-32k-0.5k-prefix-cache90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "450"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "32"
- "--max-model-len"
- "33792"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[16,32,48,64,80,96,112,128], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp": true}'
benchmarks:
perf: *_benchmark_TPOT50_32K_0_5K_pc90
- name: "Kimi-K2.6-W4A8-TOPT20-32k-0.5k"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "450"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "32"
- "--max-model-len"
- "33792"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[16,32,48,64,80,96,112,128], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp": true}'
benchmarks:
perf: *_benchmark_TPOT20_32K_0_5K
- name: "Kimi-K2.6-W4A8-TOPT20-32k-0.5k-pc90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "450"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "32"
- "--max-model-len"
- "33792"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[16,32,48,64,80,96,112,128], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp": true}'
benchmarks:
perf: *_benchmark_TPOT20_32K_0_5K_pc90

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.6-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "800"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "24"
- "--max-model-len"
- "6144"
- "--max-num-batched-tokens"
- "4096"
- "--gpu-memory-utilization"
- "0.85"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1433.4454
threshold: 0.97

View File

@@ -0,0 +1,331 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_NZ: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--allowed-local-media-path"
- "/"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "42"
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
_benchmark_acc: &_benchmark_acc
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 90
threshold: 10
_benchmark_TPOT50_3_5k_1_5k: &_benchmark_TPOT50_3_5k_1_5k
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 392
max_out_len: 1500
batch_size: 98
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT50_16K_1K: &_benchmark_TPOT50_16K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 56
max_out_len: 1024
batch_size: 14
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT20_16K_1K: &_benchmark_TPOT20_16K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 1024
batch_size: 8
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT50_64K_1K_pc90: &_benchmark_TPOT50_64K_1K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 56
max_out_len: 1024
batch_size: 14
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT20_64K_1K_pc90: &_benchmark_TPOT20_64K_1K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT50_128K_1K_pc90: &_benchmark_TPOT50_128K_1K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
_benchmark_TPOT20_128K_1K_pc90: &_benchmark_TPOT20_128K_1K_pc90
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.6-W4A8-Case"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "32768"
- "--no-enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "500"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[48,64,80,96,112,128], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
benchmarks:
perf: *_benchmark_TPOT50_3_5k_1_5k
- name: "Kimi-K2.6-W4A8-TOPT50-16k-1k"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "32768"
- "--no-enable-prefix-caching"
- "--enable-chunked-prefill"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "16"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,12,16,20], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
benchmarks:
perf: *_benchmark_TPOT50_16K_1K
- name: "Kimi-K2.6-W4A8-TOPT20-16k-1k"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "18000"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "1"
- "--no-enable-prefix-caching"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[8], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":7}'
- "--tool-call-parser"
- "kimi_k2"
- "--reasoning-parser"
- "kimi_k2"
- "--enable-auto-tool-choice"
benchmarks:
perf: *_benchmark_TPOT20_16K_1K
- name: "Kimi-K2.6-W4A8-TOPT50-64k-1k-pc90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "400"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--tool-call-parser"
- "kimi_k2"
- "--reasoning-parser"
- "kimi_k2"
- "--enable-auto-tool-choice"
- "--api-server-count"
- "1"
- "--aggregate-engine-logging"
- "--max-num-seqs"
- "3"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "8192"
- "--enable-prefix-caching"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[8,12], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
benchmarks:
perf: *_benchmark_TPOT50_64K_1K_pc90
- name: "Kimi-K2.6-W4A8-TOPT20-64k-1k-pc90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
HCCL_BUFFSIZE: "400"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--tool-call-parser"
- "kimi_k2"
- "--reasoning-parser"
- "kimi_k2"
- "--enable-auto-tool-choice"
- "--quantization"
- "ascend"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "16"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "8"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "8192"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "1024"
- "--enable-prefix-caching"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,16], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
benchmarks:
perf: *_benchmark_TPOT20_64K_1K_pc90
- name: "Kimi-K2.6-W4A8-TOPT50-128k-1k-pc90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "16"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
benchmarks:
perf: *_benchmark_TPOT50_128K_1K_pc90
- name: "Kimi-K2.6-W4A8-TOPT20-128k-1k-pc90"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-chunked-prefill"
- "--max-num-seqs"
- "16"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "8192"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.6-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
benchmarks:
perf: *_benchmark_TPOT20_128K_1K_pc90
<<: *_benchmark_acc

View File

@@ -0,0 +1,330 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.7-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
model: "Eco-Tech/Kimi-K2.7-Code-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "800"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "24"
- "--max-model-len"
- "6144"
- "--max-num-batched-tokens"
- "4096"
- "--gpu-memory-utilization"
- "0.85"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1585.1248
threshold: 0.97
- name: "Kimi-K2.7-W4A8-in3.5k-out1.5k-TPOT20-0-128-32"
model: "Eco-Tech/Kimi-K2.7-Code-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "800"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "24"
- "--max-model-len"
- "6144"
- "--max-num-batched-tokens"
- "4096"
- "--gpu-memory-utilization"
- "0.85"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1585.1248
threshold: 0.97
- name: "Kimi-K2.7-W4A8-in64k-out1k-TPOT50-90-24-6"
model: "Eco-Tech/Kimi-K2.7-Code-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "1300"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "4"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "8192"
- "--gpu-memory-utilization"
- "0.87"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 24
max_out_len: 1024
batch_size: 6
request_rate: 0
baseline: 167.6767
threshold: 0.97
- name: "Kimi-K2.7-W4A8-in128k-out1k-TPOT50-90-24-6"
model: "Eco-Tech/Kimi-K2.7-Code-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
HCCL_BUFFSIZE: "1500"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "0"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "10"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--gpu-memory-utilization"
- "0.92"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 8
baseline: 96.67
threshold: 10
temperature: 1.0
top_p: 1
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 24
max_out_len: 1024
batch_size: 6
request_rate: 0
baseline: 95.3761
threshold: 0.97
- name: "Kimi-K2.7-W4A8-in254k-out1k"
model: "Eco-Tech/Kimi-K2.7-Code-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
HCCL_BUFFSIZE: "300"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "1"
- "--max-model-len"
- "262144"
- "--max-num-batched-tokens"
- "8192"
- "--gpu-memory-utilization"
- "0.95"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in254k-bs16-prefix99-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 7.5611
threshold: 0.97

View File

@@ -0,0 +1,61 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.5-w8a8"
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
envs:
HCCL_BUFFSIZE: "512"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM-ASCEND_ENABLE_FUSED_MC2: "2"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "128"
- "--max-model-len"
- "153600"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.8"
- "--quantization"
- "ascend"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1500
batch_size: 20
request_rate: 0
baseline: 730.0832
threshold: 0.97

View File

@@ -0,0 +1,85 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1200"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
TASK_QUEUE_ENABLE: "1"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--host"
- "0.0.0.0"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--async-scheduling"
- "--max-num-seqs"
- "128"
- "--safetensors-load-strategy"
- 'prefetch'
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--enable-auto-tool-choice"
- "--tool-call-parser"
- "minimax_m2"
- "--speculative_config"
- '{"method":"eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"enable_cpu_binding": true, "enable_npugraph_ex": true, "enable_static_kernel": true,"enable_fused_mc2":true,"weight_nz_mode":true,"enable_flashcomm1":true}'
_benchmarks_3500: &benchmarks_3500
perf_50:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 760
max_out_len: 1500
batch_size: 190
request_rate: 0
baseline: 4573.02
threshold: 0.97
perf_20:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1500
batch_size: 48
request_rate: 0
baseline: 2229.147
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.7-3500"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "70000"
- "--gpu-memory-utilization"
- "0.8"
- "--no-enable-prefix-caching"
benchmarks:
<<: *benchmarks_3500

View File

@@ -0,0 +1,92 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen2.5-VL-7B-Instruct-epd"
model: "Qwen/Qwen2.5-VL-7B-Instruct"
service_mode: "epd"
envs:
ENCODE_PORT: "DEFAULT_PORT"
PD_PORT: "DEFAULT_PORT"
PROXY_PORT: "DEFAULT_PORT"
epd_server_cmds:
- - "--port"
- "$ENCODE_PORT"
- "--model"
- "Qwen/Qwen2.5-VL-7B-Instruct"
- "--gpu-memory-utilization"
- "0.01"
- "--tensor-parallel-size"
- "1"
- "--enforce-eager"
- "--no-enable-prefix-caching"
- "--max-model-len"
- "10000"
- "--max-num-batched-tokens"
- "10000"
- "--max-num-seqs"
- "1"
- "--ec-transfer-config"
- '{"ec_connector_extra_config":{"shared_storage_path":"/dev/shm/epd/storage"},"ec_connector":"ECExampleConnector","ec_role": "ec_producer"}'
- - "--port"
- "$PD_PORT"
- "--model"
- "Qwen/Qwen2.5-VL-7B-Instruct"
- "--gpu-memory-utilization"
- "0.95"
- "--tensor-parallel-size"
- "1"
- "--enforce-eager"
- "--max-model-len"
- "10000"
- "--max-num-batched-tokens"
- "10000"
- "--max-num-seqs"
- "128"
- "--ec-transfer-config"
- '{"ec_connector_extra_config":{"shared_storage_path":"/dev/shm/epd/storage"},"ec_connector":"ECExampleConnector","ec_role": "ec_consumer"}'
epd_proxy_args:
- "--host"
- "127.0.0.1"
- "--port"
- "$PROXY_PORT"
- "--encode-servers-urls"
- "http://localhost:$ENCODE_PORT"
- "--decode-servers-urls"
- "http://localhost:$PD_PORT"
- "--prefill-servers-urls"
- "disable"
test_content:
benchmarks:
warm_up:
case_type: performance
dataset_path: vllm-ascend/textvqa-perf-1080p
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
num_prompts: 50
max_out_len: 20
batch_size: 32
request_rate: 0
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/textvqa-lite
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
max_out_len: 2048
batch_size: 128
baseline: 82.05
threshold: 5
perf:
case_type: performance
dataset_path: vllm-ascend/textvqa-perf-1080p
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
num_prompts: 512
max_out_len: 256
batch_size: 128
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,59 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "qwen3-14b"
model: "vllm-ascend/Qwen3-14B-w8a8sc-310-vllm-tp1"
envs:
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--tensor-parallel-size"
- "1"
- "--gpu-memory-utilization"
- "0.8"
- "--max-num-seqs"
- "32"
- "--dtype"
- "float16"
- "--quantization"
- "ascend"
- "--max-model-len"
- "20480"
- "--no-enable-prefix-caching"
- "--load_format"
- "sharded_state"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[1,4,8,16,24,32]}'
- "--additional-config"
- '{"ascend_compilation_config":{"enable_npugraph_ex":false, "fuse_norm_quant": false}}'
- "--default-chat-template-kwargs"
- '{"enable_thinking": false}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in512-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 512
batch_size: 1
request_rate: 0
baseline: 10.7
threshold: 0.95
acc:
case_type: accuracy
dataset_path: vllm-ascend/SuperGLUE
request_conf: vllm_api_general_chat
dataset_conf: SuperGLUE/SuperGLUE_BoolQ_gen_0_shot_cot_str
max_out_len: 10240
batch_size: 32
baseline: 89.3
threshold: 1
temperature: 0
top_p: 0.95
ignore_eos: false

View File

@@ -0,0 +1,105 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
VLLM_USE_V1: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "1800"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--async-scheduling"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--data-parallel-size-local"
- "4"
- "--data-parallel-start-rank"
- "0"
- "--data-parallel-rpc-port"
- "2345"
- "--max-num-seqs"
- "120"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "16384"
- "--gpu-memory-utilization"
- "0.9"
- "--enable-expert-parallel "
- "--port"
- "$SERVER_PORT"
- "--no-enable-prefix-caching"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--host"
- "0.0.0.0"
- "--data-parallel-address"
- "0.0.0.0"
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k_lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 38912
num_prompts: 32
batch_size: 32
baseline: 100
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3_235B_Accuracy_PIECEWISE_gsm8k_lite"
model: "Eco-Tech/Qwen3-235B-A22B-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "PIECEWISE"}'
- "--hf-overrides"
- '{"rope_parameters": {"rope_type":"yarn","rope_theta": 1000000.0,"factor":4.3,"original_max_position_embeddings":32768}}'
- "--additional-config"
- '{"ascend_scheduler_config":{"enabled":false},"pa_shape_list": [4,8,16,32,48,64,96,128,160,192]}'
benchmarks:
<<: *benchmarks
- name: "Qwen3_235B_Accuracy_FULL_Graph_gsm8k_lite"
model: "Eco-Tech/Qwen3-235B-A22B-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--hf-overrides"
- '{"rope_parameters": {"rope_type":"yarn","rope_theta": 1000000.0,"factor":4.3,"original_max_position_embeddings":32768}}'
- "--additional-config"
- '{"ascend_scheduler_config":{"enabled":false},"pa_shape_list": [4,8,16,32,48,64,96,128,160,192]}'
benchmarks:
<<: *benchmarks
- name: "Qwen3_235B_Accuracy_EPLB_gsm8k_lite"
model: "Eco-Tech/Qwen3-235B-A22B-w8a8-QuaRot"
envs:
<<: *envs
DYNAMIC_EPLB: "true"
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,57 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "qwen3-32b"
model: "vllm-ascend/Qwen3-32B-w8a8sc-310-vllm-tp4"
envs:
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--tensor-parallel-size"
- "4"
- "--gpu-memory-utilization"
- "0.8"
- "--max-num-seqs"
- "32"
- "--dtype"
- "float16"
- "--quantization"
- "ascend"
- "--max-model-len"
- "20480"
- "--no-enable-prefix-caching"
- "--load_format"
- "sharded_state"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[8,32]}'
- "--additional-config"
- '{"ascend_compilation_config":{"enable_npugraph_ex":false, "fuse_norm_quant": false}}'
- "--default-chat-template-kwargs"
- '{"enable_thinking": false}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in512-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 512
batch_size: 8
request_rate: 0
baseline: 105.26
threshold: 0.95
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 10240
batch_size: 32
baseline: 95.07
threshold: 1
temperature: 0
top_p: 0.95

View File

@@ -0,0 +1,55 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
#workflow暂缺A2yaml
# - name: qwen3-32b
# os: linux-aarch64-a2b3-4
# config_file_path: Qwen3-32B.yaml
test_cases:
- name: "Qwen3-32B-TP4"
model: "Qwen/Qwen3-32B"
envs:
TASK_QUEUE_ENABLE: "1"
OMP_PROC_BIND: "false"
HCCL_OP_EXPANSION_MODE: "AIV"
PAGED_ATTENTION_MASK_LEN: "5500"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "36864"
- "--max-num-batched-tokens"
- "36864"
- "--block-size"
- "128"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_weight_nz_layout":true}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
num_prompts: 32
batch_size: 32
baseline: 95
threshold: 5
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1500
batch_size: 20
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,58 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "qwen3-8b"
model: "vllm-ascend/Qwen3-8B-w8a8sc-310-vllm-tp1"
envs:
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--tensor-parallel-size"
- "1"
- "--gpu-memory-utilization"
- "0.8"
- "--max-num-seqs"
- "32"
- "--dtype"
- "float16"
- "--quantization"
- "ascend"
- "--max-model-len"
- "20480"
- "--no-enable-prefix-caching"
- "--load_format"
- "sharded_state"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[1,4,8,16,24,32]}'
- "--additional-config"
- '{"ascend_compilation_config":{"enable_npugraph_ex":false, "fuse_norm_quant": false}}'
- "--default-chat-template-kwargs"
- '{"enable_thinking": false}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in512-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 512
batch_size: 1
request_rate: 0
baseline: 18
threshold: 0.95
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 10240
batch_size: 32
baseline: 93.3
threshold: 1
temperature: 0
top_p: 0.95
ignore_eos: false

View File

@@ -0,0 +1,74 @@
# ==========================================
# Shared Configurations
# ==========================================
#workflow暂缺A2yaml
# - name: Qwen3.5-122B-A10B-W8A8-single-A2
# os: linux-aarch64-a2b3-8
# config_file_path: Qwen3.5-122B-A10B-W8A8-A2.yaml
_envs: &envs
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--tensor-parallel-size"
- "8"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "262144"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--data-parallel-size"
- "1"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--enable-expert-parallel"
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./profiling", "torch_profiler_with_stack": false}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,20,24,28,32,36,40,44,48,52,56,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128,132,136,140,144,148,152,156,160,164,168,172,176,180,184,188,192,196,200,204,208,212,216,220,224,228,232,236,240,244,248,252,256,260,264,268,272,276,280,284,286,292,296,300,304,308,312,316,320,324,328,332,336,340], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
_benchmarks: &benchmarks
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 256
max_out_len: 1500
batch_size: 64
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-122B-A10B-W8A8-single-A2"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,281 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
_envs: &envs
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "4"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--no-enable-prefix-caching"
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding": true, "enable_shared_expert_dp": true}'
- "--api-server-count"
- "1"
_benchmark_acc: &_benchmark_acc
acc_GPQA:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 85
threshold: 5
thinking: true
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 1.5
repetition_penalty: 1.0
ignore_eos: false
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
thinking: true
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 1.5
repetition_penalty: 1.0
ignore_eos: false
_benchmark_TPOT50_3_5k_1_5K: &_benchmark_TPOT50_3_5k_1_5K
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 320
max_out_len: 1500
batch_size: 80
request_rate: 0
baseline: 1052.9
threshold: 0.95
_benchmark_TPOT50_16K_1K: &_benchmark_TPOT50_16K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 96
max_out_len: 1024
batch_size: 24
request_rate: 0
baseline: 400.183
threshold: 0.95
_benchmark_TPOT20_16K_1K: &_benchmark_TPOT20_16K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 1024
batch_size: 8
request_rate: 0
baseline: 325.5
threshold: 0.95
_benchmark_TPOT50_32K_0_5K: &_benchmark_TPOT50_32K_0_5K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 36
max_out_len: 512
batch_size: 9
request_rate: 0
baseline: 123.8
threshold: 0.95
_benchmark_TPOT20_32K_0_5K: &_benchmark_TPOT20_32K_0_5K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 512
batch_size: 4
request_rate: 0
baseline: 113.8
threshold: 0.95
_benchmark_TPOT50_64K_1K: &_benchmark_TPOT50_64K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 1024
batch_size: 8
request_rate: 0
baseline: 114.1
threshold: 0.95
_benchmark_TPOT20_64K_1K: &_benchmark_TPOT20_64K_1K
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0
baseline: 99.28
threshold: 0.95
test_cases:
- name: "Qwen3.5-122B-A10B-W8A8-A3"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-expert-parallel"
- "--max-model-len"
- "8000"
- "--max-num-batched-tokens"
- "8000"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
benchmarks:
perf: *_benchmark_TPOT50_3_5k_1_5K
- name: "Qwen3.5-122B-A10B-W8A8-TPOT50-16k-1k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-expert-parallel"
- "--max-model-len"
- "18432"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
benchmarks:
perf: *_benchmark_TPOT50_16K_1K
- name: "Qwen3.5-122B-A10B-W8A8-TPOT20-16k-1k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "18432"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5}'
benchmarks:
perf: *_benchmark_TPOT20_16K_1K
- name: "Qwen3.5-122B-A10B-W8A8-TPOT50-32k-0.5k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-expert-parallel"
- "--max-model-len"
- "34304"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
benchmarks:
perf: *_benchmark_TPOT50_32K_0_5K
- name: "Qwen3.5-122B-A10B-W8A8-TPOT20-32k-0.5k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "34304"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
benchmarks:
perf: *_benchmark_TPOT20_32K_0_5K
- name: "Qwen3.5-122B-A10B-W8A8-TPOT50-64k-1k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-expert-parallel"
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
benchmarks:
perf: *_benchmark_TPOT50_64K_1K
<<: *_benchmark_acc
- name: "Qwen3.5-122B-A10B-W8A8-TPOT20-64k-1k"
model: "Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5}'
benchmarks:
perf: *_benchmark_TPOT20_64K_1K

View File

@@ -0,0 +1,549 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "1024"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "DEFAULT_PORT"
test_cases:
- name: "Qwen3.5-27B-w8a8-in3.5k-140-35"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--tensor-parallel-size"
- "2"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "128000"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--mm-processor-cache-gb"
- "0"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 140
max_out_len: 1500
batch_size: 35
request_rate: 0
baseline: 610.22
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in16k-56-14"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "18432"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--no-enable-prefix-caching"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 56
max_out_len: 1024
batch_size: 14
request_rate: 0
baseline: 221.22
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in32k-56-14"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "34304"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 56
max_out_len: 512
batch_size: 14
request_rate: 0
baseline: 197.00
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in64k-16-4"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--no-enable-prefix-caching"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0
baseline: 56.17
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in64k-48-12"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "67584"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 48
max_out_len: 1024
batch_size: 12
request_rate: 0
baseline: 195.87
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in128k-28-7"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "133120"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 28
max_out_len: 1024
batch_size: 7
request_rate: 0
baseline: 101.12
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in16k-16-4"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "18432"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--no-enable-prefix-caching"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0
baseline: 163.77
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in32k-8-2"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "34304"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--no-enable-prefix-caching"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 512
batch_size: 2
request_rate: 0
baseline: 53.82
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in32k-16-4"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "34304"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 512
batch_size: 4
request_rate: 0
baseline: 149.23
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in64k-8-2"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--no-enable-prefix-caching"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 48.52
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in64k-8-2-90"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "67584"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 87.3
threshold: 0.97
- name: "Qwen3.5-27B-w8a8-in128k-4-1"
model: "Eco-Tech/Qwen3.5-27B-w8a8-mtp"
envs:
<<: *envs
server_cmd:
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "133120"
- "--enable-prefix-caching"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 36.44
threshold: 0.97

View File

@@ -0,0 +1,61 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w8a8-mtp"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "6000"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--safetensors-load-strategy"
- "lazy"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 56.357
threshold: 0.97

View File

@@ -0,0 +1,402 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
_envs: &envs
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "600"
_server_cmd: &server_cmd
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "128"
- "--quantization"
- "ascend"
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--safetensors-load-strategy"
- "lazy"
_benchmarks: &benchmarks
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0
baseline: 165.23
threshold: 0.97
_benchmarks_bs144: &benchmarks_bs144
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 144
max_out_len: 1024
batch_size: 36
request_rate: 0
baseline: 630.88
threshold: 0.97
_benchmarks_bs36: &benchmarks_bs36
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 36
max_out_len: 1024
batch_size: 9
request_rate: 0
baseline: 336.80
threshold: 0.97
_benchmarks_bs48: &benchmarks_bs48
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 48
max_out_len: 512
batch_size: 12
request_rate: 0
baseline: 180.60
threshold: 0.97
_benchmarks_bs16: &benchmarks_bs16
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 512
batch_size: 4
request_rate: 0
baseline: 118.17
threshold: 0.97
_benchmarks_bs160: &benchmarks_bs160
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 512
batch_size: 40
request_rate: 0
baseline: 689.71
threshold: 0.97
_benchmarks_bs32: &benchmarks_bs32
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in32768_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 512
batch_size: 8
request_rate: 0
baseline: 303.07
threshold: 0.97
_benchmarks_bs8: &benchmarks_bs8
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 93.10
threshold: 0.97
_benchmarks_bs32_in65536: &benchmarks_bs32_in65536
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 32
max_out_len: 1024
batch_size: 8
request_rate: 0
baseline: 160.55
threshold: 0.97
_benchmarks_bs136: &benchmarks_bs136
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 136
max_out_len: 1024
batch_size: 34
request_rate: 0
baseline: 618.66
threshold: 0.97
_benchmarks_bs16_in65536: &benchmarks_bs16_in65536
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 0
baseline: 198.87
threshold: 0.97
_benchmarks_bs80: &benchmarks_bs80
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1024
batch_size: 20
request_rate: 0
baseline: 335.95
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
#131072 1024 0.9 80
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs80"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "133120"
benchmarks:
<<: *benchmarks_bs80
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs2"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "133120"
benchmarks:
<<: *benchmarks
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs144"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "18432"
benchmarks:
<<: *benchmarks_bs144
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs36"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "18432"
benchmarks:
<<: *benchmarks_bs36
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs48"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "34304"
benchmarks:
<<: *benchmarks_bs48
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "34304"
benchmarks:
<<: *benchmarks_bs16
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs160"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "34304"
benchmarks:
<<: *benchmarks_bs160
#65536 1024 0 32
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs32"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "34304"
benchmarks:
<<: *benchmarks_bs32
#65536 1024 0 8
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs8"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "67584"
benchmarks:
<<: *benchmarks_bs8
#65536 1024 0 32
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs32_in65536"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--no-enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "67584"
benchmarks:
<<: *benchmarks_bs32_in65536
#67584 65536 1024 0.9 136
- name: "Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs136"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "67584"
benchmarks:
<<: *benchmarks_bs136
#67584 65536 1024 0.9 16
- name: "Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16_in65536"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enable-prefix-caching"
- "--speculative_config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 5, "enforce_eager": true}'
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,6,12,18,24,30,36,42,48,54,72,78,84,90,96,102,108,144,192], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--max-model-len"
- "67584"
benchmarks:
<<: *benchmarks_bs16_in65536

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w8a8-3.5k-1.5k-0-50-single"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- --data-parallel-size
- "1"
- --tensor-parallel-size
- "16"
- --enable-expert-parallel
- --max-model-len
- "6144"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "128"
- --gpu-memory-utilization
- "0.9"
- --compilation-config
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- --speculative_config
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- --trust-remote-code
- --async-scheduling
- --allowed-local-media-path
- "/"
- --quantization
- "ascend"
- --mm-processor-cache-gb
- "0"
- --additional-config
- '{"enable_cpu_binding":true}'
benchmarks:
perf_50:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 640
max_out_len: 1500
batch_size: 160
request_rate: 5
baseline: 2663.3728
threshold: 0.97
perf_20:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1421.89
threshold: 0.97

View File

@@ -0,0 +1,91 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.5-397B-A17B-w8a8-64k-1k-90-50-single"
model: "Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp"
envs:
VLLM_USE_MODELSCOPE: "true"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "2048"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- --data-parallel-size
- "1"
- --tensor-parallel-size
- "16"
- --quantization
- "ascend"
- --allowed-local-media-path
- "/"
- --seed
- "1024"
- --enable-prefix-caching
- --enable-expert-parallel
- --max-model-len
- "140000"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "32"
- --gpu-memory-utilization
- "0.9"
- --compilation-config
- '{"cudagraph_capture_sizes":[1,4,8,16,24,32,48,64,96,108,128,192],"cudagraph_mode":"FULL_DECODE_ONLY"}'
- --speculative_config
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- --additional-config
- '{"enable_cpu_binding":true}'
- --trust-remote-code
- --async-scheduling
- --mm-processor-cache-gb
- "0"
- --mm-encoder-tp-mode
- "data"
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
temperature: 0.6
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 0.0
repetition_penalty: 1.0
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 96
max_out_len: 1024
batch_size: 24
request_rate: 5
baseline: 374.5353
threshold: 0.97
perf_128k:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 80
max_out_len: 1024
batch_size: 20
request_rate: 5
baseline: 345.3004
threshold: 0.97

View File

@@ -0,0 +1,183 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.6-35B-A3B-in16k-out1k-0-128-32"
model: "Eco-Tech/Qwen3.6-35B-A3B-w8a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--tensor-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "36"
- "--max-model-len"
- "20000"
- "--max-num-batched-tokens"
- "8192"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "1024"
- "--no-enable-prefix-caching"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,20,24,28,40,48,64,76,80,88,96,100,104,108,112,116,120,136,140,144],"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--profiler-config"
- '{"profiler": "torch","torch_profiler_dir": "./profiling","torch_profiler_with_stack": false}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in16384_bs200_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1024
batch_size: 32
request_rate: 0.6
baseline: 548.2128
threshold: 0.97
- name: "Qwen3.6-35B-A3B-in984k-out1k-0-4-2"
model: "Eco-Tech/Qwen3.6-35B-A3B-w8a8"
envs:
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "200"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_GDN_FAST_PATH: "1"
VLLM_ASCEND_GDN_MAX_PADDING_RATIO: "2.0"
VLLM_ASCEND_GDN_MAX_H_OVERALLOC_RATIO: "2.0"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--hf-overrides"
- '{"text_config":{"rope_parameters":{"rope_type":"yarn","factor":4.0,"original_max_position_embeddings":262144}}}'
- "--port"
- "$SERVER_PORT"
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--max-model-len"
- "1010000"
- "--max-num-batched-tokens"
- "16384"
- "--max-num-seqs"
- "128"
- "--gpu-memory-utilization"
- "0.9"
- "--no-enable-prefix-caching"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,24,32,48,56,64,72,84,96,108,112,128,160,172,196,200,212,232,256,160,172,196,200,212,232,256,272,288,312,328,344,360,384,400,416,432,448,480,512],"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp", "num_speculative_tokens": 3, "enforce_eager": true}'
- "--trust-remote-code"
- "--async-scheduling"
- "--quantization"
- "ascend"
- "--allowed-local-media-path"
- "/"
- "--mm-processor-cache-gb"
- "0"
- "--additional-config"
- '{"enable_cpu_binding":true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in1007616_bs16_prefix0_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 2
request_rate: 0
baseline: 2.2225
threshold: 0.97
- name: "Qwen3.6-35B-A3B-in128k-out1k-90-60-15"
model: "Eco-Tech/Qwen3.6-35B-A3B-w8a8"
envs:
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_DISABLE_COMPILE_CACHE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--tensor-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "36"
- "--max-model-len"
- "140000"
- "--max-num-batched-tokens"
- "16384"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "1024"
- "--enable-prefix-caching"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,4,8,12,16,20,24,28,40,48,64,76,80,88,96,100,104,108,112,116,120,136,140,144],"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--profiler-config"
- '{"profiler": "torch","torch_profiler_dir": "./profiling","torch_profiler_with_stack": false}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 60
max_out_len: 1024
batch_size: 15
request_rate: 0
baseline: 223.2484
threshold: 0.97

View File

@@ -0,0 +1,67 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.6-35B-A3B"
model: "Eco-Tech/Qwen3.6-35B-A3B-w8a8"
envs:
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--tensor-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-num-seqs"
- "50"
- "--max-model-len"
- "65536"
- "--max-num-batched-tokens"
- "8192"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "1024"
- "--no-enable-prefix-caching"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,8,16,24,28,40,52,64,76,88,96,100,112,120,132,144,160,176,184,192,196,200,204,208,212,216,220],"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--profiler-config"
- '{"profiler": "torch","torch_profiler_dir": "./profiling","torch_profiler_with_stack": false}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 90
threshold: 10
temperature: 1.0
top_p: 0.95
top_k: 20
min_p: 0.0
presence_penalty: 1.5
repetition_penalty: 1.0

View File

@@ -0,0 +1,78 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
VLLM_USE_V1: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
_server_cmd: &server_cmd
- "--async-scheduling"
- "--tensor-parallel-size"
- "2"
- "--data-parallel-size"
- "4"
- "--data-parallel-size-local"
- "4"
- "--data-parallel-start-rank"
- "0"
- "--data-parallel-rpc-port"
- "2345"
- "--max-num-seqs"
- "40"
- "--max-model-len"
- "40960"
- "--max-num-batched-tokens"
- "16384"
- "--gpu-memory-utilization"
- "0.9"
- "--enable-expert-parallel "
- "--port"
- "$SERVER_PORT"
- "--no-enable-prefix-caching"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--host"
- "0.0.0.0"
- "--data-parallel-address"
- "0.0.0.0"
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 85
threshold: 5
temperature: 0.6
top_p: 0.95
top_k: 20
ignore_eos: false
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "vllm_Qwen3_235B_Accuracy_FULL_Graph_GPQA"
model: "Eco-Tech/Qwen3-235B-A22B-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,144 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
VLLM_USE_V1: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--tensor-parallel-size"
- "4"
- "--trust-remote-code"
- "--reasoning-parser"
- "qwen3"
- "--gpu-memory-utilization"
- "0.9"
- "--block-size"
- "128"
- "--no-enable-prefix-caching"
_benchmarks_gsm8k_lite: &benchmarks_gsm8k_lite
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k_lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 10240
num_prompts: 32
batch_size: 32
baseline: 100
threshold: 10
_benchmarks_aime: &benchmarks_aime
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2024
request_conf: vllm_api_general_chat
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
max_out_len: 32768
num_prompts: 32
batch_size: 32
baseline: 83.3
threshold: 10
_benchmarks_perf1: &benchmarks_perf1
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 400
max_out_len: 1500
batch_size: 46
request_rate: 0
baseline: 942.23
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3_32B_w8a8_Accuracy_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
OMP_PROC_BIND: "false"
ASCEND_RT_VISIBLE_DEVICES: "0,1,2,3"
TASK_QUEUE_ENABLE: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "35840"
- "--max-num-batched-tokens"
- "40960"
- "--distributed_executor_backend"
- "mp"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks_gsm8k_lite
- name: "Qwen3_32b_accuracy_aime2024_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "36784"
- "--max-num-batched-tokens"
- "36784"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"enable_weight_nz_layout":true}'
benchmarks:
<<: *benchmarks_aime
- name: "Qwen3_32B_w8a8_Performance_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
OMP_PROC_BIND: "false"
ASCEND_RT_VISIBLE_DEVICES: "0,1,2,3"
TASK_QUEUE_ENABLE: "1"
server_cmd: *server_perf_cmd
server_cmd_extra:
- "--max-model-len"
- "36784"
- "--max-num-batched-tokens"
- "36784"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"enable_weight_nz_layout":true}'
benchmarks:
<<: *benchmarks_perf1
- name: "single_Qwen3_32B_w8a8_server_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
OMP_PROC_BIND: "false"
ASCEND_RT_VISIBLE_DEVICES: "0,1,2,3"
TASK_QUEUE_ENABLE: "1"
server_cmd: *server_perf_cmd
server_cmd_extra:
- "--max-model-len"
- "36784"
- "--max-num-batched-tokens"
- "36784"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"enable_weight_nz_layout":true}'
- "--enforce-eager"
benchmarks:
<<: *benchmarks_perf1

View File

@@ -0,0 +1,79 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
VLLM_USE_V1: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--distributed_executor_backend"
- "mp"
- "--max-num-seqs"
- "256"
- "--max-model-len"
- "5500"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--reasoning-parser"
- "qwen3"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks_perf1: &benchmarks_perf1
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 304
max_out_len: 1500
batch_size: 76
request_rate: 0
baseline: 1326.599
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3_32b_int8_Feature_stack_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "5500"
- "--enforce-eager"
- "--additional-config"
- '{"ascend_scheduler_config":{"enabled":false},"enable_weight_nz_layout":true}'
- name: "Qwen3_32B_w8a8_Performance_3500_1500_A3"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
<<: *envs
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "40960"
- "--block-size"
- "128"
- "--async-scheduling"
- "--additional-config"
- '{"pa_shape_list":[32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,129,130,131,132,133,134,135,136,137,138,139,140,141,142,143,144,145,146,147,148,149,150,151,152,153,154,155,156,157,158,159,160,161,162,163,164,165,166,167,168,169,170,171,172,173,174,175,176,177,178,179,180,181,182,183,184,185,186,187,188,189,190,191,192,193,194,195,196,197,198,199,200,201,202,203,204,205,206,207,208,209,210,211,212,213,214,215,216,217,218,219,220,221,222,223,224,225,226,227,228,229,230,231,232,233,234,235,236,237,238,239,240,241,242,243,244,245,246,247,248,249,250,251,252,253,254,255,256]}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[1,5,8,72,76,80,100,120,140,144,160,192,216,240,252,288,320,336,360,384,400,408,416,420,432,480,540,576,600]}'
benchmarks:
<<: *benchmarks_perf1

View File

@@ -0,0 +1,113 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
TASK_QUEUE_ENABLE: "1"
VLLM_USE_V1: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FLASHCOMM: "1"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--distributed_executor_backend"
- "mp"
- "--trust-remote-code"
- "--reasoning-parser"
- "deepseek_r1"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks_gsm8k_lite: &benchmarks_gsm8k_lite
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k_lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 10240
num_prompts: 32
batch_size: 32
baseline: 100
threshold: 10
_benchmarks_perf1: &benchmarks_perf1
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 340
max_out_len: 1500
batch_size: 60
request_rate: 0
baseline: 724.59
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwq_32b_Feature_A3"
model: "Qwen/QwQ-32B"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-seqs"
- "256"
- "--max-model-len"
- "32768"
- "--max-num-batched-tokens"
- "32768"
- "--enforce-eager"
- "--additional-config"
- '{"ascend_scheduler_config":{"enabled":false}}'
- name: "Qwq_32B_gsm8k_Accuracy_A3"
model: "Qwen/QwQ-32B"
envs:
<<: *envs
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_PREFETCH_MLP: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "35840"
- "--max-num-batched-tokens"
- "40960"
- "--block-size"
- "128"
- "--async-scheduling"
- "--no-enable-prefix-caching"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks_gsm8k_lite
- name: "Qwq_Performance_A3_3500_1500_A3"
model: "Qwen/QwQ-32B"
envs:
<<: *envs
OMP_PROC_BIND: "false"
ASCEND_RT_VISIBLE_DEVICES: "0,1,2,3"
TASK_QUEUE_ENABLE: "1"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "5500"
- "--max-num-batched-tokens"
- "40960"
- "--block-size"
- "128"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [ 1,5,8,72,76,80,100,120,140,144,160,192,216,240,252,288,320,336,360,384,400,408,416,420,432,480,540,576,600 ]}'
- "--additional-config"
- '{"pa_shape_list": [32,33,34,35,36,37,38,39,40,41,42,43,44,45,46,47,48,49,50,51,52,53,54,55,56,57,58,59,60,61,62,63,64,65,66,67,68,69,70,71,72,73,74,75,76,77,78,79,80,81,82,83,84,85,86,87,88,89,90,91,92,93,94,95,96,97,98,99,100,101,102,103,104,105,106,107,108,109,110,111,112,113,114,115,116,117,118,119,120,121,122,123,124,125,126,127,128,129,130,131,132,133,134,135,136,137,138,139,140,141,142,143,144,145,146,147,148,149,150,151,152,153,154,155,156,157,158,159,160,161,162,163,164,165,166,167,168,169,170,171,172,173,174,175,176,177,178,179,180,181,182,183,184,185,186,187,188,189,190,191,192,193,194,195,196,197,198,199,200,201,202,203,204,205,206,207,208,209,210,211,212,213,214,215,216,217,218,219,220,221,222,223,224,225,226,227,228,229,230,231,232,233,234,235,236,237,238,239,240,241,242,243,244,245,246,247,248,249,250,251,252,253,254,255,256 ]}'
benchmarks:
<<: *benchmarks_perf1

View File

@@ -0,0 +1,219 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_IF_IP: "127.0.0.1"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--host"
- "0.0.0.0"
- "--data-parallel-size"
- "1"
- "--tensor-parallel-size"
- "2"
- "--max-num-seqs"
- "32"
- "--gpu-memory-utilization"
- "0.9"
- "--trust-remote-code"
- "--async-scheduling"
- "--allowed-local-media-path"
- "/"
- "--quantization"
- "ascend"
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--mm-processor-cache-type"
- "shm"
- "--additional-config"
- '{"enable_cpu_binding": true}'
- "--api-server-count"
- "4"
_benchmarks_acc: &benchmarks_acc
acc:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 84.34
threshold: 5
temperature: 0.6
top_p: 0.95 v54
ignore_eos: false
thinking: true
_benchmarks_3500_50: &benchmarks_3500_50
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 60
max_out_len: 1500
batch_size: 15
request_rate: 0
baseline: 701.55
threshold: 0.97
_benchmarks_3500_20: &benchmarks_3500_20
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs8000-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 48
max_out_len: 1500
batch_size: 12
request_rate: 0
baseline: 115.03
threshold: 0.97
_benchmarks_prefix_64k_50: &benchmarks_prefix_64k_50
perf_prefix_90:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 1
threshold: 0.1
perf_64k:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in65536_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 72
max_out_len: 1024
batch_size: 18
request_rate: 0
baseline: 284.9
threshold: 0.97
_benchmarks_prefix_128k_50: &benchmarks_prefix_128k_50
perf_prefix_90:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_qwen
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 1
threshold: 0.1
perf_128k:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs100-qwen3
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 24
max_out_len: 1024
batch_size: 6
request_rate: 0
baseline: 95.24
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Qwen3.6-27B-acc"
model: "Eco-Tech/Qwen3.6-27B-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,40,44,48,52,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128]}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- "--enable-prefix-caching"
benchmarks:
<<: *benchmarks_acc
- name: "Qwen3.6-27B-3500-50"
model: "Eco-Tech/Qwen3.6-27B-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "18432"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,40,44,48,52,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128]}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
benchmarks:
<<: *benchmarks_3500_50
- name: "Qwen3.6-27B-3500-20"
model: "Eco-Tech/Qwen3.6-27B-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "65536"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,6,12,18,24,30,36,42,48,54,60,66,72,78,84,90,96,102,108,114,120,126,132,138,144,150,156,162,168,174,180,186,192]}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 5}'
benchmarks:
<<: *benchmarks_3500_20
- name: "Qwen3.6-27B-prefix_64k_50"
model: "Eco-Tech/Qwen3.6-27B-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "67584"
- "--max-num-batched-tokens"
- "65536"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,6,12,18,24,30,36,42,48,54,60,66,72,78,84,90,96,102,108,114,120,126,132,138,144,150,156,162,168,174,180,186,192]}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 5}'
- "--enable-prefix-caching"
benchmarks:
<<: *benchmarks_prefix_64k_50
- name: "Qwen3.6-27B-prefix_128k_50"
model: "Eco-Tech/Qwen3.6-27B-w8a8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "16384"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,40,44,48,52,60,64,68,72,76,80,84,88,92,96,100,104,108,112,116,120,124,128]}'
- "--speculative-config"
- '{"method": "qwen3_5_mtp","num_speculative_tokens": 3}'
- "--enable-prefix-caching"
benchmarks:
<<: *benchmarks_prefix_128k_50

View File

@@ -0,0 +1,47 @@
import pytest
from tests.e2e.conftest import RemoteOpenAIServer
from tests.e2e.weekly.single_node.engine_func_test_robot.utility.http_client import (
HTTPClient,
)
env_dict: dict = {}
server_args: list = [
"--served-model-name",
"auto",
"--max-model-len",
"65536",
"--tensor-parallel-size",
"2",
"--enable-expert-parallel",
"--allowed-local-media-path",
"/",
"--limit-mm-per-prompt.video",
"1",
"--limit-mm-per-prompt.image",
"5",
"--enable-auto-tool-choice",
"--tool-call-parser",
"hermes",
"--safetensors-load-strategy",
"prefetch",
]
@pytest.fixture(scope="session")
def api_client(request):
model = "Qwen/Qwen3-VL-30B-A3B-Instruct"
with RemoteOpenAIServer(model, server_args, server_port=8000, env_dict=env_dict, auto_port=False) as server:
yield HTTPClient(base_url=server.url_root)
def pytest_addoption(parser):
parser.addoption("--thinkTagOutput", action="store", type=str, default="false", required=False)
parser.addoption("--engineArchitecture", action="store", default="single", choices=["pd", "single"])
parser.addoption("--maxModelLength", action="store", default="128")
parser.addoption("--model", action="store", default="qwen")
parser.addoption("--imageNum", action="store", type=int, default=1)
parser.addoption("--videoNum", action="store", type=int, default=1)
parser.addoption("--audioNum", action="store", type=int, default=1)

View File

@@ -0,0 +1 @@
# chat_template_kwargs field test package

View File

@@ -0,0 +1,184 @@
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
def test_chat_template_kwargs_string_non_stream(api_client, request):
"""Non-streaming: chat_template_kwargs is a string instead of an object, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": "invalid_string",
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_string_stream(api_client, request):
"""Streaming: chat_template_kwargs is a string, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": "invalid_string",
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_array_non_stream(api_client, request):
"""Non-streaming: chat_template_kwargs is an array instead of an object, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": ["item1", "item2"],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_array_stream(api_client, request):
"""Streaming: chat_template_kwargs is an array, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": ["item1", "item2"],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_integer_non_stream(api_client, request):
"""Non-streaming: chat_template_kwargs is an integer, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": 123,
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_integer_stream(api_client, request):
"""Streaming: chat_template_kwargs is an integer, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": 123,
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_boolean_non_stream(api_client, request):
"""Non-streaming: chat_template_kwargs is a boolean, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": True,
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_boolean_stream(api_client, request):
"""Streaming: chat_template_kwargs is a boolean, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": False,
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 400 and error code is 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_chat_template_kwargs_nested_invalid_type_non_stream(api_client, request):
"""Non-streaming: chat_template_kwargs contains a nested invalid value type, so a 400 error should be returned."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"valid_param": "value",
"invalid_param": [1, 2, 3], # Some engines may not support array values for parameters
},
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: should be 400 if the engine validates strictly, or 200 if it ignores invalid values
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
def test_chat_template_kwargs_nested_invalid_type_stream(api_client, request):
"""Streaming: chat_template_kwargs contains a nested invalid value type."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"valid_param": "value",
"invalid_param": {"nested": [1, 2, 3]},
},
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 or 400
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"

View File

@@ -0,0 +1,244 @@
import pytest
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_very_long_value(api_client, request, stream):
"""chat_template_kwargs value is an extremely long string; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"custom_param": "a" * 10000}, # Extremely long string value
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 if the engine accepts it, or 400 if it exceeds the limit
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_many_keys(api_client, request, stream):
"""chat_template_kwargs contains many key-value pairs; boundary test."""
# Build an object with many keys
kwargs = {f"param_{i}": f"value_{i}" for i in range(100)}
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": kwargs,
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 or 400
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_unicode_keys(api_client, request, stream):
"""chat_template_kwargs contains Unicode key names; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"中文参数": "value",
"日本語パラメータ": "value",
"emoji_参数": "value",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 or 400 depending on whether the engine supports non-ASCII key names
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_special_chars_in_keys(api_client, request, stream):
"""chat_template_kwargs key names contain special characters; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"param-with-dash": "value",
"param_with_underscore": "value",
"param.with.dot": "value",
"param:with:colon": "value",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 or 400
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_numeric_string_values(api_client, request, stream):
"""chat_template_kwargs values are numeric strings; boundary type-conversion test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"number_as_string": "12345",
"float_as_string": "3.14159",
"bool_as_string": "true",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_mixed_types_values(api_client, request, stream):
"""chat_template_kwargs values have mixed types (number, boolean, null); boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"int_value": 42,
"float_value": 3.14,
"bool_value": True,
"null_value": None,
"string_value": "text",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 or 400 depending on how the engine handles non-string values
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_reserved_words_keys(api_client, request, stream):
"""chat_template_kwargs uses reserved words or internal keywords as key names; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"model": "overridden_model", # Key that may conflict with request parameters
"messages": "overridden", # Key that may conflict with request parameters
"stream": True, # Key that may conflict with request parameters
"temperature": 2.0, # Key that may conflict with generation parameters
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 if the engine isolates namespaces correctly, or 400 if there is a conflict
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_empty_string_values(api_client, request, stream):
"""chat_template_kwargs values are empty strings; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"empty_string": "",
"whitespace_only": " ",
"null_string": "null",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_deeply_nested_object(api_client, request, stream):
"""chat_template_kwargs is a deeply nested object; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"level1": {"level2": {"level3": {"level4": {"level5": {"deep_value": "found"}}}}}},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200 if the engine flattens nested objects, or 400 if it rejects nesting
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_case_sensitive_keys(api_client, request, stream):
"""chat_template_kwargs key names are case-sensitive; boundary test."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {
"Add_Generation_Prompt": True, # Different from the standard snake_case form
"ADD_GENERATION_PROMPT": True, # All uppercase
"add_generation_prompt": True, # Standard lowercase
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code should be 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)

View File

@@ -0,0 +1,230 @@
import pytest
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_null(api_client, request, stream):
"""chat_template_kwargs is null; the request should respond normally."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": None,
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: finish_reason is stop or length
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_empty_object(api_client, request, stream):
"""chat_template_kwargs is an empty object {}; the optional field should be handled normally."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: finish_reason is valid
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_with_add_generation_prompt(api_client, request, stream):
"""Set add_generation_prompt in chat_template_kwargs to control generation prompt insertion."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"add_generation_prompt": True},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_with_custom_system_prompt(api_client, request, stream):
"""Set custom system-prompt-related parameters in chat_template_kwargs."""
request_body = {
"model": "auto",
"messages": [
{"role": "system", "content": "你是AI助手"},
{"role": "user", "content": "你好"},
],
"chat_template_kwargs": {"enable_system_prompt": True},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_date_params(api_client, request, stream):
"""chat_template_kwargs contains date-related parameters; some models support dynamic dates."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "今天是星期几"}],
"chat_template_kwargs": {"date": "2025-04-01", "time": "10:00:00"},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_multiple_params(api_client, request, stream):
"""chat_template_kwargs contains multiple valid parameters."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "请简单回答"}],
"chat_template_kwargs": {
"add_generation_prompt": True,
"tools_prompt": "default",
"custom_var": "custom_value",
},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_with_tools_prompt(api_client, request, stream):
"""Set tools_prompt in chat_template_kwargs to control the tool-call prompt format."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "需要查询天气"}],
"tools": [
{
"type": "function",
"function": {
"name": "get_weather",
"description": "获取天气信息",
"parameters": {
"type": "object",
"properties": {"location": {"type": "string"}},
"required": ["location"],
},
},
}
],
"chat_template_kwargs": {"tools_prompt": "tool_instruction"},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_bos_token(api_client, request, stream):
"""Set add_special_tokens or bos_token-related parameters in chat_template_kwargs."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"add_special_tokens": True},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_skip_special_tokens(api_client, request, stream):
"""Set skip_special_tokens in chat_template_kwargs."""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"skip_special_tokens": False},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)

View File

@@ -0,0 +1,316 @@
"""
Tests for the thinking and enable_thinking fields.
- thinking: used by DeepSeek/DS model families.
- enable_thinking: used by Qwen model families.
When the field is true, validate that the think tags are complete.
When the field is false, validate that no think tags are present.
Follow the validation rules from the think_tag directory.
"""
import pytest
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
def is_qwen_model(model_name):
"""Return whether the model belongs to the Qwen family, case-insensitively."""
return model_name and "qwen" in model_name.lower()
def is_deepseek_model(model_name):
"""Return whether the model belongs to the DeepSeek/DS family, case-insensitively."""
if not model_name:
return False
model_lower = model_name.lower()
return "deepseek" in model_lower or "ds" in model_lower
def should_check_think_tag(request):
"""Return whether think-tag validation should be performed."""
return request.config.getoption("--thinkTagOutput").strip().lower() == "true"
# ==================== Qwen Model Tests - enable_thinking ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_qwen_enable_thinking_true(api_client, request, stream):
"""Qwen model: enable_thinking=true enables thinking mode; validate complete think tags."""
model = request.config.getoption("--model")
if not is_qwen_model(model):
pytest.skip(f"current model {model} is not in the Qwen family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "请用最简单的一句话介绍你是谁。"}],
"chat_template_kwargs": {"enable_thinking": True},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: think tags are complete when enable_thinking=true
if should_check_think_tag(request):
assertion.assert_think_tag_present(response.content.decode("utf-8"), "enable_thinking=true")
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_qwen_enable_thinking_false(api_client, request, stream):
"""Qwen model: enable_thinking=false disables thinking mode; validate that no think tags are present."""
model = request.config.getoption("--model")
if not is_qwen_model(model):
pytest.skip(f"current model {model} is not in the Qwen family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "请用最简单的一句话介绍你是谁。"}],
"chat_template_kwargs": {"enable_thinking": False},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: no think tags are present when enable_thinking=false
if should_check_think_tag(request):
assertion.assert_no_think_tag(response.content.decode("utf-8"), "enable_thinking=false")
# ==================== DeepSeek/DS Model Tests - thinking ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_deepseek_thinking_true(api_client, request, stream):
"""DeepSeek/DS model: thinking=true enables thinking mode; validate complete think tags."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "请用最简单的一句话介绍你是谁。"}],
"chat_template_kwargs": {"thinking": True},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: think tags are complete when thinking=true
if should_check_think_tag(request):
assertion.assert_think_tag_present(response.content.decode("utf-8"), "thinking=true")
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_deepseek_thinking_false(api_client, request, stream):
"""DeepSeek/DS model: thinking=false disables thinking mode; validate that no think tags are present."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "请用最简单的一句话介绍你是谁。"}],
"chat_template_kwargs": {"thinking": False},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check 1: status code is 200
assertion.assert_status_code_200(response)
# Check 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Check 3: no think tags are present when thinking=false
if should_check_think_tag(request):
assertion.assert_no_think_tag(response.content.decode("utf-8"), "thinking=false")
# ==================== Inapplicable Model Tests - Abnormal Scenarios ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_qwen_field_on_deepseek(api_client, request, stream):
"""Abnormal: use the enable_thinking field on a DeepSeek model."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"enable_thinking": True}, # Use the Qwen field on DeepSeek
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: either 200 if the engine ignores unknown fields, or 400 if it validates strictly
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_deepseek_field_on_qwen(api_client, request, stream):
"""Abnormal: use the thinking field on a Qwen model."""
model = request.config.getoption("--model")
if not is_qwen_model(model):
pytest.skip(f"current model {model} is not in the Qwen family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"thinking": True}, # Use the DeepSeek field on Qwen
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: either 200 if the engine ignores unknown fields, or 400 if it validates strictly
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
# ==================== Boundary Tests ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_qwen_enable_thinking_null(api_client, request, stream):
"""Abnormal: enable_thinking is null for a Qwen model."""
model = request.config.getoption("--model")
if not is_qwen_model(model):
pytest.skip(f"current model {model} is not in the Qwen family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"enable_thinking": None},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
assertion.assert_status_code_200(response)
def test_chat_template_kwargs_deepseek_thinking_null_non_stream(api_client, request):
"""Abnormal: thinking is null for a DeepSeek model in non-streaming mode; error code is 400."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"thinking": None},
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
assertion.assert_status_code_200(response)
def test_chat_template_kwargs_deepseek_thinking_null_stream(api_client, request):
"""Abnormal: thinking is null for a DeepSeek model in streaming mode; status code is 200 and error code is 400."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"thinking": None},
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: status code is 200 and error code is 400
assertion.assert_status_code_200(response)
assertion.assert_error_code_400(response)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_qwen_enable_thinking_string(api_client, request, stream):
"""Abnormal: enable_thinking is a string for a Qwen model."""
model = request.config.getoption("--model")
if not is_qwen_model(model):
pytest.skip(f"current model {model} is not in the Qwen family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"enable_thinking": "true"},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_chat_template_kwargs_deepseek_thinking_string(api_client, request, stream):
"""Abnormal: thinking is a string for a DeepSeek model."""
model = request.config.getoption("--model")
if not is_deepseek_model(model):
pytest.skip(f"current model {model} is not in the DeepSeek/DS family; skipping this test")
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好"}],
"chat_template_kwargs": {"thinking": "true"},
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
assert response.status_code in [
200,
400,
], f"status code should be 200 or 400, got {response.status_code}"

View File

@@ -0,0 +1 @@
# Content field test package

View File

@@ -0,0 +1,264 @@
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
def test_content_integer_non_stream(api_client, request):
"""Non-streaming: content is integer type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": 12345}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_integer_stream(api_client, request):
"""Streaming: content is integer type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": 12345}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_object_non_stream(api_client, request):
"""Non-streaming: content is object type (non-standard multimodal format), should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": {"text": "hello", "extra": "data"}}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_object_stream(api_client, request):
"""Streaming: content is object type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": {"text": "hello", "extra": "data"}}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_boolean_non_stream(api_client, request):
"""Non-streaming: content is boolean type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": True}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_boolean_stream(api_client, request):
"""Streaming: content is boolean type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": False}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_invalid_format_non_stream(api_client, request):
"""Non-streaming: content is array but format does not conform to OpenAI multimodal spec (string array),
should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": ["invalid", "array", "format"]}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_invalid_format_stream(api_client, request):
"""Streaming: content is array but format does not conform to OpenAI multimodal spec, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": ["invalid", "array", "format"]}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
# ==================== Content Array Abnormal Tests ====================
def test_content_array_missing_type_non_stream(api_client, request):
"""Non-streaming: content array object missing type field, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"text": "你好"}]}], # missing type field
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_missing_type_stream(api_client, request):
"""Streaming: content array object missing type field, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"text": "你好"}]}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_missing_text_non_stream(api_client, request):
"""Non-streaming: content array object type is text but missing text field, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text"}]}], # missing text field
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_invalid_type_non_stream(api_client, request):
"""Non-streaming: content array object type is invalid, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "invalid_type", "text": "你好"}]}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_invalid_type_stream(api_client, request):
"""Streaming: content array object type is invalid, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "unknown", "text": "你好"}]}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_text_field_null_non_stream(api_client, request):
"""Non-streaming: content array object text field is null, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": None}]}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_text_field_integer_non_stream(api_client, request):
"""Non-streaming: content array object text field is integer type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": 12345}]}],
"stream": False,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)
def test_content_array_text_field_integer_stream(api_client, request):
"""Streaming: content array object text field is integer type, should return 400 error"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": 12345}]}],
"stream": True,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code 400, error code 400
assertion.assert_status_code_400(response)
assertion.assert_error_code_400(response)

View File

@@ -0,0 +1,307 @@
import pytest
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_whitespace_only(api_client, request, stream):
"""Content contains only whitespace characters (spaces, tabs, newlines), boundary case handling"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": " \t\n\n "}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400 (depends on engine implementation)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_single_char(api_client, request, stream):
"""Content is a single character, minimum valid content boundary"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "?"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_with_null_bytes(api_client, request, stream):
"""Content contains null bytes \x00, boundary security test"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "Hello\x00World"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400 (depends on how engine handles null bytes)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_json_escape_sequences(api_client, request, stream):
"""Content contains JSON escape characters, boundary test"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": 'Line1\nLine2\tTabbed"Quoted"\\Backslash'}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_unicode_edge_cases(api_client, request, stream):
"""Content contains Unicode boundary characters (e.g., zero-width characters, combining characters)"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": "零宽空格:\u200b 零宽连接符:\u200d 从右向左符:\u202e 组合字符:é",
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_rare_unicode_blocks(api_client, request, stream):
"""Content contains rare Unicode block characters (emoji variants, math symbols, etc.)"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": "数学:∀∃∈∉ 表情变体:👨🏻‍💻 盲文:⠓⠑⠇⠇⠕ 箭头:↳↴↵",
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_rtl_languages(api_client, request, stream):
"""Content contains right-to-left languages (Arabic, Hebrew, etc.)"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "مرحبا بالعالم (Arabic) שלום עולם (Hebrew)"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_mixed_encoding_simulation(api_client, request, stream):
"""Content simulates mixed encoding scenario (correctly encoded UTF-8)"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "Mixed: English中文العربية日本語🌍"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# ==================== Content Array Format Boundary Tests ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_type_field_case_sensitive(api_client, request, stream):
"""Content array format type field case sensitivity boundary test"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "TEXT", "text": "你好"}]}], # uppercase TEXT
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 (insensitive) or 400 (sensitive)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_extra_fields(api_client, request, stream):
"""Content array format contains extra fields, boundary test"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [{"type": "text", "text": "你好", "extra_field": "extra_value"}],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 (if engine ignores extra fields) or 400 (if strict validation)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_text_whitespace_only(api_client, request, stream):
"""Content array format text field contains only whitespace characters, boundary test"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": " \t\n "}]}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400 (depends on engine implementation)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_very_long_text(api_client, request, stream):
"""Content array format text field is extremely long text, boundary test"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": "A" * 5000}]}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_many_objects(api_client, request, stream):
"""Content array format contains many text objects, boundary test"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [{"type": "text", "text": f"分段{i}"} for i in range(50)],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_unicode_text(api_client, request, stream):
"""Content array format text field contains Unicode characters, boundary test"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [{"type": "text", "text": "中文🇨🇳日本語🗾العربية🌍"}],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)

View File

@@ -0,0 +1,385 @@
import pytest
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import assertion
from tests.e2e.weekly.single_node.engine_func_test_robot.utility import (
request_helper as helper,
)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_simple_string(api_client, request, stream):
"""Content is a plain string, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好,请简单介绍一下自己"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Checkpoint 3: finish_reason is valid
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_empty_string(api_client, request, stream):
"""Content is an empty string, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": ""}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Checkpoint 3: finish_reason is stop or length
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_null(api_client, request, stream):
"""Content is null, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": None}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Check: error code 400, or finish_reason stop/length both pass
if assertion.has_error_code(response):
# Error code exists, validate it is 400
assertion.assert_error_code_400(response)
else:
# No error code, check finish_reason is stop or length
if stream:
assertion.assert_stream_has_done(response.text)
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_empty(api_client, request, stream):
"""Content is an empty array [], request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": []}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Checkpoint 3: finish_reason is stop or length
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_missing(api_client, request, stream):
"""Message object missing content field, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user"
# missing content field
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Checkpoint 3: finish_reason is stop or length
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_with_special_chars(api_client, request, stream):
"""Content contains special characters (punctuation, symbols, etc.), request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "Hello! 你好~ @#$%^&*()_+-=[]{}|;':\",./<>?"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_multiline_text(api_client, request, stream):
"""Content contains multiline text (newline characters), request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "第一行\n第二行\n\n空行后的第三行"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_with_emoji(api_client, request, stream):
"""Content contains emoji, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": "你好👋 很高兴见到你😊 这是一颗星星⭐"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_unicode_chinese(api_client, request, stream):
"""Content contains Chinese characters and Unicode characters, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": " apples 中文测试 日本語テスト 한국어"}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_long_text(api_client, request, stream):
"""Content is a long text (approx. 1000 characters), request should succeed normally"""
long_content = "这是测试文本。" * 100
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": long_content}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_code_snippet(api_client, request, stream):
"""Content is a code snippet, request should succeed normally"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": "```python\ndef hello():\n print('Hello World')\n```请解释这段代码",
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# ==================== Content Array Format Tests ====================
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_text_objects(api_client, request, stream):
"""Content is an array of multiple text objects (OpenAI multimodal standard format)"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [
{"type": "text", "text": "你好"},
{"type": "text", "text": "你是谁?"},
],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
# Checkpoint 3: finish_reason is valid
if stream:
finish_reason = assertion.assert_stream_single_finish_reason(response.text)
else:
finish_reason = response.json()["choices"][0]["finish_reason"]
assertion.assert_finish_reason_valid(finish_reason)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_single_text_object(api_client, request, stream):
"""Content is an array with a single text object"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [{"type": "text", "text": "请简单介绍一下自己"}],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_empty_text(api_client, request, stream):
"""Content is an array format but text is an empty string"""
request_body = {
"model": "auto",
"messages": [{"role": "user", "content": [{"type": "text", "text": ""}]}],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint: status code should be 200 or 400 (depends on engine implementation)
assert response.status_code in [
200,
400,
], f"Status code should be 200 or 400, got {response.status_code}"
@pytest.mark.parametrize("stream", [False, True], ids=["non_stream", "stream"])
def test_content_array_many_text_objects(api_client, request, stream):
"""Content is an array containing many text objects (boundary test)"""
request_body = {
"model": "auto",
"messages": [
{
"role": "user",
"content": [
{"type": "text", "text": "第一部分内容。"},
{"type": "text", "text": "第二部分内容。"},
{"type": "text", "text": "第三部分内容。"},
{"type": "text", "text": "第四部分内容。"},
],
}
],
"stream": stream,
"max_tokens": 512,
}
response = helper.send_request(api_client, "/v1/chat/completions", request_body)
# Checkpoint 1: status code 200
assertion.assert_status_code_200(response)
# Checkpoint 2: streaming response contains [DONE]
if stream:
assertion.assert_stream_has_done(response.text)

View File

@@ -0,0 +1,195 @@
import json
import regex as re
# think tag definitions
THINK_OPEN = "<think>"
THINK_CLOSE = "</think>"
class Check:
@staticmethod
def equal(a, b, msg=""):
assert a == b, msg
@staticmethod
def not_equal(a, b, msg=""):
assert a != b, msg
@staticmethod
def is_true(v, msg=""):
assert v, msg
@staticmethod
def is_in(v, seq, msg=""):
assert v in seq, msg
check = Check()
def assert_status_code_200(response, msg=""):
"""Verify HTTP status code is 200"""
check.equal(response.status_code, 200, f"{msg}Response status code is not 200")
def assert_status_code_400(response, msg=""):
"""Verify HTTP status code is 400"""
check.equal(response.status_code, 400, f"{msg}Response status code is not 400")
def assert_finish_reason_stop(finish_reason, msg=""):
"""Verify finish_reason is stop"""
check.equal(finish_reason, "stop", f"{msg}finish_reason is not stop")
def assert_finish_reason_valid(finish_reason, msg=""):
"""Verify finish_reason is stop or length"""
check.is_in(finish_reason, ["stop", "length"], f"{msg}finish_reason is not stop or length")
def assert_stream_has_done(response_text, msg=""):
"""Verify streaming response contains [DONE]"""
check.is_true(
re.search(r"^data:\s*\[DONE\](?:\n|$)", response_text, re.M),
f"{msg}Streaming response does not contain [DONE]",
)
def assert_stream_single_finish_reason(response_text, msg=""):
"""Verify streaming response has exactly one finish_reason, return its value"""
finish_reasons = re.findall(r'finish_reason":\s*"([^"]+)"', response_text, re.M)
check.equal(len(finish_reasons), 1, f"{msg}Streaming response has multiple finish_reason values")
return finish_reasons[0] if finish_reasons else None
def assert_think_tag_present(response_text, msg=""):
"""Verify complete think tag pairs exist"""
think_open_count = response_text.count(THINK_OPEN)
think_close_count = response_text.count(THINK_CLOSE)
check.equal(
think_open_count,
think_close_count,
f"{msg}think tags are not balanced, OPEN: {think_open_count}, CLOSE: {think_close_count}",
)
check.equal(think_open_count, 1, f"{msg}No think tag present")
def assert_no_think_tag(response_text, msg=""):
"""Verify think tags do not exist"""
check.equal(response_text.count(THINK_OPEN), 0, f"{msg}think tag exists")
def assert_json_response_content(response_text, msg=""):
"""Verify response content is valid JSON (after filtering think tags)"""
pattern = rf"\s*{re.escape(THINK_OPEN)}[\s\S]*?{re.escape(THINK_CLOSE)}"
json_str = re.sub(pattern, "", response_text)
match = re.search(r"(\{.*\})\s*(?:$|`|```)$", json_str, re.S)
check.is_true(match, f"{msg}Content is not in JSON format")
if match:
json.loads(match.group(1))
def has_error_code(response):
"""Determine if the response contains an error code"""
content_type = response.headers.get("Content-Type", "")
if "application/json" in content_type:
response_json = response.json()
error_code = response_json.get("error", {}).get("code") or response_json.get("code")
return error_code is not None
elif "text/event-stream" in content_type or "text/plain" in content_type:
match = re.search(r"\"code\"\s?:\s?(\d+)", response.text, re.M)
return match is not None
return False
def assert_error_code_400(response, msg=""):
"""Verify error code is 400"""
if "application/json" in response.headers.get("Content-Type", ""):
error_code = response.json().get("error", {}).get("code") or response.json().get("code")
check.equal(error_code, 400, f"{msg}Error code is not 400")
elif "text/event-stream" in response.headers.get("Content-Type", ""):
match = re.search(r"\"code\"\s?:\s?(\d+)", response.text, re.M)
if match:
check.equal(int(match.group(1)), 400, f"{msg}Streaming response error code is not 400")
def assert_error_code_422(response, msg=""):
"""Verify error code is 422 Unprocessable Entity (data validation failure)"""
if "application/json" in response.headers.get("Content-Type", ""):
error_code = response.json().get("error", {}).get("code") or response.json().get("code")
check.equal(error_code, 422, f"{msg}Error code is not 422 (Unprocessable Entity - data validation failure)")
elif "text/event-stream" in response.headers.get("Content-Type", ""):
match = re.search(r"\"code\"\s?:\s?(\d+)", response.text, re.M)
if match:
check.equal(int(match.group(1)), 422, f"{msg}Streaming response error code is not 422")
def assert_error_code_not_500(response, msg=""):
"""If response body contains an error code, verify it is not 500"""
content_type = response.headers.get("Content-Type", "")
if "application/json" in content_type:
error_code = response.json().get("error", {}).get("code") or response.json().get("code")
if error_code is not None:
check.not_equal(error_code, 500, f"{msg}Error code should not be 500, actual: {error_code}")
elif "text/event-stream" in content_type:
match = re.search(r"\"code\"\s?:\s?(\d+)", response.text, re.M)
if match:
error_code = int(match.group(1))
check.not_equal(error_code, 500, f"{msg}Error code should not be 500, actual: {error_code}")
def assert_image_edit_response_fields(response, msg=""):
"""Verify completeness of response fields for image edit API
Args:
response: HTTP response object
msg: Prefix for error messages
"""
resp_json = response.json()
# Verify top-level fields
check.is_true("created" in resp_json, f"{msg}Response should contain created field")
check.is_true("data" in resp_json, f"{msg}Response should contain data field")
check.is_true("output_format" in resp_json, f"{msg}Response should contain output_format field")
check.is_true("size" in resp_json, f"{msg}Response should contain size field")
# Verify data array
data = resp_json.get("data", [])
check.is_true(len(data) > 0, f"{msg}data should contain at least one result")
# Verify fields of each data array element
for idx, item in enumerate(data):
has_b64 = "b64_json" in item and item["b64_json"]
has_url = "url" in item and item["url"]
check.is_true(has_b64 or has_url, f"{msg}data[{idx}] should contain b64_json or url field")
check.is_true("revised_prompt" in item, f"{msg}data[{idx}] should contain revised_prompt field")
return resp_json
def assert_top_logprobs_count(response, top_logprobs_value, msg=""):
"""Verify the number of top_logprobs in logprobs"""
content_type = response.headers.get("Content-Type", "")
if "application/json" in content_type:
logprobs_content_list = response.json()["choices"][0]["logprobs"]["content"]
for item_dict in logprobs_content_list:
check.equal(
len(item_dict.get("top_logprobs")),
top_logprobs_value,
f"{msg}logprobs top_logprobs length is not {top_logprobs_value}",
)
elif "text/event-stream" in content_type:
chunk_list = re.findall(r"^data:\s*(.*)(?:\n|$)", response.text, re.M)[1:-1]
for chunk_item in chunk_list:
chunk_json = json.loads(chunk_item)
content = chunk_json["choices"][0]["delta"].get("content", "")
if content:
logprobs_content_list = chunk_json["choices"][0]["logprobs"]["content"]
for item_dict in logprobs_content_list:
check.equal(
len(item_dict.get("top_logprobs")),
top_logprobs_value,
f"{msg}Streaming logprobs top_logprobs length is not {top_logprobs_value}",
)

View File

@@ -0,0 +1,25 @@
import requests
from requests.exceptions import RequestException
class HTTPClient:
def __init__(self, base_url=None, timeout=36000):
self.base_url = base_url.rstrip("/") if base_url else ""
self.timeout = timeout
def get(self, endpoint, params=None, headers=None):
url = f"{self.base_url}/{endpoint.lstrip('/')}"
try:
response = requests.get(url, params=params, headers=headers, timeout=self.timeout)
response.raise_for_status()
return response
except RequestException as e:
raise AssertionError(f"GET {url} failed: {str(e)}")
def post(self, endpoint, json=None, data=None, files=None, headers=None):
url = f"{self.base_url}/{endpoint.lstrip('/')}"
try:
response = requests.post(url, json=json, data=data, files=files, headers=headers, timeout=self.timeout)
return response
except RequestException as e:
raise AssertionError(f"POST {url} failed: {str(e)}")

View File

@@ -0,0 +1,3 @@
def send_request(api_client, uri, request_body):
"""Send request and return response object"""
return api_client.post(uri, json=request_body, headers={"Content-Type": "application/json"})

View File

@@ -0,0 +1,143 @@
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
# Copyright 2023 The vLLM team.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
import json
from typing import Any
import openai
import pytest
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import MooncakeLauncher, RemoteOpenAIServer
from tools.aisbench import maybe_download_from_modelscope, run_aisbench_cases
MODELS = [
"vllm-ascend/Qwen3-30B-A3B-W8A8",
]
eagle_model = maybe_download_from_modelscope("vllm-ascend/Qwen3-a3B_eagle3")
TENSOR_PARALLELS = [1, 4]
prompts = [
"Janet\u2019s ducks lay 16 eggs per day. She eats three for breakfast every morning and bakes muffins for her "
"friends every day with four. She sells the remainder at the farmers' market daily for $2 per fresh duck egg. "
"How much in dollars does she make every day at the farmers' market?",
]
api_keyword_args = {
"max_tokens": 10,
}
mooncake_json = {
"local_hostname": "localhost",
"metadata_server": "P2PHANDSHAKE",
"protocol": "ascend",
"device_name": "",
"master_server_address": "",
"global_segment_size": 30000000000,
}
aisbench_cases = [
{
"case_type": "accuracy",
"dataset_path": "vllm-ascend/gsm8k-lite",
"request_conf": "vllm_api_general_chat",
"dataset_conf": "gsm8k/gsm8k_gen_0_shot_cot_chat_prompt",
"max_out_len": 32768,
"batch_size": 32,
"baseline": 95,
"threshold": 5,
}
]
@pytest.mark.asyncio
@pytest.mark.parametrize("model", MODELS)
@pytest.mark.parametrize("tp_size", TENSOR_PARALLELS)
async def test_models(model: str, tp_size: int) -> None:
port = get_open_port()
mooncake_port = get_open_port()
mooncake_metrics_port = get_open_port()
mooncake_json["master_server_address"] = f"127.0.0.1:{mooncake_port}"
with open("mooncake.json", "w") as f:
json.dump(mooncake_json, f)
env_dict = {
"PYTHONHASHSEED": "0",
"ASCEND_CONNECT_TIMEOUT": "10000",
"ASCEND_TRANSFER_TIMEOUT": "10000",
"VLLM_USE_V1": "1",
"OMP_PROC_BIND": "false",
"HCCL_OP_EXPANSION_MODE": "AIV",
"HCCL_BUFFSIZE": "1024",
"OMP_NUM_THREADS": "1",
"PYTORCH_NPU_ALLOC_CONF": "expandable_segments:True",
"VLLM_ASCEND_ENABLE_NZ": "2",
"MOONCAKE_CONFIG_PATH": "mooncake.json",
}
if tp_size != 1:
env_dict["VLLM_ASCEND_ENABLE_FLASHCOMM1"] = "1"
kv_transfer_config = {
"kv_connector": "AscendStoreConnector",
"kv_role": "kv_both",
"kv_connector_extra_config": {"register_buffer": True, "use_layerwise": False, "mooncake_rpc_port": "0"},
}
speculative_config = {"method": "eagle3", "model": eagle_model, "num_speculative_tokens": 3}
server_args = [
"--trust-remote-code",
"--max-num-seqs",
"100",
"--max-model-len",
"37364",
"--max-num-batched-tokens",
"16384",
"--tensor-parallel-size",
str(tp_size),
"--enable-expert-parallel",
"--port",
str(port),
"--distributed_executor_backend",
"mp",
"--quantization",
"ascend",
"--compilation-config",
'{"cudagraph_mode": "FULL_DECODE_ONLY"}',
"--gpu-memory-utilization",
"0.95",
"--speculative-config",
json.dumps(speculative_config),
"--kv-transfer-config",
json.dumps(kv_transfer_config),
]
request_keyword_args: dict[str, Any] = {
**api_keyword_args,
}
with (
MooncakeLauncher(mooncake_port, mooncake_metrics_port),
RemoteOpenAIServer(model, server_args, server_port=port, env_dict=env_dict, auto_port=False) as server,
):
client = server.get_async_client()
for _ in range(2):
batch = await client.completions.create(
model=model,
prompt=prompts,
**request_keyword_args,
)
choices: list[openai.types.CompletionChoice] = batch.choices
assert choices[0].text, "empty response"
# aisbench test
run_aisbench_cases(model, port, aisbench_cases)
run_aisbench_cases(model, port, aisbench_cases)