init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

File diff suppressed because it is too large Load Diff

422
tests/e2e/coverage.md Normal file
View File

@@ -0,0 +1,422 @@
The coverage of e2e is as follows:
## 1-Card Tests
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
| _310p/test_classification_310p.py | test_qwen_pooling_classify_correctness | Howeee/Qwen2.5-1.5B-apeach | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | |
| _310p/test_dense_model_310p.py | test_qwen3_5_dense_tp1_fp16 | Qwen/Qwen3.5-4B | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| _310p/test_dense_model_310p.py | test_qwen3_5_dense_tp1_fp16_aclgraph | Qwen/Qwen3.5-4B | ✅ | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_fp16 | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_fp16_aclgraph | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_w8a8 | vllm-ascend/Qwen3-8B-W8A8 | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| _310p/test_embedding_310p.py | test_bge_m3_correctness | BAAI/bge-m3 | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
| _310p/test_embedding_310p.py | test_embed_models_correctness | Qwen/Qwen3-Embedding-0.6B<br>intfloat/multilingual-e5-small | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
| _310p/test_scoring_310p.py | test_cross_encoder_score_1_to_1 | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| _310p/test_scoring_310p.py | test_cross_encoder_score_1_to_N | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| _310p/test_scoring_310p.py | test_cross_encoder_score_N_to_N | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| _310p/test_vl_model_310p.py | test_qwen3_vl_8b_tp1_fp16 | Qwen/Qwen3-VL-8B-Instruct | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_accuracy.py | test_default_full_and_piecewise_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_accuracy.py | test_full_decode_only_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_accuracy.py | test_full_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_accuracy.py | test_npugraph_ex_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_accuracy.py | test_npugraph_ex_with_static_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_logprobs_bitwise_batch_invariance_bs1_vs_bsN | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_logprobs_without_batch_invariance_should_fail | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_simple_generation | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | |
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_v1_generation_is_deterministic_across_batch_sizes_with_needle | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
| aclgraph/test_aclgraph_mem.py | test_aclgraph_mem_use | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| compile/test_graphex_norm_quant_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| compile/test_graphex_qknorm_rope_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| compile/test_norm_quant_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| lora/test_ilama_lora.py | test_ilama_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| lora/test_llama32_lora.py | test_llama_lora | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| lora/test_lora_with_spec_decode.py | test_batch_inference_correctness | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| lora/test_qwen35_densemodel_lora.py | test_qwen35_text_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| lora/test_qwen3_multi_loras.py | test_multi_loras_with_tp_sync | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | |
| lora/test_qwen3_reranker_lora.py | test_reranker_models_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | |
| model_runner_v2/test_basic.py | test_egale_spec_decoding | Qwen/Qwen3-0.6B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
| model_runner_v2/test_basic.py | test_qwen3_dense_eager_mode | Qwen/Qwen3-0.6B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | ✅ | |
| model_runner_v2/test_basic.py | test_qwen3_dense_graph_mode | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
| pooling/test_classification.py | test_qwen_pooling_classify_correctness | Howeee/Qwen2.5-1.5B-apeach | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | |
| pooling/test_embedding.py | test_bge_m3_correctness | BAAI/bge-m3 | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
| pooling/test_embedding.py | test_causal_embed_models_using_prefix_caching_correctness | Qwen/Qwen3-Embedding-0.6B | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | ✅ | |
| pooling/test_embedding.py | test_embed_models_correctness | Qwen/Qwen3-Embedding-0.6B<br>intfloat/multilingual-e5-small | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
| pooling/test_scoring.py | test_cross_encoder_score_1_to_1 | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| pooling/test_scoring.py | test_cross_encoder_score_1_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| pooling/test_scoring.py | test_cross_encoder_score_N_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| pooling/test_scoring.py | test_embedding_score_1_to_1 | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| pooling/test_scoring.py | test_embedding_score_1_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| pooling/test_scoring.py | test_embedding_score_N_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
| spec_decode/test_dflash.py | test_dflash_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
| spec_decode/test_draft_parallel.py | test_parallel_drafting_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
| spec_decode/test_eagle.py | test_qwen3_vl_eagle | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| spec_decode/test_eagle.py | test_qwen_eagle3_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| spec_decode/test_extract_hidden_states.py | test_extract_hidden_states_aclgraph_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | |
| spec_decode/test_extract_hidden_states.py | test_extract_hidden_states_eager_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | |
| spec_decode/test_mtp_eagle_correctness.py | test_deepseek_mtp | wemaster/deepseek_mtp_main_random_bf16 | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| spec_decode/test_ngram.py | test_ngram | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | |
| spec_decode/test_ngram_npu.py | test_ngram_npu_async_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| spec_decode/test_suffix.py | test_suffix_acceptance | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | |
| test_attention_fa3.py | test_fa3_vs_fia_logprobs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| test_attention_fa3.py | test_fa3_vs_fia_mixed_lengths | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | ✅ |
| test_attention_fa3.py | test_fa3_vs_fia_single_prompt | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| test_attention_fa3.py | test_fa3_vs_fia_with_chunkprefill | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| test_batch_invariant.py | test_logprobs_bitwise_batch_invariance_bs1_vs_bsN | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
| test_batch_invariant.py | test_logprobs_without_batch_invariance_should_fail | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
| test_batch_invariant.py | test_simple_generation | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | |
| test_batch_invariant.py | test_v1_generation_is_deterministic_across_batch_sizes_with_needle | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
| test_camem.py | test_end_to_end | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | |
| test_completion_with_prompt_embeds.py | test_mixed_prompt_embeds_and_text | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ |
| test_cpu_offloading.py | test_cpu_offloading | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| test_guided_decoding.py | test_guided_json_completion | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_guided_decoding.py | test_guided_regex | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_minicpm.py | test_minicpm | OpenBMB/MiniCPM4-0.5B<br>openbmb/MiniCPM-2B-sft-bf16 | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_multi_instance.py | test_two_instances_on_single_card | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_multistream_overlap_shared_expert.py | test_models_with_multistream_overlap_shared_expert | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_0_6b.py | test_dense_default_full_and_piecewise_graph | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_qwen3_5_0_8b.py | test_mamba_ssm_multimodal_reasoning_mtp_full_decode_only | Qwen/Qwen3.5-0.8B | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_8b_w8a8.py | test_dense_w8a8_eagle3_full_graph | RedHatAI/Qwen3-8B-speculator.eagle3<br>vllm-ascend/Qwen3-8B-W8A8 | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | |
| test_qwen3_embedding_0_6b.py | test_embedding_full_decode_only | Qwen/Qwen3-Embedding-0.6B | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
| test_sampler.py | test_qwen3_exponential_overlap | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_sampler.py | test_qwen3_prompt_logprobs | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | |
| test_sampler.py | test_qwen3_topk | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_vlm.py | test_multimodal_audio | Qwen/Qwen2-Audio-7B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_vlm.py | test_multimodal_vl | openai-mirror/whisper-large-v3-turbo | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_vlm.py | test_multimodal_vl_language_model_only | Qwen/Qwen3-VL-8B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_vlm.py | test_whisper | openai-mirror/whisper-large-v3-turbo | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_xlite.py | test_models_with_xlite_decode_only | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| test_xlite.py | test_models_with_xlite_full_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
## 2-Card Tests
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
| aclgraph/test_aclgraph_capture_replay.py | test_models_aclgraph_capture_replay_metrics_dp2 | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| aclgraph/test_full_graph_mode.py | test_qwen3_moe_full_decode_only_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| aclgraph/test_full_graph_mode.py | test_qwen3_moe_full_graph_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| lora/test_ilama_lora_tp2.py | test_ilama_lora_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| lora/test_llama32_lora_tp2.py | test_llama_lora_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| spec_decode/test_spec_decode.py | test_eagle3_sp_acceptance | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | ✅ | |
| spec_decode/test_spec_decode.py | test_p_eagle_acceptance | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
| spec_decode/test_spec_decode.py | test_qwen3_eagle3_pcp2_tp1 | - | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| test_data_parallel.py | test_qwen3_inference_dp2 | Qwen/Qwen3-30B-A3B<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_deepseek_multistream_moe.py | test_deepseek_multistream_moe_tp2 | vllm-ascend/DeepSeek-V3-Pruning | | | ✅ | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_external_launcher.py | test_qwen3_external_launcher | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_external_launcher.py | test_qwen3_external_launcher_with_matmul_allreduce | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
| test_external_launcher.py | test_qwen3_external_launcher_with_sleepmode | Qwen/Qwen3-8B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_external_launcher.py | test_qwen3_external_launcher_with_sleepmode_level2 | Qwen/Qwen3-8B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_external_launcher.py | test_qwen3_moe_external_launcher_ep_tp2 | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_flashcomm_distributed.py | test_deepseek_v2_lite_fc1_tp2 | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_flashcomm_distributed.py | test_qwen3_dense_fc1_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_flashcomm_distributed.py | test_qwen3_dense_prefetch_mlp_weight_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_flashcomm_distributed.py | test_qwen3_moe_fc2_oshard_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | |
| test_gpt_oss_distributed.py | test_gpt_oss_distributed_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_moe_routing_replay.py | test_qwen3_moe_routing_replay | Qwen/Qwen3-30B-A3B<br>Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_offline_weight_load.py | test_qwen3_offline_load_and_sleepmode_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_prefix_caching.py | test_models_prefix_cache_tp2 | Qwen/Qwen3-8B<br>deepseek-ai/DeepSeek-V2-Lite-Chat | | ✅ | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | |
| test_qwen3_30b_a3b.py | test_moe_tp_ep_eplb_full_decode_only | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_6_27b_fia.py | test_qwen3_6_27b_multimodel_fia_eager | Qwen/Qwen3.6-27B/ | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_6_27b_fia.py | test_qwen3_6_27b_multimodel_fia_acl_graph | Qwen/Qwen3.6-27B/ | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_vl_30b_a3b_instruct.py | test_multimodal_reasoning_pp_full_decode_only | Qwen/Qwen3-VL-30B-A3B-Instruct | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_sequence_parallelism_moe.py | test_sequence_parallelism_moe_patterns | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_shared_expert_dp.py | test_deepseek_v2_lite_enable_shared_expert_dp_tp2 | deepseek-ai/DeepSeek-V2-Lite | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_sp_pass.py | test_qwen3_vl_sp_tp2 | Qwen/Qwen3-VL-2B-Instruct | | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
## 4-Card Tests
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp2_fp16 | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp4_w8a8 | vllm-ascend/Qwen3-32B-W8A8 | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| _310p/test_moe_model_310p.py | test_qwen3_5_moe_tp4_fp16 | Qwen/Qwen3.5-35B-A3B | ✅ | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| _310p/test_moe_model_310p.py | test_qwen3_moe_tp2_w8a8 | vllm-ascend/Qwen3-30B-A3B-W8A8 | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| _310p/test_moe_model_310p.py | test_qwen3_moe_tp4_fp16 | Qwen/Qwen3-30B-A3B | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| _310p/test_vl_model_310p.py | test_qwen3_vl_8b_tp2_fp16 | Qwen/Qwen3-VL-8B-Instruct | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| context_parallel/test_accuracy.py | test_accuracy_dcp_only_eager | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_accuracy.py | test_accuracy_dcp_only_graph | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_accuracy.py | test_accuracy_pcp_only | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_accuracy.py | test_models_long_sequence_cp_kv_interleave_size_output_between_tp_and_cp | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_accuracy.py | test_models_long_sequence_output_between_tp_and_cp | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_dcp_basic | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_dcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_dcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_deepseek_v4_w4a8_dsa_cp_basic_greedy | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | ✅ | | ✅ | | ✅ | |
| context_parallel/test_basic.py | test_models_pcp_dcp_basic | Qwen/Qwen3-Next-80B-A3B-Instruct<br>deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/DeepSeek-V3.2-W8A8-Pruning<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_models_pcp_dcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_models_pcp_dcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_pcp_basic | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_pcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_pcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
| context_parallel/test_basic.py | test_qwen3_5_4b_multimodal_single_and_multi_image | Qwen/Qwen3.5-4B | | | | | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | |
| context_parallel/test_basic.py | test_qwen3_vl_8b_multimodal_single_and_multi_image | Qwen/Qwen3-VL-8B-Instruct | | | | | | | | ✅ | ✅ | | | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | |
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_mixed_length_prompts_including_1_token | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | ✅ |
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_basic | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_default_full_and_piecewise | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_full_graph | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_empty_kvcache | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | ✅ | | | ✅ | |
| context_parallel/test_mtp.py | test_dcp_mtp3_full_graph | - | | | | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_mtp.py | test_pcp_dcp_mtp1_eager | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_eager | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_full_graph | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_piecewise_graph | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_mtp.py | test_pcp_eagle3_eager | - | | | | | | | | | ✅ | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | |
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_basic | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_default_full_and_piecewise | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_full_graph | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| spec_decode/test_mtp_qwen3_next.py | test_qwen3_next_mtp_acceptance_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| test_data_parallel_tp2.py | test_qwen3_inference_dp2_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_deepseek_v3_2_w8a8_pruning.py | test_moe_w8a8_tp_pp_ep_full_decode_only | vllm-ascend/DeepSeek-V3.2-W8A8-Pruning | | | ✅ | | | | | | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_deepseek_v3_2_w8a8_pruning.py | test_pd_disaggregation_w8a8_sfa_dsa_full_decode_only | vllm-ascend/DeepSeek-V3.2-W8A8-Pruning | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_deepseek_v4.py | test_deepseek_v4_w4a8_tp4_basic_greedy | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_deepseek_v4.py | test_deepseek_v4_w4a8_tp4_index_cache_freq4 | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_pipeline_parallel.py | test_models_pp2_dp2 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_pipeline_parallel.py | test_models_pp2_tp2 | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| test_profiling_chunk_performance.py | test_profiling_chunk_ttft_performance | - | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | ✅ | | | | | | |
| test_qwen3_5.py | test_qwen3_5_27b_distributed_mp_tp4 | Qwen/Qwen3.5-27B | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4 | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3 | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3_flashcomm | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| test_qwen3_next.py | test_qwen3_next_distributed_mp_flash_comm_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_next.py | test_qwen3_next_distributed_mp_full_decode_only_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_next.py | test_qwen3_next_distributed_mp_graph_mode_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_next.py | test_qwen3_next_distributed_mp_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
| test_qwen3_next.py | test_qwen3_next_w8a8dynamic_distributed_tp4_ep | vllm-ascend/Qwen3-Next-80B-A3B-Instruct-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
## Nightly Tests
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
| 310p/single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule_v310.py | test_recurrent_gated_delta_rule_v310 | - | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| multi_node/external_dp/config/GLM5_1-W8A8-EP-external.yaml | multi-node-glm-5.1-w8a8-ep-external-dp | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | |
| multi_node/external_dp/scripts/test_external_dp.py | test_external_dp | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| multi_node/internal_dp/config/DeepSeek-R1-W8A8-EPLB.yaml | test DeepSeek-R1-W8A8 disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | ✅ | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/DeepSeek-R1-W8A8-longseq.yaml | test DeepSeek-R1-W8A8-longseq disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | ✅ | | | ✅ | ✅ | | | | | | |
| multi_node/internal_dp/config/DeepSeek-R1-W8A8.yaml | test DeepSeek-R1-W8A8 disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/DeepSeek-V3.1-BF16.yaml | test DeepSeek-V3.1-BF16 on A3 | unsloth/DeepSeek-V3.1-BF16 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | |
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-A3-dual-nodes.yaml | test DeepSeek-V3.2-W8A8 on A3 | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-EP.yaml | test DeepSeek-V3.2-W8A8-EP disaggregated_prefill | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/GLM5_1-W8A8-A2-dual-nodes.yaml | multi-node-GLM-5.1-w8a8-A2 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/GLM5_1-W8A8-A3-dual-nodes.yaml | multi-node-GLM-5.1-w8a8-A3 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/GLM5_1-W8A8-EP.yaml | multi-node-GLM-5.1-w8a8-EP | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/Kimi-K2_5-W4A8-A2-dual-nodes.yaml | test Kimi-K2.5-W4A8 A2 dual nodes | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-A22B-A2.yaml | test Qwen3-235B-A22B multi-dp on A2 | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-A22B.yaml | test Qwen3-235B-A22B multi-dp | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-W8A8-EPLB.yaml | test Qwen3-235B-A22B-W8A8 disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | ✅ | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-W8A8-longseq.yaml | test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-W8A8.yaml | test Qwen3-235B-A22B-W8A8 disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| multi_node/internal_dp/config/Qwen3-235B-disagg-pd.yaml | test Qwen3-235B-A22B disaggregated_prefill | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| multi_node/internal_dp/config/Qwen3-VL-235B-disagg-pd.yaml | test Qwen3-VL-235B-A22B disaggregated_prefill | Qwen/Qwen3-VL-235B-A22B-Instruct | | | | | | | | ✅ | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-EPLB | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-aclgraph | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-single | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/DeepSeek-V3.2-W8A8.yaml | DeepSeek-V3.2-W8A8-TP8-DP2 | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/DeepSeek-V4-Flash-W8A8-A3.yaml | DeepSeek-V4-Flash-W8A8-A3 | Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/GLM-4.7.yaml | GLM-4.7-TP8-DP2-decodegraph | Eco-Tech/GLM-4.7-W8A8-floatmtp | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/models/configs/Hy3-preview.yaml | Hy3-preview-TP16-EP-MTP | Tencent-Hunyuan/Hy3-preview | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Kimi-K2-Thinking.yaml | Kimi-K2-Thinking-TP16-Case | moonshotai/Kimi-K2-Thinking | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/models/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-Case | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/MTPX-DeepSeek-R1-0528-W8A8.yaml | MTPX-DeepSeek-R1-0528-W8A8-mtp2 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/MTPX-DeepSeek-R1-0528-W8A8.yaml | MTPX-DeepSeek-R1-0528-W8A8-mtp3 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/MiniMax-M2.5-w8a8-QuaRot-A2.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Prefix-Cache-DeepSeek-R1-0528-W8A8.yaml | prefix-cache-deepseek-r1-0528-w8a8 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/models/configs/Prefix-Cache-Qwen3-32B-Int8.yaml | prefix-cache-qwen3-32b-w8a8 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-EPLB | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-full_graph | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-piecewise | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3-30B-A3B-W4A8-llm-compressor.yaml | Qwen3-30B-A3B-W4A8-llm-compressor | vllm-ascend/Qwen3-30B-A3B-Instruct-2507-quantized.w4a8 | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3-30B-A3B-W8A8.yaml | Qwen3-30B-A3B-W8A8-TP1 | vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/models/configs/Qwen3-30B-QuaRot-eagle3.yaml | Qwen3-30B-QuaRot | vllm-ascend/Qwen3-30B-A3B-W8A8-QuaRot | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | |
| single_node/models/configs/Qwen3-32B-Int8-A2.yaml | Qwen3-32B-W8A8-aclgraph-a2 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3-32B-Int8-A2.yaml | Qwen3-32B-W8A8-single-a2 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3-32B-Int8.yaml | Qwen3-32B-W8A8-aclgraph-a3 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3-32B-Int8.yaml | Qwen3-32B-W8A8-single-a3 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3-32B-QuaRot-eagle3.yaml | Qwen3-32B-QuaRot | vllm-ascend/Qwen3-32B-W8A8-QuaRot | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | |
| single_node/models/configs/Qwen3-VL-235B-A22B-Instruct-W8A8.yaml | Qwen3-VL-235B-A22B-Instruct-W8A8 | Eco-Tech/Qwen3-VL-235B-A22B-Instruct-w8a8-QuaRot | | | | | | | | ✅ | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3-VL-32B-Instruct-W8A8.yaml | Qwen3-VL-32B-Instruct-W8A8 | Eco-Tech/Qwen3-VL-32B-Instruct-w8a8-QuaRot | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-A3 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/models/configs/Qwen3.5-27B-w8a8-A2.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3.yaml | Qwen3.5-397B-A17B-w8a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/models/configs/Qwen3.5-397B-A17B-w4a8-mtp-A2.yaml | Qwen3.5-397B-A17B-w4a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w4a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/ops/multicard_ops_a2/test_matmul_allreduce_add_rmsnorm.py | test_matmul_allreduce_add_rmsnorm_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine_bf16.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine_w4a8.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_base | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_dynamic_eplb | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_with_mc2_mask | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_add_rms_norm_bias.py | test_quant_fpx_linear | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_boundary_conditions | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_random_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_zero_values | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_bgmv_expand.py | test_bgmv_expand | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_bgmv_shrink.py | test_bgmv_shrink | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_causal_conv1d_310.py | test_ascend_causal_conv1d_310_fn | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| single_node/ops/singlecard_ops/test_causal_conv1d_310.py | test_causal_conv1d_310_update | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_copy_and_expand_eagle_inputs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_large_tokens_per_request | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_large_tokens_shift_true | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_minimal_case | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_no_rejected_tokens | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_dequant_swiglu_quant.py | test_npu_dequant_swiglu_quant_with_limit | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_bulk_dma_alignment | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_extreme_large_batch | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_large_batch_multi_row | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_non_bulk_dma_fallback | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_non_default_params | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_output_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_small_batch_optimization | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_vs_reference | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_fused_moe.py | test_select_experts | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_fused_moe.py | test_select_experts_invalid_scoring_func | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_fused_moe.py | test_token_dispatcher_with_all_gather | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_fused_moe.py | test_token_dispatcher_with_all_gather_quant | - | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_gating_top_k_softmax.py | test_quant_fpx_linear | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_gmm_swiglu_quant_weight_nz_tensor_list.py | test_gmm_swiglu_quant_weight_nz_tensor_list | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_grouped_matmul_swiglu_quant.py | test_grouped_matmul_swiglu_quant_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_hamming_dist_top_k.py | test_hamming_dist_top_k | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_hamming_dist_top_k.py | test_hamming_dist_top_k_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_mla_preprocess.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_mla_preprocess_nq.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_mla_preprocess_qdown.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/test_moe_init_routing_custom.py | test_moe_init_routing_custom | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_attrs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_basic | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_decode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_discard | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_exact_match | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_full_capacity | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_k1 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_minimal | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_no_valid_sampled | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_padding | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_prefill | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_npu_hc_pre.py | test_npu_hc_pre_v1_v2_bf16_3d_input | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_npu_hc_pre.py | test_npu_hc_pre_v1_v2_bf16_4d_input | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_npu_moe_gating_top_k.py | test_npu_moe_gating_topk_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule.py | test_recurrent_gated_delta_rule | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule.py | test_recurrent_gated_delta_rule_no_accepted | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule_310.py | test_fused_recurrent_gated_delta_rule_310 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_bf16_shape | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_with_expected_output | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_transpose_kv_cache_by_block.py | test_transpose_kv_cache_by_block | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/test_vocabparallelembedding.py | test_get_masked_input_and_mask | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_apply_penalties_triton.py | test_apply_all_penalties_v1_vs_ascend | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_different_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_edge_cases | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_no_bad_words | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_token_limit | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_batch_memcpy.py | test_batch_memcpy | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| single_node/ops/singlecard_ops/triton/test_bincount.py | test_bincount_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_ascend_causal_conv1d | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d_update_qwen3_next_shape | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d_update_with_batch_gather | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | |
| single_node/ops/singlecard_ops/triton/test_chunk_gated_delta_rule.py | test_chunk_gated_delta_rule_310_state_layout_matches_vllm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_chunk_gated_delta_rule.py | test_triton_fusion_ops | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_clear_ssm_states.py | test_clear_ssm_states_ref_parity | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_slot_mapping.py | test_compute_slot_mapping_npu_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_deterministic | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_dtypes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_edge_cases | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_compute_topk_logprobs.py | test_compute_topk_logprobs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_fused_gdn_gating.py | test_fused_gdn_gating_310p_parity_precision | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_fused_qkvzba_split_reshape_cat.py | test_fused_qkvzba_split_reshape_cat | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_fused_recurrent_gated_delta_rule.py | test_fused_recurrent_gated_delta_rule_310_state_layout_matches_vllm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_fused_recurrent_gated_delta_rule.py | test_fused_recurrent_gated_delta_rule_310p_parity_precision | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_fused_sigmoid_gating_delta_rule.py | test_triton_fusion_ops | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_l2norm.py | test_l2norm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_log_softmax.py | test_topk_log_softmax_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_min_p.py | test_apply_min_p_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_mrope.py | test_mrotary_embedding_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_muls_add.py | test_muls_add_triton_correctness | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_penality.py | test_apply_penalties | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_post_update.py | test_post_update | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
| single_node/ops/singlecard_ops/triton/test_prepare_inputs_padded.py | test_prepare_inputs_padded | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_rejection_sample.py | test_rejection_random_sample | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_rejection_sample.py | test_rejection_sampler_block_verify_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel_siso | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel_with_cos_sin_cache | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_mrope.py | test_split_qkv_rmsnorm_mrope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_rope.py | test_split_qkv_rmsnorm_rope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_rope.py | test_split_qkv_rmsnorm_rope_with_bias | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_split_qkv_tp_rmsnorm_rope.py | test_split_qkv_tp_rmsnorm_rope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/ops/singlecard_ops/triton/test_temperature.py | test_temperature_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
## Weekly Tests
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
| multi_node/internal_dp/config/DeepSeek-V3.yaml | test DeepSeek-V3 disaggregated_prefill | vllm-ascend/DeepSeek-V3-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-EP_weekly.yaml | weekly test DeepSeek-V3.2-W8A8-EP disaggregated_prefill | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
| multi_node/internal_dp/config/GLM-4.7-W8A8C8-Mooncake-Layerwise.yaml | test GLM-4.7-W8A8C8 PD separation with mooncake layerwise connector | vllm-ascend/GLM-4.7-W8A8C8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
| single_node/configs/DeepSeek-V3.2-W8A8_A3_weekly.yaml | DeepSeek-V3.2-W8A8-weekly | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/GLM-5.yaml | GLM-5-TP16-DP1-decodegraph | Eco-Tech/GLM-5-w4a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | ✅ | | | | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs10 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs20 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs32 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs8 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT20-32k-0.5k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT20-32k-0.5k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT50-32k-0.5k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT50-32k-0.5k-prefix-cache90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-Case | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-128k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-16k-1k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-64k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-128k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-16k-1k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-64k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-W8A8-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-32-8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-64-16 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in16k-120-30 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in16k-16-4 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-36-9 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-4-1-90 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-80-20 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in64k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in64k-72-18 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/Qwen2.5-VL-7B-Instruct-EPD.yaml | Qwen2.5-VL-7B-Instruct-epd | Qwen/Qwen2.5-VL-7B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/configs/Qwen3-32B.yaml | Qwen3-32B-TP4 | Qwen/Qwen3-32B | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A2.yaml | Qwen3.5-122B-A10B-W8A8-single-A2 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-A3 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-16k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-32k-0.5k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-64k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-16k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-32k-0.5k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-64k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in128k-28-7 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in128k-4-1 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in16k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in16k-56-14 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-56-14 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-8-2 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-48-12 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-8-2 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-8-2-90 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3.yaml | Qwen3.5-397B-A17B-w8a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs136 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs144 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs32_in65536 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs48 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs8 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs80 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs160 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16_in65536 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs2 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs32 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs36 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |

View File

@@ -20,7 +20,6 @@ trap clean_venv EXIT
function install_system_packages() {
if command -v apt-get >/dev/null; then
sed -i 's|ports.ubuntu.com|mirrors.tuna.tsinghua.edu.cn|g' /etc/apt/sources.list
apt-get update -y && apt-get install -y gcc g++ cmake libnuma-dev wget git curl jq
elif command -v yum >/dev/null; then
yum update -y && yum install -y gcc g++ cmake numactl-devel wget git curl jq
@@ -30,31 +29,56 @@ function install_system_packages() {
}
function config_pip_mirror() {
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
pip config set global.index-url http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple
pip config set global.trusted-host cache-service.nginx-pypi-cache.svc.cluster.local
if [ -f /etc/os-release ]; then
. /etc/os-release
case "$ID" in
ubuntu|debian)
sed -Ei 's@(ports|archive).ubuntu.com@cache-service.nginx-pypi-cache.svc.cluster.local:8081@g' /etc/apt/sources.list
;;
openEuler|centos|rhel|fedora)
sed -Ei 's@https?://[^/]+/(openeuler|centos|fedora)@http://cache-service.nginx-pypi-cache.svc.cluster.local:8081/\1@g' /etc/yum.repos.d/*.repo
;;
esac
fi
}
function install_binary_test() {
install_system_packages
config_pip_mirror
install_system_packages
create_vllm_venv
pip install -r ${SCRIPT_DIR}/../../docs/requirements-docs.txt
PIP_VLLM_VERSION=$(get_version pip_vllm_version)
VLLM_VERSION=$(get_version vllm_version)
PIP_VLLM_ASCEND_VERSION=$(get_version pip_vllm_ascend_version)
_info "====> Install vllm==${PIP_VLLM_VERSION} and vllm-ascend ${PIP_VLLM_ASCEND_VERSION}"
# Setup extra-index-url for x86 & torch_npu dev version
pip config set global.extra-index-url "https://download.pytorch.org/whl/cpu/ https://mirrors.huaweicloud.com/ascend/repos/pypi"
# Setup extra-index-url for public PyPI mirror, Ascend packages, and PyTorch CPU wheels.
local pip_extra_index_urls=(
"https://mirrors.huaweicloud.com/repository/pypi/variant"
"https://mirrors.huaweicloud.com/ascend/repos/pypi"
"https://download.pytorch.org/whl/cpu/"
)
local IFS=" "
pip config set global.extra-index-url "${pip_extra_index_urls[*]}"
pip install vllm=="$(get_version pip_vllm_version)"
pip install vllm-ascend=="$(get_version pip_vllm_ascend_version)"
# The vLLM version already in pypi, we install from pypi.
pip install --default-timeout=300 --retries 3 vllm=="${PIP_VLLM_VERSION}"
pip install vllm-ascend=="${PIP_VLLM_ASCEND_VERSION}"
pip list | grep vllm
# Verify the installation
_info "====> Run offline example test"
pip install modelscope
python3 "${SCRIPT_DIR}/../../examples/offline_inference_npu.py"
cd ${SCRIPT_DIR}/../../examples && python3 ./offline_inference_npu.py
cd -
}

File diff suppressed because it is too large Load Diff

View File

@@ -17,16 +17,11 @@
# Adapted from vllm-project/vllm/blob/main/tests/models/utils.py
#
from typing import Dict, List, Optional, Sequence, Tuple, Union
from collections.abc import Sequence
from vllm_ascend.utils import vllm_version_is
from vllm.logprobs import PromptLogprobs, SampleLogprobs
if vllm_version_is("0.10.2"):
from vllm.sequence import PromptLogprobs, SampleLogprobs
else:
from vllm.logprobs import PromptLogprobs, SampleLogprobs
TokensText = Tuple[List[int], str]
TokensText = tuple[list[int], str]
def check_outputs_equal(
@@ -42,16 +37,18 @@ def check_outputs_equal(
"""
assert len(outputs_0_lst) == len(outputs_1_lst)
for prompt_idx, (outputs_0,
outputs_1) in enumerate(zip(outputs_0_lst,
outputs_1_lst)):
for prompt_idx, (outputs_0, outputs_1) in enumerate(zip(outputs_0_lst, outputs_1_lst)):
output_ids_0, output_str_0 = outputs_0
output_ids_1, output_str_1 = outputs_1
# The text and token outputs should exactly match
fail_msg = (f"Test{prompt_idx}:"
f"\n{name_0}:\t{output_str_0!r}"
f"\n{name_1}:\t{output_str_1!r}")
fail_msg = (
f"Test{prompt_idx}:"
f"\n{name_0}:\t{output_str_0!r}"
f"\n{name_1}:\t{output_str_1!r}"
f"\n{name_0}:\t{output_ids_0!r}"
f"\n{name_1}:\t{output_ids_1!r}"
)
assert output_str_0 == output_str_1, fail_msg
assert output_ids_0 == output_ids_1, fail_msg
@@ -63,9 +60,7 @@ def check_outputs_equal(
# * List of top sample logprobs for each sampled token
#
# Assumes prompt logprobs were not requested.
TokensTextLogprobs = Tuple[List[int], str, Optional[Union[List[Dict[int,
float]],
SampleLogprobs]]]
TokensTextLogprobs = tuple[list[int], str, list[dict[int, float]] | SampleLogprobs | None]
# Representation of generated sequence as a tuple of
# * Token ID list
@@ -74,6 +69,9 @@ TokensTextLogprobs = Tuple[List[int], str, Optional[Union[List[Dict[int,
# * Optional list of top prompt logprobs for each prompt token
#
# Allows prompt logprobs to be requested.
TokensTextLogprobsPromptLogprobs = Tuple[
List[int], str, Optional[Union[List[Dict[int, float]], SampleLogprobs]],
Optional[Union[List[Optional[Dict[int, float]]], PromptLogprobs]]]
TokensTextLogprobsPromptLogprobs = tuple[
list[int],
str,
list[dict[int, float]] | SampleLogprobs | None,
list[dict[int, float] | None] | PromptLogprobs | None,
]

View File

@@ -0,0 +1,21 @@
model_name: "PaddlePaddle/ERNIE-4.5-21B-A3B-PT"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,flexible-extract"
value: 0.71
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "Tencent-Hunyuan/Hunyuan-A13B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 32768
gpu_memory_utilization: 0.90
enforce_eager: true
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.37
- name: "exact_match,flexible-extract"
value: 0.28
num_fewshot: 5
limit: 1000
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "OpenGVLab/InternVL3_5-8B-hf"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 40960
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.58
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,23 @@
model_name: "LLM-Research/Llama-3.2-3B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.71
- name: "exact_match,flexible-extract"
value: 0.76
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,24 @@
model_name: "nv-community/Minitron-8B-Base"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.9
enforce_eager: true
trust_remote_code: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.5436
- name: "exact_match,flexible-extract"
value: 0.5451
limit: 1000
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,32 @@
model_name: "mistralai/Mixtral-8x7B-Instruct-v0.1"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: bfloat16
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: true
enforce_eager: true
block_size: 128
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "200"
VLLM_ASCEND_ENABLE_MLAPO: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.45
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -0,0 +1,21 @@
model_name: "LLM-Research/Molmo-7B-D-0924"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.71
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,23 @@
model_name: "Qwen/Qwen2-Audio-7B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.44
- name: "exact_match,flexible-extract"
value: 0.45
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "Qwen/Qwen2.5-Math-RM-72B"
model_type: "vllm-rm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.9
trust_remote_code: false
# system_prompt controls the <|im_start|>system block passed to the reward model.
system_prompt: "Please reason step by step, and put your final answer within \\boxed{}."
tasks:
- name: "gsm8k_correctness"
dataset: "AI-ModelScope/gsm8k"
split: "test"
dataset_config: "main"
metrics:
- name: "accuracy"
value: 0.80
limit: 200
batch_size: 4

View File

@@ -0,0 +1,25 @@
model_name: "vllm-ascend/Qwen3-30B-A3B-W8A8"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
quantization: ascend
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.9
- name: "exact_match,flexible-extract"
value: 0.8
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -1,6 +1,15 @@
model_name: "Qwen/Qwen3-30B-A3B"
runner: "linux-aarch64-a2-2"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.6
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "gsm8k"
metrics:
@@ -12,9 +21,8 @@ tasks:
metrics:
- name: "acc,none"
value: 0.84
num_fewshot: 5
gpu_memory_utilization: 0.6
enable_expert_parallel: True
tensor_parallel_size: 2
apply_chat_template: False
fewshot_as_multiturn: False
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,25 @@
model_name: "vllm-ascend/Qwen3-8B-W8A8"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
quantization: ascend
enable_thinking: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.80
- name: "exact_match,flexible-extract"
value: 0.82
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,24 @@
model_name: "Qwen/Qwen3-8B"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
enable_thinking: false
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.765
- name: "exact_match,flexible-extract"
value: 0.81
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "Qwen/Qwen3-ASR-1.7B"
model_type: "vllm-asr"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: false
tasks:
- name: "librispeech_test_clean"
dataset: "openslr/librispeech_asr"
split: "test"
dataset_config: "clean"
metrics:
- name: "wer"
value: 0.035
limit: 500

View File

@@ -0,0 +1,23 @@
model_name: "Qwen/Qwen3-Next-80B-A3B-Instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
enforce_eager: true
tasks:
- name: "ceval-valid_accountant"
metrics:
- name: "acc,none"
value: 0.98
num_fewshot: 5
apply_chat_template: true
fewshot_as_multiturn: true
batch_size: 1

View File

@@ -0,0 +1,22 @@
model_name: "Qwen/Qwen3-Omni-30B-A3B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 4
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.60
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,22 @@
model_name: "Qwen/Qwen3-VL-30B-A3B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 2
dtype: auto
max_model_len: 128000
gpu_memory_utilization: 0.7
trust_remote_code: false
enable_expert_parallel: true
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.58
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,22 @@
model_name: "vllm-ascend/Qwen3-VL-8B-Instruct-W8A8"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.8
trust_remote_code: false
quantization: ascend
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.52
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -0,0 +1,21 @@
model_name: "Qwen/Qwen3-VL-8B-Instruct"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 8192
gpu_memory_utilization: 0.7
trust_remote_code: false
tasks:
- name: "mmmu_val"
metrics:
- name: "acc,none"
value: 0.55
num_fewshot: 0
apply_chat_template: true
fewshot_as_multiturn: false
batch_size: 32

View File

@@ -1,4 +1,14 @@
DeepSeek-V2-Lite.yaml
Qwen3-8B-Base.yaml
Qwen2.5-VL-7B-Instruct.yaml
Qwen3-30B-A3B.yaml
Qwen3-30B-A3B.yaml
Qwen3-8B.yaml
Qwen2-Audio-7B-Instruct.yaml
Qwen3-VL-30B-A3B-Instruct.yaml
Qwen3-VL-8B-Instruct.yaml
Qwen3-Omni-30B-A3B-Instruct.yaml
InternVL3_5-8B-hf.yaml
ERNIE-4.5-21B-A3B-PT.yaml
gemma-3-4b-it.yaml
internlm3-8b-instruct.yaml
Molmo-7B-D-0924.yaml
llava-onevision-qwen2-0.5b-ov-hf.yaml
Llama-3.2-3B-Instruct.yaml
Qwen3-ASR-1.7B.yaml

View File

@@ -0,0 +1,24 @@
model_name: "LLM-Research/gemma-3-4b-it"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.7
trust_remote_code: false
enforce_eager: true
tasks:
- name: "gsm8k"
metrics:
- name: "exact_match,strict-match"
value: 0.59
- name: "exact_match,flexible-extract"
value: 0.59
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "Shanghai_AI_Laboratory/internlm3-8b-instruct"
model_type: "vllm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: "bfloat16"
max_model_len: 2048
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.42
num_fewshot: 5
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -0,0 +1,21 @@
model_name: "llava-hf/llava-onevision-qwen2-0.5b-ov-hf"
model_type: "vllm-vlm"
hardware: "Atlas A2 Series"
serve:
tensor_parallel_size: 1
dtype: auto
max_model_len: 4096
gpu_memory_utilization: 0.8
trust_remote_code: true
tasks:
- name: "ceval-valid"
metrics:
- name: "acc,none"
value: 0.42
num_fewshot: 0
apply_chat_template: false
fewshot_as_multiturn: false
batch_size: "auto"

View File

@@ -21,7 +21,7 @@ def pytest_addoption(parser):
parser.addoption(
"--config",
action="store",
default="./tests/e2e/models/configs/Qwen3-8B-Base.yaml",
default="./tests/e2e/models/configs/Qwen3-8B.yaml",
help="Path to the model config YAML file",
)
parser.addoption(
@@ -55,16 +55,12 @@ def report_dir(pytestconfig):
def pytest_generate_tests(metafunc):
if "config_filename" in metafunc.fixturenames:
if metafunc.config.getoption("--config-list-file"):
rel_path = metafunc.config.getoption("--config-list-file")
config_list_file = Path(rel_path).resolve()
config_dir = config_list_file.parent
with open(config_list_file, encoding="utf-8") as f:
configs = [
config_dir / line.strip() for line in f
if line.strip() and not line.startswith("#")
]
configs = [config_dir / line.strip() for line in f if line.strip() and not line.startswith("#")]
metafunc.parametrize("config_filename", configs)
else:
single_config = metafunc.config.getoption("--config")

View File

@@ -1,30 +1,33 @@
# {{ model_name }}
- **vLLM Version**: vLLM: {{ vllm_version }} ([{{ vllm_commit[:7] }}](https://github.com/vllm-project/vllm/commit/{{ vllm_commit }})), **vLLM Ascend Version**: {{ vllm_ascend_version }} ([{{ vllm_ascend_commit[:7] }}](https://github.com/vllm-project/vllm-ascend/commit/{{ vllm_ascend_commit }}))
- **Software Environment**: **CANN**: {{ cann_version }}, **PyTorch**: {{ torch_version }}, **torch-npu**: {{ torch_npu_version }}
- **Software Environment**: **CANN**: {{ cann_version }}, **PyTorch**: {{ torch_version }}, **TorchNPU**: {{ torch_npu_version }}
- **Hardware Environment**: {{ hardware }}
- **Parallel mode**: {{ parallel_mode }}
- **Execution mode**: {{ execution_model }}
{% if show_command is not defined or show_command %}
**Command**:
```bash
export MODEL_ARGS={{ model_args }}
lm_eval --model {{ model_type }} --model_args $MODEL_ARGS --tasks {{ datasets }} \
{% if apply_chat_template is defined and (apply_chat_template|string|lower in ["true", "1"]) -%}
lm_eval --model {{ model_type }} --model_args $MODEL_ARGS \
--tasks {{ datasets }} \
{%- if apply_chat_template is defined and (apply_chat_template|string|lower in ["true", "1"]) %}
--apply_chat_template \
{%- endif %}
{% if fewshot_as_multiturn is defined and (fewshot_as_multiturn|string|lower in ["true", "1"]) -%}
{%- if fewshot_as_multiturn is defined and (fewshot_as_multiturn|string|lower in ["true", "1"]) %}
--fewshot_as_multiturn \
{%- endif %}
{% if num_fewshot is defined and num_fewshot != "N/A" -%}
{%- if num_fewshot is defined and num_fewshot != "N/A" %}
--num_fewshot {{ num_fewshot }} \
{%- endif %}
{% if limit is defined and limit != "N/A" -%}
{%- if limit is defined and limit != "N/A" %}
--limit {{ limit }} \
{%- endif %}
--batch_size {{ batch_size }}
--batch_size {{ batch_size }}
```
{% endif %}
| Task | Metric | Value | Stderr |
|-----------------------|-------------|----------:|-------:|

View File

@@ -0,0 +1,290 @@
import io
import os
import string
from dataclasses import dataclass
import jiwer # type: ignore[import-untyped]
import numpy as np
import pytest
import scipy.io.wavfile as wav_io # type: ignore[import-untyped]
import soundfile as sf # type: ignore[import-untyped]
import yaml
from datasets import Audio
from jinja2 import Environment, FileSystemLoader
from modelscope.msdatasets import MsDataset # type: ignore[import-untyped]
from vllm.utils.network_utils import get_open_port
from tests.e2e.conftest import RemoteOpenAIServer
# Allow up to 10% relative deviation from the declared ground-truth WER.
# ASR results have higher variance than classification tasks, so we use a
# more generous tolerance than the 5% used in test_lm_eval_correctness.py.
RTOL = 0.03
TEST_DIR = os.path.dirname(__file__)
_PUNCT_TABLE = str.maketrans("", "", string.punctuation)
@dataclass
class EnvConfig:
vllm_version: str
vllm_commit: str
vllm_ascend_version: str
vllm_ascend_commit: str
cann_version: str
torch_version: str
torch_npu_version: str
@pytest.fixture
def env_config() -> EnvConfig:
return EnvConfig(
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
cann_version=os.getenv("CANN_VERSION", "unknown"),
torch_version=os.getenv("TORCH_VERSION", "unknown"),
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
)
def build_serve_args(eval_config: dict) -> list[str]:
"""Convert the serve: section of the YAML into a vllm serve CLI args list.
Example — serve: {tensor_parallel_size: 2, dtype: auto} becomes:
["--tensor-parallel-size", "2", "--dtype", "auto"]
"""
serve_cfg = eval_config.get("serve", {})
flag_map = {
"tensor_parallel_size": "--tensor-parallel-size",
"dtype": "--dtype",
"max_model_len": "--max-model-len",
"gpu_memory_utilization": "--gpu-memory-utilization",
"trust_remote_code": "--trust-remote-code",
"enforce_eager": "--enforce-eager",
"quantization": "--quantization",
}
args: list[str] = []
for key, flag in flag_map.items():
value = serve_cfg.get(key)
if value is None:
continue
if isinstance(value, bool):
if value:
args.append(flag)
else:
args.extend([flag, str(value)])
return args
def audio_to_wav_bytes(audio_array: np.ndarray, sample_rate: int) -> bytes:
"""Convert a numpy audio array to in-memory WAV bytes at the given sample rate."""
buf = io.BytesIO()
# Ensure int16 encoding for maximum API compatibility.
if audio_array.dtype != np.int16:
if np.issubdtype(audio_array.dtype, np.floating):
audio_array = np.clip(audio_array, -1.0, 1.0)
audio_array = (audio_array * 32767).astype(np.int16)
else:
audio_array = audio_array.astype(np.int16)
wav_io.write(buf, sample_rate, audio_array)
return buf.getvalue()
def normalize_text(text: str) -> str:
"""Normalize text for WER calculation: lowercase, strip punctuation, collapse whitespace."""
text = text.lower()
text = text.translate(_PUNCT_TABLE)
text = " ".join(text.split())
return text
def transcribe_batch(client, model_name: str, audio_items: list[dict], language: str) -> list[str]:
"""Call /v1/audio/transcriptions for a list of audio items.
Each item in audio_items must have keys: audio_array (np.ndarray), sample_rate (int).
Returns the raw transcription strings in the same order.
"""
hypotheses: list[str] = []
for item in audio_items:
wav_bytes = audio_to_wav_bytes(item["audio_array"], item["sample_rate"])
response = client.audio.transcriptions.create(
model=model_name,
file=("audio.wav", wav_bytes, "audio/wav"),
language=language,
)
hypotheses.append(response.text)
return hypotheses
def generate_asr_report(
eval_config: dict,
report_data: dict,
report_dir: str,
env_config: EnvConfig,
) -> None:
"""Write a Markdown accuracy report using the same Jinja2 template as lm_eval tests."""
env = Environment(loader=FileSystemLoader(TEST_DIR))
template = env.get_template("report_template.md")
serve_cfg = eval_config.get("serve", {})
tp_size = serve_cfg.get("tensor_parallel_size", 1)
ep_enabled = serve_cfg.get("enable_expert_parallel", False)
enforce_eager = serve_cfg.get("enforce_eager", False)
parallel_mode = f"TP{tp_size}"
if ep_enabled:
parallel_mode += " + EP"
execution_model = "Eager" if enforce_eager else "ACLGraph"
model_args_str = ",".join(f"{k}={v}" for k, v in serve_cfg.items())
report_content = template.render(
vllm_version=env_config.vllm_version,
vllm_commit=env_config.vllm_commit,
vllm_ascend_version=env_config.vllm_ascend_version,
vllm_ascend_commit=env_config.vllm_ascend_commit,
cann_version=env_config.cann_version,
torch_version=env_config.torch_version,
torch_npu_version=env_config.torch_npu_version,
hardware=eval_config.get("hardware", "unknown"),
model_name=eval_config["model_name"],
model_args=f"'{model_args_str}'",
model_type=eval_config.get("model_type", "vllm-asr"),
datasets=",".join(t["name"] for t in eval_config["tasks"]),
apply_chat_template=False,
fewshot_as_multiturn=False,
limit=eval_config.get("limit", "N/A"),
batch_size=eval_config.get("batch_size", 8),
num_fewshot="N/A",
rows=report_data["rows"],
parallel_mode=parallel_mode,
execution_model=execution_model,
show_command=False,
)
report_path = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
os.makedirs(os.path.dirname(report_path), exist_ok=True)
with open(report_path, "w", encoding="utf-8") as f:
f.write(report_content)
def test_asr_eval_param(config_filename, tp_size, report_dir, env_config):
"""Parametrised ASR accuracy test driven by a YAML config file.
Skips automatically when the config's model_type is not "vllm-asr".
"""
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
if eval_config.get("model_type", "vllm") != "vllm-asr":
pytest.skip(f"Skipping non-ASR config (model_type={eval_config.get('model_type', 'vllm')})")
model_name: str = eval_config["model_name"]
language: str = eval_config.get("language", "en")
limit: int | None = eval_config.get("limit", None)
batch_size: int = eval_config.get("batch_size", 8)
# Build serve args, letting --tp-size CLI flag override the YAML value.
serve_args = build_serve_args(eval_config)
if tp_size and tp_size != "1":
# Drop any --tensor-parallel-size already in serve_args, then append
# the CLI-supplied value so it takes precedence over the YAML setting.
it = iter(serve_args)
serve_args = [a for a in it if a != "--tensor-parallel-size" or not next(it, None)]
serve_args += ["--tensor-parallel-size", str(tp_size)]
print(f"\nStarting vllm serve for {model_name}")
print(f" serve args: {serve_args}")
success = True
report_data: dict[str, list[dict]] = {"rows": []}
server_port = get_open_port()
serve_args = serve_args + ["--port", str(server_port)]
with RemoteOpenAIServer(model_name, serve_args, server_port=server_port, auto_port=False) as server:
client = server.get_client()
for task in eval_config["tasks"]:
task_name: str = task["name"]
dataset_name: str = task["dataset"]
split: str = task["split"]
dataset_config_name: str | None = task.get("dataset_config")
audio_col: str = task.get("audio_column", "audio")
text_col: str = task.get("text_column", "text")
split_expr = f"{split}[:{limit}]" if limit is not None else split
print(f"\nLoading dataset via modelscope: {dataset_name} / {dataset_config_name} ({split_expr})")
ds = MsDataset.load(
dataset_name,
subset_name=dataset_config_name,
split=split_expr,
)
if limit is not None:
ds = ds.select(range(min(limit, len(ds))))
# Disable automatic audio decoding so we can use soundfile instead
# of torchcodec (which requires CUDA libs unavailable on Ascend NPU).
if hasattr(ds, "cast_column"):
ds = ds.cast_column(audio_col, Audio(decode=False))
print(f" {len(ds)} samples to evaluate")
# Collect audio items and references in batches.
all_hypotheses: list[str] = []
all_references: list[str] = []
for batch_start in range(0, len(ds), batch_size):
batch = ds.select(range(batch_start, min(batch_start + batch_size, len(ds))))
audio_items = []
for sample in batch:
raw = sample[audio_col]
if isinstance(raw, dict) and "bytes" in raw and raw["bytes"] is not None:
audio_array, sample_rate = sf.read(io.BytesIO(raw["bytes"]))
elif isinstance(raw, dict) and "path" in raw and raw["path"] is not None:
audio_array, sample_rate = sf.read(raw["path"])
else:
# Already decoded (e.g. MsDataset with native decoding)
audio_array = raw["array"]
sample_rate = raw["sampling_rate"]
audio_items.append({"audio_array": audio_array, "sample_rate": sample_rate})
references = [sample[text_col] for sample in batch]
hypotheses = transcribe_batch(client, model_name, audio_items, language)
all_hypotheses.extend(hypotheses)
all_references.extend(references)
if (batch_start // batch_size + 1) % 5 == 0:
print(f" processed {batch_start + len(batch)}/{len(ds)} samples …")
# Normalise both sides before WER calculation.
norm_hypotheses = [normalize_text(h) for h in all_hypotheses]
norm_references = [normalize_text(r) for r in all_references]
measured_wer = round(jiwer.wer(norm_references, norm_hypotheses), 4)
print(f"\n{task_name} WER = {measured_wer:.4f}")
for metric in task["metrics"]:
if metric["name"] != "wer":
continue
ground_truth = metric["value"]
# Pass if measured WER is at or below the threshold (better is OK);
# allow up to RTOL relative degradation above the threshold.
task_success = measured_wer <= ground_truth * (1 + RTOL)
success = success and task_success
status = "" if task_success else ""
print(f"{task_name} | wer: ground_truth={ground_truth} | measured={measured_wer} | {status}")
report_data["rows"].append(
{
"task": task_name,
"metric": "wer",
"value": f"{status}{measured_wer}",
"stderr": "N/A",
}
)
generate_asr_report(eval_config, report_data, report_dir, env_config)
assert success, "One or more ASR tasks exceeded the WER tolerance. See output above."

View File

@@ -7,7 +7,7 @@ import pytest
import yaml
from jinja2 import Environment, FileSystemLoader
RTOL = 0.03
RTOL = 0.05
TEST_DIR = os.path.dirname(__file__)
@@ -24,33 +24,39 @@ class EnvConfig:
@pytest.fixture
def env_config() -> EnvConfig:
return EnvConfig(vllm_version=os.getenv('VLLM_VERSION', 'unknown'),
vllm_commit=os.getenv('VLLM_COMMIT', 'unknown'),
vllm_ascend_version=os.getenv('VLLM_ASCEND_VERSION',
'unknown'),
vllm_ascend_commit=os.getenv('VLLM_ASCEND_COMMIT',
'unknown'),
cann_version=os.getenv('CANN_VERSION', 'unknown'),
torch_version=os.getenv('TORCH_VERSION', 'unknown'),
torch_npu_version=os.getenv('TORCH_NPU_VERSION',
'unknown'))
return EnvConfig(
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
cann_version=os.getenv("CANN_VERSION", "unknown"),
torch_version=os.getenv("TORCH_VERSION", "unknown"),
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
)
def build_model_args(eval_config, tp_size):
trust_remote_code = eval_config.get("trust_remote_code", False)
max_model_len = eval_config.get("max_model_len", 4096)
serve_cfg = eval_config.get("serve", {})
trust_remote_code = serve_cfg.get("trust_remote_code", False)
max_model_len = serve_cfg.get("max_model_len", 4096)
dtype = serve_cfg.get("dtype", "auto")
model_args = {
"pretrained": eval_config["model_name"],
"tensor_parallel_size": tp_size,
"dtype": "auto",
"dtype": dtype,
"trust_remote_code": trust_remote_code,
"max_model_len": max_model_len,
}
for s in [
"max_images", "gpu_memory_utilization", "enable_expert_parallel",
"tensor_parallel_size", "enforce_eager"
"max_images",
"gpu_memory_utilization",
"enable_expert_parallel",
"tensor_parallel_size",
"enforce_eager",
"enable_thinking",
"quantization",
]:
val = eval_config.get(s, None)
val = serve_cfg.get(s, None)
if val is not None:
model_args[s] = val
@@ -66,7 +72,7 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
model_args = build_model_args(eval_config, tp_size)
parallel_mode = f"TP{model_args.get('tensor_parallel_size', 1)}"
if model_args.get('enable_expert_parallel', False):
if model_args.get("enable_expert_parallel", False):
parallel_mode += " + EP"
execution_model = f"{'Eager' if model_args.get('enforce_eager', False) else 'ACLGraph'}"
@@ -82,7 +88,7 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
hardware=eval_config.get("hardware", "unknown"),
model_name=eval_config["model_name"],
model_args=f"'{','.join(f'{k}={v}' for k, v in model_args.items())}'",
model_type=eval_config.get("model", "vllm"),
model_type=eval_config.get("model_type", "vllm"),
datasets=",".join([task["name"] for task in eval_config["tasks"]]),
apply_chat_template=eval_config.get("apply_chat_template", True),
fewshot_as_multiturn=eval_config.get("fewshot_as_multiturn", True),
@@ -91,24 +97,27 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
num_fewshot=eval_config.get("num_fewshot", "N/A"),
rows=report_data["rows"],
parallel_mode=parallel_mode,
execution_model=execution_model)
execution_model=execution_model,
)
report_output = os.path.join(
report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
report_output = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
os.makedirs(os.path.dirname(report_output), exist_ok=True)
with open(report_output, 'w', encoding='utf-8') as f:
with open(report_output, "w", encoding="utf-8") as f:
f.write(report_content)
def test_lm_eval_correctness_param(config_filename, tp_size, report_dir,
env_config):
def test_lm_eval_correctness_param(config_filename, tp_size, report_dir, env_config):
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
if eval_config.get("model_type", "vllm") == "vllm-asr":
pytest.skip("Skipping ASR config, use test_asr_eval.py instead")
model_args = build_model_args(eval_config, tp_size)
success = True
report_data: dict[str, list[dict]] = {"rows": []}
eval_params = {
"model": eval_config.get("model", "vllm"),
"model": eval_config.get("model_type", "vllm"),
"model_args": model_args,
"tasks": [task["name"] for task in eval_config["tasks"]],
"apply_chat_template": eval_config.get("apply_chat_template", True),
@@ -133,25 +142,26 @@ def test_lm_eval_correctness_param(config_filename, tp_size, report_dir,
metric_name = metric["name"]
ground_truth = metric["value"]
measured_value = round(task_result[metric_name], 4)
task_success = bool(
np.isclose(ground_truth, measured_value, rtol=RTOL))
task_success = bool(np.isclose(ground_truth, measured_value, rtol=RTOL))
success = success and task_success
print(f"{task_name} | {metric_name}: "
f"ground_truth={ground_truth} | measured={measured_value} | "
f"success={'' if task_success else ''}")
print(
f"{task_name} | {metric_name}: "
f"ground_truth={ground_truth} | measured={measured_value} | "
f"success={'' if task_success else ''}"
)
report_data["rows"].append({
"task":
task_name,
"metric":
metric_name,
"value":
f"{measured_value}" if success else f"{measured_value}",
"stderr":
task_result[
metric_name.replace(',', '_stderr,') if metric_name ==
"acc,none" else metric_name.replace(',', '_stderr,')]
})
report_data["rows"].append(
{
"task": task_name,
"metric": metric_name,
"value": f"{measured_value}" if success else f"{measured_value}",
"stderr": task_result[
metric_name.replace(",", "_stderr,")
if metric_name == "acc,none"
else metric_name.replace(",", "_stderr,")
],
}
)
generate_report(tp_size, eval_config, report_data, report_dir, env_config)
assert success

View File

@@ -0,0 +1,255 @@
import os
from dataclasses import dataclass
import pytest
import regex as re
import yaml
from jinja2 import Environment, FileSystemLoader
from modelscope.msdatasets import MsDataset # type: ignore[import-untyped]
from tests.e2e.conftest import VllmRunner
# Allow up to 5 % relative degradation from the declared ground-truth accuracy.
RTOL = 0.05
TEST_DIR = os.path.dirname(__file__)
# Default system prompt for Qwen2.5-Math-RM style models.
_DEFAULT_SYSTEM_PROMPT = "Please reason step by step, and put your final answer within \\boxed{}."
@dataclass
class EnvConfig:
vllm_version: str
vllm_commit: str
vllm_ascend_version: str
vllm_ascend_commit: str
cann_version: str
torch_version: str
torch_npu_version: str
@pytest.fixture
def env_config() -> EnvConfig:
return EnvConfig(
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
cann_version=os.getenv("CANN_VERSION", "unknown"),
torch_version=os.getenv("TORCH_VERSION", "unknown"),
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
)
def format_rm_input(system_prompt: str, problem: str, solution: str) -> str:
"""Format a (problem, solution) pair using the Qwen chat template."""
return (
f"<|im_start|>system\n{system_prompt}<|im_end|>\n"
f"<|im_start|>user\n{problem}<|im_end|>\n"
f"<|im_start|>assistant\n{solution}<|im_end|>"
)
def perturb_answer(solution: str) -> str:
"""Create an obviously wrong solution for a GSM8K-style answer string.
GSM8K answers end with ``#### <number>``. We replace that number with
``correct * 3 + 137`` so the final answer is clearly incorrect while the
reasoning chain looks plausible.
"""
match = re.search(r"####\s*([\d,]+(?:\.\d+)?)", solution)
if match:
num_str = match.group(1).replace(",", "")
try:
correct_num = float(num_str)
wrong_num = int(correct_num * 3 + 137)
return solution[: match.start()] + f"#### {wrong_num}"
except ValueError:
pass
# Fallback: append an unmistakably wrong sentinel answer.
return solution + "\n#### -999999"
def extract_reward_score(reward_output) -> float:
"""Extract a scalar score from VllmRunner.reward() output for one sample.
VllmRunner.reward() returns list[list[float]] or list[Tensor]; for a reward
model with a single output the inner list has one element. For a token-level
reward model the output is a 2-D tensor [seq_len, 1]; in both cases we take
the last element (final-step score).
"""
if isinstance(reward_output, (list, tuple)):
return float(reward_output[-1])
# Tensor (e.g. shape [seq_len, 1] from a token-level reward model)
return float(reward_output.flatten()[-1].item())
def generate_rm_report(
eval_config: dict,
report_data: dict,
report_dir: str,
env_config: EnvConfig,
) -> None:
"""Write a Markdown accuracy report using the shared Jinja2 template."""
jinja_env = Environment(loader=FileSystemLoader(TEST_DIR))
template = jinja_env.get_template("report_template.md")
serve_cfg = eval_config.get("serve", {})
tp_size = serve_cfg.get("tensor_parallel_size", 1)
ep_enabled = serve_cfg.get("enable_expert_parallel", False)
enforce_eager = serve_cfg.get("enforce_eager", False)
parallel_mode = f"TP{tp_size}"
if ep_enabled:
parallel_mode += " + EP"
execution_model = "Eager" if enforce_eager else "ACLGraph"
model_args_str = ",".join(f"{k}={v}" for k, v in serve_cfg.items())
report_content = template.render(
vllm_version=env_config.vllm_version,
vllm_commit=env_config.vllm_commit,
vllm_ascend_version=env_config.vllm_ascend_version,
vllm_ascend_commit=env_config.vllm_ascend_commit,
cann_version=env_config.cann_version,
torch_version=env_config.torch_version,
torch_npu_version=env_config.torch_npu_version,
hardware=eval_config.get("hardware", "unknown"),
model_name=eval_config["model_name"],
model_args=f"'{model_args_str}'",
model_type=eval_config.get("model_type", "vllm-rm"),
datasets=",".join(t["name"] for t in eval_config["tasks"]),
apply_chat_template=False,
fewshot_as_multiturn=False,
limit=eval_config.get("limit", "N/A"),
batch_size=eval_config.get("batch_size", 4),
num_fewshot="N/A",
rows=report_data["rows"],
parallel_mode=parallel_mode,
execution_model=execution_model,
)
report_path = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
os.makedirs(os.path.dirname(report_path), exist_ok=True)
with open(report_path, "w", encoding="utf-8") as f:
f.write(report_content)
def test_rm_eval_param(config_filename, tp_size, report_dir, env_config):
"""Parametrised reward-model accuracy test driven by a YAML config file.
Skips automatically when the config's model_type is not "vllm-rm".
"""
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
if eval_config.get("model_type", "vllm") != "vllm-rm":
pytest.skip(f"Skipping non-RM config (model_type={eval_config.get('model_type', 'vllm')})")
model_name: str = eval_config["model_name"]
limit: int | None = eval_config.get("limit", None)
batch_size: int = eval_config.get("batch_size", 4)
system_prompt: str = eval_config.get("system_prompt", _DEFAULT_SYSTEM_PROMPT)
serve_cfg: dict = eval_config.get("serve", {})
# CLI --tp-size takes precedence over the YAML tensor_parallel_size.
effective_tp = int(tp_size) if (tp_size and tp_size != "1") else int(serve_cfg.get("tensor_parallel_size", 1))
runner_kwargs: dict = {
k: v
for k, v in {
"runner": "pooling",
"dtype": serve_cfg.get("dtype", "auto"),
"tensor_parallel_size": effective_tp,
"enforce_eager": serve_cfg.get("enforce_eager", False),
"max_model_len": serve_cfg.get("max_model_len"),
"gpu_memory_utilization": serve_cfg.get("gpu_memory_utilization"),
}.items()
if v is not None
}
print(f"\nLoading reward model: {model_name}")
print(f" VllmRunner kwargs: {runner_kwargs}")
success = True
report_data: dict[str, list[dict]] = {"rows": []}
with VllmRunner(model_name, **runner_kwargs) as vllm_model:
for task in eval_config["tasks"]:
task_name: str = task["name"]
dataset_name: str = task["dataset"]
split: str = task["split"]
dataset_config_name: str | None = task.get("dataset_config")
task_type: str = task.get("task_type", "correctness")
# Column names for "correctness" tasks (e.g. GSM8K).
problem_col: str = task.get("problem_column", "question")
solution_col: str = task.get("solution_column", "answer")
# Column names for "pairwise" tasks (e.g. reward-bench).
prompt_col: str = task.get("prompt_column", "prompt")
chosen_col: str = task.get("chosen_column", "chosen")
rejected_col: str = task.get("rejected_column", "rejected")
split_expr = f"{split}[:{limit}]" if limit is not None else split
print(f"\nLoading dataset via ModelScope: {dataset_name} / {dataset_config_name} ({split_expr})")
# MsDataset may bypass the HF_HUB_OFFLINE lock; patch temporarily.
ds = MsDataset.load(
dataset_name,
subset_name=dataset_config_name,
split=split_expr,
)
print(f" {len(ds)} samples to evaluate (task_type={task_type})")
correct_count = 0
total_count = 0
for batch_start in range(0, len(ds), batch_size):
batch = ds.select(range(batch_start, min(batch_start + batch_size, len(ds))))
if task_type == "pairwise":
positive_texts = [format_rm_input(system_prompt, s[prompt_col], s[chosen_col]) for s in batch]
negative_texts = [format_rm_input(system_prompt, s[prompt_col], s[rejected_col]) for s in batch]
else:
positive_texts = [format_rm_input(system_prompt, s[problem_col], s[solution_col]) for s in batch]
negative_texts = [
format_rm_input(system_prompt, s[problem_col], perturb_answer(s[solution_col])) for s in batch
]
pos_rewards = vllm_model.reward(positive_texts)
neg_rewards = vllm_model.reward(negative_texts)
for pos_r, neg_r in zip(pos_rewards, neg_rewards):
if extract_reward_score(pos_r) > extract_reward_score(neg_r):
correct_count += 1
total_count += 1
if (batch_start // batch_size + 1) % 5 == 0:
print(f" processed {batch_start + len(batch)}/{len(ds)} samples …")
measured_accuracy = round(correct_count / total_count, 4) if total_count > 0 else 0.0
print(f"\n{task_name} accuracy = {measured_accuracy:.4f}")
for metric in task["metrics"]:
if metric["name"] != "accuracy":
continue
ground_truth = metric["value"]
# Pass if measured accuracy meets or exceeds the threshold
# (allow up to RTOL relative degradation).
task_success = measured_accuracy >= ground_truth * (1 - RTOL)
success = success and task_success
status = "" if task_success else ""
print(f"{task_name} | accuracy: ground_truth={ground_truth} | measured={measured_accuracy} | {status}")
report_data["rows"].append(
{
"task": task_name,
"metric": "accuracy",
"value": f"{status}{measured_accuracy}",
"stderr": "N/A",
}
)
generate_rm_report(eval_config, report_data, report_dir, env_config)
assert success, "One or more RM tasks did not meet the accuracy threshold. See output above."

View File

@@ -0,0 +1,167 @@
"""
chunk_fwd_o correctness tests on Ascend 310P via torch.ops._C_ascend binding.
"""
import pytest
import torch
import torch_npu # noqa: F401
from vllm_ascend.utils import enable_custom_op
CHUNK_SIZE = 64
def npu_chunk_fwd_o(q, k, v, h, g, scale):
enable_custom_op()
return torch.ops._C_ascend.chunk_fwd_o(
q,
k,
v,
h,
scale,
g=g,
g_gamma=None,
cu_seqlens=None,
chunk_indices=None,
chunk_size=CHUNK_SIZE,
transpose_state_layout=False,
)
def golden_chunk_fwd_o(q, k, v, h_state, g, scale):
"""CPU fp32 reference.
Per chunk c (CS tokens starting at t0):
attn = q[c] @ k[c].T [CS, CS]
gate[i,j] = exp(min(0, g[j] - g[i])) * (j<=i) [CS, CS]
attn_masked = attn * gate
h_work = q[c] @ h_state[c] [CS, Dv]
v_work = attn_masked @ v[c] [CS, Dv]
o[c] = scale * (v_work + exp(g[c]) * h_work)
"""
q, k, v, g = q.float(), k.float(), v.float(), g.float()
h_state = h_state.float()
B, H_k, L, D_k = q.shape
H_v, D_v = v.shape[1], v.shape[3]
CS = CHUNK_SIZE
NT = L // CS
head_groups = H_v // H_k
o = torch.zeros(B, H_v, L, D_v)
for b in range(B):
for hv in range(H_v):
hk = hv // head_groups
for c in range(NT):
t0 = c * CS
q_c = q[b, hk, t0 : t0 + CS]
k_c = k[b, hk, t0 : t0 + CS]
v_c = v[b, hv, t0 : t0 + CS]
g_c = g[b, hv, t0 : t0 + CS]
h_c = h_state[b, hv, c * D_k : (c + 1) * D_k]
attn = q_c @ k_c.T
g_row = g_c.unsqueeze(1)
g_col = g_c.unsqueeze(0)
gate = torch.exp(torch.clamp(g_col - g_row, max=0.0))
causal = torch.tril(torch.ones(CS, CS))
attn_masked = attn * gate * causal
h_work = q_c @ h_c
v_work = attn_masked @ v_c
g_exp = torch.exp(g_c).unsqueeze(1)
o[b, hv, t0 : t0 + CS] = scale * (v_work + g_exp * h_work)
return o
class TestChunkFwdO310:
"""chunk_fwd_o kernel correctness on Ascend 310P."""
@pytest.mark.parametrize(
"B,Hk,Hv,L,Dk,Dv",
[
(1, 2, 2, 128, 128, 128),
(1, 4, 4, 256, 128, 128),
],
)
def test_constant_inputs(self, B, Hk, Hv, L, Dk, Dv):
"""Constant q=k=v, h=0, g=0 => analytically verifiable output."""
scale = 1.0 / (Dk**0.5)
NC = L // CHUNK_SIZE
c = 0.01
q = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
k = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
v = torch.full((B, Hv, L, Dv), c, dtype=torch.float16).npu()
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16).npu()
g = torch.zeros(B, Hv, L, dtype=torch.float32).npu()
o = npu_chunk_fwd_o(q, k, v, h, g, scale)
oc = o.cpu().float()
assert torch.isnan(oc).sum() == 0, "output has NaN"
assert torch.isinf(oc).sum() == 0, "output has Inf"
attn_val = c * c * Dk
for i in range(min(CHUNK_SIZE, 8)):
expected = scale * (i + 1) * attn_val * c
actual = oc[0, 0, i, 0].item()
rel_err = abs(actual - expected) / max(abs(expected), 1e-10)
assert rel_err < 0.10, f"row {i}: actual={actual:.8f} expected={expected:.8f} rel_err={rel_err:.2f}"
@pytest.mark.parametrize(
"B,Hk,Hv,L,Dk,Dv",
[
(1, 2, 2, 128, 128, 128),
(1, 4, 4, 256, 128, 128),
],
)
def test_random_inputs_no_nan(self, B, Hk, Hv, L, Dk, Dv):
"""Random small inputs: no NaN/Inf in output."""
torch.manual_seed(42)
scale = 1.0 / (Dk**0.5)
NC = L // CHUNK_SIZE
q = (torch.randn(B, Hk, L, Dk) * 0.01).half().npu()
k = (torch.randn(B, Hk, L, Dk) * 0.01).half().npu()
v = (torch.randn(B, Hv, L, Dv) * 0.01).half().npu()
h = (torch.randn(B, Hv, NC * Dk, Dv) * 0.01).half().npu()
g = torch.randn(B, Hv, L, dtype=torch.float32).npu() * 0.001
o = npu_chunk_fwd_o(q, k, v, h, g, scale)
oc = o.cpu().float()
assert torch.isnan(oc).sum() == 0, "output has NaN"
assert torch.isinf(oc).sum() == 0, "output has Inf"
assert oc.abs().max() > 0, "output is all zeros"
def test_g_zero_reduces_to_standard_attention(self):
"""g=0 => gate=1, so kernel = scale*(causal_attn@v + q@h)."""
torch.manual_seed(123)
B, Hk, Hv, L, Dk, Dv = 1, 2, 2, 128, 128, 128
scale = 1.0 / (Dk**0.5)
NC = L // CHUNK_SIZE
q = (torch.randn(B, Hk, L, Dk) * 0.01).half()
k = (torch.randn(B, Hk, L, Dk) * 0.01).half()
v = (torch.randn(B, Hv, L, Dv) * 0.01).half()
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16)
g = torch.zeros(B, Hv, L, dtype=torch.float32)
o_npu = npu_chunk_fwd_o(q.npu(), k.npu(), v.npu(), h.npu(), g.npu(), scale)
o_ref = golden_chunk_fwd_o(q, k, v, h, g, scale)
cos = torch.nn.functional.cosine_similarity(o_npu.cpu().float().flatten(), o_ref.flatten(), dim=0).item()
assert cos > 0.999, f"cosine {cos:.4f} too low for g=0 h=0 case"
def test_chunk_boundary_independence(self):
"""Each chunk should produce the same output for identical data."""
B, Hk, Hv, L, Dk, Dv = 1, 2, 2, 128, 128, 128
scale = 1.0 / (Dk**0.5)
NC = L // CHUNK_SIZE
c = 0.02
q = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
k = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
v = torch.full((B, Hv, L, Dv), c, dtype=torch.float16).npu()
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16).npu()
g = torch.zeros(B, Hv, L, dtype=torch.float32).npu()
o = npu_chunk_fwd_o(q, k, v, h, g, scale).cpu().float()
chunk0 = o[0, 0, :CHUNK_SIZE, :]
chunk1 = o[0, 0, CHUNK_SIZE:, :]
cos = torch.nn.functional.cosine_similarity(chunk0.flatten(), chunk1.flatten(), dim=0).item()
assert cos > 0.999, f"chunks differ: cosine={cos:.6f}"

View File

@@ -0,0 +1,160 @@
"""
chunk_gated_delta_rule_fwd_h correctness tests on Ascend 310P
via torch.ops._C_ascend binding.
"""
import pytest
import torch
import torch_npu # noqa: F401
from vllm_ascend.utils import enable_custom_op
CHUNK_SIZE = 64
def npu_chunk_gdr_fwd_h(k, w, u, g, initial_state=None, chunk_size=64):
enable_custom_op()
return torch.ops._C_ascend.chunk_gated_delta_rule_fwd_h(
k,
w,
u,
g=g,
initial_state=initial_state,
output_final_state=False,
chunk_size=chunk_size,
save_new_value=True,
)
def cpu_reference(k, w, u, g, initial_state=None, chunk_size=64):
"""CPU fp32 reference matching kernel semantics."""
k, w, u, g = k.float(), w.float(), u.float(), g.float()
B, Hg, T, K = k.shape
HV, V = u.shape[1], u.shape[3]
NT = T // chunk_size
h = initial_state.float().clone() if initial_state is not None else torch.zeros(B, HV, K, V)
h_chunks = [h.clone()]
v_new = torch.zeros_like(u)
for c in range(NT):
t0 = c * chunk_size
W_chunk = w[:, :, t0 : t0 + chunk_size, :]
ws = torch.einsum("bhik,bhkv->bhiv", W_chunk, h)
g_chunk = g[:, :, t0 : t0 + chunk_size]
v_update = torch.zeros(B, HV, chunk_size, V)
for i in range(chunk_size):
gi_cum = g_chunk[:, :, -1] - g_chunk[:, :, i]
vn = u[:, :, t0 + i, :] - ws[:, :, i, :]
v_new[:, :, t0 + i, :] = vn
v_update[:, :, i, :] = gi_cum.unsqueeze(-1).exp() * vn
K_chunk = k[:, :, t0 : t0 + chunk_size, :]
h_work = torch.einsum("bhik,bhiv->bhkv", K_chunk, v_update)
h = h * g_chunk[:, :, -1:].unsqueeze(-1).exp() + h_work
h_chunks.append(h.clone())
return h_chunks, v_new
def cosine(a, b):
a, b = a.flatten().double(), b.flatten().double()
if a.norm() == 0 and b.norm() == 0:
return 1.0
if a.norm() == 0 or b.norm() == 0:
return 0.0
return torch.nn.functional.cosine_similarity(a.unsqueeze(0), b.unsqueeze(0)).item()
class TestChunkGatedDeltaRuleFwdH310:
"""chunk_gated_delta_rule_fwd_h kernel correctness on Ascend 310P."""
@pytest.mark.parametrize(
"B,Hg,HV,T,K,V",
[
(1, 1, 1, 128, 128, 128),
(1, 2, 2, 128, 128, 128),
],
)
def test_h_state_correctness(self, B, Hg, HV, T, K, V):
torch.manual_seed(42)
DTYPE = torch.float16
k = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
w = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
u = torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1
g = (-torch.rand(B, HV, T) * 0.1).float()
init = torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01
h_ref, _ = cpu_reference(k, w, u, g, init, CHUNK_SIZE)
h_out, _, _ = npu_chunk_gdr_fwd_h(
k.npu(),
w.npu(),
u.npu(),
g.npu(),
initial_state=init.npu(),
chunk_size=CHUNK_SIZE,
)
h_npu = h_out.cpu().float()
NT = T // CHUNK_SIZE
for c in range(min(NT + 1, h_npu.shape[2])):
ref = h_ref[c].flatten()
npu = h_npu[0, :, c].flatten()
cos = cosine(npu, ref)
assert cos >= 0.99, f"h[{c}] cos={cos:.6f} too low"
@pytest.mark.parametrize(
"B,Hg,HV,T,K,V",
[
(1, 1, 1, 128, 128, 128),
(1, 2, 2, 128, 128, 128),
],
)
def test_v_new_correctness(self, B, Hg, HV, T, K, V):
torch.manual_seed(42)
DTYPE = torch.float16
k = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
w = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
u = torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1
g = (-torch.rand(B, HV, T) * 0.1).float()
init = torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01
_, vn_ref = cpu_reference(k, w, u, g, init, CHUNK_SIZE)
_, vn_out, _ = npu_chunk_gdr_fwd_h(
k.npu(),
w.npu(),
u.npu(),
g.npu(),
initial_state=init.npu(),
chunk_size=CHUNK_SIZE,
)
vn_npu = vn_out.cpu().float()
NT = T // CHUNK_SIZE
for c in range(NT):
t0, t1 = c * CHUNK_SIZE, (c + 1) * CHUNK_SIZE
ref = vn_ref[:, :, t0:t1].flatten()
npu = vn_npu[:, :, t0:t1].flatten()
cos = cosine(npu, ref)
assert cos >= 0.99, f"v_new chunk {c} cos={cos:.6f} too low"
def test_no_nan(self):
torch.manual_seed(42)
B, Hg, HV, T, K, V = 1, 1, 1, 128, 128, 128
DTYPE = torch.float16
k = (torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1).npu()
w = (torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1).npu()
u = (torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1).npu()
g = (-torch.rand(B, HV, T).float() * 0.1).npu()
init = (torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01).npu()
h_out, vn_out, _ = npu_chunk_gdr_fwd_h(
k,
w,
u,
g,
initial_state=init,
chunk_size=CHUNK_SIZE,
)
assert torch.isnan(h_out.cpu()).sum() == 0, "h_out has NaN"
assert torch.isnan(vn_out.cpu()).sum() == 0, "vn_out has NaN"
assert torch.isinf(h_out.cpu()).sum() == 0, "h_out has Inf"
assert torch.isinf(vn_out.cpu()).sum() == 0, "vn_out has Inf"

View File

@@ -0,0 +1,178 @@
"""Test V310 kernel via ctypes API against golden CPU reference."""
import ctypes
import os
import pytest
import torch
import torch_npu
torch_npu.npu.set_compile_mode(jit_compile=False)
_CANN = os.environ.get("ASCEND_HOME_PATH", "/usr/local/Ascend/ascend-toolkit/latest")
_CUST = f"{_CANN}/opp/vendors/custom_transformer/op_api/lib"
_LIB_PATHS = [_CUST, f"{_CANN}/lib64", f"{_CANN}/aarch64-linux/lib64"]
def _find_lib(name, paths):
for p in paths:
full = os.path.join(p, name)
if os.path.exists(full):
return full
return name
_acl = ctypes.CDLL(_find_lib("libnnopbase.so", _LIB_PATHS))
_opapi = ctypes.CDLL(_find_lib("libcust_opapi.so", _LIB_PATHS))
_acl.aclCreateTensor.restype = ctypes.c_void_p
_acl.aclCreateTensor.argtypes = [
ctypes.POINTER(ctypes.c_int64),
ctypes.c_uint64,
ctypes.c_int,
ctypes.POINTER(ctypes.c_int64),
ctypes.c_int64,
ctypes.c_int,
ctypes.POINTER(ctypes.c_int64),
ctypes.c_uint64,
ctypes.c_void_p,
]
_acl.aclDestroyTensor.argtypes = [ctypes.c_void_p]
_DTYPE_MAP = {torch.float16: 1, torch.float32: 0, torch.int32: 3}
def mk(t):
if t is None:
return None
shape, strides, ndim = list(t.shape), list(t.stride()), len(t.shape)
return ctypes.c_void_p(
_acl.aclCreateTensor(
(ctypes.c_int64 * ndim)(*shape),
ndim,
_DTYPE_MAP[t.dtype],
(ctypes.c_int64 * ndim)(*strides),
ctypes.c_int64(0),
2,
(ctypes.c_int64 * ndim)(*shape),
ndim,
ctypes.c_void_p(t.data_ptr()),
)
)
def call_v310(query, key, value, beta, state, seq_lens, indices, g, nat, scale):
out = torch.empty_like(value)
ws_size = ctypes.c_uint64(0)
executor = ctypes.c_void_p(0)
ret = _opapi.aclnnRecurrentGatedDeltaRuleV310GetWorkspaceSize(
mk(query),
mk(key),
mk(value),
mk(beta),
mk(state),
mk(seq_lens),
mk(indices),
mk(g),
None,
mk(nat),
ctypes.c_float(scale),
mk(out),
ctypes.byref(ws_size),
ctypes.byref(executor),
)
assert ret == 0, f"GetWorkspaceSize failed: {ret}"
ws_ptr = ctypes.c_void_p(0)
if ws_size.value > 0:
ws = torch.empty(ws_size.value, dtype=torch.uint8, device=query.device)
ws_ptr = ctypes.c_void_p(ws.data_ptr())
stream = torch.npu.current_stream().npu_stream
ret = _opapi.aclnnRecurrentGatedDeltaRuleV310(ws_ptr, ws_size, executor, ctypes.c_void_p(stream))
assert ret == 0, f"Execute failed: {ret}"
torch.npu.synchronize()
return out
def golden(query, key, value, state, beta, scale, seq_lens, indices, g, nat):
k = key.float()
q = query.float()
v = value.float()
S = state.clone().float()
T, nv, Dv = v.shape
nk = q.shape[1]
g_f = torch.ones(T, nv) if g is None else g.float().exp()
beta_f = beta.float()
o = torch.empty_like(v, dtype=torch.float32)
q = q * scale
seq_start = 0
for i in range(len(seq_lens)):
init_idx = indices[seq_start + nat[i] - 1] if nat is not None else indices[seq_start]
for head in range(nv):
s = S[init_idx][head].clone()
for t in range(seq_start, seq_start + seq_lens[i]):
qi = q[t][head // (nv // nk)]
ki = k[t][head // (nv // nk)]
vi = v[t][head]
s = s * g_f[t][head]
x = (s * ki.unsqueeze(-2)).sum(dim=-1)
y = (vi - x) * beta_f[t][head]
s = s + y[:, None] * ki[None, :]
S[indices[t]][head] = s
o[t][head] = (s * qi.unsqueeze(-2)).sum(dim=-1)
seq_start += seq_lens[i]
return o, S
@pytest.mark.parametrize(
"batch_size,mtp,nk,nv,dk,dv,num_slots",
[
(1, 1, 8, 16, 128, 128, 444),
(2, 2, 8, 16, 128, 128, 444),
(4, 2, 4, 4, 64, 64, 32),
],
)
def test_recurrent_gated_delta_rule_v310(batch_size, mtp, nk, nv, dk, dv, num_slots):
torch.manual_seed(42)
scale = dk**-0.5
seq_lens = torch.ones(batch_size, dtype=torch.int32) * mtp
T = int(seq_lens.sum())
state = torch.rand(num_slots, nv, dv, dk, dtype=torch.float16)
indices = torch.randperm(num_slots, dtype=torch.int32)[:T]
nat = torch.ones(batch_size, dtype=torch.int32)
query = torch.nn.functional.normalize(torch.randn(T, nk, dk), dim=-1).to(torch.float16)
key = torch.nn.functional.normalize(torch.randn(T, nk, dk), dim=-1).to(torch.float16)
value = torch.randn(T, nv, dv, dtype=torch.float16)
beta = torch.rand(T, nv, dtype=torch.float16)
g = torch.rand(T, nv, dtype=torch.float32)
out_gold, state_gold = golden(query, key, value, state, beta, scale, seq_lens, indices, g, nat)
state_npu = state.clone().npu()
out_npu = call_v310(
query.npu(),
key.npu(),
value.npu(),
beta.npu(),
state_npu,
seq_lens.npu(),
indices.npu(),
g.npu(),
nat.npu(),
scale,
)
touched = indices.long()
torch.testing.assert_close(
out_npu.float().cpu(),
out_gold,
rtol=3e-3,
atol=2e-3,
equal_nan=True,
)
torch.testing.assert_close(
state_npu.float().cpu()[touched],
state_gold.float()[touched],
rtol=3e-3,
atol=2e-3,
equal_nan=True,
)

View File

View File

@@ -0,0 +1 @@
"""External DP nightly test package."""

View File

@@ -0,0 +1,308 @@
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 0
tp_size: 16
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 1
dp_rank_start: 1
tp_size: 16
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "2048"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "128"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --additional-config
- '{"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "60"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "128"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
benchmarks:
perf_1M_1k_prefix99_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1
batch_size: 1
request_rate: 0
baseline: 0
threshold: 0.95
perf_1M_1k_prefix99:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1024
batch_size: 1
request_rate: 0
baseline: 11.93
threshold: 0.95

View File

@@ -0,0 +1,324 @@
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
num_nodes: 4
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0, 1 ]
decoder: [ 2, 3 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 0
tp_size: 2
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 8
dp_rank_start: 8
tp_size: 2
dp_address: "${NODE_2_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
env_prefill: &env_prefill
<<: *env_common
HCCL_CONNECT_TIMEOUT: "6000"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_BUFFSIZE: "1800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 1
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "4096"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --quantization
- "ascend"
- --block-size
- "32"
- --enforce-eager
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- node_index: 2
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
- node_index: 3
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "204800"
- --max-num-batched-tokens
- "120"
- --max-num-seqs
- "30"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
perf_TPOT50_128k_1_prefix_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 1
batch_size: 4
request_rate: 0
baseline: 0
threshold: 0.95
perf_TPOT50_128k_1_prefix90:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1024
batch_size: 48
request_rate: 1
baseline: 869.13
threshold: 0.95
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,176 @@
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 16
dp_size_local: 16
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_RPC_TIMEOUT: "3600"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "1500"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --speculative-config
- '{"num_speculative_tokens": 1,"method": "mtp"}'
- --trust-remote-code
- --block-size
- "32"
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.9"
- --quantization
- "ascend"
- --enforce-eager
- --additional-config
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "1048576"
- --max-num-batched-tokens
- "240"
- --max-num-seqs
- "60"
- --async-scheduling
- --block-size
- "32"
- --no-enable-prefix-caching
- --no-disable-hybrid-kv-cache-manager
- --model-loader-extra-config
- '{"enable_multithread_load": true, "num_threads": 128}'
- --trust-remote-code
- --tokenizer-mode
- "deepseek_v4"
- --tool-call-parser
- "deepseek_v4"
- --enable-auto-tool-choice
- --reasoning-parser
- "deepseek_v4"
- --gpu-memory-utilization
- "0.95"
- --quantization
- "ascend"
- --speculative-config
- '{"num_speculative_tokens": 3,"method": "mtp"}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
- --additional-config
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
temperature: 1.0
top_p: 1.0
thinking: "true"
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10

View File

@@ -0,0 +1,335 @@
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 4
npu_per_node: 16
special_dependencies:
transformers: "5.2.0"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0, 1]
decoder: [2, 3]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 2
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_2_IP}"
- node_index: 3
port_start: 7100
dp_rpc_port: 12321
dp_size: 8
dp_size_local: 4
dp_rank_start: 4
tp_size: 4
dp_address: "${NODE_2_IP}"
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
VLLM_TORCH_PROFILER_WITH_STACK: "0"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_INTRA_PCIE_ENABLE: "1"
HCCL_INTRA_ROCE_ENABLE: "0"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
templates:
- node_index: 0
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 1
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "131072"
- --additional-config
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "64"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.95"
- --enforce-eager
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 2
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
- node_index: 3
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: "1"
TASK_QUEUE_ENABLE: "1"
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- ${PORT}
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --speculative-config
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--seed"
- "1024"
- --max-model-len
- "202752"
- --max-num-batched-tokens
- "32"
- --additional-config
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
- --max-num-batched-tokens
- "4096"
- --trust-remote-code
- --max-num-seqs
- "8"
- --quantization
- ascend
- --gpu-memory-utilization
- "0.92"
- --async-scheduling
- --enable-auto-tool-choice
- --tool-call-parser
- glm47
- --reasoning-parser
- glm45
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1500
batch_size: 40
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,161 @@
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
model: "Eco-Tech/Kimi-K2.6-w4a8"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 4
dp_rank_start: 0
tp_size: 4
dp_address: "${NODE_1_IP}"
env_common: &env_common
SERVER_PORT: "${PORT}"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
HCCL_EXEC_TIMEOUT: "204"
HCCL_CONNECT_TIMEOUT: "120"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
VLLM_SERVER_DEV_MODE: "1"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
VLLM_USE_MODELSCOPE: "true"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "512"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "800"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --safetensors-load-strategy
- 'prefetch'
- --enable-expert-parallel
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "8192"
- --max-num-seqs
- "16"
- --enforce-eager
- --trust-remote-code
- --gpu-memory-utilization
- "0.94"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --allowed-local-media-path
- "/"
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --safetensors-load-strategy
- 'prefetch'
- --seed
- "1024"
- --max-model-len
- "68000"
- --max-num-batched-tokens
- "256"
- --max-num-seqs
- "16"
- --trust-remote-code
- --gpu-memory-utilization
- "0.92"
- --speculative-config
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 60
max_out_len: 1024
batch_size: 15
request_rate: 0.4
baseline: 347.4475
threshold: 0.97
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 93.33
threshold: 10
temperature: 1.0
top_p: 1

View File

@@ -0,0 +1,197 @@
test_name: "Minimax_m2.7_in3_5_tpot50"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
num_nodes: 2
npu_per_node: 16
routing:
type: "disaggregated_prefill"
groups:
prefiller: [ 0 ]
decoder: [ 1 ]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 8
dp_address: "${NODE_1_IP}"
env_common: &env_common
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "${PORT}"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
PYTHONHASHSEED: "0"
env_prefill: &env_prefill
<<: *env_common
HCCL_BUFFSIZE: "1024"
env_decode: &env_decode
<<: *env_common
HCCL_BUFFSIZE: "2048"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_prefill
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --enforce-eager
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "55880",
"engine_id": "0",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}} }'
- node_index: 1
envs:
<<: *env_decode
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --enable-expert-parallel
- --max-model-len
- "199608"
- --max-num-batched-tokens
- "16384"
- --max-num-seqs
- "24"
- --trust-remote-code
- --gpu-memory-utilization
- "0.8"
- --quantization
- "ascend"
- --speculative-config
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- --async-scheduling
- --compilation-config
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- --additional-config
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "56900",
"engine_id": "1",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}}'
benchmarks:
perf_warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2
max_out_len: 1
batch_size: 2
request_rate: 0
baseline: 0
threshold: 0.97
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1024
batch_size: 40
request_rate: 0
baseline: 717.5332
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 90
threshold: 10
temperature: 1
top_p: 1
top_k: 40
ignore_eos: false

View File

@@ -0,0 +1,360 @@
# External DP Config Template
This document shows how to write YAML configs consumed by
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
`server_cmd_template` contains only the arguments after
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
`model` automatically.
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
and roles are selected by `routing.groups`.
## Generic DP Template
Use this template for generic external data parallel serving. This mode uses
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
independent vLLM instances instead of external DP rank arguments.
```yaml
test_name: "test Qwen3-30B-A3B generic external dp"
model: "Qwen/Qwen3-30B-A3B"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "generic_dp"
groups:
worker: [0, 1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 4
dp_size_local: 2
dp_rank_start: 2
tp_size: 1
dp_address: "${NODE_0_IP}"
templates:
- node_index: 0
envs: &generic_env
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "1024"
SERVER_PORT: "${PORT}"
server_cmd_template: &generic_server_cmd
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --max-model-len
- "4096"
- --trust-remote-code
- --enable-expert-parallel
- node_index: 1
envs:
<<: *generic_env
server_cmd_template: *generic_server_cmd
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 4
max_out_len: 16
batch_size: 1
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
num_prompts: 4
max_out_len: 16
batch_size: 1
baseline: 0
threshold: 100
```
## Disaggregated Prefill Template
Use this template for PD disaggregation. `routing.groups` decides which config
entries run as prefillers or decoders. The framework derives the PD proxy script
from `routing.type`, so do not write `proxy_*` fields in YAML.
```yaml
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
num_nodes: 2
npu_per_node: 16
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
# cluster_hosts:
# - "172.22.0.xxx"
# - "172.22.0.xxx"
routing:
type: "disaggregated_prefill"
groups:
prefiller: [0]
decoder: [1]
config:
- node_index: 0
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_0_IP}"
- node_index: 1
port_start: 7100
dp_rpc_port: 12321
dp_size: 2
dp_size_local: 2
dp_rank_start: 0
tp_size: 1
dp_address: "${NODE_1_IP}"
env_common: &env_common
HCCL_OP_
VLLM_USE_MODELSCOPE: "true"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "10"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
HCCL_BUFFSIZE: "256"
SERVER_PORT: "${PORT}"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
templates:
- node_index: 0
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
- node_index: 1
envs:
<<: *env_common
server_cmd_template:
- --host
- "0.0.0.0"
- --port
- $SERVER_PORT
- --data-parallel-size
- ${DP_SIZE}
- --data-parallel-rank
- ${DP_RANK}
- --data-parallel-address
- ${DP_ADDRESS}
- --data-parallel-rpc-port
- ${DP_RPC_PORT}
- --tensor-parallel-size
- ${TP_SIZE}
- --trust-remote-code
- --quantization
- ascend
- --enable-expert-parallel
- --kv-transfer-config
- '{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 1
},
"decode": {
"dp_size": 2,
"tp_size": 1
}
}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
max_out_len: 128
batch_size: 4
request_rate: 1
baseline: 1
threshold: 0.1
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 48
batch_size: 4
baseline: 0
threshold: 100
```
## Field Notes
- `test_name`: Human-readable test name. It is also used when writing benchmark
result metadata.
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
- `num_nodes`: Number of config entries and templates expected.
- `npu_per_node`: Device capacity validation for each node.
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
needs fixed hosts.
- `routing.type`: Supported values are `generic_dp` and
`disaggregated_prefill`.
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
`kv_consumer` for decoder templates.
- `config[].dp_size`: Global DP size for this DP group.
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
- `config[].dp_rank_start`: First global DP rank owned by this node.
- `config[].dp_address`: DP master address. For one global DP group, use
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
address for prefiller nodes and the decoder master address for decoder nodes.
- `templates`: One template per config entry. The framework expands one command
per local DP rank.
The framework injects distributed network envs at startup:
```text
HCCL_IF_IP
HCCL_SOCKET_IFNAME
GLOO_SOCKET_IFNAME
TP_SOCKET_IFNAME
LOCAL_IP
NIC_NAME
MASTER_IP
```
The framework also derives proxy metadata from `routing.type`:
```text
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
```
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
for benchmark requests.
## Template Variables
The following variables are available in `envs` and `server_cmd_template`:
```text
${MODEL}
${PORT_START}
${PORT}
${DP_SIZE}
${DP_SIZE_LOCAL}
${DP_RANK_START}
${DP_RANK}
${LOCAL_RANK}
${TP_SIZE}
${CP_SIZE}
${SP_SIZE}
${PP_SIZE}
${DP_ADDRESS}
${DP_RPC_PORT}
${VISIBLE_DEVICES}
${NODE_INDEX}
${CONFIG_INDEX}
${NODE_0_IP}, ${NODE_1_IP}, ...
${LOCAL_IP}
${MASTER_IP}
${LWS_WORKER_INDEX}
```
Command arguments can also reference rendered environment variables with
shell-style `$VARNAME`, for example:
```yaml
envs:
SERVER_PORT: "${PORT}"
server_cmd_template:
- --port
- $SERVER_PORT
```
## Checks Before Running
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
- Make sure each config index is assigned to exactly one routing group.
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
`--enable-expert-parallel`.
- Set `--max-model-len` large enough for benchmark input tokens plus
`max_out_len`.

View File

@@ -0,0 +1 @@
"""External DP nightly test helpers."""

View File

@@ -0,0 +1,449 @@
import logging
import os
from dataclasses import dataclass, field
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.scripts.utils import (
load_yaml_mapping,
resolve_cluster_ips,
)
from tests.e2e.nightly.multi_node.scripts.utils import (
resolve_current_node_index as resolve_node_index,
)
logger = logging.getLogger(__name__)
ROUTING_GENERIC_DP = "generic_dp"
ROUTING_DISAGGREGATED_PREFILL = "disaggregated_prefill"
PROXY_SCRIPT_BY_ROUTING_TYPE = {
ROUTING_GENERIC_DP: "examples/external_online_dp/dp_load_balance_proxy_server.py",
ROUTING_DISAGGREGATED_PREFILL: "examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
}
CLUSTER_PLACEHOLDER_RE = re.compile(r"\$\{(NODE_(\d+)_IP|LOCAL_IP|MASTER_IP|LWS_WORKER_INDEX)\}")
@dataclass(frozen=True)
class RoutingConfig:
"""Proxy routing metadata shared by all external DP ranks."""
type: str
proxy_node_index: int
proxy_host: str
proxy_port: int
proxy_script: str
groups: dict[str, list[int]]
@dataclass(frozen=True)
class NodeInfo:
"""Per-node external DP server topology loaded from one config entry."""
ip: str
port_start: int
dp_rpc_port: int
dp_size: int
dp_size_local: int
dp_rank_start: int
tp_size: int
dp_address: str
cp_size: int = 1
sp_size: int = 1
pp_size: int = 1
@property
def devices_per_rank(self) -> int:
return self.tp_size * self.cp_size * self.sp_size * self.pp_size
@property
def devices_per_node(self) -> int:
return self.dp_size_local * self.devices_per_rank
@dataclass(frozen=True)
class NodeTemplate:
"""Per-node env and argument template for launching vLLM servers."""
envs: dict[str, Any]
server_cmd_template: list[str]
@dataclass(frozen=True)
class RankInfo:
"""One concrete vLLM server rank expanded from a node config."""
node_index: int
role: str
local_rank: int
dp_rank: int
host: str
port: int
visible_devices: str
dp_size: int
dp_size_local: int
tp_size: int
cp_size: int
sp_size: int
pp_size: int
dp_address: str
dp_rpc_port: int
port_start: int
@dataclass(frozen=True)
class ExternalDPConfig:
"""Top-level external DP test config after YAML anchors are merged."""
test_name: str
model: str
num_nodes: int
npu_per_node: int
cluster_hosts: list[str] | None
cluster_ips: list[str]
routing: RoutingConfig
nodes: list[NodeInfo]
launch_templates: list[NodeTemplate]
benchmark_cases: list[dict[str, Any]] = field(default_factory=list)
special_dependencies: dict[str, str] = field(default_factory=dict)
@property
def is_disaggregated_prefill(self) -> bool:
return self.routing.type == ROUTING_DISAGGREGATED_PREFILL
def replace_cluster_placeholders(
value: Any,
*,
cluster_ips: list[str],
local_ip: str | None = None,
current_node_index: int | None = None,
) -> Any:
if isinstance(value, dict):
return {
key: replace_cluster_placeholders(
val,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for key, val in value.items()
}
if isinstance(value, list):
return [
replace_cluster_placeholders(
item,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=current_node_index,
)
for item in value
]
if not isinstance(value, str):
return value
def repl(match: re.Match[str]) -> str:
token = match.group(1)
node_index = match.group(2)
if node_index is not None:
idx = int(node_index)
if idx >= len(cluster_ips):
raise ValueError(f"Cluster placeholder ${{{token}}} is out of range")
return cluster_ips[idx]
if token == "MASTER_IP":
return cluster_ips[0]
if token == "LOCAL_IP":
if local_ip is None:
return match.group(0)
return local_ip
if token == "LWS_WORKER_INDEX":
if current_node_index is None:
return os.environ.get("LWS_WORKER_INDEX", match.group(0))
return str(current_node_index)
return match.group(0)
return CLUSTER_PLACEHOLDER_RE.sub(repl, value)
def resolve_current_node_index(config: ExternalDPConfig) -> int:
return resolve_node_index(config.cluster_ips)
class ExternalDPConfigLoader:
"""Load, normalize, and validate external DP YAML files."""
@classmethod
def from_yaml(
cls,
yaml_path: str | None = None,
*,
cluster_ips: list[str] | None = None,
) -> ExternalDPConfig:
raw_config = cls._load_yaml(yaml_path)
cls._validate_root(raw_config)
num_nodes = int(raw_config["num_nodes"])
resolved_cluster_ips = cls._resolve_cluster_ips(raw_config, num_nodes, cluster_ips)
model = str(raw_config["model"])
routing = cls._parse_routing(raw_config["routing"], resolved_cluster_ips)
nodes = cls._parse_nodes(raw_config, resolved_cluster_ips)
launch_templates = cls._parse_templates(raw_config)
benchmark_cases = cls._parse_benchmarks(raw_config)
config = ExternalDPConfig(
test_name=str(raw_config.get("test_name", "external_dp_test")),
model=model,
num_nodes=num_nodes,
npu_per_node=int(raw_config["npu_per_node"]),
cluster_hosts=raw_config.get("cluster_hosts"),
cluster_ips=resolved_cluster_ips,
routing=routing,
nodes=nodes,
launch_templates=launch_templates,
benchmark_cases=benchmark_cases,
special_dependencies=dict(raw_config.get("special_dependencies", {})),
)
cls._validate_config(config)
return config
@staticmethod
def _load_yaml(yaml_path: str | None) -> dict[str, Any]:
default_config_name = "GLM5_1-W8A8-EP-external.yaml"
default_config_base_path = "tests/e2e/nightly/multi_node/external_dp/config/"
return load_yaml_mapping(
yaml_path,
default_name=default_config_name,
default_base_path=default_config_base_path,
description="external DP config",
)
@staticmethod
def _validate_root(config: dict[str, Any]) -> None:
required = ["model", "num_nodes", "npu_per_node", "routing", "config", "templates", "benchmarks"]
missing = [key for key in required if key not in config]
if missing:
raise KeyError(f"Missing required external DP config fields: {missing}")
if int(config["num_nodes"]) <= 0:
raise ValueError("num_nodes must be greater than 0")
@staticmethod
def _resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
cluster_ips: list[str] | None,
) -> list[str]:
return resolve_cluster_ips(
raw_config,
num_nodes,
cluster_ips,
dns_log_message="Resolving external DP cluster IPs via LWS DNS",
)
@staticmethod
def _parse_routing(raw_routing: dict[str, Any], cluster_ips: list[str]) -> RoutingConfig:
routing_type = str(raw_routing["type"])
if routing_type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {routing_type}")
proxy_node_index = 0
proxy_port = 1999
if proxy_node_index >= len(cluster_ips) or proxy_node_index < 0:
raise ValueError("routing.proxy_node_index out of range")
local_ip = cluster_ips[proxy_node_index]
routing = replace_cluster_placeholders(
raw_routing,
cluster_ips=cluster_ips,
local_ip=local_ip,
current_node_index=proxy_node_index,
)
return RoutingConfig(
type=routing_type,
proxy_node_index=proxy_node_index,
proxy_host=local_ip,
proxy_port=proxy_port,
proxy_script=PROXY_SCRIPT_BY_ROUTING_TYPE[routing_type],
groups={
str(name): [int(index) for index in indices] for name, indices in routing.get("groups", {}).items()
},
)
@staticmethod
def _parse_nodes(raw_config: dict[str, Any], cluster_ips: list[str]) -> list[NodeInfo]:
nodes: list[NodeInfo] = []
for index, raw_node in enumerate(raw_config["config"]):
raw_node_index = raw_node.get("node_index")
if raw_node_index is not None and int(raw_node_index) != index:
raise ValueError(f"config[{index}].node_index must equal {index}")
node = replace_cluster_placeholders(
raw_node,
cluster_ips=cluster_ips,
local_ip=cluster_ips[index],
current_node_index=index,
)
nodes.append(
NodeInfo(
ip=cluster_ips[index],
port_start=int(node["port_start"]),
dp_rpc_port=int(node["dp_rpc_port"]),
dp_size=int(node.get("dp_size", 1)),
dp_size_local=int(node.get("dp_size_local", 1)),
dp_rank_start=int(node.get("dp_rank_start", 0)),
tp_size=int(node.get("tp_size", 1)),
cp_size=int(node.get("cp_size", 1)),
sp_size=int(node.get("sp_size", 1)),
dp_address=str(node["dp_address"]),
pp_size=int(node.get("pp_size", 1)),
)
)
return nodes
@staticmethod
def _parse_templates(raw_config: dict[str, Any]) -> list[NodeTemplate]:
templates: list[NodeTemplate] = []
for index, raw_template in enumerate(raw_config["templates"]):
envs = raw_template.get("envs")
server_cmd_template = raw_template.get("server_cmd_template")
if envs is None or server_cmd_template is None:
raise KeyError(f"templates[{index}] must contain envs and server_cmd_template")
if not isinstance(server_cmd_template, list):
raise TypeError(f"templates[{index}].server_cmd_template must be a list")
templates.append(
NodeTemplate(
envs=dict(envs),
server_cmd_template=[str(arg) for arg in server_cmd_template],
)
)
return templates
@staticmethod
def _parse_benchmarks(raw_config: dict[str, Any]) -> list[dict[str, Any]]:
benchmark_cases: list[dict[str, Any]] = []
for name, case in (raw_config.get("benchmarks") or {}).items():
case_with_name = dict(case)
case_with_name["case_name"] = name
benchmark_cases.append(case_with_name)
return benchmark_cases
@classmethod
def _validate_config(cls, config: ExternalDPConfig) -> None:
cls._validate_config_sizes(config)
cls._validate_routing(config)
cls._validate_node_parallel_config(config)
@staticmethod
def _validate_config_sizes(config: ExternalDPConfig) -> None:
if len(config.nodes) != config.num_nodes:
raise AssertionError(f"config size ({len(config.nodes)}) != num_nodes ({config.num_nodes})")
if len(config.launch_templates) != config.num_nodes:
raise AssertionError(f"templates size ({len(config.launch_templates)}) != num_nodes ({config.num_nodes})")
if config.cluster_hosts and len(config.cluster_hosts) != config.num_nodes:
raise AssertionError("cluster_hosts size mismatch")
@staticmethod
def _validate_routing(config: ExternalDPConfig) -> None:
if config.routing.type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
raise ValueError(f"Unsupported routing.type: {config.routing.type}")
groups = config.routing.groups
if config.routing.type == ROUTING_GENERIC_DP and not groups.get("worker"):
raise ValueError("generic_dp routing requires routing.groups.worker")
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL and (
not groups.get("prefiller") or not groups.get("decoder")
):
raise ValueError("disaggregated_prefill routing requires prefiller and decoder groups")
seen_group_indices: dict[int, str] = {}
for group_name, indices in groups.items():
for index in indices:
if index < 0 or index >= config.num_nodes:
raise ValueError(f"routing.groups.{group_name} index out of range: {index}")
if index in seen_group_indices:
raise ValueError(f"node index {index} appears in both {seen_group_indices[index]} and {group_name}")
seen_group_indices[index] = group_name
if config.routing.proxy_node_index < 0 or config.routing.proxy_node_index >= config.num_nodes:
raise ValueError("routing.proxy_node_index out of range")
@staticmethod
def _validate_node_parallel_config(config: ExternalDPConfig) -> None:
for node_index, node in enumerate(config.nodes):
parallel_sizes = {
"dp_size": node.dp_size,
"dp_size_local": node.dp_size_local,
"tp_size": node.tp_size,
"cp_size": node.cp_size,
"sp_size": node.sp_size,
"pp_size": node.pp_size,
}
invalid_sizes = {name: value for name, value in parallel_sizes.items() if value < 1}
if invalid_sizes:
raise ValueError(f"node {node_index} parallel sizes must be >= 1: {invalid_sizes}")
if node.dp_rank_start < 0:
raise ValueError(f"node {node_index} dp_rank_start must be >= 0")
if node.devices_per_node > config.npu_per_node:
raise ValueError(
f"node {node_index} uses {node.devices_per_node} NPUs, but npu_per_node is {config.npu_per_node}"
)
if node.dp_rank_start + node.dp_size_local > node.dp_size:
raise ValueError(f"node {node_index} dp rank range exceeds dp_size")
class RankResolver:
"""Expand node-level configs into concrete vLLM server ranks."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def resolve(self) -> list[RankInfo]:
role_by_node_index = self._role_by_node_index()
ranks: list[RankInfo] = []
for node_index, node_info in enumerate(self.config.nodes):
role = role_by_node_index[node_index]
ranks.extend(self._expand_node(node_index, role, node_info))
return ranks
def _role_by_node_index(self) -> dict[int, str]:
role_by_index: dict[int, str] = {}
for role, node_indices in self.config.routing.groups.items():
for index in node_indices:
role_by_index[index] = role
missing = [index for index in range(self.config.num_nodes) if index not in role_by_index]
if missing:
raise ValueError(f"routing.groups does not assign role for node indices: {missing}")
return role_by_index
@staticmethod
def _expand_node(node_index: int, role: str, node_info: NodeInfo) -> list[RankInfo]:
ranks: list[RankInfo] = []
for local_rank in range(node_info.dp_size_local):
dp_rank = node_info.dp_rank_start + local_rank
port = node_info.port_start + local_rank
device_range = range(
local_rank * node_info.devices_per_rank,
(local_rank + 1) * node_info.devices_per_rank,
)
visible_devices = ",".join(str(device) for device in device_range)
ranks.append(
RankInfo(
node_index=node_index,
role=role,
local_rank=local_rank,
dp_rank=dp_rank,
host=node_info.ip,
port=port,
visible_devices=visible_devices,
dp_size=node_info.dp_size,
dp_size_local=node_info.dp_size_local,
tp_size=node_info.tp_size,
cp_size=node_info.cp_size,
sp_size=node_info.sp_size,
pp_size=node_info.pp_size,
dp_address=node_info.dp_address,
dp_rpc_port=node_info.dp_rpc_port,
port_start=node_info.port_start,
)
)
return ranks

View File

@@ -0,0 +1,435 @@
import logging
import os
import subprocess
import sys
import time
from collections.abc import Iterable
from dataclasses import dataclass
from pathlib import Path
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ROUTING_GENERIC_DP,
ExternalDPConfig,
NodeTemplate,
RankInfo,
replace_cluster_placeholders,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
format_server_cmd,
is_http_ready,
start_logged_process,
terminate_process_tree,
wait_http_ready,
wait_http_unready,
)
from tests.e2e.nightly.multi_node.scripts.utils import get_net_interface
logger = logging.getLogger(__name__)
SERVER_READY_TIMEOUT_SECONDS = 3600
TEMPLATE_VAR_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
ENV_VAR_RE = re.compile(r"(?<!\$)\$([A-Za-z_][A-Za-z0-9_]*)")
@dataclass(frozen=True)
class ServerCommand:
"""Rendered command, env, and printable command line."""
cmd: list[str]
env: dict[str, str]
display_cmd: str
RankProcess = tuple[subprocess.Popen, RankInfo, Path]
class ServerCommandBuilder:
"""Render rank templates into vLLM serve commands."""
def __init__(self, config: ExternalDPConfig):
self.config = config
def build(self, rank: RankInfo, template: NodeTemplate) -> ServerCommand:
variables = self._build_variables(rank)
rendered_env = self._render_envs(template.envs, rank, variables)
rendered_args = [
self._render_string(
arg,
rank=rank,
braced_variables=variables,
unbraced_variables=rendered_env,
allow_missing_unbraced=False,
)
for arg in template.server_cmd_template
]
cmd = ["vllm", "serve", self.config.model, *rendered_args]
env = {key: str(value) for key, value in rendered_env.items()}
display_cmd = format_server_cmd(cmd, env)
logger.info(
"External DP server command node=%s rank=%s: %s",
rank.node_index,
rank.local_rank,
display_cmd,
)
return ServerCommand(cmd=cmd, env=env, display_cmd=display_cmd)
def build_all(self, ranks: list[RankInfo]) -> list[ServerCommand]:
return [self.build(rank, self.config.launch_templates[rank.node_index]) for rank in ranks]
def _build_variables(self, rank: RankInfo) -> dict[str, str]:
return {
"MODEL": self.config.model,
"PORT_START": str(rank.port_start),
"PORT": str(rank.port),
"DP_SIZE": str(rank.dp_size),
"DP_SIZE_LOCAL": str(rank.dp_size_local),
"DP_RANK_START": str(rank.dp_rank - rank.local_rank),
"DP_RANK": str(rank.dp_rank),
"LOCAL_RANK": str(rank.local_rank),
"TP_SIZE": str(rank.tp_size),
"CP_SIZE": str(rank.cp_size),
"SP_SIZE": str(rank.sp_size),
"PP_SIZE": str(rank.pp_size),
"DP_ADDRESS": rank.dp_address,
"DP_RPC_PORT": str(rank.dp_rpc_port),
"VISIBLE_DEVICES": rank.visible_devices,
"NODE_INDEX": str(rank.node_index),
"CONFIG_INDEX": str(rank.node_index),
}
def _render_envs(
self,
envs: dict[str, Any],
rank: RankInfo,
variables: dict[str, str],
) -> dict[str, str]:
rendered_envs: dict[str, str] = {}
for key, value in envs.items():
if isinstance(value, str):
value = self._render_string(
value,
rank=rank,
braced_variables=variables,
unbraced_variables={**os.environ, **rendered_envs},
allow_missing_unbraced=True,
)
rendered_envs[str(key)] = str(value)
return rendered_envs
def _render_string(
self,
value: str,
*,
rank: RankInfo,
braced_variables: dict[str, str],
unbraced_variables: dict[str, str],
allow_missing_unbraced: bool,
) -> str:
value = replace_cluster_placeholders(
value,
cluster_ips=self.config.cluster_ips,
local_ip=rank.host,
current_node_index=rank.node_index,
)
value = self._render_variables(
value,
braced_variables,
pattern=TEMPLATE_VAR_RE,
allow_missing=False,
)
return self._render_variables(
value,
unbraced_variables,
pattern=ENV_VAR_RE,
allow_missing=allow_missing_unbraced,
)
@staticmethod
def _render_variables(
value: str,
variables: dict[str, str],
*,
pattern: re.Pattern[str],
allow_missing: bool,
) -> str:
def repl(match: re.Match[str]) -> str:
key = match.group(1)
if key not in variables:
if allow_missing:
return ""
raise KeyError(f"Unknown external DP template variable: {key}")
return variables[key]
return pattern.sub(repl, value)
class ExternalDPServerManager:
"""Start and stop the external DP ranks owned by the current node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.command_builder = ServerCommandBuilder(config)
self.dist_envs = build_dist_envs(
config.cluster_ips[current_node_index],
config.cluster_ips[0],
)
self.rank_processes: list[RankProcess] = []
def start_current_node(self) -> None:
local_ranks = [rank for rank in self.ranks if rank.node_index == self.current_node_index]
logger.info("Starting %d external DP ranks on node %d", len(local_ranks), self.current_node_index)
try:
for rank in local_ranks:
template = self.config.launch_templates[rank.node_index]
template = type(template)(
envs={**template.envs, **self.dist_envs},
server_cmd_template=template.server_cmd_template,
)
server_cmd = self.command_builder.build(rank, template)
log_file = self._rank_log_file(rank)
process = start_logged_process(server_cmd.cmd, server_cmd.env, log_file)
self.rank_processes.append((process, rank, log_file))
wait_ranks_ready(
local_ranks,
timeout=SERVER_READY_TIMEOUT_SECONDS,
rank_processes=self.rank_processes,
)
except Exception:
self.cleanup()
raise
def __enter__(self):
self.start_current_node()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
for process, rank, _log_file in reversed(self.rank_processes):
logger.info(
"Stopping external DP rank node=%d rank=%d pid=%d",
rank.node_index,
rank.local_rank,
process.pid,
)
terminate_process_tree(process.pid)
self.rank_processes.clear()
def _rank_log_file(self, rank: RankInfo) -> Path:
return self.log_root / f"node-{rank.node_index}" / f"rank-{rank.local_rank}.log"
class ExternalDPProxyLauncher:
"""Launch the external DP proxy on the configured proxy node."""
def __init__(
self,
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
current_node_index: int,
log_root: Path,
):
self.config = config
self.ranks = ranks
self.current_node_index = current_node_index
self.log_root = log_root
self.pid: int | None = None
def start(self) -> None:
if self.current_node_index != self.config.routing.proxy_node_index:
logger.info("Current node is not proxy node, skip launching external DP proxy")
return
cmd = build_proxy_server_cmd(self.config, self.ranks)
log_file = self.log_root / f"node-{self.current_node_index}" / "proxy.log"
process = start_logged_process(cmd, {}, log_file)
self.pid = process.pid
logger.info("External DP proxy launched: %s", proxy_server_health_url(self.config))
def wait_ready(self, timeout: int = 300) -> None:
wait_http_ready(proxy_server_health_url(self.config), timeout=timeout)
logger.info("External DP proxy ready: %s", proxy_server_health_url(self.config))
def __enter__(self):
self.start()
return self
def __exit__(self, exc_type, exc_value, traceback):
self.cleanup()
def cleanup(self) -> None:
if self.pid is None:
return
logger.info("Stopping external DP proxy pid=%d", self.pid)
terminate_process_tree(self.pid)
self.pid = None
def build_all_server_commands(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[ServerCommand]:
return ServerCommandBuilder(config).build_all(ranks)
def build_dist_envs(cur_ip: str, master_ip: str) -> dict[str, str]:
nic_name = get_net_interface(cur_ip)
return {
"HCCL_IF_IP": cur_ip,
"HCCL_SOCKET_IFNAME": nic_name,
"GLOO_SOCKET_IFNAME": nic_name,
"TP_SOCKET_IFNAME": nic_name,
"LOCAL_IP": cur_ip,
"NIC_NAME": nic_name,
"MASTER_IP": master_ip,
}
def build_proxy_server_cmd(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[str]:
routing = config.routing
cmd = [sys.executable, routing.proxy_script, "--host", routing.proxy_host, "--port", str(routing.proxy_port)]
if routing.type == ROUTING_GENERIC_DP:
worker_ranks = [rank for rank in ranks if rank.role == "worker"]
if not worker_ranks:
raise ValueError("generic_dp proxy requires worker ranks")
cmd.extend(["--dp-hosts", *[rank.host for rank in worker_ranks]])
cmd.extend(["--dp-ports", *[str(rank.port) for rank in worker_ranks]])
return cmd
if routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefiller_ranks = [rank for rank in ranks if rank.role == "prefiller"]
decoder_ranks = [rank for rank in ranks if rank.role == "decoder"]
if not prefiller_ranks or not decoder_ranks:
raise ValueError("disaggregated_prefill proxy requires prefiller and decoder ranks")
cmd.extend(["--prefiller-hosts", *[rank.host for rank in prefiller_ranks]])
cmd.extend(["--prefiller-ports", *[str(rank.port) for rank in prefiller_ranks]])
cmd.extend(["--decoder-hosts", *[rank.host for rank in decoder_ranks]])
cmd.extend(["--decoder-ports", *[str(rank.port) for rank in decoder_ranks]])
return cmd
raise ValueError(f"Unsupported routing.type: {routing.type}")
def proxy_server_health_url(config: ExternalDPConfig) -> str:
return f"http://{config.routing.proxy_host}:{config.routing.proxy_port}/healthcheck"
def rank_health_url(rank: RankInfo) -> str:
return f"http://{rank.host}:{rank.port}/health"
def master_rank_health_url(ranks: list[RankInfo]) -> str:
for rank in ranks:
if rank.node_index == 0 and rank.local_rank == 0:
return rank_health_url(rank)
raise RuntimeError("External DP master rank was not found")
def rank_label(rank: RankInfo) -> str:
return f"node={rank.node_index} rank={rank.local_rank} role={rank.role} url={rank_health_url(rank)}"
def format_http_status(label: str, url: str) -> str:
status = "ready" if is_http_ready(url, timeout=1.0) else "waiting"
return f"{label}={status} url={url}"
def _format_rank_statuses(
ranks: list[RankInfo],
rank_ready: dict[RankInfo, bool],
) -> str:
parts = []
for rank in ranks:
status = "ready" if rank_ready[rank] else "waiting"
parts.append(f" {rank_label(rank)} status={status}")
return "\n".join(parts)
def _raise_if_rank_process_exited(rank_processes: list[RankProcess] | None) -> None:
if not rank_processes:
return
exited = []
for process, rank, log_file in rank_processes:
returncode = process.poll()
if returncode is not None:
exited.append(f"{rank_label(rank)} pid={process.pid} returncode={returncode} log={log_file}")
if exited:
raise RuntimeError("External DP rank process exited before ready: " + "; ".join(exited))
def wait_ranks_ready(
ranks: Iterable[RankInfo],
timeout: int,
rank_processes: list[RankProcess] | None = None,
) -> None:
ranks = list(ranks)
rank_ready = {rank: False for rank in ranks}
deadline = time.monotonic() + timeout
last_log_time = 0.0
while True:
_raise_if_rank_process_exited(rank_processes)
all_ready = True
unhealthy_after_ready = []
for rank in ranks:
is_ready = is_http_ready(rank_health_url(rank), timeout=1.0)
if is_ready:
if not rank_ready[rank]:
logger.info("[READY] External DP rank %s", rank_label(rank))
rank_ready[rank] = True
continue
all_ready = False
if rank_ready[rank]:
unhealthy_after_ready.append(rank)
if unhealthy_after_ready:
failed = "; ".join(rank_label(rank) for rank in unhealthy_after_ready)
raise RuntimeError(f"External DP rank became unhealthy after ready: {failed}")
if all_ready:
return
now = time.monotonic()
if now - last_log_time >= 30:
logger.info(
"Polling external DP ranks: ready=%d/%d\n%s",
sum(rank_ready.values()),
len(ranks),
_format_rank_statuses(ranks, rank_ready),
)
last_log_time = now
if now >= deadline:
pending = [rank for rank in ranks if not rank_ready[rank]]
pending_labels = "; ".join(rank_label(rank) for rank in pending)
raise TimeoutError(f"Timed out waiting for external DP ranks ready: {pending_labels}")
time.sleep(5)
def wait_master_rank_stopped(ranks: list[RankInfo], timeout: int) -> None:
url = master_rank_health_url(ranks)
wait_http_ready(url, timeout=SERVER_READY_TIMEOUT_SECONDS)
logger.info("Hanging until master external DP rank stops: %s", url)
wait_http_unready(url, timeout=timeout)

View File

@@ -0,0 +1,163 @@
import logging
import os
import subprocess
import sys
import threading
import time
from collections.abc import Callable
from contextlib import contextmanager
from pathlib import Path
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ExternalDPConfig,
ExternalDPConfigLoader,
RankResolver,
resolve_current_node_index,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import (
ExternalDPProxyLauncher,
ExternalDPServerManager,
build_all_server_commands,
format_http_status,
master_rank_health_url,
proxy_server_health_url,
wait_master_rank_stopped,
wait_ranks_ready,
)
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
collect_logs,
write_benchmark_results_json,
)
from tools.aisbench import run_aisbench_cases
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
logger = logging.getLogger(__name__)
DEFAULT_LOG_ROOT = Path("/tmp/external_dp_logs")
def _install_special_dependencies(config: ExternalDPConfig) -> None:
for package, version in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{package}=={version}",
]
subprocess.call(command)
@contextmanager
def _heartbeat(
task_name: str,
*,
interval: int = 30,
status_fn: Callable[[], str] | None = None,
):
start_time = time.monotonic()
stop_event = threading.Event()
def report_progress() -> None:
while not stop_event.wait(interval):
elapsed = int(time.monotonic() - start_time)
status = ""
if status_fn is not None:
try:
status = f" {status_fn()}"
except Exception as exc: # pragma: no cover - diagnostic only
status = f" status_error={exc!r}"
logger.info("%s still running: elapsed=%ds%s", task_name, elapsed, status)
logger.info("%s started", task_name)
thread = threading.Thread(target=report_progress, daemon=True)
thread.start()
try:
yield
finally:
stop_event.set()
thread.join(timeout=1)
elapsed = int(time.monotonic() - start_time)
logger.info("%s finished: elapsed=%ds", task_name, elapsed)
def _format_benchmark_cases(config: ExternalDPConfig) -> str:
names = [str(case.get("case_name", "<unnamed>")) for case in config.benchmark_cases]
return ", ".join(names) if names else "<none>"
def _archive_rank_logs(log_root: Path, current_node_index: int) -> None:
log_prefix = os.environ.get("LOG_PREFIX")
if not log_prefix:
return
node_log_dir = log_root / f"node-{current_node_index}"
output_tar = Path(log_prefix) / f"node_{current_node_index}_external_dp_logs.tar.gz"
collect_logs(node_log_dir, output_tar)
def test_external_dp() -> None:
config = ExternalDPConfigLoader.from_yaml()
_install_special_dependencies(config)
ranks = RankResolver(config).resolve()
current_node_index = resolve_current_node_index(config)
log_root = Path(os.environ.get("EXTERNAL_DP_LOG_DIR", str(DEFAULT_LOG_ROOT)))
max_wait_seconds = int(os.environ.get("EXTERNAL_DP_MAX_WAIT_SECONDS", "3600"))
is_master = current_node_index == 0
server_manager = ExternalDPServerManager(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
proxy_launcher = ExternalDPProxyLauncher(
config=config,
ranks=ranks,
current_node_index=current_node_index,
log_root=log_root,
)
try:
with server_manager, proxy_launcher:
if is_master:
wait_ranks_ready(ranks, timeout=max_wait_seconds)
proxy_launcher.wait_ready()
target = f"http://{config.routing.proxy_host}:{config.routing.proxy_port}"
logger.info(
"Running AISBench cases: model=%s target=%s cases=[%s]",
config.model,
target,
_format_benchmark_cases(config),
)
with _heartbeat(
"Running AISBench",
status_fn=lambda: format_http_status("proxy", proxy_server_health_url(config)),
):
results = run_aisbench_cases(
model=config.model,
port=config.routing.proxy_port,
aisbench_cases=config.benchmark_cases,
host_ip=config.routing.proxy_host,
)
logger.info("AISBench completed: results=%d", len(results or []))
all_commands = build_all_server_commands(config, ranks)
write_benchmark_results_json(
config=config,
ranks=ranks,
commands=all_commands,
results=results,
)
wait_ranks_ready(ranks, timeout=30)
else:
master_url = master_rank_health_url(ranks)
with _heartbeat(
"Waiting for master external DP rank to stop",
status_fn=lambda: format_http_status("master", master_url),
):
wait_master_rank_stopped(ranks, timeout=max_wait_seconds)
finally:
_archive_rank_logs(log_root, current_node_index)

View File

@@ -0,0 +1,236 @@
import logging
import os
import shlex
import signal
import subprocess
import tarfile
import time
import urllib.error
import urllib.request
from pathlib import Path
from typing import TYPE_CHECKING, Any
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
ROUTING_DISAGGREGATED_PREFILL,
ExternalDPConfig,
RankInfo,
)
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
build_task_entry,
extract_hardware,
filter_environment,
get_vllm_version,
write_results_json,
)
logger = logging.getLogger(__name__)
if TYPE_CHECKING:
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import ServerCommand
SENSITIVE_ENV_TOKENS = ("TOKEN", "SECRET", "PASSWORD", "ACCESS_KEY")
def format_server_cmd(cmd: list[str], env: dict[str, str] | None = None) -> str:
env_parts: list[str] = []
for key, value in sorted((env or {}).items()):
display_value = "***" if any(token in key.upper() for token in SENSITIVE_ENV_TOKENS) else str(value)
env_parts.append(f"{key}={shlex.quote(display_value)}")
return " ".join([*env_parts, shlex.join(cmd)])
def start_logged_process(cmd: list[str], env: dict[str, str], log_file: Path) -> subprocess.Popen:
log_file.parent.mkdir(parents=True, exist_ok=True)
merged_env = {**os.environ, **env}
with log_file.open("ab") as f:
f.write(f"Starting command: {format_server_cmd(cmd, env)}\n".encode())
f.flush()
return subprocess.Popen(
cmd,
stdout=f,
stderr=subprocess.STDOUT,
env=merged_env,
start_new_session=True,
)
def terminate_process_tree(pid: int, timeout: int = 30) -> None:
try:
import psutil
except ModuleNotFoundError:
try:
os.killpg(pid, signal.SIGTERM)
except ProcessLookupError:
return
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
try:
os.kill(pid, 0)
except ProcessLookupError:
return
time.sleep(0.2)
try:
os.killpg(pid, signal.SIGKILL)
except ProcessLookupError:
return
return
try:
parent = psutil.Process(pid)
except psutil.NoSuchProcess:
return
children = parent.children(recursive=True)
for process in children:
process.terminate()
parent.terminate()
gone, alive = psutil.wait_procs([parent, *children], timeout=timeout)
del gone
for process in alive:
process.kill()
def is_http_ready(url: str, timeout: float = 5.0) -> bool:
try:
with urllib.request.urlopen(url, timeout=timeout) as response:
return 200 <= response.status < 300
except (urllib.error.URLError, TimeoutError, OSError):
return False
def wait_http_ready(url: str, timeout: int, interval: float = 2.0) -> None:
deadline = time.monotonic() + timeout
last_error: Exception | None = None
while time.monotonic() < deadline:
try:
with urllib.request.urlopen(url, timeout=5) as response:
if 200 <= response.status < 300:
return
except (urllib.error.URLError, TimeoutError, OSError) as exc:
last_error = exc
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP ready: {url}; last_error={last_error}")
def wait_http_unready(url: str, timeout: int, interval: float = 5.0) -> None:
deadline = time.monotonic() + timeout
while time.monotonic() < deadline:
if not is_http_ready(url):
return
time.sleep(interval)
raise TimeoutError(f"Timed out waiting for HTTP unready: {url}")
def collect_logs(src_dir: Path, output_tar: Path) -> None:
if not src_dir.exists():
return
output_tar.parent.mkdir(parents=True, exist_ok=True)
with tarfile.open(output_tar, "w:gz") as tar:
tar.add(src_dir, arcname=src_dir.name)
def _common_command_envs(commands: list["ServerCommand"]) -> dict[str, str]:
if not commands:
return {}
common_keys = set(commands[0].env)
for command in commands[1:]:
common_keys.intersection_update(command.env)
common_envs: dict[str, str] = {}
for key in sorted(common_keys):
values = {command.env[key] for command in commands}
if len(values) == 1:
common_envs[key] = next(iter(values))
return common_envs
def _extract_dtype(config: ExternalDPConfig, commands: list["ServerCommand"]) -> str:
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = any("--quantization ascend" in command.display_cmd for command in commands)
return "w8a8" if has_w8a8 and has_quant_ascend else "bf16"
def _extract_features(commands: list["ServerCommand"]) -> list[str]:
if not commands:
return []
features: list[str] = []
command_args = [command.cmd for command in commands]
command_displays = [" ".join(shlex.quote(arg) for arg in command.cmd) for command in commands]
if any("--async-scheduling" in cmd for cmd in command_args):
features.append("async_scheduling")
if any("--enable-expert-parallel" in cmd for cmd in command_args):
features.append("expert_parallel")
if any("--speculative-config" in cmd for cmd in command_args):
features.append("speculative")
if any("cudagraph_mode" in display for display in command_displays):
features.append("aclgraph")
feature_envs = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
for env_key, feature_name in feature_envs.items():
values = [str(command.env.get(env_key, "0")) for command in commands]
if any(value not in ("0", "", "false", "False") for value in values):
features.append(feature_name)
return features
def _build_serve_cmd(
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
) -> dict[str, Any]:
entries: dict[str, str] = {}
for rank, command in zip(ranks, commands):
prefix = rank.role
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL:
prefix = "prefill" if rank.role == "prefiller" else "decode"
entries[f"{prefix}-node{rank.node_index}-rank{rank.local_rank}"] = command.display_cmd
key = "external_dp_pd" if config.routing.type == ROUTING_DISAGGREGATED_PREFILL else "external_dp"
return {key: entries}
def build_benchmark_results(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
) -> dict[str, Any]:
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
tasks = [build_task_entry(key, case, result) for (key, case), result in zip(valid_items, results)]
runner = os.environ.get("VLLM_CI_RUNNER", "")
common_envs = _common_command_envs(commands)
return {
"model_name": config.model,
"hardware": extract_hardware(runner),
"dtype": _extract_dtype(config, commands),
"feature": _extract_features(commands),
"vllm_version": get_vllm_version(),
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config, ranks, commands),
"environment": filter_environment(common_envs),
}
def write_benchmark_results_json(
*,
config: ExternalDPConfig,
ranks: list[RankInfo],
commands: list["ServerCommand"],
results: list[Any],
output_dir: Path | None = None,
) -> Path:
output = build_benchmark_results(config=config, ranks=ranks, commands=commands, results=results)
job_name = os.environ.get("BENCHMARK_JOB_NAME", "") or config.test_name.replace(" ", "-")
return write_results_json(output, job_name=job_name, output_dir=output_dir)

View File

@@ -0,0 +1,196 @@
test_name: "test DeepSeek-R1-W8A8 disaggregated_prefill"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 10
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_DETERMINISTIC: True
TASK_QUEUE_ENABLE: 1
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0, L2:0"
DYNAMIC_EPLB: true
VLLM_ENGINE_READY_TIMEOUT_S: 3000
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 4
--max-model-len 36864
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 4
--max-model-len 36864
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 32
--data-parallel-size-local 16
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 1
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 28
--max-model-len 36864
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"multistream_overlap_shared_expert":true,"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--headless
--data-parallel-size 32
--data-parallel-size-local 16
--data-parallel-start-rank 16
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 1
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 28
--max-model-len 36864
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.9
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 32,
"tp_size": 1
}
}
}'
--additional-config
'{"multistream_overlap_shared_expert":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 512
baseline: 95
threshold: 5

View File

@@ -0,0 +1,114 @@
test_name: "test DeepSeek-R1-W8A8-longseq disaggregated_prefill"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 768
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_DETERMINISTIC: True
TASK_QUEUE_ENABLE: 1
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0"
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
VLLM_ENGINE_READY_TIMEOUT_S: 3000
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 1
--decode-context-parallel-size 8
--prefill-context-parallel-size 2
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--enforce-eager
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 32
--max-model-len 32768
--max-num-batched-tokens 16384
--trust-remote-code
--gpu-memory-utilization 0.85
--enable-chunked-prefill
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--decode-context-parallel-size 2
--prefill-context-parallel-size 1
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--enable-expert-parallel
--seed 1024
--quantization ascend
--max-num-seqs 32
--max-model-len 32768
--max-num-batched-tokens 256
--trust-remote-code
--gpu-memory-utilization 0.85
--compilation_config '{"cudagraph_capture_sizes":[4,8,16,32],"cudagraph_mode": "FULL_DECODE_ONLY"}'
--enable-chunked-prefill
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
--additional-config '{"recompute_scheduler_enable":true}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
num_prompts: 360
max_out_len: 4096
batch_size: 32
baseline: 95
threshold: 5

View File

@@ -0,0 +1,85 @@
test_name: "test DeepSeek-V3.1-BF16 on A3"
model: "unsloth/DeepSeek-V3.1-BF16"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 2048
SERVER_PORT: 8080
OMP_PROC_BIND: false
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
OMP_NUM_THREADS: 1
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_BALANCE_SCHEDULING: 1
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve unsloth/DeepSeek-V3.1-BF16
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--tensor-parallel-size 8
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13399
--no-enable-prefix-caching
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 4096
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.95
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
--additional_config '{"enable_multistream_moe": true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve unsloth/DeepSeek-V3.1-BF16
--headless
--data-parallel-size 4
--tensor-parallel-size 8
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13399
--no-enable-prefix-caching
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 4096
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.95
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
--additional_config '{"enable_multistream_moe": true}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 512
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 512
baseline: 95
threshold: 10

View File

@@ -0,0 +1,127 @@
test_name: "test DeepSeek-V3.2-W8A8 on A3"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
ASCEND_A3_EBA_ENABLE: 1
VLLM_ENGINE_READY_TIMEOUT_S: 3000
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13399
--tensor-parallel-size 8
--quantization ascend
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 90000
--max-num-batched-tokens 4096
--no-enable-prefix-caching
--gpu-memory-utilization 0.85
--trust-remote-code
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
-
envs:
<<: *env_common
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--headless
--data-parallel-size 4
--data-parallel-rpc-port 13399
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--tensor-parallel-size 8
--quantization ascend
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 90000
--max-num-batched-tokens 4096
--no-enable-prefix-caching
--gpu-memory-utilization 0.85
--trust-remote-code
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
benchmarks:
perf_short_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 3000
batch_size: 512
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_long_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 3000
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_short:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 3000
batch_size: 256
request_rate: 11.2
baseline: 305.2903
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 128
baseline: 95
threshold: 10
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 80000
batch_size: 32
baseline: 57
threshold: 10

View File

@@ -0,0 +1,268 @@
test_name: "test DeepSeek-V3.2-W8A8-EP disaggregated_prefill"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 10
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: 360
VLLM_TORCH_PROFILER_WITH_STACK: 0
ASCEND_AGGREGATE_ENABLE: 1
ASCEND_TRANSPORT_PRINT: 1
ACL_OP_INIT_MODE: 1
ASCEND_A3_ENABLE: 1
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
VLLM_ENGINE_READY_TIMEOUT_S: 3000
HCCL_CONNECT_TIMEOUT: 1200
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
deployment:
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-start-rank 0
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 16
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 133000
--max-num-batched-tokens 8192
--trust-remote-code
--gpu-memory-utilization 0.90
--enforce-eager
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-start-rank 1
--data-parallel-size-local 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 16
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 133000
--max-num-batched-tokens 8192
--trust-remote-code
--gpu-memory-utilization 0.90
--enforce-eager
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 1100
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-model-len 133000
--max-num-batched-tokens 42
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
--trust-remote-code
--max-num-seqs 14
--gpu-memory-utilization 0.90
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 1100
server_cmd: >
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
--seed 1024
--quantization ascend
--max-model-len 133000
--max-num-batched-tokens 42
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
--trust-remote-code
--max-num-seqs 14
--gpu-memory-utilization 0.90
--no-enable-prefix-caching
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
--tokenizer-mode deepseek_v32
--reasoning-parser deepseek_v3
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 16
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
benchmarks:
perf_short_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1500
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_long_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1024
batch_size: 1
request_rate: 11.2
baseline: 1
threshold: 0.97
perf_short:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1
threshold: 0.97
perf_long:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1024
batch_size: 4
request_rate: 1
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 64
baseline: 96.88
threshold: 10

View File

@@ -0,0 +1,102 @@
test_name: "multi-node-GLM-5.1-W8A8C8-MTP-A3_64k/128k"
model: "Eco-Tech/GLM-5.1-w8a8c8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
VLLM_USE_MODELSCOPE: "true"
OMP_NUM_THREADS: "1"
HCCL_BUFFSIZE: "400"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
SERVER_PORT: 8077
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8c8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--data-parallel-rpc-port 12981
--tensor-parallel-size 4
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
--seed 1024
--tool-call-parser glm47
--reasoning-parser glm45
--enable-auto-tool-choice
--max-num-seqs 6
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.92
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8c8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--data-parallel-rpc-port 12981
--tensor-parallel-size 4
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
--seed 1024
--tool-call-parser glm47
--reasoning-parser glm45
--enable-auto-tool-choice
--max-num-seqs 6
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.9
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
benchmarks:
perf_128k_warmup:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 1
max_out_len: 1
batch_size: 1
request_rate: 0
baseline: 0
threshold: 0.97
perf_128k:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 256
max_out_len: 1024
batch_size: 64
request_rate: 0
baseline: 290.8308
threshold: 0.97

View File

@@ -0,0 +1,84 @@
test_name: "multi-node-GLM-5.1-w8a8-A2"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: 0
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_USE_MODELSCOPE: true
VLLM_ENGINE_READY_TIMEOUT_S: 3000
VLLM_RPC_TIMEOUT: 600
SERVER_PORT: 8078
special_dependencies:
transformers: "5.2.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 8
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 64
--max-model-len 38000
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--no-enable-prefix-caching
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 8
--data-parallel-size-local 1
--data-parallel-start-rank 1
--headless
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--enable-expert-parallel
--seed 1024
--max-num-seqs 64
--max-model-len 38000
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--no-enable-prefix-caching
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 72
max_out_len: 1500
batch_size: 18
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,98 @@
test_name: "multi-node-GLM-5.1-w8a8-A3"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
VLLM_USE_MODELSCOPE: true
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_BALANCE_SCHEDULING: 0
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "600"
SERVER_PORT: 8080
special_dependencies:
transformers: "5.2.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-rpc-port 13389
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--enable-chunked-prefill
--enable-prefix-caching
--async-scheduling
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 72348
batch_size: 32
baseline: 90
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 12
max_out_len: 1024
batch_size: 3
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,240 @@
test_name: "multi-node-GLM-5.1-w8a8-EP"
model: "Eco-Tech/GLM-5.1-w8a8"
num_nodes: 4
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: 1024
VLLM_TORCH_PROFILER_WITH_STACK: 0
ASCEND_AGGREGATE_ENABLE: 1
ASCEND_TRANSPORT_PRINT: 1
ACL_OP_INIT_MODE: 1
ASCEND_A3_ENABLE: 1
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
HCCL_CONNECT_TIMEOUT: "1200"
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
VLLM_ASCEND_ENABLE_FUSED_MC2: 1
special_dependencies:
transformers: "5.2.0"
disaggregated_prefill:
enabled: true
prefiller_host_index: [0, 1]
decoder_host_index: [2, 3]
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--tensor-parallel-size 8
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 131072
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
--max-num-batched-tokens 4096
--trust-remote-code
--max-num-seqs 64
--quantization ascend
--gpu-memory-utilization 0.95
--enforce-eager
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--tensor-parallel-size 8
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 131072
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
--max-num-batched-tokens 4096
--trust-remote-code
--max-num-seqs 64
--quantization ascend
--gpu-memory-utilization 0.95
--enforce-eager
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 10543
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 202752
--max-num-batched-tokens 32
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
--trust-remote-code
--max-num-seqs 8
--gpu-memory-utilization 0.92
--quantization ascend
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
VLLM_ASCEND_ENABLE_MLAPO: 1
TASK_QUEUE_ENABLE: 1
server_cmd: >
vllm serve Eco-Tech/GLM-5.1-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 8
--data-parallel-size-local 4
--data-parallel-start-rank 4
--headless
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 10543
--tensor-parallel-size 4
--enable-expert-parallel
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
--profiler-config
'{"profiler": "torch",
"torch_profiler_dir": "./vllm_profile",
"torch_profiler_with_stack": false}'
--seed 1024
--max-model-len 202752
--max-num-batched-tokens 32
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
--trust-remote-code
--max-num-seqs 8
--gpu-memory-utilization 0.92
--quantization ascend
--enable-auto-tool-choice
--tool-call-parser glm47
--reasoning-parser glm45
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"use_ascend_direct": true,
"prefill": {
"dp_size": 4,
"tp_size": 8
},
"decode": {
"dp_size": 8,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 160
max_out_len: 1500
batch_size: 40
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,91 @@
test_name: "multi-node-GLM-5.2-w8a8-A3"
model: "Eco-Tech/GLM-5.2-w8a8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
VLLM_USE_MODELSCOPE: true
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 200
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
VLLM_RPC_TIMEOUT: "600"
SERVER_PORT: 8080
special_dependencies:
transformers: "5.12.0"
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
deployment:
- envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.2-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-rpc-port 13389
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/GLM-5.2-w8a8
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--tensor-parallel-size 16
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-rpc-port 13389
--headless
--data-parallel-address $MASTER_IP
--enable-expert-parallel
--seed 1024
--max-num-seqs 16
--max-model-len 133120
--max-num-batched-tokens 4096
--trust-remote-code
--gpu-memory-utilization 0.95
--quantization ascend
--additional-config '{"multistream_overlap_shared_expert":true}'
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 72348
batch_size: 32
baseline: 90
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 12
max_out_len: 1024
batch_size: 3
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,90 @@
test_name: "test Kimi-K2.5-W4A8 A2 dual nodes"
model: "Eco-Tech/Kimi-K2.5-W4A8"
num_nodes: 2
npu_per_node: 8
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
HCCL_INTRA_PCIE_ENABLE: 1
HCCL_INTRA_ROCE_ENABLE: 0
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
TASK_QUEUE_ENABLE: 1
HCCL_BUFFSIZE: 512
VLLM_ASCEND_ENABLE_MLAPO: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
VLLM_USE_MODELSCOPE: true
SERVER_PORT: 8080
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/Kimi-K2.5-W4A8
--host 0.0.0.0
--port $SERVER_PORT
--quantization ascend
--allowed-local-media-path /
--trust-remote-code
--no-enable-prefix-caching
--seed 42
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--enable-expert-parallel
--max-num-seqs 64
--max-model-len 51200
--max-num-batched-tokens 8192
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
--mm-processor-cache-gb 0
--mm-encoder-tp-mode data
-
envs:
<<: *env_common
server_cmd: >
vllm serve Eco-Tech/Kimi-K2.5-W4A8
--host 0.0.0.0
--headless
--port $SERVER_PORT
--quantization ascend
--allowed-local-media-path /
--trust-remote-code
--no-enable-prefix-caching
--seed 42
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--enable-expert-parallel
--max-num-seqs 64
--max-model-len 51200
--max-num-batched-tokens 8192
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
--mm-processor-cache-gb 0
--mm-encoder-tp-mode data
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 320
max_out_len: 1500
batch_size: 80
trust_remote_code: True
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,74 @@
test_name: "test Qwen3-235B-A22B multi-dp on A2"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 8
env_common: &env_common
VLLM_USE_MODELSCOPE: true
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
TASK_QUEUE_ENABLE: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 128
--max-model-len 40960
--max-num-batched-tokens 2048
--trust-remote-code
--gpu-memory-utilization 0.9
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--headless
--data-parallel-size 2
--data-parallel-size-local 1
--data-parallel-start-rank 1
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 128
--max-model-len 40960
--max-num-batched-tokens 2048
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 256
request_rate: 4.8
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 256
baseline: 96
threshold: 10

View File

@@ -0,0 +1,77 @@
test_name: "test Qwen3-235B-A22B multi-dp"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
TASK_QUEUE_ENABLE: 1
VLLM_USE_MODELSCOPE: true
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--headless
--data-parallel-size 4
--data-parallel-size-local 2
--data-parallel-start-rank 2
--data-parallel-address $MASTER_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 512
baseline: 95
threshold: 3

View File

@@ -0,0 +1,93 @@
test_name: "test Qwen3-235B-A22B-W8A8 EPLB"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
DYNAMIC_EPLB: true
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":50,"algorithm_execution_interval":5}}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":600,"algorithm_execution_interval":50}}'
benchmarks:

View File

@@ -0,0 +1,100 @@
test_name: "test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
DYNAMIC_EPLB: true
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 1
--decode-context-parallel-size 2
--prefill-context-parallel-size 2
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--seed 1024
--enforce-eager
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"dynamic_eplb":true}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--decode-context-parallel-size 2
--prefill-context-parallel-size 1
--tensor-parallel-size 8
--cp-kv-cache-interleave-size 128
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--compilation_config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 1,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
--additional-config
'{"dynamic_eplb":true}'
benchmarks:

View File

@@ -0,0 +1,89 @@
test_name: "test Qwen3-235B-A22B-W8A8 disaggregated_prefill"
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
HCCL_OP_EXPANSION_MODE: AIV
VLLM_USE_MODELSCOPE: true
TASK_QUEUE_ENABLE: 1
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
NUMEXPR_MAX_THREADS: 128
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--quantization ascend
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--quantization ascend
--max-num-seqs 16
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 2,
"tp_size": 8
}
}
}'
benchmarks:

View File

@@ -0,0 +1,118 @@
test_name: "test Qwen3-235B-A22B disaggregated_prefill"
model: "Qwen/Qwen3-235B-A22B"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
HCCL_BUFFSIZE: 1024
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
VLLM_ASCEND_ENABLE_FUSED_MC2: 2
TASK_QUEUE_ENABLE: 1
SERVER_PORT: 8080
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 8
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
--no-enable-prefix-caching
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-235B-A22B"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 4
--data-parallel-start-rank 0
--data-parallel-address $LOCAL_IP
--data-parallel-rpc-port 13389
--tensor-parallel-size 4
--seed 1024
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--enable-expert-parallel
--trust-remote-code
--gpu-memory-utilization 0.9
--no-enable-prefix-caching
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30100",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 2800
max_out_len: 1500
batch_size: 700
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 7680
batch_size: 512
baseline: 97
threshold: 10

View File

@@ -0,0 +1,108 @@
test_name: "test Qwen3-VL-235B-A22B disaggregated_prefill"
model: "Qwen/Qwen3-VL-235B-A22B-Instruct"
num_nodes: 2
npu_per_node: 16
env_common: &env_common
VLLM_USE_MODELSCOPE: true
HCCL_BUFFSIZE: 1024
SERVER_PORT: 8080
OMP_PROC_BIND: false
OMP_NUM_THREADS: 1
HCCL_OP_EXPANSION_MODE: "AIV"
TASK_QUEUE_ENABLE: 1
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
disaggregated_prefill:
enabled: true
prefiller_host_index: [0]
decoder_host_index: [1]
deployment:
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 2
--data-parallel-size-local 2
--tensor-parallel-size 8
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_producer",
"kv_port": "30000",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
-
envs:
<<: *env_common
server_cmd: >
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
--host 0.0.0.0
--port $SERVER_PORT
--data-parallel-size 4
--data-parallel-size-local 4
--tensor-parallel-size 4
--seed 1024
--enable-expert-parallel
--max-num-seqs 32
--max-model-len 8192
--max-num-batched-tokens 8192
--trust-remote-code
--no-enable-prefix-caching
--gpu-memory-utilization 0.9
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
--kv-transfer-config
'{"kv_connector": "MooncakeConnectorV1",
"kv_role": "kv_consumer",
"kv_port": "30200",
"kv_connector_extra_config": {
"prefill": {
"dp_size": 2,
"tp_size": 8
},
"decode": {
"dp_size": 4,
"tp_size": 4
}
}
}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/textvqa-perf-1080p
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
num_prompts: 2800
max_out_len: 1500
batch_size: 64
request_rate: 11.2
baseline: 1
threshold: 0.97
acc:
case_type: accuracy
dataset_path: vllm-ascend/textvqa-lite
request_conf: vllm_api_stream_chat
dataset_conf: textvqa/textvqa_gen_base64
max_out_len: 7680
batch_size: 64
baseline: 85
threshold: 5

View File

@@ -0,0 +1,331 @@
import logging
import os
import subprocess
from dataclasses import dataclass
from typing import Any
import regex as re
from tests.e2e.nightly.multi_node.scripts.utils import (
get_available_port,
get_net_interface,
load_yaml_mapping,
resolve_cluster_ips,
resolve_current_node_index,
setup_logger,
)
setup_logger()
logger = logging.getLogger(__name__)
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
DEFAULT_SERVER_PORT = 8080
@dataclass(frozen=True)
class NodeInfo:
index: int
ip: str
server_cmd: str
envs: dict[str, Any] | None = None
headless: bool = False
def __post_init__(self):
if not self.ip:
raise ValueError("NodeInfo.ip must not be empty")
def __str__(self) -> str:
return f"NodeInfo(\n index={self.index},\n ip={self.ip},\n headless={self.headless},\n)"
class DisaggregatedPrefillCfg:
def __init__(self, raw_cfg: dict, num_nodes: int):
self.prefiller_indices: list[int] = raw_cfg.get("prefiller_host_index", [])
self.decoder_indices: list[int] = raw_cfg.get("decoder_host_index", [])
if not self.decoder_indices:
raise RuntimeError("decoder_host_index must be provided")
self._validate(num_nodes)
self.decode_start_index = self.decoder_indices[0]
self.num_prefillers = len(self.prefiller_indices)
self.num_decoders = len(self.decoder_indices)
def _validate(self, num_nodes: int):
overlap = set(self.prefiller_indices) & set(self.decoder_indices)
if overlap:
raise AssertionError(f"Prefiller and decoder overlap: {overlap}")
all_indices = self.prefiller_indices + self.decoder_indices
if any(i >= num_nodes for i in all_indices):
raise ValueError("Disaggregated prefill index out of range")
def is_prefiller(self, index: int) -> bool:
return index in self.prefiller_indices
def is_decoder(self, index: int) -> bool:
return index in self.decoder_indices
def master_ip_for_node(self, index: int, nodes: list[NodeInfo]) -> str:
if self.is_prefiller(index):
return nodes[0].ip
return nodes[self.decode_start_index].ip
class DistEnvBuilder:
def __init__(
self,
*,
cur_node: NodeInfo,
master_ip: str,
):
self.cur_ip = cur_node.ip
self.nic_name = get_net_interface(self.cur_ip)
self.master_ip = master_ip
self.base_envs = dict(cur_node.envs or {})
def build(self) -> dict:
envs = dict(self.base_envs)
envs.update(
{
"HCCL_IF_IP": self.cur_ip,
"HCCL_SOCKET_IFNAME": self.nic_name,
"GLOO_SOCKET_IFNAME": self.nic_name,
"TP_SOCKET_IFNAME": self.nic_name,
"LOCAL_IP": self.cur_ip,
"NIC_NAME": self.nic_name,
"MASTER_IP": self.master_ip,
}
)
return {k: str(v) for k, v in envs.items()}
class ProxyLauncher:
def __init__(
self,
*,
nodes: list[NodeInfo],
envs: dict,
proxy_port: int,
cur_index: int,
disagg_cfg: DisaggregatedPrefillCfg | None = None,
):
self.nodes = nodes
self.cfg = disagg_cfg
self.server_port = envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
self.proxy_port = proxy_port
self.proxy_script = envs.get(
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
"examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
)
self.envs = envs
self.is_master = cur_index == 0
self.cur_ip = nodes[cur_index].ip
self.process: subprocess.Popen[bytes] | None = None
def __enter__(self):
if not self.is_master or self.cfg is None:
logger.info("Not launching proxy on non-master node")
return self
prefiller_ips = [self.nodes[i].ip for i in self.cfg.prefiller_indices if not self.nodes[i].headless]
decoder_ips = [self.nodes[i].ip for i in self.cfg.decoder_indices if not self.nodes[i].headless]
cmd = [
"python",
self.proxy_script,
"--host",
self.cur_ip,
"--port",
str(self.proxy_port),
"--prefiller-hosts",
*prefiller_ips,
"--prefiller-ports",
*[str(self.server_port)] * len(prefiller_ips),
"--decoder-hosts",
*decoder_ips,
"--decoder-ports",
*[str(self.server_port)] * len(decoder_ips),
]
logger.info("Launching proxy: %s", " ".join(cmd))
self.process = subprocess.Popen(cmd, env={**os.environ, **self.envs})
return self
def __exit__(self, exc_type, exc, tb):
if not self.process:
return
logger.info("Stopping proxy server...")
self.process.terminate()
try:
self.process.wait(timeout=5)
except subprocess.TimeoutExpired:
self.process.kill()
class MultiNodeConfig:
def __init__(
self,
*,
model: str,
test_name: str,
nodes: list[NodeInfo],
npu_per_node: int,
disaggregated_prefill: dict | None,
benchmark_cases: list[dict],
special_dependencies: dict,
):
self.model = model
self.test_name = test_name
self.nodes = nodes
self.npu_per_node = npu_per_node
self.benchmark_cases = benchmark_cases
self.cur_index = self._resolve_cur_index()
self.cur_node = self.nodes[self.cur_index]
self.special_dependencies = special_dependencies
self.disagg_cfg = DisaggregatedPrefillCfg(disaggregated_prefill, len(nodes)) if disaggregated_prefill else None
master_ip = (
self.disagg_cfg.master_ip_for_node(self.cur_index, self.nodes) if self.disagg_cfg else self.nodes[0].ip
)
self.proxy_port = get_available_port()
self.envs = DistEnvBuilder(
cur_node=self.cur_node,
master_ip=master_ip,
).build()
logger.info("Node %d envs: %s", self.cur_index, self.envs)
self.server_cmd = self._expand_env(self.cur_node.server_cmd)
def _resolve_cur_index(self) -> int:
return resolve_current_node_index([node.ip for node in self.nodes])
def _expand_env(self, cmd: str) -> str:
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
def repl(m):
key = m.group(1) or m.group(2)
return self.envs.get(key, m.group(0))
return pattern.sub(repl, cmd)
@property
def world_size(self) -> int:
return len(self.nodes) * self.npu_per_node
@property
def is_master(self) -> bool:
return self.cur_index == 0
@property
def server_port(self) -> int:
return self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
@property
def master_ip(self) -> str:
return self.nodes[0].ip
@property
def benchmark_endpoint(self) -> tuple[str, int]:
"""
Endpoint used by benchmark clients.
"""
master_ip = self.nodes[0].ip
server_port = self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
if self.disagg_cfg:
return master_ip, self.proxy_port
return master_ip, server_port
class MultiNodeConfigLoader:
"""Load MultiNodeConfig from yaml file."""
DEFAULT_CONFIG_NAME = "DeepSeek-V3.yaml"
@classmethod
def from_yaml(cls, yaml_path: str | None = None) -> MultiNodeConfig:
config = cls._load_yaml(yaml_path)
cls._validate_root(config)
nodes = cls._parse_nodes(config)
benchmarks = cls._parse_benchmarks(config)
return MultiNodeConfig(
model=config["model"],
test_name=config.get("test_name", "untitled_test"),
nodes=nodes,
npu_per_node=config.get("npu_per_node", 16),
disaggregated_prefill=config.get("disaggregated_prefill"),
special_dependencies=config.get("special_dependencies", {}),
benchmark_cases=list(benchmarks.values()),
)
@classmethod
def _load_yaml(cls, yaml_path: str | None) -> dict:
return load_yaml_mapping(
yaml_path,
default_name=cls.DEFAULT_CONFIG_NAME,
default_base_path=DEFAULT_CONFIG_BASE_PATH,
description="config",
)
@staticmethod
def _validate_root(cfg: dict):
required = ["model", "deployment", "num_nodes", "npu_per_node", "benchmarks"]
missing = [k for k in required if k not in cfg]
if missing:
raise KeyError(f"Missing required config fields: {missing}")
@classmethod
def _parse_nodes(cls, cfg: dict) -> list[NodeInfo]:
num_nodes = cfg["num_nodes"]
deployments = cfg["deployment"]
if len(deployments) != num_nodes:
raise AssertionError(f"deployment size ({len(deployments)}) != num_nodes ({num_nodes})")
for idx, deploy in enumerate(deployments):
if deploy.get("envs") is None:
raise KeyError(f"deployment[{idx}].envs is required for multi-node configs")
cluster_ips = cls._resolve_cluster_ips(cfg, num_nodes)
nodes: list[NodeInfo] = []
for idx, deploy in enumerate(deployments):
cmd = deploy.get("server_cmd", "")
envs = deploy["envs"]
nodes.append(
NodeInfo(
index=idx,
ip=cluster_ips[idx],
server_cmd=cmd,
envs=envs,
headless="--headless" in cmd,
)
)
return nodes
@staticmethod
def _parse_benchmarks(cfg: dict) -> dict:
benchmarks = cfg.get("benchmarks") or {}
for name, case in benchmarks.items():
case["case_name"] = name
return benchmarks
@staticmethod
def _resolve_cluster_ips(cfg: dict, num_nodes: int) -> list[str]:
return resolve_cluster_ips(
cfg,
num_nodes,
cluster_hosts_log_message=(
"Using cluster_hosts from config. This typically indicates that your current environment is a "
"non-Kubernetes environment."
),
dns_log_message="Resolving cluster IPs via DNS...",
)

View File

@@ -0,0 +1,207 @@
import json
import logging
import os
import shlex
import subprocess
import sys
from typing import Any
import pytest
import vllm
from tests.e2e.conftest import RemoteOpenAIServer
from tests.e2e.nightly.multi_node.internal_dp.scripts.multi_node_config import (
MultiNodeConfig,
MultiNodeConfigLoader,
ProxyLauncher,
)
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
build_task_entry,
extract_hardware,
filter_environment,
write_results_json,
)
from tools.aisbench import run_aisbench_cases
logger = logging.getLogger(__name__)
_FEATURE_ENVS: dict[str, str] = {
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
}
def _extract_dtype(config: MultiNodeConfig) -> str:
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and any node uses --quantization ascend."""
has_w8a8 = "w8a8" in config.model.lower()
has_quant_ascend = any("--quantization ascend" in node.server_cmd for node in config.nodes)
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
def _cmd_to_list(server_cmd: list[str] | str) -> list[str]:
"""Normalize server_cmd to a list of argument strings."""
if isinstance(server_cmd, str):
try:
return shlex.split(server_cmd)
except ValueError:
return server_cmd.split()
return list(server_cmd)
def _extract_server_cmd_value(cmd_list: list[str], flag: str) -> str | None:
"""Return the value following `flag` in a command list, or None."""
try:
idx = cmd_list.index(flag)
return cmd_list[idx + 1]
except (ValueError, IndexError):
return None
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
"""Extract and JSON-parse the value following `flag` in a command list."""
val = _extract_server_cmd_value(cmd_list, flag)
if not val:
return {}
try:
return json.loads(val)
except (json.JSONDecodeError, ValueError):
return {}
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
"""Extract enabled feature names from server_cmd and environment variables."""
cmd_list = _cmd_to_list(server_cmd)
features: list[str] = []
# Features from --additional-config JSON
additional = _parse_json_flag(cmd_list, "--additional-config")
if additional.get("enable_weight_nz_layout"):
features.append("weight_nz_layout")
wp = additional.get("weight_prefetch_config") or {}
if isinstance(wp, dict) and wp.get("enabled"):
features.append("weight_prefetch")
tc = additional.get("torchair_graph_config") or {}
if isinstance(tc, dict) and tc.get("enabled"):
features.append("torchair_graph")
asc = additional.get("ascend_scheduler_config") or {}
if isinstance(asc, dict) and asc.get("enabled"):
features.append("ascend_scheduler")
# Features from --compilation-config JSON
compilation = _parse_json_flag(cmd_list, "--compilation-config")
if compilation.get("cudagraph_mode"):
features.append("aclgraph")
# Features from --speculative-config JSON
speculative = _parse_json_flag(cmd_list, "--speculative-config")
if speculative:
features.append(speculative.get("method", "speculative"))
# Features from direct flags
if "--enable-expert-parallel" in cmd_list:
features.append("expert_parallel")
# Features from environment variables
for env_key, feature_name in _FEATURE_ENVS.items():
val = str(envs.get(env_key, "0"))
if val not in ("0", "", "false", "False"):
features.append(feature_name)
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
features.append("flashcomm2")
return features
def _build_serve_cmd(config: MultiNodeConfig) -> dict[str, Any]:
"""Build serve_cmd dict: pd format for disaggregated, dp format for multi-node."""
if config.disagg_cfg:
pd: dict[str, str] = {}
for node in config.nodes:
idx = node.index
if config.disagg_cfg.is_prefiller(idx):
n = config.disagg_cfg.prefiller_indices.index(idx)
pd[f"prefill-{n}"] = node.server_cmd
elif config.disagg_cfg.is_decoder(idx):
n = config.disagg_cfg.decoder_indices.index(idx)
pd[f"decode-{n}"] = node.server_cmd
return {"pd": pd}
return {"dp": {f"node{node.index}": node.server_cmd for node in config.nodes}}
def _save_benchmark_results_json(config: MultiNodeConfig, results: list[Any]) -> None:
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
runner = os.environ.get("VLLM_CI_RUNNER", "")
# Filter out None benchmark cases; results align with the non-None ones in order
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
tasks = [build_task_entry(key, case_cfg, result) for (key, case_cfg), result in zip(valid_items, results)]
output: dict[str, Any] = {
"model_name": config.model,
"hardware": extract_hardware(runner),
"dtype": _extract_dtype(config),
"feature": _extract_features(config.nodes[0].server_cmd, config.envs),
"vllm_version": vllm.__version__,
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
"tasks": tasks,
"serve_cmd": _build_serve_cmd(config),
"environment": filter_environment(config.envs),
}
job_name = os.environ.get("BENCHMARK_JOB_NAME", "")
write_results_json(output, job_name=job_name)
@pytest.mark.asyncio
async def test_multi_node() -> None:
config = MultiNodeConfigLoader.from_yaml()
if config.special_dependencies:
for k, v in config.special_dependencies.items():
command = [
sys.executable,
"-m",
"pip",
"install",
f"{k}=={v}",
]
subprocess.call(command)
with (
ProxyLauncher(
nodes=config.nodes,
disagg_cfg=config.disagg_cfg,
envs=config.envs,
proxy_port=config.proxy_port,
cur_index=config.cur_index,
) as proxy,
RemoteOpenAIServer(
model=config.model,
vllm_serve_args=config.server_cmd,
server_port=config.server_port,
server_host=config.master_ip,
env_dict=config.envs,
auto_port=False,
proxy_port=proxy.proxy_port,
disaggregated_prefill=config.disagg_cfg,
nodes_info=config.nodes,
max_wait_seconds=2800,
) as server,
):
host, port = config.benchmark_endpoint
if config.is_master:
results = run_aisbench_cases(
model=config.model,
port=port,
aisbench_cases=config.benchmark_cases,
host_ip=host,
)
_save_benchmark_results_json(config, results)
else:
# We should keep listening on the master node's server url determining when to exit.
server.hang_until_terminated(f"http://{host}:{config.server_port}/health")

View File

@@ -0,0 +1,28 @@
import os
from tests.e2e.nightly.multi_node.scripts.utils import (
get_all_ipv4,
get_available_port,
get_cluster_ips,
get_net_interface,
setup_logger,
temp_env,
)
DISAGGEGATED_PREFILL_PORT = 5333
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or DEFAULT_CONFIG_BASE_PATH
DEFAULT_SERVER_PORT = 8080
__all__ = [
"CONFIG_BASE_PATH",
"DEFAULT_CONFIG_BASE_PATH",
"DEFAULT_SERVER_PORT",
"DISAGGEGATED_PREFILL_PORT",
"get_all_ipv4",
"get_available_port",
"get_cluster_ips",
"get_net_interface",
"setup_logger",
"temp_env",
]

View File

@@ -0,0 +1 @@

View File

@@ -0,0 +1,128 @@
import json
import logging
from pathlib import Path
from typing import Any
logger = logging.getLogger(__name__)
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
INFRA_ENV_KEYS = {
"HCCL_IF_IP",
"HCCL_SOCKET_IFNAME",
"GLOO_SOCKET_IFNAME",
"TP_SOCKET_IFNAME",
"LOCAL_IP",
"NIC_NAME",
"MASTER_IP",
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
}
PERF_METRIC_RENAME: dict[str, str] = {
"Benchmark Duration": "Benchmark_Duration(BD)",
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
"Input Token Throughput": "Input_Token_Throughput(ITT)",
"Output Token Throughput": "Output_Token_Throughput(OTT)",
"Total Token Throughput": "Total_Token_Throughput(TTT)",
}
def extract_hardware(runner: str) -> str:
runner_lower = runner.lower()
for label in ("a3", "a2"):
if label in runner_lower:
return label.upper()
return runner
def get_vllm_version() -> str:
try:
import vllm
return vllm.__version__
except Exception:
return ""
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
if result == "":
return False
case_type = case_config.get("case_type")
baseline = case_config.get("baseline")
threshold = case_config.get("threshold")
if baseline is None or threshold is None:
return True
if case_type == "accuracy" and isinstance(result, (int, float)):
return abs(float(result) - float(baseline)) <= float(threshold)
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
try:
throughput_val = float(throughput_str.replace("token/s", "").strip())
return throughput_val >= float(threshold) * float(baseline)
except (ValueError, AttributeError):
return False
return True
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
dataset_path = case_config.get("dataset_path", "")
dataset_conf = case_config.get("dataset_conf", "")
if dataset_path:
task_name = dataset_path.split("/", 1)[-1]
elif dataset_conf:
task_name = dataset_conf.split("/")[0]
else:
task_name = case_key
case_type = case_config.get("case_type", "unknown")
metrics: dict[str, float] = {}
if result == "":
pass
elif case_type == "accuracy" and isinstance(result, (int, float)):
metrics["accuracy"] = round(float(result), 4)
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
_, result_json = result
for metric_name, metric_data in result_json.items():
if not isinstance(metric_data, dict):
continue
total_str = metric_data.get("total", "")
try:
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
except (ValueError, AttributeError):
pass
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
target: dict[str, Any] = {}
if case_config.get("baseline") is not None:
target["baseline"] = case_config["baseline"]
if case_config.get("threshold") is not None:
target["threshold"] = case_config["threshold"]
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
if target:
entry["target"] = target
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
return entry
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
return {key: value for key, value in envs.items() if key not in exclude}
def write_results_json(
output: dict[str, Any],
*,
job_name: str,
output_dir: Path | None = None,
) -> Path:
if output_dir is None:
output_dir = Path("/root/.cache/benchmark_results") / job_name
output_dir.mkdir(parents=True, exist_ok=True)
output_path = output_dir / f"{job_name}.json"
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
logger.info("Benchmark results saved to PVC at %s", output_path)
print(f"Benchmark results saved to PVC at {output_path}")
return output_path

View File

@@ -0,0 +1,174 @@
apiVersion: leaderworkerset.x-k8s.io/v1
kind: LeaderWorkerSet
metadata:
name: {{ lws_name | default("vllm") }}
namespace: vllm-project
spec:
replicas: {{ replicas | default(1) }}
leaderWorkerTemplate:
size: {{ size | default(2) }}
restartPolicy: None
leaderTemplate:
metadata:
labels:
role: leader
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-leader
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: VLLM_ASCEND_VERSION
value: {{ vllm_ascend_ref | default("main") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
ports:
- containerPort: 8080
# readinessProbe:
# tcpSocket:
# port: 8080
# initialDelaySeconds: 15
# periodSeconds: 10
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
workerTemplate:
spec:
tolerations:
- key: "dedicated"
operator: "Equal"
value: "night"
effect: "NoSchedule"
containers:
- name: vllm-worker
imagePullPolicy: Always
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
env:
- name: CONFIG_YAML_PATH
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
- name: CONFIG_BASE_PATH
value: "{{ config_base_path | default("") }}"
- name: LOG_PREFIX
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
- name: WORKSPACE
value: "/vllm-workspace"
- name: FAIL_TAG
value: {{ fail_tag | default("FAIL_TAG") }}
- name: IS_PR_TEST
value: "{{ is_pr_test | default("false") }}"
- name: VLLM_ASCEND_REF
value: {{ vllm_ascend_ref | default("main") }}
- name: VLLM_ASCEND_REMOTE_URL
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
- name: BENCHMARK_JOB_NAME
value: {{ benchmark_job_name | default("") }}
- name: VLLM_CI_RUNNER
value: {{ runner | default("linux-aarch64-a3-0") }}
- name: AOP_MULTI_ENABLED
value: "{{ aop_multi_enabled }}"
- name: GOOD_TABLE
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
command:
- sh
- -c
- |
bash /root/.cache/tests/run.sh
resources:
limits:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
memory: 512Gi
ephemeral-storage: 100Gi
requests:
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
ephemeral-storage: 100Gi
cpu: 125
volumeMounts:
- mountPath: /root/.cache
name: shared-volume
- mountPath: /usr/local/Ascend/driver/tools
name: driver-tools
- mountPath: /dev/shm
name: dshm
volumes:
- name: dshm
emptyDir:
medium: Memory
sizeLimit: 512Gi
- name: shared-volume
persistentVolumeClaim:
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
- name: driver-tools
hostPath:
path: /usr/local/Ascend/driver/tools
---
apiVersion: v1
kind: Service
metadata:
name: {{ lws_name | default("vllm") }}-leader
namespace: vllm-project
spec:
ports:
- name: http
port: 8080
protocol: TCP
targetPort: 8080
selector:
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
role: leader
type: ClusterIP

View File

@@ -0,0 +1,466 @@
#!/bin/bash
set -euo pipefail
# Color definitions
GREEN="\033[0;32m"
BLUE="\033[0;34m"
YELLOW="\033[0;33m"
RED="\033[0;31m"
NC="\033[0m" # No Color
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
else
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
fi
fi
# Configuration
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
# cann and atb environment setup
source /usr/local/Ascend/ascend-toolkit/set_env.sh
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
set +eu
source /usr/local/Ascend/nnal/atb/set_env.sh
set -eu
# Home path for aisbench
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
# Logging configurations
export VLLM_LOGGING_LEVEL="INFO"
# Reduce glog verbosity for mooncake
export GLOG_minloglevel=1
# Set transformers to offline mode to avoid downloading models during tests
export HF_HUB_OFFLINE="1"
# Default is 600s
export VLLM_ENGINE_READY_TIMEOUT_S=1800
# Function to print section headers
print_section() {
echo -e "\n${BLUE}=== $1 ===${NC}"
}
print_failure() {
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
exit 1
}
# Function to print success messages
print_success() {
echo -e "${GREEN}$1${NC}"
}
# Function to print error messages and exit
print_error() {
echo -e "${RED}✗ ERROR: $1${NC}"
exit 1
}
show_vllm_info() {
cd "$WORKSPACE"
echo "Installed vLLM-related Python packages:"
pip list | grep vllm || echo "No vllm packages found."
echo ""
echo "============================"
echo "vLLM Git information"
echo "============================"
cd vllm
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm"
fi
cd ..
echo ""
echo "============================"
echo "vLLM-Ascend Git information"
echo "============================"
cd vllm-ascend
if [ -d .git ]; then
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
echo "Commit hash: $(git rev-parse HEAD)"
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
echo "Message: $(git log -1 --pretty=format:'%s')"
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
echo "Remote: $(git remote -v | head -n1)"
echo ""
else
echo "No .git directory found in vllm-ascend"
fi
cd ..
}
check_npu_info() {
echo "====> Check NPU info"
npu-smi info
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
}
check_and_config() {
echo "====> Configure mirrors and git proxy"
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
}
install_extra_components() {
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
return 1
fi
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
echo "Failed to download custom_ops wheel"
return 1
fi
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
source /usr/local/Ascend/ascend-toolkit/set_env.sh
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
echo "====> Extra components installation completed"
}
checkout_src() {
echo "====> Checkout source code"
mkdir -p "$WORKSPACE"
cd "$WORKSPACE"
pip uninstall -y vllm-ascend || true
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
rm -rf "$WORKSPACE/vllm-ascend"
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
cd "$WORKSPACE/vllm-ascend"
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
if [ -n "$PR_REF" ]; then
git fetch --depth 1 origin "$PR_REF"
git checkout FETCH_HEAD
else
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
git checkout "$VLLM_ASCEND_REF"
fi
git submodule update --init --recursive
fi
}
install_vllm_ascend() {
echo "====> Install vllm-ascend"
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
pip install -e "$WORKSPACE/vllm-ascend"
}
install_aisbench() {
echo "====> Install AISBench benchmark"
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
cp -r /tmp/aisbench-backup "$BENCH_DIR"
cd "$BENCH_DIR"
pip install -e . \
-r requirements/api.txt \
-r requirements/extra.txt
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
}
show_triton_ascend_info() {
echo "====> Check triton ascend info"
clang -v
which bishengir-compile
pip show triton-ascend
}
kill_npu_processes() {
pgrep python3 | xargs -r kill -9
pgrep VLLM | xargs -r kill -9
sleep 4
}
run_tests_with_log() {
set +e
kill_npu_processes
mkdir -p "${LOG_PREFIX}"
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
ret=$?
echo "pytest exit code: ret=${ret}"
set -e
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
if [ $ret -eq 0 ]; then
print_success "All tests passed!"
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
else
echo "Leader: waiting 10s for worker logs..."
sleep 10
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
set +e; aop_pipeline; set -e
fi
local done_file="${LOG_PREFIX}/aop_done"
touch "$done_file"
echo "Leader: notifying workers (${done_file})"
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
exit 1
fi
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
if [ $ret -eq 0 ]; then
echo "Worker: test passed, waiting for leader..."
local wait_timeout=30
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
sleep 1
wait_timeout=$((wait_timeout - 1))
done
fi
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
local release="${LOG_PREFIX}/aop_done"
mkdir -p "$coord"
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
cd "$WORKSPACE/vllm-ascend"
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--coord-dir "${coord}" \
--release-file "${release}"
while [ ! -f "$release" ]; do sleep 5; done
echo "Worker: release signal received, exiting"
exit 1
else
echo "Worker: leader finished successfully, exiting"
fi
fi
}
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
aop_pipeline() {
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
local table="${GOOD_TABLE:-}"
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
local case_name="${BENCHMARK_JOB_NAME#*-}"
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
case_name="${CONFIG_YAML_PATH%.yaml}"
fi
echo "============================================"
echo " AOP Pipeline (Pod) - START"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Case name : ${case_name}"
echo " Rules file : ${rules}"
echo " Table file : ${table}"
echo " Log prefix : ${LOG_PREFIX}"
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
echo "============================================"
# ---- Step 1: Classify ----
echo ""
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
echo " Rules content:"
if [ -f "$rules" ]; then
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
else
echo " (rules file not found)"
fi
echo ""
echo " Pod logs found:"
local found_any=0
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
echo " - ${f} ($(wc -l < "$f") lines)"
found_any=1
fi
done
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
local env_count=0
if [ -f "$rules" ]; then
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
if [ -f "$f" ]; then
local n
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
n=${n%%[!0-9]*}
echo " Scan ${f}: ${n} matches"
env_count=$((env_count + n))
if [ "$n" -gt 0 ]; then
echo " Matched lines:"
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
fi
fi
done
fi
echo " Classify result: env_count=${env_count}"
if [ "$found_any" -eq 0 ]; then
echo " Decision: no pod logs → SKIP"
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
return 1
fi
if [ "$env_count" -gt 0 ]; then
echo " Decision: env_failure → SKIP"
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
return 1
fi
# ---- Step 2: Check age ----
echo ""
echo "--- [2/3] Check commit age ---"
echo " Looking up: ${case_name}"
local skip_age=0
if [ ! -f "$table" ]; then
echo " Table file not found: ${table}"
echo " Decision: no table → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Only consider success rows
local success_rows
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
if [ -z "$success_rows" ]; then
echo " No success row found for '${case_name}'"
echo " Decision: no success entry → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# Pick most recent success row
local best_date=""
while IFS= read -r row; do
local d
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
[ -z "$d" ] && continue
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
best_date="$d"
fi
done <<< "$success_rows"
if [ -z "$best_date" ]; then
echo " No valid date in success rows"
echo " Decision: no date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
local last_ts now_ts age_days
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
echo " Date parse failed: ${best_date}"
echo " Decision: invalid date → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
now_ts=$(date +%s)
age_days=$(( (now_ts - last_ts) / 86400 ))
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
if [ "$age_days" -gt 3 ]; then
echo " Decision: old commit (> 3 days) → SKIP"
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
return 1
fi
# ---- Step 3: Bisect ----
echo ""
echo "--- [3/3] Run bisect ---"
echo " Scene : multi_node"
echo " Config : ${CONFIG_YAML_PATH}"
echo " Bad commit : HEAD"
echo " Name : ${case_name}"
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
echo " Coord dir : ${coord}"
# Wait for all workers to signal ready
echo " Waiting for workers..."
for i in $(seq 1 30); do
local ready_count=0
for f in "${coord}"/worker_ready_*; do
[ -e "$f" ] && ready_count=$((ready_count + 1))
done
echo " [${i}/30] ready workers: ${ready_count}"
if [ "$ready_count" -ge 1 ]; then break; fi
sleep 2
done
cd "$WORKSPACE/vllm-ascend"
local bisect_rc=0
python -m tests.e2e.nightly.bisect.auto_bisect \
--scene multi_node \
--config-yaml "${CONFIG_YAML_PATH}" \
--bad-commit HEAD \
--good-table "${table}" \
--name "${case_name}" \
--coord-dir "${coord}" || bisect_rc=$?
echo " bisect completed (exit code: ${bisect_rc})"
echo "=== AOP Pipeline (Pod) - END ==="
return 1
}
clear_logs() {
print_section "Clearing logs from previous runs"
rm -fr "$HOME/ascend/log" || true
}
backup_ascend_logs() {
if [ -n "${LOG_PREFIX:-}" ]; then
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
mkdir -p "$dest"
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
echo "Ascend logs backed up to $dest"
fi
}
main() {
trap backup_ascend_logs EXIT
check_npu_info
clear_logs
check_and_config
if [[ "$IS_PR_TEST" == "true" ]]; then
checkout_src
install_vllm_ascend
install_aisbench
fi
show_vllm_info
show_triton_ascend_info
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
install_extra_components
fi
cd "$WORKSPACE/vllm-ascend"
run_tests_with_log
}
main "$@"

View File

@@ -0,0 +1,183 @@
import logging
import os
import socket
import time
from contextlib import contextmanager
from pathlib import Path
from typing import Any
import yaml
logger = logging.getLogger(__name__)
@contextmanager
def temp_env(env_dict: dict[str, Any]):
old_env = {}
for key, value in env_dict.items():
old_env[key] = os.environ.get(key)
os.environ[key] = str(value)
try:
yield
finally:
for key, value in old_env.items():
if value is None:
os.environ.pop(key, None)
else:
os.environ[key] = value
def setup_logger() -> None:
logging.basicConfig(
level=logging.INFO,
format="[%(asctime)s] [%(levelname)s] %(message)s",
datefmt="%Y-%m-%d %H:%M:%S",
)
def load_yaml_mapping(
yaml_path: str | None,
*,
default_name: str,
default_base_path: str,
description: str,
) -> dict[str, Any]:
if not yaml_path:
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
path = Path(yaml_path)
if not path.is_absolute() and not path.exists():
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
path = Path(base_path) / yaml_path
logger.info("Loading %s yaml: %s", description, path)
with path.open(encoding="utf-8") as f:
data = yaml.safe_load(f)
if not isinstance(data, dict):
raise TypeError(f"{description} must be a mapping: {path}")
return data
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
def resolve(dns: str) -> str:
delay = base_delay
for attempt in range(retries):
try:
return socket.gethostbyname(dns)
except socket.gaierror:
if attempt == retries - 1:
raise
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError(f"Unable to resolve DNS: {dns}")
return resolve
def get_cluster_dns_list(world_size: int) -> list[str]:
if world_size < 1:
raise ValueError(f"world_size must be >= 1, got {world_size}")
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
if not leader_dns:
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
parts = leader_dns.split(".")
if len(parts) < 3:
raise ValueError(f"invalid leader DNS format: {leader_dns}")
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
return [leader_dns, *worker_dns_list]
def get_cluster_ips(world_size: int = 2) -> list[str]:
resolver = dns_resolver()
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
def resolve_cluster_ips(
raw_config: dict[str, Any],
num_nodes: int,
explicit_cluster_ips: list[str] | None = None,
*,
cluster_hosts_log_message: str | None = None,
dns_log_message: str = "Resolving cluster IPs via DNS...",
) -> list[str]:
if explicit_cluster_ips is not None:
if len(explicit_cluster_ips) != num_nodes:
raise AssertionError("cluster_ips size mismatch")
return explicit_cluster_ips
cluster_hosts = raw_config.get("cluster_hosts")
if cluster_hosts:
if cluster_hosts_log_message:
logger.info(cluster_hosts_log_message)
if len(cluster_hosts) != num_nodes:
raise AssertionError("cluster_hosts size mismatch")
return list(cluster_hosts)
logger.info(dns_log_message)
return get_cluster_ips(num_nodes)
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
for port in range(start_port, end_port):
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
try:
s.bind(("", port))
return port
except OSError:
continue
raise RuntimeError("No available port found")
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
delay = base_delay
for attempt in range(retries):
try:
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
s.connect(("8.8.8.8", 80))
return s.getsockname()[0]
except Exception:
try:
return socket.gethostbyname(socket.gethostname())
except Exception:
if attempt == retries - 1:
raise RuntimeError("Failed to determine local IP address")
time.sleep(delay)
delay = min(delay * 1.5, 5)
raise RuntimeError("Failed to determine local IP address")
def get_net_interface(ip: str | None = None) -> str:
import psutil
if ip is None:
ip = get_cur_ip()
for iface, addrs in psutil.net_if_addrs().items():
for addr in addrs:
if addr.family == socket.AF_INET and addr.address == ip:
return iface
raise RuntimeError(f"No network interface found for IP {ip}")
def get_all_ipv4() -> list[str]:
ipv4s = {"127.0.0.1"}
hostname = socket.gethostname()
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
ipv4s.add(info[4][0])
return list(ipv4s)
def resolve_current_node_index(cluster_ips: list[str]) -> int:
worker_index = os.environ.get("LWS_WORKER_INDEX")
if worker_index:
return int(worker_index)
local_ips = set(get_all_ipv4())
for index, ip in enumerate(cluster_ips):
if ip in local_ips:
return index
raise RuntimeError("Unable to determine current node index")

View File

@@ -0,0 +1,59 @@
#!/bin/bash
# ============================================================
# aop_capture.sh - Capture test results from log files
#
# Called from _e2e_nightly_single_node.yaml [AOP] steps.
# Writes outputs to $GITHUB_OUTPUT.
#
# Usage: aop_capture.sh <yaml_outcome> <pytest_outcome>
# ============================================================
set -euo pipefail
YAML_OUTCOME="$1"
PYTEST_OUTCOME="$2"
LOG_DIR="/tmp/test-logs"
echo "============================================"
echo " Test Result Summary"
echo " YAML-driven : ${YAML_OUTCOME:-skipped}"
echo " Pytest-driven: ${PYTEST_OUTCOME:-skipped}"
echo "============================================"
parse_log() {
local log_file="$1"
local prefix="$2"
if [ ! -f "$log_file" ]; then
return 0
fi
echo ""
echo "--- ${prefix} tail (last 40 lines) ---"
tail -n 40 "$log_file"
echo "--- end ---"
local summary
summary=$(grep -E '=+.*(passed|failed|error).*=+' "$log_file" | tail -1 || true)
echo "${prefix}_summary=${summary}" >> "$GITHUB_OUTPUT"
echo "${prefix}_summary: ${summary}"
local failures
failures=$(grep -c 'FAILED' "$log_file" || true)
echo "${prefix}_failures=${failures}" >> "$GITHUB_OUTPUT"
}
parse_log "${LOG_DIR}/pytest-driven.log" "pytest"
parse_log "${LOG_DIR}/yaml-test.log" "yaml"
# Final verdict + which test failed
if [ "$YAML_OUTCOME" = "failure" ] || [ "$PYTEST_OUTCOME" = "failure" ]; then
echo "result=failure" >> "$GITHUB_OUTPUT"
FAILED=""
[ "$YAML_OUTCOME" = "failure" ] && FAILED="${FAILED}yaml,"
[ "$PYTEST_OUTCOME" = "failure" ] && FAILED="${FAILED}pytest,"
echo "failed_test=${FAILED%,}" >> "$GITHUB_OUTPUT"
elif [ "$YAML_OUTCOME" = "success" ] || [ "$PYTEST_OUTCOME" = "success" ]; then
echo "result=success" >> "$GITHUB_OUTPUT"
else
echo "result=skipped" >> "$GITHUB_OUTPUT"
fi

View File

@@ -0,0 +1,60 @@
#!/bin/bash
# ============================================================
# aop_classify.sh - Check if failure is environmental
#
# Args: $1 = failed_test (from capture: "yaml", "pytest", "yaml,pytest")
#
# Only scans the log files that actually failed.
# Writes failure_type to $GITHUB_OUTPUT.
# ============================================================
set -euo pipefail
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
RULES="$SCRIPT_DIR/rules-env.txt"
LOG_DIR="/tmp/test-logs"
FAILED_TEST="${1:-yaml,pytest}"
check_log() {
local log_file="$1"
local label="$2"
if [ ! -f "$log_file" ] || [ ! -s "$log_file" ]; then
echo " [$label] log empty/missing -> skipped"
return 0
fi
local count
count=$(grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -ciEf - "$log_file" 2>/dev/null || echo 0)
echo " [$label] env patterns matched: ${count}"
if [ "$count" -gt 0 ]; then
echo " [$label] --- matches ---"
grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -niEf - "$log_file" | head -10
return 1
fi
return 0
}
echo "=== Failure Classification ==="
echo "[DEBUG] rules file : ${RULES}"
echo "[DEBUG] rules exists: $(test -f "$RULES" && echo yes || echo no)"
echo "[DEBUG] rules lines : $(grep -c . "$RULES" 2>/dev/null || echo 0)"
echo "[DEBUG] rules content:"
cat -n "$RULES" 2>/dev/null || echo "(file not found)"
ENV_FOUND=0
if [[ "$FAILED_TEST" == *pytest* ]]; then
check_log "${LOG_DIR}/pytest-driven.log" "pytest-driven" || ENV_FOUND=1
fi
if [[ "$FAILED_TEST" == *yaml* ]]; then
check_log "${LOG_DIR}/yaml-test.log" "yaml-test" || ENV_FOUND=1
fi
if [ "$ENV_FOUND" -eq 1 ]; then
echo "=== Result: env_failure ==="
echo "failure_type=env_failure" >> "$GITHUB_OUTPUT"
else
echo "=== Result: not_env_failure ==="
echo "failure_type=not_env_failure" >> "$GITHUB_OUTPUT"
fi

View File

@@ -0,0 +1,98 @@
#!/bin/bash
# ============================================================
# aop_commit_age.sh - Look up last successful time in good_table.csv
#
# CSV format (good_table.csv):
# name,yaml/path,link,status,vLLM Git information,vLLM-Ascend Git information,time
#
# Finds rows matching config_name where status=success,
# picks the most recent "time" column, and calculates age.
#
# Args: $1 = config_name
# $2 = csv_path
#
# Writes to $GITHUB_OUTPUT:
# commit_age_days - days since last success
# is_old - true if > 3 days
# last_status - status from table
# last_date - date from table
# ============================================================
set -euo pipefail
CONFIG_NAME="${1:-}"
CSV_PATH="${GOOD_TABLE:-$2}"
if [ -z "$CONFIG_NAME" ]; then
echo "ERROR: no config name provided"
exit 1
fi
echo ">>> Looking up config : ${CONFIG_NAME}"
echo ">>> CSV path : ${CSV_PATH}"
if [ ! -f "$CSV_PATH" ]; then
echo ">>> CSV not found → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
# Find matching rows, only consider success rows (match name column only)
ROWS=$(grep "^${CONFIG_NAME}," "$CSV_PATH" | grep -F ',success,' || true)
if [ -z "$ROWS" ]; then
echo ">>> No success row for '${CONFIG_NAME}' → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
# Pick most recent success row
BEST_ROW=""
BEST_DATE=""
while IFS= read -r row; do
date_str=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
[ -z "$date_str" ] && continue
if [ -z "$BEST_DATE" ] || [[ "$date_str" > "$BEST_DATE" ]]; then
BEST_ROW="$row"
BEST_DATE="$date_str"
fi
done <<< "$ROWS"
if [ -z "$BEST_ROW" ]; then
echo ">>> No valid date in success rows → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
exit 0
fi
LAST_STATUS=$(echo "$BEST_ROW" | awk -F',' '{print $4}' | xargs)
LAST_DATE="$BEST_DATE"
echo ">>> Matched row: ${BEST_ROW}"
echo "last_status=${LAST_STATUS}" >> "$GITHUB_OUTPUT"
echo "last_date=${LAST_DATE}" >> "$GITHUB_OUTPUT"
LAST_TS=$(date -d "$LAST_DATE" +%s 2>/dev/null || true)
if [ -z "$LAST_TS" ]; then
echo ">>> Could not parse date: ${LAST_DATE} → skip"
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
exit 0
fi
NOW=$(date +%s)
AGE_DAYS=$(( (NOW - LAST_TS) / 86400 ))
echo "commit_age_days=${AGE_DAYS}" >> "$GITHUB_OUTPUT"
if [ "$AGE_DAYS" -gt 3 ]; then
echo "is_old=true" >> "$GITHUB_OUTPUT"
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (> 3 days) → old"
else
echo "is_old=false" >> "$GITHUB_OUTPUT"
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (<= 3 days) → recent"
fi

View File

@@ -0,0 +1,93 @@
#!/bin/bash
# ============================================================
# aop_process.sh - Handle a recent real failure + auto bisect
#
# Args:
# $1 failure_type
# $2 commit_age_days
# $3 runner
# $4 tests
# $5 config_file_path
# $6 pytest_summary
# $7 yaml_summary
# $8 scene (single_node | multi_node)
# $9 bad_commit (commit SHA, default HEAD)
# $10 num_nodes (multi_node only)
# $11 coord_dir (multi_node only)
# $12 case_name (optional)
# ============================================================
set -euo pipefail
FT="${1:-unknown}"
AGE="${2:-?}"
RUNNER="${3:-?}"
TESTS="${4:-}"
CONFIG="${5:-}"
PYTEST_SUMMARY="${6:-}"
YAML_SUMMARY="${7:-}"
SCENE="${8:-single_node}"
BAD_COMMIT="${9:-HEAD}"
NUM_NODES="${10:-}"
COORD_DIR="${11:-}"
NAME="${12:-}"
echo "================================================"
echo " PROCESS - needs attention"
echo " Failure type : ${FT}"
echo " Commit age : ${AGE} days"
echo " Runner : ${RUNNER}"
echo " Tests : ${TESTS:-N/A}"
echo " Config : ${CONFIG:-N/A}"
echo " Scene : ${SCENE}"
echo " Bad commit : ${BAD_COMMIT}"
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
echo " YAML : ${YAML_SUMMARY:-N/A}"
echo "================================================"
echo "::group::Failed test details"
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log /tmp/test-logs/multi-node.log; do
if [ -f "$f" ]; then
grep -A 10 'FAILED' "$f" || true
fi
done
echo "::endgroup::"
# =====================================================
# Auto bisect
# =====================================================
# Extract case_name if not provided (single_node requires it)
if [ -z "$NAME" ] && [ "$SCENE" = "single_node" ]; then
if [ -n "$TESTS" ]; then
# py-driven: tests/e2e/.../test_xxx.py → test_xxx
NAME=$(basename "$TESTS" .py)
elif [ -n "$CONFIG" ]; then
# YAML-driven: Qwen3-32B-Int8.yaml → Qwen3-32B-Int8
NAME=$(basename "$CONFIG" .yaml)
fi
if [ -z "$NAME" ]; then
echo "WARNING: could not extract case_name, bisect may fail"
else
echo "Extracted name: ${NAME}"
fi
fi
GOOD_TABLE="${GOOD_TABLE:-}"
BISECT_CMD=(
python -m tests.e2e.nightly.bisect.auto_bisect
--scene "${SCENE}"
--bad-commit "${BAD_COMMIT}"
--good-table "${GOOD_TABLE}"
)
[ -n "$CONFIG" ] && BISECT_CMD+=(--config-yaml "$CONFIG")
[ -n "$NAME" ] && BISECT_CMD+=(--name "$NAME")
[ -n "$NUM_NODES" ] && BISECT_CMD+=(--num-nodes "$NUM_NODES")
[ -n "$COORD_DIR" ] && BISECT_CMD+=(--coord-dir "$COORD_DIR")
echo ""
echo "=== Running auto bisect ==="
echo "${BISECT_CMD[@]}"
"${BISECT_CMD[@]}"

View File

@@ -0,0 +1,39 @@
#!/bin/bash
# ============================================================
# aop_skip.sh - Log skip reason and show failure details
#
# Args: failure_type last_status last_date age_days
# pytest_summary yaml_summary
# ============================================================
set -euo pipefail
FT="${1:-unknown}"
LAST_STATUS="${2:-?}"
LAST_DATE="${3:-?}"
AGE="${4:-?}"
PYTEST_SUMMARY="${5:-}"
YAML_SUMMARY="${6:-}"
case "$FT" in
env_failure) REASON="environment issue" ;;
*) REASON="last run > 3 days ago" ;;
esac
echo "================================================"
echo " SKIP - no further action"
echo " Failure type : ${FT}"
echo " Last status : ${LAST_STATUS}"
echo " Last date : ${LAST_DATE}"
echo " Age (days) : ${AGE}"
echo " Reason : ${REASON}"
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
echo " YAML : ${YAML_SUMMARY:-N/A}"
echo "================================================"
echo "::group::Failed test details"
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log; do
if [ -f "$f" ]; then
grep -A 10 'FAILED' "$f" || true
fi
done
echo "::endgroup::"

View File

@@ -0,0 +1,6 @@
# Environment failure patterns (network / hardware / infra)
# One regex per line. Lines starting with # are comments.
# Used by aop_classify.sh via grep -Ef
RuntimeError: Timeout
TimeoutError: Timed out waiting for engine core processes to start

View File

@@ -0,0 +1,155 @@
#!/usr/bin/env python3
#
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
# This file is a part of the vllm-ascend project.
#
"""Update /root/.cache/vllm-ascend/main/nightly/good_table.csv with a
successful test entry. Creates the file (with header) if it does not exist;
replaces the existing row for the same test name if it does.
CSV columns:
name, yaml/path, link, status,
vLLM Git information, vLLM-Ascend Git information, time
"""
import argparse
import csv
import os
import subprocess
from datetime import datetime, timedelta, timezone
HEADER = [
"name",
"yaml/path",
"link",
"status",
"vLLM Git information",
"vLLM-Ascend Git information",
"time",
]
def git_head(repo_dir: str) -> str:
try:
return subprocess.check_output(
["git", "rev-parse", "HEAD"],
cwd=repo_dir,
stderr=subprocess.DEVNULL,
text=True,
).strip()
except Exception:
return "N/A"
def current_timestamp() -> str:
tz = timezone(timedelta(hours=8))
ts = datetime.now(tz).strftime("%Y-%m-%d %H:%M:%S %z")
# Reformat +0800 → +08:00 to match existing CSV entries
return ts[:-2] + ":" + ts[-2:]
def load_rows(csv_path: str) -> list[list[str]]:
if not os.path.isfile(csv_path):
return []
with open(csv_path, newline="", encoding="utf-8") as f:
reader = csv.reader(f)
rows = list(reader)
# Drop the header row if present
if rows and rows[0] == HEADER:
rows = rows[1:]
return rows
def save_rows(csv_path: str, rows: list[list[str]]) -> None:
os.makedirs(os.path.dirname(csv_path), exist_ok=True)
with open(csv_path, "w", newline="", encoding="utf-8") as f:
writer = csv.writer(f)
writer.writerow(HEADER)
writer.writerows(rows)
_DEFAULT_SINGLE_NODE_CONFIG_BASE = "tests/e2e/nightly/single_node/models/configs"
_DEFAULT_MULTI_NODE_CONFIG_BASES = (
"tests/e2e/nightly/multi_node/internal_dp/config",
"tests/e2e/nightly/multi_node/external_dp/config",
)
def resolve_test_path(
test_path: str,
config_base_path: str,
scene: str = "single_node",
repo_dir: str = ".",
) -> str:
"""Return the full relative path for the yaml/path CSV column.
Upper-level workflows pass config_file_path as a bare filename
(e.g. ``Qwen3.5-27B-w8a8-A2.yaml``). When no directory component is
present we prepend the config base path so the CSV matches the format
used by the existing hand-curated good_table entries.
"""
if os.sep in test_path or "/" in test_path:
return test_path
if config_base_path.strip():
return f"{config_base_path.strip()}/{test_path}"
if scene == "multi_node":
for base in _DEFAULT_MULTI_NODE_CONFIG_BASES:
if os.path.isfile(os.path.join(repo_dir, base, test_path)):
return f"{base}/{test_path}"
return f"{_DEFAULT_MULTI_NODE_CONFIG_BASES[0]}/{test_path}"
return f"{_DEFAULT_SINGLE_NODE_CONFIG_BASE}/{test_path}"
def main() -> None:
parser = argparse.ArgumentParser(description="Update good_table.csv on test success")
parser.add_argument("--cache-csv", required=True)
parser.add_argument("--test-name", required=True)
parser.add_argument("--test-path", required=True)
parser.add_argument("--config-base-path", default="")
parser.add_argument("--scene", default="single_node", choices=["single_node", "multi_node"])
parser.add_argument("--run-link", required=True)
parser.add_argument("--vllm-dir", default="/vllm-workspace/vllm")
parser.add_argument("--vllm-ascend-dir", default="/vllm-workspace/vllm-ascend")
parser.add_argument("--vllm-ascend-version", default="")
parser.add_argument("--vllm-version", default="")
args = parser.parse_args()
vllm_hash = args.vllm_version.strip() or git_head(args.vllm_dir)
vllm_ascend_hash = args.vllm_ascend_version.strip() or git_head(args.vllm_ascend_dir)
timestamp = current_timestamp()
test_path = resolve_test_path(args.test_path, args.config_base_path, args.scene, args.vllm_ascend_dir)
new_row = [
args.test_name,
test_path,
args.run_link,
"success",
vllm_hash,
vllm_ascend_hash,
timestamp,
]
is_new = not os.path.isfile(args.cache_csv)
rows = load_rows(args.cache_csv)
rows = [r for r in rows if r and r[0] != args.test_name]
rows.append(new_row)
save_rows(args.cache_csv, rows)
action = "Created" if is_new else "Updated"
print(f">>> {action} {args.cache_csv}: name={args.test_name} status=success time={timestamp}")
if __name__ == "__main__":
main()

View File

@@ -0,0 +1,84 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "10"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "36864"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "16"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"num_speculative_tokens": 1, "method": "mtp"}'
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
_benchmarks_acc: &benchmarks_acc
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10
_benchmarks_perf: &benchmarks_perf
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 400
max_out_len: 1500
batch_size: 1000
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-R1-0528-W8A8-single"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--enforce-eager"
benchmarks:
- name: "DeepSeek-R1-0528-W8A8-aclgraph"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks_acc
<<: *benchmarks_perf

View File

@@ -0,0 +1,81 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V3.2-W8A8-DCP-replicated-indexer"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
VLLM_ASCEND_ENABLE_NZ: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "20"
HCCL_BUFFSIZE: "768"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_SERVER_DEV_MODE: "1"
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
ASCEND_LAUNCH_BLOCKING: "0"
ASCEND_ENABLE_USE_FABRIC_MEM: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_MLAPO: "0"
PYTHONHASHSEED: "0"
ASCEND_A3_ENABLE: "1"
VLLM_ENGINE_READY_TIMEOUT_S: "10000"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
TASK_QUEUE_ENABLE: "1"
CPU_AFFINITY_CONF: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "1024"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "1"
- "--pipeline-parallel-size"
- "1"
- "--tensor-parallel-size"
- "16"
- "--prefill-context-parallel-size"
- "1"
- "--decode-context-parallel-size"
- "16"
- "--cp-kv-cache-interleave-size"
- "1"
- "--block-size"
- "128"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.95"
- "--api-server-count"
- "1"
- "--safetensors-load-strategy"
- "prefetch"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 16, 64, 128]}'
- "--additional-config"
- '{"enable_dsa_cp": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}, "multistream_overlap_shared_expert": true, "enable_mc2_hierarchy_comm": false, "enable_sparse_sfa_c8": true, "enable_sparse_li_c8": true, "enable_cpu_binding": true, "recompute_scheduler_enable": false}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
test_content: []
benchmarks:
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 8192
batch_size: 32
baseline: 95
threshold: 5

View File

@@ -0,0 +1,80 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V3.2-W8A8-TP8-DP2"
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "256"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "67000"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "8"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--async-scheduling"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.95"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[1,2,4,8,16,24,32,40,48], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
- "--reasoning-parser"
- "deepseek_v3"
- "--tokenizer_mode"
- "deepseek_v32"
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 86.67
temperature: 1.0
top_p: 0.95
thinking: true
threshold: 10
perf_2:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 16
max_out_len: 1500
batch_size: 4
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,80 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "DeepSeek-V4-Flash-W8A8-A3"
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
special_dependencies:
transformers: "5.9.0"
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_BUFFSIZE: "1024"
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
ASCEND_LAUNCH_BLOCKING: "0"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
server_cmd:
- "--enable-prefix-caching"
- "--max-model-len"
- "1048576"
- "--max-num-batched-tokens"
- "10240"
- "--gpu-memory-utilization"
- "0.9"
- "--max-num-seqs"
- "64"
- "--data-parallel-size"
- "4"
- "--tensor-parallel-size"
- "4"
- "--enable-expert-parallel"
- "--tokenizer-mode"
- "deepseek_v4"
- "--tool-call-parser"
- "deepseek_v4"
- "--enable-auto-tool-choice"
- "--reasoning-parser"
- "deepseek_v4"
- "--safetensors-load-strategy"
- "prefetch"
- "--quantization"
- "ascend"
- "--api-server-count"
- "1"
- "--speculative-config"
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
- "--port"
- "$SERVER_PORT"
- "--block-size"
- "128"
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--async-scheduling"
- "--additional-config"
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":"true","enable_shared_expert_dp":true,"multistream_overlap_shared_expert":true}'
benchmarks:
acc-gpqa:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
max_out_len: 65536
batch_size: 32
baseline: 86.36
threshold: 5
thinking: true
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 64
max_out_len: 1024
batch_size: 16
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,66 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
HCCL_OP_EXPANSION_MODE: "AIV"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "16"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"mtp"}'
- "--additional-config"
- '{"enable_shared_expert_dp": true, "ascend_fusion_config": {"fusion_ops_gmmswigluquant": false}}'
_benchmarks: &benchmarks
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 8
baseline: 95
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "GLM-4.7-TP8-DP2-decodegraph"
model: "Eco-Tech/GLM-4.7-W8A8-floatmtp"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--compilation-config"
- '{"cudagraph_capture_sizes": [1,2,4,8,16,32,64,128,256,512], "cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,82 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "GLM-5.1-W8A8-PrefillMC2"
model: "Eco-Tech/GLM-5.1-w8a8" #need update
envs:
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_USE_V1: "1"
HCCL_BUFFSIZE: "1800"
ASCEND_AGGREGATE_ENABLE: "1"
ASCEND_TRANSPORT_PRINT: "1"
ACL_OP_INIT_MODE: "1"
ASCEND_A3_ENABLE: "1"
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "10240"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "32"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--async-scheduling"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--gpu-memory-utilization"
- "0.94"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp", "enforce_eager": true}'
- "--additional_config"
- '{"enable_prefill_mc2": true}'
- "--reasoning-parser"
- "glm45"
- "--tool-call-parser"
- "glm47"
benchmarks:
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 8192
batch_size: 32
baseline: 96.88
temperature: 1.0
top_p: 0.95
thinking: true
threshold: 5
perf_2:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 64
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,58 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
VLLM_USE_MODELSCOPE: "true"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1024"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
_server_cmd: &server_cmd
- "--tensor-parallel-size"
- "16"
- "--enable-expert-parallel"
- "--enable-ep-weight-filter"
- "--tool-call-parser"
- "hy_v3"
- "--reasoning-parser"
- "hy_v3"
- "--enable-auto-tool-choice"
- "--max-model-len"
- "32768"
- "--max-num-seqs"
- "8"
- "--port"
- "$SERVER_PORT"
- "--speculative-config"
- '{"method": "mtp", "num_speculative_tokens": 1}'
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
_benchmarks: &benchmarks
acc_gsm8k:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_4_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 8
baseline: 93.07
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Hy3-preview-TP16-EP-MTP"
model: "Tencent-Hunyuan/Hy3-preview"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,52 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2-Thinking-TP16-Case"
model: "moonshotai/Kimi-K2-Thinking"
envs:
HCCL_BUFFSIZE: "1024"
TASK_QUEUE_ENABLE: "1"
OMP_PROC_BIND: "false"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "16"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "12"
- "--gpu-memory-utilization"
- "0.9"
- "--trust-remote-code"
- "--enable-expert-parallel"
- "--no-enable-prefix-caching"
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 4096
batch_size: 32
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs400
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 256
batch_size: 64
trust_remote_code: true
request_rate: 11.2
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,91 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
HCCL_BUFFSIZE: "512"
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
TASK_QUEUE_ENABLE: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_NZ: "1"
_server_cmd: &server_cmd
- "--enable-expert-parallel"
- "--enable-prefix-caching"
- "--enable-chunked-prefill"
- "--allowed-local-media-path"
- "/"
- "--tensor-parallel-size"
- "4"
- "--data-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "133120"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "16"
- "--quantization"
- "ascend"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--seed"
- "42"
- "--compilation-config"
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--speculative-config"
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
- "--additional-config"
- '{"enable_shared_expert_dp":true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
_benchmarks: &benchmarks
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 65536
temperature: 0.0
top_p: 1
top_k: -1
repetition_penalty: 1.0
batch_size: 32
baseline: 95
threshold: 10
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 8
max_out_len: 1024
batch_size: 2
trust_remote_code: true
request_rate: 0
baseline: 1
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.5-W4A8-Case"
model: "Eco-Tech/Kimi-K2.5-W4A8"
envs:
<<: *envs
server_cmd: *server_cmd
benchmarks:
<<: *benchmarks

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "Kimi-K2.6-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
model: "Eco-Tech/Kimi-K2.6-w4a8"
envs:
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_PROC_BIND: "false"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
HCCL_BUFFSIZE: "800"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM_ASCEND_ENABLE_MLAPO: "1"
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
DYNAMIC_EPLB: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--port"
- "$SERVER_PORT"
- "--allowed-local-media-path"
- "/"
- "--trust-remote-code"
- "--safetensors-load-strategy"
- 'prefetch'
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--no-enable-prefix-caching"
- "--enable-expert-parallel"
- "--max-num-seqs"
- "24"
- "--max-model-len"
- "6144"
- "--max-num-batched-tokens"
- "4096"
- "--gpu-memory-utilization"
- "0.85"
- "--seed"
- "42"
- "--async-scheduling"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
- "--profiler-config"
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
- "--mm-processor-cache-gb"
- "0"
- "--mm-encoder-tp-mode"
- "data"
- "--speculative-config"
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
benchmarks:
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 128
max_out_len: 1500
batch_size: 32
request_rate: 0
baseline: 1433.4454
threshold: 0.97

View File

@@ -0,0 +1,91 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
OMP_NUM_THREADS: "100"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
VLLM_RPC_TIMEOUT: "3600000"
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "3600000"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
_server_cmd: &server_cmd
- "--quantization"
- "ascend"
- "--seed"
- "1024"
- "--no-enable-prefix-caching"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "40960"
- "--max-num-seqs"
- "14"
- "--trust-remote-code"
_benchmarks_gsm8k: &benchmarks_gsm8k
acc:
case_type: accuracy
dataset_path: vllm-ascend/gsm8k-lite
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 95
threshold: 10
_benchmarks_aime: &benchmarks_aime
acc:
case_type: accuracy
dataset_path: vllm-ascend/aime2024
request_conf: vllm_api_general_chat
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
max_out_len: 32768
batch_size: 32
baseline: 86.67
threshold: 10
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp2"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "4096"
- "--speculative-config"
- '{"num_speculative_tokens": 2, "method": "mtp"}'
- "--gpu-memory-utilization"
- "0.92"
benchmarks:
<<: *benchmarks_gsm8k
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp3"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
<<: *envs
HCCL_OP_EXPANSION_MODE: "AIV"
server_cmd: *server_cmd
server_cmd_extra:
- "--max-num-batched-tokens"
- "2048"
- "--speculative-config"
- '{"num_speculative_tokens": 3, "method": "mtp"}'
- "--gpu-memory-utilization"
- "0.9"
- "--compilation-config"
- '{"cudagraph_capture_sizes": [56], "cudagraph_mode": "FULL_DECODE_ONLY"}'
benchmarks:
<<: *benchmarks_aime

View File

@@ -0,0 +1,88 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.5-w8a8"
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
envs:
HCCL_BUFFSIZE: "512"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
HCCL_INTRA_PCIE_ENABLE: "1"
HCCL_INTRA_ROCE_ENABLE: "0"
OMP_PROC_BIND: "false"
VLLM_TORCH_PROFILER_WITH_STACK: "0"
VLLM_TORCH_PROFILER_DIR: "./profile"
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--port"
- "$SERVER_PORT"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--quantization"
- "ascend"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--model-loader-extra-config"
- '{"enable_multithread_load":true,"num_threads":16}'
- "--speculative_config"
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
- "--enable-expert-parallel"
- "--enable-chunked-prefill"
- "--enable-prefix-caching"
- "--max-num-seqs"
- "100"
- "--max-model-len"
- "196608"
- "--seed"
- "1024"
- "--max-num-batched-tokens"
- "6144"
- "--enable-auto-tool-choice"
- "--tool-call-parser"
- "minimax_m2"
- "--reasoning-parser"
- "minimax_m2_append_think"
- "--enable-force-include-usage"
- "--profiler-config"
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,16,40,80,160,256,400]}'
benchmarks:
acc:
case_type: accuracy
dataset_path: vllm-ascend/gpqa
request_conf: vllm_api_general_chat
dataset_conf: gsm8k/gpqa_gen_0_shot_str
max_out_len: 131072
batch_size: 64
baseline: 83
threshold: 5
bos_token_id: 200019
do_sample: true
eos_token_id: 200020
temperature: 1.0
top_p: 0.95
top_k: 40
transformers_version: 4.46.1
ignore_eos: false
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 360
max_out_len: 1500
batch_size: 120
request_rate: 0
baseline: 2042
threshold: 0.97

View File

@@ -0,0 +1,73 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.5-w8a8"
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
envs:
HCCL_BUFFSIZE: "512"
HCCL_OP_EXPANSION_MODE: "AIV"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
TASK_QUEUE_ENABLE: "1"
VLLM-ASCEND_ENABLE_NZ: "1"
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
VLLM_USE_MODELSCOPE: "true"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "1"
- "--port"
- "$SERVER_PORT"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.85"
- "--quantization"
- "ascend"
- "--no-enable-prefix-caching"
- "--additional-config"
- '{"enable_cpu_binding":true}'
- "--speculative_config"
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
- "--enable-expert-parallel"
- "--max-num-seqs"
- "128"
- "--max-num-batched-tokens"
- "16384"
- "--max-model-len"
- "196608"
- "--compilation-config"
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
benchmarks:
acc_aime2025:
case_type: accuracy
dataset_path: vllm-ascend/aime2025
request_conf: vllm_api_general_chat
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
max_out_len: 131072
batch_size: 32
baseline: 90
threshold: 10
bos_token_id: 200019
do_sample: true
eos_token_id: 200020
temperature: 1.0
top_p: 0.95
top_k: 40
transformers_version: 4.46.1
ignore_eos: false
perf:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 512
max_out_len: 1500
batch_size: 128
request_rate: 0
baseline: 1116
threshold: 0.97

View File

@@ -0,0 +1,85 @@
# ==========================================
# Shared Configurations
# ==========================================
_envs: &envs
SERVER_PORT: "DEFAULT_PORT"
HCCL_OP_EXPANSION_MODE: "AIV"
HCCL_BUFFSIZE: "1200"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
OMP_NUM_THREADS: "1"
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
TASK_QUEUE_ENABLE: "1"
_server_cmd: &server_cmd
- "--port"
- "$SERVER_PORT"
- "--host"
- "0.0.0.0"
- "--tensor-parallel-size"
- "8"
- "--data-parallel-size"
- "2"
- "--enable-expert-parallel"
- "--async-scheduling"
- "--max-num-seqs"
- "128"
- "--safetensors-load-strategy"
- 'prefetch'
- "--max-num-batched-tokens"
- "16384"
- "--trust-remote-code"
- "--quantization"
- "ascend"
- "--enable-auto-tool-choice"
- "--tool-call-parser"
- "minimax_m2"
- "--speculative-config"
- '{"method":"eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
- "--compilation-config"
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
- "--additional-config"
- '{"enable_cpu_binding": true, "enable_npugraph_ex": true, "enable_static_kernel": true,"enable_fused_mc2":true,"weight_nz_mode":true,"enable_flashcomm1":true}'
_benchmarks_3500: &benchmarks_3500
perf_50:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 760
max_out_len: 1500
batch_size: 190
request_rate: 0
baseline: 4573.02
threshold: 0.97
perf_20:
case_type: performance
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 192
max_out_len: 1500
batch_size: 48
request_rate: 0
baseline: 2229.147
threshold: 0.97
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "MiniMax-M2.7-3500"
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
envs:
<<: *envs
server_cmd: *server_cmd
server_cmd_extra:
- "--max-model-len"
- "70000"
- "--gpu-memory-utilization"
- "0.8"
- "--no-enable-prefix-caching"
benchmarks:
<<: *benchmarks_3500

View File

@@ -0,0 +1,78 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "prefix-cache-deepseek-r1-0528-w8a8"
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
envs:
OMP_NUM_THREADS: "10"
OMP_PROC_BIND: "false"
HCCL_BUFFSIZE: "1024"
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
SERVER_PORT: "DEFAULT_PORT"
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
server_cmd:
- "--quantization"
- "ascend"
- "--data-parallel-size"
- "2"
- "--tensor-parallel-size"
- "8"
- "--enable-expert-parallel"
- "--port"
- "$SERVER_PORT"
- "--seed"
- "1024"
- "--max-model-len"
- "5200"
- "--max-num-batched-tokens"
- "4096"
- "--max-num-seqs"
- "16"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
- "--speculative-config"
- '{"num_speculative_tokens": 1, "method": "mtp"}'
test_content:
- "benchmark_comparisons"
benchmark_comparisons_args:
- metric: "TTFT"
baseline: "prefix0"
target: "prefix75"
ratio: 0.5
operator: "<"
benchmarks:
warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1024-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 1000
baseline: 0
threshold: 0.97
prefix0:
case_type: performance
dataset_path: vllm-ascend/prefix0-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 18
baseline: 1
threshold: 0.97
prefix75:
case_type: performance
dataset_path: vllm-ascend/prefix75-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 18
baseline: 1
threshold: 0.97

View File

@@ -0,0 +1,70 @@
# ==========================================
# ACTUAL TEST CASES
# ==========================================
test_cases:
- name: "prefix-cache-qwen3-32b-w8a8"
model: "vllm-ascend/Qwen3-32B-W8A8"
envs:
TASK_QUEUE_ENABLE: "1"
HCCL_OP_EXPANSION_MODE: "AIV"
SERVER_PORT: "DEFAULT_PORT"
server_cmd:
- "--quantization"
- "ascend"
- "--reasoning-parser"
- "qwen3"
- "--tensor-parallel-size"
- "4"
- "--port"
- "$SERVER_PORT"
- "--max-model-len"
- "8192"
- "--max-num-batched-tokens"
- "8192"
- "--max-num-seqs"
- "256"
- "--trust-remote-code"
- "--gpu-memory-utilization"
- "0.9"
- "--additional-config"
- '{"enable_weight_nz_layout": true}'
test_content:
- "benchmark_comparisons"
benchmark_comparisons_args:
- metric: "TTFT"
baseline: "prefix0"
target: "prefix75"
ratio: 0.4
operator: "<"
benchmarks:
warm_up:
case_type: performance
dataset_path: vllm-ascend/GSM8K-in1024-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 1000
baseline: 0
threshold: 0.97
prefix0:
case_type: performance
dataset_path: vllm-ascend/prefix0-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 48
baseline: 1
threshold: 0.97
prefix75:
case_type: performance
dataset_path: vllm-ascend/prefix75-in3500-bs210
request_conf: vllm_api_stream_chat
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
num_prompts: 210
max_out_len: 1
batch_size: 48
baseline: 1
threshold: 0.97

Some files were not shown because too many files have changed in this diff Show More