File diff suppressed because it is too large
Load Diff
422
tests/e2e/coverage.md
Normal file
422
tests/e2e/coverage.md
Normal file
@@ -0,0 +1,422 @@
|
||||
The coverage of e2e is as follows:
|
||||
|
||||
## 1-Card Tests
|
||||
|
||||
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
|
||||
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
||||
| _310p/test_classification_310p.py | test_qwen_pooling_classify_correctness | Howeee/Qwen2.5-1.5B-apeach | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_5_dense_tp1_fp16 | Qwen/Qwen3.5-4B | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_5_dense_tp1_fp16_aclgraph | Qwen/Qwen3.5-4B | ✅ | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_fp16 | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_fp16_aclgraph | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp1_w8a8 | vllm-ascend/Qwen3-8B-W8A8 | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| _310p/test_embedding_310p.py | test_bge_m3_correctness | BAAI/bge-m3 | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| _310p/test_embedding_310p.py | test_embed_models_correctness | Qwen/Qwen3-Embedding-0.6B<br>intfloat/multilingual-e5-small | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| _310p/test_scoring_310p.py | test_cross_encoder_score_1_to_1 | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| _310p/test_scoring_310p.py | test_cross_encoder_score_1_to_N | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| _310p/test_scoring_310p.py | test_cross_encoder_score_N_to_N | BAAI/bge-reranker-v2-m3 | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| _310p/test_vl_model_310p.py | test_qwen3_vl_8b_tp1_fp16 | Qwen/Qwen3-VL-8B-Instruct | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_accuracy.py | test_default_full_and_piecewise_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_accuracy.py | test_full_decode_only_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_accuracy.py | test_full_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_accuracy.py | test_npugraph_ex_res_consistency | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_accuracy.py | test_npugraph_ex_with_static_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_logprobs_bitwise_batch_invariance_bs1_vs_bsN | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
|
||||
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_logprobs_without_batch_invariance_should_fail | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
|
||||
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_simple_generation | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | |
|
||||
| aclgraph/test_aclgraph_batch_invariant.py | test_aclgraph_v1_generation_is_deterministic_across_batch_sizes_with_needle | - | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| aclgraph/test_aclgraph_mem.py | test_aclgraph_mem_use | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| compile/test_graphex_norm_quant_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| compile/test_graphex_qknorm_rope_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| compile/test_norm_quant_fusion.py | test_rmsnorm_quant_fusion | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| lora/test_ilama_lora.py | test_ilama_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| lora/test_llama32_lora.py | test_llama_lora | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| lora/test_lora_with_spec_decode.py | test_batch_inference_correctness | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| lora/test_qwen35_densemodel_lora.py | test_qwen35_text_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| lora/test_qwen3_multi_loras.py | test_multi_loras_with_tp_sync | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| lora/test_qwen3_reranker_lora.py | test_reranker_models_lora | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| model_runner_v2/test_basic.py | test_egale_spec_decoding | Qwen/Qwen3-0.6B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
|
||||
| model_runner_v2/test_basic.py | test_qwen3_dense_eager_mode | Qwen/Qwen3-0.6B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | ✅ | |
|
||||
| model_runner_v2/test_basic.py | test_qwen3_dense_graph_mode | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
|
||||
| pooling/test_classification.py | test_qwen_pooling_classify_correctness | Howeee/Qwen2.5-1.5B-apeach | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | |
|
||||
| pooling/test_embedding.py | test_bge_m3_correctness | BAAI/bge-m3 | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| pooling/test_embedding.py | test_causal_embed_models_using_prefix_caching_correctness | Qwen/Qwen3-Embedding-0.6B | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| pooling/test_embedding.py | test_embed_models_correctness | Qwen/Qwen3-Embedding-0.6B<br>intfloat/multilingual-e5-small | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| pooling/test_scoring.py | test_cross_encoder_score_1_to_1 | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| pooling/test_scoring.py | test_cross_encoder_score_1_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| pooling/test_scoring.py | test_cross_encoder_score_N_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| pooling/test_scoring.py | test_embedding_score_1_to_1 | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| pooling/test_scoring.py | test_embedding_score_1_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| pooling/test_scoring.py | test_embedding_score_N_to_N | BAAI/bge-reranker-v2-m3<br>dengcao/ms-marco-MiniLM-L6-v2<br>sentence-transformers/all-MiniLM-L12-v2 | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | |
|
||||
| spec_decode/test_dflash.py | test_dflash_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_draft_parallel.py | test_parallel_drafting_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_eagle.py | test_qwen3_vl_eagle | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| spec_decode/test_eagle.py | test_qwen_eagle3_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_extract_hidden_states.py | test_extract_hidden_states_aclgraph_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_extract_hidden_states.py | test_extract_hidden_states_eager_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_mtp_eagle_correctness.py | test_deepseek_mtp | wemaster/deepseek_mtp_main_random_bf16 | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_ngram.py | test_ngram | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| spec_decode/test_ngram_npu.py | test_ngram_npu_async_acceptance | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_suffix.py | test_suffix_acceptance | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_attention_fa3.py | test_fa3_vs_fia_logprobs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| test_attention_fa3.py | test_fa3_vs_fia_mixed_lengths | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | ✅ |
|
||||
| test_attention_fa3.py | test_fa3_vs_fia_single_prompt | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| test_attention_fa3.py | test_fa3_vs_fia_with_chunkprefill | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| test_batch_invariant.py | test_logprobs_bitwise_batch_invariance_bs1_vs_bsN | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
|
||||
| test_batch_invariant.py | test_logprobs_without_batch_invariance_should_fail | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | ✅ | ✅ | |
|
||||
| test_batch_invariant.py | test_simple_generation | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | |
|
||||
| test_batch_invariant.py | test_v1_generation_is_deterministic_across_batch_sizes_with_needle | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| test_camem.py | test_end_to_end | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | |
|
||||
| test_completion_with_prompt_embeds.py | test_mixed_prompt_embeds_and_text | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ |
|
||||
| test_cpu_offloading.py | test_cpu_offloading | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| test_guided_decoding.py | test_guided_json_completion | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_guided_decoding.py | test_guided_regex | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_minicpm.py | test_minicpm | OpenBMB/MiniCPM4-0.5B<br>openbmb/MiniCPM-2B-sft-bf16 | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_multi_instance.py | test_two_instances_on_single_card | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_multistream_overlap_shared_expert.py | test_models_with_multistream_overlap_shared_expert | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_0_6b.py | test_dense_default_full_and_piecewise_graph | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_5_0_8b.py | test_mamba_ssm_multimodal_reasoning_mtp_full_decode_only | Qwen/Qwen3.5-0.8B | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_8b_w8a8.py | test_dense_w8a8_eagle3_full_graph | RedHatAI/Qwen3-8B-speculator.eagle3<br>vllm-ascend/Qwen3-8B-W8A8 | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_embedding_0_6b.py | test_embedding_full_decode_only | Qwen/Qwen3-Embedding-0.6B | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_sampler.py | test_qwen3_exponential_overlap | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_sampler.py | test_qwen3_prompt_logprobs | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | |
|
||||
| test_sampler.py | test_qwen3_topk | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_vlm.py | test_multimodal_audio | Qwen/Qwen2-Audio-7B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_vlm.py | test_multimodal_vl | openai-mirror/whisper-large-v3-turbo | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_vlm.py | test_multimodal_vl_language_model_only | Qwen/Qwen3-VL-8B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_vlm.py | test_whisper | openai-mirror/whisper-large-v3-turbo | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_xlite.py | test_models_with_xlite_decode_only | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| test_xlite.py | test_models_with_xlite_full_mode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
|
||||
## 2-Card Tests
|
||||
|
||||
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
|
||||
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
||||
| aclgraph/test_aclgraph_capture_replay.py | test_models_aclgraph_capture_replay_metrics_dp2 | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| aclgraph/test_full_graph_mode.py | test_qwen3_moe_full_decode_only_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| aclgraph/test_full_graph_mode.py | test_qwen3_moe_full_graph_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| lora/test_ilama_lora_tp2.py | test_ilama_lora_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| lora/test_llama32_lora_tp2.py | test_llama_lora_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_spec_decode.py | test_eagle3_sp_acceptance | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | ✅ | |
|
||||
| spec_decode/test_spec_decode.py | test_p_eagle_acceptance | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | ✅ | |
|
||||
| spec_decode/test_spec_decode.py | test_qwen3_eagle3_pcp2_tp1 | - | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_data_parallel.py | test_qwen3_inference_dp2 | Qwen/Qwen3-30B-A3B<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_deepseek_multistream_moe.py | test_deepseek_multistream_moe_tp2 | vllm-ascend/DeepSeek-V3-Pruning | | | ✅ | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_external_launcher.py | test_qwen3_external_launcher | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_external_launcher.py | test_qwen3_external_launcher_with_matmul_allreduce | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | |
|
||||
| test_external_launcher.py | test_qwen3_external_launcher_with_sleepmode | Qwen/Qwen3-8B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_external_launcher.py | test_qwen3_external_launcher_with_sleepmode_level2 | Qwen/Qwen3-8B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_external_launcher.py | test_qwen3_moe_external_launcher_ep_tp2 | Qwen/Qwen3-0.6B | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_flashcomm_distributed.py | test_deepseek_v2_lite_fc1_tp2 | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_flashcomm_distributed.py | test_qwen3_dense_fc1_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_flashcomm_distributed.py | test_qwen3_dense_prefetch_mlp_weight_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_flashcomm_distributed.py | test_qwen3_moe_fc2_oshard_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | | | | |
|
||||
| test_gpt_oss_distributed.py | test_gpt_oss_distributed_tp2 | - | | | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_moe_routing_replay.py | test_qwen3_moe_routing_replay | Qwen/Qwen3-30B-A3B<br>Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_offline_weight_load.py | test_qwen3_offline_load_and_sleepmode_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_prefix_caching.py | test_models_prefix_cache_tp2 | Qwen/Qwen3-8B<br>deepseek-ai/DeepSeek-V2-Lite-Chat | | ✅ | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | |
|
||||
| test_qwen3_30b_a3b.py | test_moe_tp_ep_eplb_full_decode_only | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_6_27b_fia.py | test_qwen3_6_27b_multimodel_fia_eager | Qwen/Qwen3.6-27B/ | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_6_27b_fia.py | test_qwen3_6_27b_multimodel_fia_acl_graph | Qwen/Qwen3.6-27B/ | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_vl_30b_a3b_instruct.py | test_multimodal_reasoning_pp_full_decode_only | Qwen/Qwen3-VL-30B-A3B-Instruct | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_sequence_parallelism_moe.py | test_sequence_parallelism_moe_patterns | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_shared_expert_dp.py | test_deepseek_v2_lite_enable_shared_expert_dp_tp2 | deepseek-ai/DeepSeek-V2-Lite | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_sp_pass.py | test_qwen3_vl_sp_tp2 | Qwen/Qwen3-VL-2B-Instruct | | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
|
||||
## 4-Card Tests
|
||||
|
||||
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
|
||||
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp2_fp16 | Qwen/Qwen3-8B | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| _310p/test_dense_model_310p.py | test_qwen3_dense_tp4_w8a8 | vllm-ascend/Qwen3-32B-W8A8 | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| _310p/test_moe_model_310p.py | test_qwen3_5_moe_tp4_fp16 | Qwen/Qwen3.5-35B-A3B | ✅ | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| _310p/test_moe_model_310p.py | test_qwen3_moe_tp2_w8a8 | vllm-ascend/Qwen3-30B-A3B-W8A8 | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| _310p/test_moe_model_310p.py | test_qwen3_moe_tp4_fp16 | Qwen/Qwen3-30B-A3B | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| _310p/test_vl_model_310p.py | test_qwen3_vl_8b_tp2_fp16 | Qwen/Qwen3-VL-8B-Instruct | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| context_parallel/test_accuracy.py | test_accuracy_dcp_only_eager | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_accuracy.py | test_accuracy_dcp_only_graph | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_accuracy.py | test_accuracy_pcp_only | Qwen/Qwen3-8B<br>vllm-ascend/DeepSeek-V2-Lite-W8A8 | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_accuracy.py | test_models_long_sequence_cp_kv_interleave_size_output_between_tp_and_cp | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_accuracy.py | test_models_long_sequence_output_between_tp_and_cp | vllm-ascend/DeepSeek-V2-Lite-W8A8 | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_dcp_basic | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_dcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_dcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_deepseek_v4_w4a8_dsa_cp_basic_greedy | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | ✅ | | ✅ | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_models_pcp_dcp_basic | Qwen/Qwen3-Next-80B-A3B-Instruct<br>deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/DeepSeek-V3.2-W8A8-Pruning<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_models_pcp_dcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_models_pcp_dcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_pcp_basic | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_pcp_full_graph | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_pcp_piece_wise | deepseek-ai/DeepSeek-V2-Lite-Chat | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | |
|
||||
| context_parallel/test_basic.py | test_qwen3_5_4b_multimodal_single_and_multi_image | Qwen/Qwen3.5-4B | | | | | | | ✅ | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | |
|
||||
| context_parallel/test_basic.py | test_qwen3_vl_8b_multimodal_single_and_multi_image | Qwen/Qwen3-VL-8B-Instruct | | | | | | | | ✅ | ✅ | | | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | |
|
||||
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_mixed_length_prompts_including_1_token | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | ✅ |
|
||||
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_basic | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_default_full_and_piecewise | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_cp_full_graph | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_chunked_prefill_cp.py | test_models_chunked_prefill_with_empty_kvcache | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | ✅ | | | ✅ | |
|
||||
| context_parallel/test_mtp.py | test_dcp_mtp3_full_graph | - | | | | | | | | | ✅ | | ✅ | | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_mtp.py | test_pcp_dcp_mtp1_eager | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_eager | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_full_graph | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_mtp.py | test_pcp_dcp_mtp3_piecewise_graph | - | | | | | | | | | ✅ | | ✅ | ✅ | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_mtp.py | test_pcp_eagle3_eager | - | | | | | | | | | ✅ | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_basic | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_default_full_and_piecewise | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| context_parallel/test_prefix_caching_cp.py | test_models_prefix_cache_with_cp_full_graph | vllm-ascend/DeepSeek-V2-Lite-W8A8<br>vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| spec_decode/test_mtp_qwen3_next.py | test_qwen3_next_mtp_acceptance_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_data_parallel_tp2.py | test_qwen3_inference_dp2_tp2 | Qwen/Qwen3-30B-A3B | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_deepseek_v3_2_w8a8_pruning.py | test_moe_w8a8_tp_pp_ep_full_decode_only | vllm-ascend/DeepSeek-V3.2-W8A8-Pruning | | | ✅ | | | | | | ✅ | ✅ | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_deepseek_v3_2_w8a8_pruning.py | test_pd_disaggregation_w8a8_sfa_dsa_full_decode_only | vllm-ascend/DeepSeek-V3.2-W8A8-Pruning | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_deepseek_v4.py | test_deepseek_v4_w4a8_tp4_basic_greedy | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_deepseek_v4.py | test_deepseek_v4_w4a8_tp4_index_cache_freq4 | gdydems/DeepSeek-V4-Flash-w4a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_pipeline_parallel.py | test_models_pp2_dp2 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_pipeline_parallel.py | test_models_pp2_tp2 | - | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| test_profiling_chunk_performance.py | test_profiling_chunk_ttft_performance | - | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| test_qwen3_5.py | test_qwen3_5_27b_distributed_mp_tp4 | Qwen/Qwen3.5-27B | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4 | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3 | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | ✅ | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_5.py | test_qwen3_5_35b_distributed_mp_tp4_full_decode_only_mtp3_flashcomm | Qwen/Qwen3.5-35B-A3B | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| test_qwen3_next.py | test_qwen3_next_distributed_mp_flash_comm_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_next.py | test_qwen3_next_distributed_mp_full_decode_only_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_next.py | test_qwen3_next_distributed_mp_graph_mode_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_next.py | test_qwen3_next_distributed_mp_tp4 | Qwen/Qwen3-Next-80B-A3B-Instruct | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| test_qwen3_next.py | test_qwen3_next_w8a8dynamic_distributed_tp4_ep | vllm-ascend/Qwen3-Next-80B-A3B-Instruct-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
|
||||
## Nightly Tests
|
||||
|
||||
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
|
||||
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
||||
| 310p/single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule_v310.py | test_recurrent_gated_delta_rule_v310 | - | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| multi_node/external_dp/config/GLM5_1-W8A8-EP-external.yaml | multi-node-glm-5.1-w8a8-ep-external-dp | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | ✅ | | | | | |
|
||||
| multi_node/external_dp/scripts/test_external_dp.py | test_external_dp | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-R1-W8A8-EPLB.yaml | test DeepSeek-R1-W8A8 disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | ✅ | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-R1-W8A8-longseq.yaml | test DeepSeek-R1-W8A8-longseq disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | ✅ | | | ✅ | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-R1-W8A8.yaml | test DeepSeek-R1-W8A8 disaggregated_prefill | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-V3.1-BF16.yaml | test DeepSeek-V3.1-BF16 on A3 | unsloth/DeepSeek-V3.1-BF16 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-A3-dual-nodes.yaml | test DeepSeek-V3.2-W8A8 on A3 | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-EP.yaml | test DeepSeek-V3.2-W8A8-EP disaggregated_prefill | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/GLM5_1-W8A8-A2-dual-nodes.yaml | multi-node-GLM-5.1-w8a8-A2 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/GLM5_1-W8A8-A3-dual-nodes.yaml | multi-node-GLM-5.1-w8a8-A3 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/GLM5_1-W8A8-EP.yaml | multi-node-GLM-5.1-w8a8-EP | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/Kimi-K2_5-W4A8-A2-dual-nodes.yaml | test Kimi-K2.5-W4A8 A2 dual nodes | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-A22B-A2.yaml | test Qwen3-235B-A22B multi-dp on A2 | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-A22B.yaml | test Qwen3-235B-A22B multi-dp | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-W8A8-EPLB.yaml | test Qwen3-235B-A22B-W8A8 disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | ✅ | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-W8A8-longseq.yaml | test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-W8A8.yaml | test Qwen3-235B-A22B-W8A8 disaggregated_prefill | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-235B-disagg-pd.yaml | test Qwen3-235B-A22B disaggregated_prefill | Qwen/Qwen3-235B-A22B | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/Qwen3-VL-235B-disagg-pd.yaml | test Qwen3-VL-235B-A22B disaggregated_prefill | Qwen/Qwen3-VL-235B-A22B-Instruct | | | | | | | | ✅ | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-EPLB | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-aclgraph | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/DeepSeek-R1-0528-W8A8.yaml | DeepSeek-R1-0528-W8A8-single | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/DeepSeek-V3.2-W8A8.yaml | DeepSeek-V3.2-W8A8-TP8-DP2 | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/DeepSeek-V4-Flash-W8A8-A3.yaml | DeepSeek-V4-Flash-W8A8-A3 | Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp | | | ✅ | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/GLM-4.7.yaml | GLM-4.7-TP8-DP2-decodegraph | Eco-Tech/GLM-4.7-W8A8-floatmtp | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/models/configs/Hy3-preview.yaml | Hy3-preview-TP16-EP-MTP | Tencent-Hunyuan/Hy3-preview | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Kimi-K2-Thinking.yaml | Kimi-K2-Thinking-TP16-Case | moonshotai/Kimi-K2-Thinking | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-Case | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | ✅ | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/MTPX-DeepSeek-R1-0528-W8A8.yaml | MTPX-DeepSeek-R1-0528-W8A8-mtp2 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/MTPX-DeepSeek-R1-0528-W8A8.yaml | MTPX-DeepSeek-R1-0528-W8A8-mtp3 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/MiniMax-M2.5-w8a8-QuaRot-A2.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Prefix-Cache-DeepSeek-R1-0528-W8A8.yaml | prefix-cache-deepseek-r1-0528-w8a8 | vllm-ascend/DeepSeek-R1-0528-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Prefix-Cache-Qwen3-32B-Int8.yaml | prefix-cache-qwen3-32b-w8a8 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-EPLB | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-full_graph | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3-235B-A22B-W8A8.yaml | Qwen3-235B-A22B-W8A8-piecewise | vllm-ascend/Qwen3-235B-A22B-W8A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3-30B-A3B-W4A8-llm-compressor.yaml | Qwen3-30B-A3B-W4A8-llm-compressor | vllm-ascend/Qwen3-30B-A3B-Instruct-2507-quantized.w4a8 | | | ✅ | | | | | | ✅ | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3-30B-A3B-W8A8.yaml | Qwen3-30B-A3B-W8A8-TP1 | vllm-ascend/Qwen3-30B-A3B-W8A8 | | | ✅ | | | | | | | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Qwen3-30B-QuaRot-eagle3.yaml | Qwen3-30B-QuaRot | vllm-ascend/Qwen3-30B-A3B-W8A8-QuaRot | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Qwen3-32B-Int8-A2.yaml | Qwen3-32B-W8A8-aclgraph-a2 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3-32B-Int8-A2.yaml | Qwen3-32B-W8A8-single-a2 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3-32B-Int8.yaml | Qwen3-32B-W8A8-aclgraph-a3 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3-32B-Int8.yaml | Qwen3-32B-W8A8-single-a3 | vllm-ascend/Qwen3-32B-W8A8 | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3-32B-QuaRot-eagle3.yaml | Qwen3-32B-QuaRot | vllm-ascend/Qwen3-32B-W8A8-QuaRot | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | |
|
||||
| single_node/models/configs/Qwen3-VL-235B-A22B-Instruct-W8A8.yaml | Qwen3-VL-235B-A22B-Instruct-W8A8 | Eco-Tech/Qwen3-VL-235B-A22B-Instruct-w8a8-QuaRot | | | | | | | | ✅ | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3-VL-32B-Instruct-W8A8.yaml | Qwen3-VL-32B-Instruct-W8A8 | Eco-Tech/Qwen3-VL-32B-Instruct-w8a8-QuaRot | | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-A3 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/models/configs/Qwen3.5-27B-w8a8-A2.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3.yaml | Qwen3.5-397B-A17B-w8a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/models/configs/Qwen3.5-397B-A17B-w4a8-mtp-A2.yaml | Qwen3.5-397B-A17B-w4a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w4a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/ops/multicard_ops_a2/test_matmul_allreduce_add_rmsnorm.py | test_matmul_allreduce_add_rmsnorm_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine_bf16.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_ffn_combine_w4a8.py | test_dispatch_ffn_combine_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_base | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_dynamic_eplb | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/multicard_ops_a3/test_dispatch_gmm_combine_decode.py | test_dispatch_gmm_combine_decode_with_mc2_mask | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_add_rms_norm_bias.py | test_quant_fpx_linear | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_boundary_conditions | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_random_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_batch_matmul_transpose.py | test_zero_values | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_bgmv_expand.py | test_bgmv_expand | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_bgmv_shrink.py | test_bgmv_shrink | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_causal_conv1d_310.py | test_ascend_causal_conv1d_310_fn | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| single_node/ops/singlecard_ops/test_causal_conv1d_310.py | test_causal_conv1d_310_update | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_copy_and_expand_eagle_inputs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_large_tokens_per_request | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_large_tokens_shift_true | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_minimal_case | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_copy_and_expand_eagle_inputs.py | test_no_rejected_tokens | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_dequant_swiglu_quant.py | test_npu_dequant_swiglu_quant_with_limit | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_bulk_dma_alignment | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_extreme_large_batch | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_large_batch_multi_row | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_non_bulk_dma_fallback | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_non_default_params | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_output_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_small_batch_optimization | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/test_fused_gdn_gating.py | test_fused_gdn_gating_vs_reference | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_moe.py | test_select_experts | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_moe.py | test_select_experts_invalid_scoring_func | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_moe.py | test_token_dispatcher_with_all_gather | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_fused_moe.py | test_token_dispatcher_with_all_gather_quant | - | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_gating_top_k_softmax.py | test_quant_fpx_linear | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_gmm_swiglu_quant_weight_nz_tensor_list.py | test_gmm_swiglu_quant_weight_nz_tensor_list | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_grouped_matmul_swiglu_quant.py | test_grouped_matmul_swiglu_quant_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_hamming_dist_top_k.py | test_hamming_dist_top_k | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_hamming_dist_top_k.py | test_hamming_dist_top_k_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_mla_preprocess.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_mla_preprocess_nq.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_mla_preprocess_qdown.py | test_mla_preprocess_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/test_moe_init_routing_custom.py | test_moe_init_routing_custom | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_attrs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_basic | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_decode | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_discard | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_exact_match | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_full_capacity | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_k1 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_minimal | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_no_valid_sampled | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_padding | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_ngram_spec_decode.py | test_ngram_spec_decode_prefill | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_npu_hc_pre.py | test_npu_hc_pre_v1_v2_bf16_3d_input | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_npu_hc_pre.py | test_npu_hc_pre_v1_v2_bf16_4d_input | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_npu_moe_gating_top_k.py | test_npu_moe_gating_topk_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule.py | test_recurrent_gated_delta_rule | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule.py | test_recurrent_gated_delta_rule_no_accepted | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_recurrent_gated_delta_rule_310.py | test_fused_recurrent_gated_delta_rule_310 | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | |
|
||||
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_bf16_shape | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_compare | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_reshape_and_cache_bnsd.py | test_reshape_and_cache_bnsd_with_expected_output | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_transpose_kv_cache_by_block.py | test_transpose_kv_cache_by_block | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/test_vocabparallelembedding.py | test_get_masked_input_and_mask | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_apply_penalties_triton.py | test_apply_all_penalties_v1_vs_ascend | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_different_shapes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_edge_cases | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_no_bad_words | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_bad_words.py | test_apply_bad_words_token_limit | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_batch_memcpy.py | test_batch_memcpy | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/triton/test_bincount.py | test_bincount_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_ascend_causal_conv1d | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d_update_qwen3_next_shape | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_causal_conv1d.py | test_causal_conv1d_update_with_batch_gather | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/triton/test_chunk_gated_delta_rule.py | test_chunk_gated_delta_rule_310_state_layout_matches_vllm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_chunk_gated_delta_rule.py | test_triton_fusion_ops | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_clear_ssm_states.py | test_clear_ssm_states_ref_parity | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_slot_mapping.py | test_compute_slot_mapping_npu_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_deterministic | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_dtypes | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_edge_cases | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_token_logprobs.py | test_topk_log_softmax_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_compute_topk_logprobs.py | test_compute_topk_logprobs | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_fused_gdn_gating.py | test_fused_gdn_gating_310p_parity_precision | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_fused_qkvzba_split_reshape_cat.py | test_fused_qkvzba_split_reshape_cat | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_fused_recurrent_gated_delta_rule.py | test_fused_recurrent_gated_delta_rule_310_state_layout_matches_vllm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_fused_recurrent_gated_delta_rule.py | test_fused_recurrent_gated_delta_rule_310p_parity_precision | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_fused_sigmoid_gating_delta_rule.py | test_triton_fusion_ops | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_l2norm.py | test_l2norm | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_log_softmax.py | test_topk_log_softmax_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_min_p.py | test_apply_min_p_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_mrope.py | test_mrotary_embedding_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_muls_add.py | test_muls_add_triton_correctness | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_penality.py | test_apply_penalties | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_post_update.py | test_post_update | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | |
|
||||
| single_node/ops/singlecard_ops/triton/test_prepare_inputs_padded.py | test_prepare_inputs_padded | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_rejection_sample.py | test_rejection_random_sample | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_rejection_sample.py | test_rejection_sampler_block_verify_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel_siso | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_rope.py | test_rotary_embedding_triton_kernel_with_cos_sin_cache | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_mrope.py | test_split_qkv_rmsnorm_mrope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_rope.py | test_split_qkv_rmsnorm_rope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_split_qkv_rmsnorm_rope.py | test_split_qkv_rmsnorm_rope_with_bias | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_split_qkv_tp_rmsnorm_rope.py | test_split_qkv_tp_rmsnorm_rope | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/ops/singlecard_ops/triton/test_temperature.py | test_temperature_kernel | - | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
|
||||
## Weekly Tests
|
||||
|
||||
| Test file | Test method | Model | 310P | Dense | MoE | Embedding | Classification | Reranker | Mamba/SSM | Multimodal Reasoning | TP | PP | EP | PCP | DCP | Context Parallel | EPLB | Dynamic EPLB | Multistream MoE | Full Graph | Full Decode Only Graph | Default FULL_AND_PIECEWISE Graph | Piecewise Graph | Eager Mode | PD disaggregation | W8A8 | W4A8 | FP16 | LoRA | Multi-LoRA | Runtime LoRA updating | Fully sharded LoRA parameterization | Spec Decode | MTP | Eagle-3 | SFA/DSA | DSA CP | Pooling runner | Score API | Classification API | Distributed executor mp | Flash Attention 3 | FIA comparison | Chunked Prefill | Prefix Caching | CPU/KV offloading | KV transfer/events | Sleep/Wake memory | Xlite Graph | CP KV Interleave | Long Sequence | FlashComm1 env | Skipped | Conditional skip | Logprobs | Batch inference | Mixed lengths |
|
||||
| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |
|
||||
| multi_node/internal_dp/config/DeepSeek-V3.yaml | test DeepSeek-V3 disaggregated_prefill | vllm-ascend/DeepSeek-V3-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| multi_node/internal_dp/config/DeepSeek-V3_2-W8A8-EP_weekly.yaml | weekly test DeepSeek-V3.2-W8A8-EP disaggregated_prefill | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | ✅ | | | | ✅ | | | | | | |
|
||||
| multi_node/internal_dp/config/GLM-4.7-W8A8C8-Mooncake-Layerwise.yaml | test GLM-4.7-W8A8C8 PD separation with mooncake layerwise connector | vllm-ascend/GLM-4.7-W8A8C8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | ✅ | ✅ | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | | | | | |
|
||||
| single_node/configs/DeepSeek-V3.2-W8A8_A3_weekly.yaml | DeepSeek-V3.2-W8A8-weekly | vllm-ascend/DeepSeek-V3.2-W8A8 | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/GLM-5.yaml | GLM-5-TP16-DP1-decodegraph | Eco-Tech/GLM-5-w4a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | ✅ | | | | | ✅ | | | | | | ✅ | ✅ | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs10 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
|
||||
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs20 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
|
||||
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs32 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
|
||||
| single_node/configs/GLM-5_1-W8A8_A3_weekly.yaml | GLM-5_1-w8a8-High—Throughput-bs8 | Eco-Tech/GLM-5.1-w8a8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | ✅ | | ✅ | | | | | ✅ | | | | | | | ✅ | ✅ | | | | | | | | | | ✅ | | | | | | | | | | | | | |
|
||||
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT20-32k-0.5k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT20-32k-0.5k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT50-32k-0.5k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5-32k-512.yaml | Kimi-K2.5-W4A8-TOPT50-32k-0.5k-prefix-cache90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-Case | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-128k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-16k-1k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT20-64k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | | ✅ | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-128k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-16k-1k | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Kimi-K2.5.yaml | Kimi-K2.5-W4A8-TOPT50-64k-1k-pc90 | Eco-Tech/Kimi-K2.5-W4A8 | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | | | | ✅ | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-W8A8-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-32-8 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in128k-64-16 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in16k-120-30 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in16k-16-4 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-36-9 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-4-1-90 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in32k-80-20 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in64k-4-1 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/MiniMax-M2.5-w8a8-QuaRot-A3.yaml | MiniMax-M2.5-w8a8-in64k-72-18 | Eco-Tech/MiniMax-M2.5-w8a8-QuaRot | | ✅ | | | | | | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/Qwen2.5-VL-7B-Instruct-EPD.yaml | Qwen2.5-VL-7B-Instruct-epd | Qwen/Qwen2.5-VL-7B-Instruct | | | | | | | | ✅ | | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/configs/Qwen3-32B.yaml | Qwen3-32B-TP4 | Qwen/Qwen3-32B | | ✅ | | | | | | | ✅ | | | | | | | | | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A2.yaml | Qwen3.5-122B-A10B-W8A8-single-A2 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-A3 | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-16k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-32k-0.5k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT20-64k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-16k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-32k-0.5k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-122B-A10B-W8A8-A3.yaml | Qwen3.5-122B-A10B-W8A8-TPOT50-64k-1k | Eco-Tech/Qwen3.5-122B-A10B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in128k-28-7 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in128k-4-1 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in16k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in16k-56-14 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-56-14 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in32k-8-2 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-16-4 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-48-12 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-8-2 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-27B-w8a8-A3.yaml | Qwen3.5-27B-w8a8-in64k-8-2-90 | Eco-Tech/Qwen3.5-27B-w8a8-mtp | | | | | | | ✅ | | ✅ | | | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | ✅ | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3.yaml | Qwen3.5-397B-A17B-w8a8-mtp | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | ✅ | | | | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs136 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs144 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs32_in65536 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs48 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs8 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-High—Throughput-bs80 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs160 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs16_in65536 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs2 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs32 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
| single_node/configs/Qwen3.5-397B-A17B-W8A8-mtp-A3_weekly.yaml | Qwen3-397B-A17B-w8a8-A3-Minimal-Delay-bs36 | Eco-Tech/Qwen3.5-397B-A17B-w8a8-mtp | | | | | | | ✅ | | ✅ | | ✅ | | | | | | | | | | | ✅ | | ✅ | | | | | | | | | | | | | | | | | | | | | | | | | | ✅ | | | | | |
|
||||
@@ -20,7 +20,6 @@ trap clean_venv EXIT
|
||||
|
||||
function install_system_packages() {
|
||||
if command -v apt-get >/dev/null; then
|
||||
sed -i 's|ports.ubuntu.com|mirrors.tuna.tsinghua.edu.cn|g' /etc/apt/sources.list
|
||||
apt-get update -y && apt-get install -y gcc g++ cmake libnuma-dev wget git curl jq
|
||||
elif command -v yum >/dev/null; then
|
||||
yum update -y && yum install -y gcc g++ cmake numactl-devel wget git curl jq
|
||||
@@ -30,31 +29,56 @@ function install_system_packages() {
|
||||
}
|
||||
|
||||
function config_pip_mirror() {
|
||||
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
|
||||
pip config set global.index-url http://cache-service.nginx-pypi-cache.svc.cluster.local/pypi/simple
|
||||
pip config set global.trusted-host cache-service.nginx-pypi-cache.svc.cluster.local
|
||||
|
||||
if [ -f /etc/os-release ]; then
|
||||
. /etc/os-release
|
||||
case "$ID" in
|
||||
ubuntu|debian)
|
||||
sed -Ei 's@(ports|archive).ubuntu.com@cache-service.nginx-pypi-cache.svc.cluster.local:8081@g' /etc/apt/sources.list
|
||||
;;
|
||||
openEuler|centos|rhel|fedora)
|
||||
sed -Ei 's@https?://[^/]+/(openeuler|centos|fedora)@http://cache-service.nginx-pypi-cache.svc.cluster.local:8081/\1@g' /etc/yum.repos.d/*.repo
|
||||
;;
|
||||
esac
|
||||
fi
|
||||
}
|
||||
|
||||
|
||||
function install_binary_test() {
|
||||
|
||||
install_system_packages
|
||||
config_pip_mirror
|
||||
install_system_packages
|
||||
create_vllm_venv
|
||||
pip install -r ${SCRIPT_DIR}/../../docs/requirements-docs.txt
|
||||
|
||||
PIP_VLLM_VERSION=$(get_version pip_vllm_version)
|
||||
VLLM_VERSION=$(get_version vllm_version)
|
||||
PIP_VLLM_ASCEND_VERSION=$(get_version pip_vllm_ascend_version)
|
||||
_info "====> Install vllm==${PIP_VLLM_VERSION} and vllm-ascend ${PIP_VLLM_ASCEND_VERSION}"
|
||||
|
||||
# Setup extra-index-url for x86 & torch_npu dev version
|
||||
pip config set global.extra-index-url "https://download.pytorch.org/whl/cpu/ https://mirrors.huaweicloud.com/ascend/repos/pypi"
|
||||
# Setup extra-index-url for public PyPI mirror, Ascend packages, and PyTorch CPU wheels.
|
||||
local pip_extra_index_urls=(
|
||||
"https://mirrors.huaweicloud.com/repository/pypi/variant"
|
||||
"https://mirrors.huaweicloud.com/ascend/repos/pypi"
|
||||
"https://download.pytorch.org/whl/cpu/"
|
||||
)
|
||||
local IFS=" "
|
||||
pip config set global.extra-index-url "${pip_extra_index_urls[*]}"
|
||||
|
||||
pip install vllm=="$(get_version pip_vllm_version)"
|
||||
pip install vllm-ascend=="$(get_version pip_vllm_ascend_version)"
|
||||
# The vLLM version already in pypi, we install from pypi.
|
||||
pip install --default-timeout=300 --retries 3 vllm=="${PIP_VLLM_VERSION}"
|
||||
|
||||
pip install vllm-ascend=="${PIP_VLLM_ASCEND_VERSION}"
|
||||
|
||||
pip list | grep vllm
|
||||
|
||||
# Verify the installation
|
||||
_info "====> Run offline example test"
|
||||
pip install modelscope
|
||||
python3 "${SCRIPT_DIR}/../../examples/offline_inference_npu.py"
|
||||
cd ${SCRIPT_DIR}/../../examples && python3 ./offline_inference_npu.py
|
||||
cd -
|
||||
|
||||
}
|
||||
|
||||
|
||||
1327
tests/e2e/generate_coverage_md.py
Normal file
1327
tests/e2e/generate_coverage_md.py
Normal file
File diff suppressed because it is too large
Load Diff
@@ -17,16 +17,11 @@
|
||||
# Adapted from vllm-project/vllm/blob/main/tests/models/utils.py
|
||||
#
|
||||
|
||||
from typing import Dict, List, Optional, Sequence, Tuple, Union
|
||||
from collections.abc import Sequence
|
||||
|
||||
from vllm_ascend.utils import vllm_version_is
|
||||
from vllm.logprobs import PromptLogprobs, SampleLogprobs
|
||||
|
||||
if vllm_version_is("0.10.2"):
|
||||
from vllm.sequence import PromptLogprobs, SampleLogprobs
|
||||
else:
|
||||
from vllm.logprobs import PromptLogprobs, SampleLogprobs
|
||||
|
||||
TokensText = Tuple[List[int], str]
|
||||
TokensText = tuple[list[int], str]
|
||||
|
||||
|
||||
def check_outputs_equal(
|
||||
@@ -42,16 +37,18 @@ def check_outputs_equal(
|
||||
"""
|
||||
assert len(outputs_0_lst) == len(outputs_1_lst)
|
||||
|
||||
for prompt_idx, (outputs_0,
|
||||
outputs_1) in enumerate(zip(outputs_0_lst,
|
||||
outputs_1_lst)):
|
||||
for prompt_idx, (outputs_0, outputs_1) in enumerate(zip(outputs_0_lst, outputs_1_lst)):
|
||||
output_ids_0, output_str_0 = outputs_0
|
||||
output_ids_1, output_str_1 = outputs_1
|
||||
|
||||
# The text and token outputs should exactly match
|
||||
fail_msg = (f"Test{prompt_idx}:"
|
||||
f"\n{name_0}:\t{output_str_0!r}"
|
||||
f"\n{name_1}:\t{output_str_1!r}")
|
||||
fail_msg = (
|
||||
f"Test{prompt_idx}:"
|
||||
f"\n{name_0}:\t{output_str_0!r}"
|
||||
f"\n{name_1}:\t{output_str_1!r}"
|
||||
f"\n{name_0}:\t{output_ids_0!r}"
|
||||
f"\n{name_1}:\t{output_ids_1!r}"
|
||||
)
|
||||
|
||||
assert output_str_0 == output_str_1, fail_msg
|
||||
assert output_ids_0 == output_ids_1, fail_msg
|
||||
@@ -63,9 +60,7 @@ def check_outputs_equal(
|
||||
# * List of top sample logprobs for each sampled token
|
||||
#
|
||||
# Assumes prompt logprobs were not requested.
|
||||
TokensTextLogprobs = Tuple[List[int], str, Optional[Union[List[Dict[int,
|
||||
float]],
|
||||
SampleLogprobs]]]
|
||||
TokensTextLogprobs = tuple[list[int], str, list[dict[int, float]] | SampleLogprobs | None]
|
||||
|
||||
# Representation of generated sequence as a tuple of
|
||||
# * Token ID list
|
||||
@@ -74,6 +69,9 @@ TokensTextLogprobs = Tuple[List[int], str, Optional[Union[List[Dict[int,
|
||||
# * Optional list of top prompt logprobs for each prompt token
|
||||
#
|
||||
# Allows prompt logprobs to be requested.
|
||||
TokensTextLogprobsPromptLogprobs = Tuple[
|
||||
List[int], str, Optional[Union[List[Dict[int, float]], SampleLogprobs]],
|
||||
Optional[Union[List[Optional[Dict[int, float]]], PromptLogprobs]]]
|
||||
TokensTextLogprobsPromptLogprobs = tuple[
|
||||
list[int],
|
||||
str,
|
||||
list[dict[int, float]] | SampleLogprobs | None,
|
||||
list[dict[int, float] | None] | PromptLogprobs | None,
|
||||
]
|
||||
|
||||
21
tests/e2e/models/configs/ERNIE-4.5-21B-A3B-PT.yaml
Normal file
21
tests/e2e/models/configs/ERNIE-4.5-21B-A3B-PT.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "PaddlePaddle/ERNIE-4.5-21B-A3B-PT"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.71
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: "auto"
|
||||
25
tests/e2e/models/configs/Hunyuan-A13B-Instruct.yaml
Normal file
25
tests/e2e/models/configs/Hunyuan-A13B-Instruct.yaml
Normal file
@@ -0,0 +1,25 @@
|
||||
model_name: "Tencent-Hunyuan/Hunyuan-A13B-Instruct"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 4
|
||||
dtype: auto
|
||||
max_model_len: 32768
|
||||
gpu_memory_utilization: 0.90
|
||||
enforce_eager: true
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.37
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.28
|
||||
|
||||
num_fewshot: 5
|
||||
limit: 1000
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
21
tests/e2e/models/configs/InternVL3_5-8B-hf.yaml
Normal file
21
tests/e2e/models/configs/InternVL3_5-8B-hf.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "OpenGVLab/InternVL3_5-8B-hf"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 40960
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "mmmu_val"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.58
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
23
tests/e2e/models/configs/Llama-3.2-3B-Instruct.yaml
Normal file
23
tests/e2e/models/configs/Llama-3.2-3B-Instruct.yaml
Normal file
@@ -0,0 +1,23 @@
|
||||
model_name: "LLM-Research/Llama-3.2-3B-Instruct"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.71
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.76
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: "auto"
|
||||
24
tests/e2e/models/configs/Minitron-8B-Base.yaml
Normal file
24
tests/e2e/models/configs/Minitron-8B-Base.yaml
Normal file
@@ -0,0 +1,24 @@
|
||||
model_name: "nv-community/Minitron-8B-Base"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.9
|
||||
enforce_eager: true
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.5436
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.5451
|
||||
|
||||
limit: 1000
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
32
tests/e2e/models/configs/Mixtral-8x7B-Instruct-v0.1.yaml
Normal file
32
tests/e2e/models/configs/Mixtral-8x7B-Instruct-v0.1.yaml
Normal file
@@ -0,0 +1,32 @@
|
||||
model_name: "mistralai/Mixtral-8x7B-Instruct-v0.1"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 4
|
||||
dtype: bfloat16
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: true
|
||||
enforce_eager: true
|
||||
block_size: 128
|
||||
|
||||
envs:
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
VLLM_USE_V1: "1"
|
||||
HCCL_BUFFSIZE: "200"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
|
||||
tasks:
|
||||
- name: "ceval-valid"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.45
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: 32
|
||||
21
tests/e2e/models/configs/Molmo-7B-D-0924.yaml
Normal file
21
tests/e2e/models/configs/Molmo-7B-D-0924.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "LLM-Research/Molmo-7B-D-0924"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "ceval-valid"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.71
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
23
tests/e2e/models/configs/Qwen2-Audio-7B-Instruct.yaml
Normal file
23
tests/e2e/models/configs/Qwen2-Audio-7B-Instruct.yaml
Normal file
@@ -0,0 +1,23 @@
|
||||
model_name: "Qwen/Qwen2-Audio-7B-Instruct"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.44
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.45
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: "auto"
|
||||
25
tests/e2e/models/configs/Qwen2.5-Math-RM-72B.yaml
Normal file
25
tests/e2e/models/configs/Qwen2.5-Math-RM-72B.yaml
Normal file
@@ -0,0 +1,25 @@
|
||||
model_name: "Qwen/Qwen2.5-Math-RM-72B"
|
||||
model_type: "vllm-rm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 4
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.9
|
||||
trust_remote_code: false
|
||||
|
||||
# system_prompt controls the <|im_start|>system block passed to the reward model.
|
||||
system_prompt: "Please reason step by step, and put your final answer within \\boxed{}."
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k_correctness"
|
||||
dataset: "AI-ModelScope/gsm8k"
|
||||
split: "test"
|
||||
dataset_config: "main"
|
||||
metrics:
|
||||
- name: "accuracy"
|
||||
value: 0.80
|
||||
|
||||
limit: 200
|
||||
batch_size: 4
|
||||
25
tests/e2e/models/configs/Qwen3-30B-A3B-W8A8.yaml
Normal file
25
tests/e2e/models/configs/Qwen3-30B-A3B-W8A8.yaml
Normal file
@@ -0,0 +1,25 @@
|
||||
model_name: "vllm-ascend/Qwen3-30B-A3B-W8A8"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 2
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
enable_expert_parallel: true
|
||||
quantization: ascend
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.9
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.8
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
@@ -1,6 +1,15 @@
|
||||
model_name: "Qwen/Qwen3-30B-A3B"
|
||||
runner: "linux-aarch64-a2-2"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 2
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.6
|
||||
trust_remote_code: false
|
||||
enable_expert_parallel: true
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
@@ -12,9 +21,8 @@ tasks:
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.84
|
||||
|
||||
num_fewshot: 5
|
||||
gpu_memory_utilization: 0.6
|
||||
enable_expert_parallel: True
|
||||
tensor_parallel_size: 2
|
||||
apply_chat_template: False
|
||||
fewshot_as_multiturn: False
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
|
||||
25
tests/e2e/models/configs/Qwen3-8B-W8A8.yaml
Normal file
25
tests/e2e/models/configs/Qwen3-8B-W8A8.yaml
Normal file
@@ -0,0 +1,25 @@
|
||||
model_name: "vllm-ascend/Qwen3-8B-W8A8"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
quantization: ascend
|
||||
enable_thinking: false
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.80
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.82
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: "auto"
|
||||
24
tests/e2e/models/configs/Qwen3-8B.yaml
Normal file
24
tests/e2e/models/configs/Qwen3-8B.yaml
Normal file
@@ -0,0 +1,24 @@
|
||||
model_name: "Qwen/Qwen3-8B"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
enable_thinking: false
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.765
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.81
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: "auto"
|
||||
21
tests/e2e/models/configs/Qwen3-ASR-1.7B.yaml
Normal file
21
tests/e2e/models/configs/Qwen3-ASR-1.7B.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "Qwen/Qwen3-ASR-1.7B"
|
||||
model_type: "vllm-asr"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
|
||||
tasks:
|
||||
- name: "librispeech_test_clean"
|
||||
dataset: "openslr/librispeech_asr"
|
||||
split: "test"
|
||||
dataset_config: "clean"
|
||||
metrics:
|
||||
- name: "wer"
|
||||
value: 0.035
|
||||
|
||||
limit: 500
|
||||
23
tests/e2e/models/configs/Qwen3-Next-80B-A3B-Instruct.yaml
Normal file
23
tests/e2e/models/configs/Qwen3-Next-80B-A3B-Instruct.yaml
Normal file
@@ -0,0 +1,23 @@
|
||||
model_name: "Qwen/Qwen3-Next-80B-A3B-Instruct"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 4
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
enable_expert_parallel: true
|
||||
enforce_eager: true
|
||||
|
||||
tasks:
|
||||
- name: "ceval-valid_accountant"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.98
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: true
|
||||
batch_size: 1
|
||||
22
tests/e2e/models/configs/Qwen3-Omni-30B-A3B-Instruct.yaml
Normal file
22
tests/e2e/models/configs/Qwen3-Omni-30B-A3B-Instruct.yaml
Normal file
@@ -0,0 +1,22 @@
|
||||
model_name: "Qwen/Qwen3-Omni-30B-A3B-Instruct"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 4
|
||||
dtype: auto
|
||||
max_model_len: 8192
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
enable_expert_parallel: true
|
||||
|
||||
tasks:
|
||||
- name: "mmmu_val"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.60
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
22
tests/e2e/models/configs/Qwen3-VL-30B-A3B-Instruct.yaml
Normal file
22
tests/e2e/models/configs/Qwen3-VL-30B-A3B-Instruct.yaml
Normal file
@@ -0,0 +1,22 @@
|
||||
model_name: "Qwen/Qwen3-VL-30B-A3B-Instruct"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 2
|
||||
dtype: auto
|
||||
max_model_len: 128000
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
enable_expert_parallel: true
|
||||
|
||||
tasks:
|
||||
- name: "mmmu_val"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.58
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
22
tests/e2e/models/configs/Qwen3-VL-8B-Instruct-W8A8.yaml
Normal file
22
tests/e2e/models/configs/Qwen3-VL-8B-Instruct-W8A8.yaml
Normal file
@@ -0,0 +1,22 @@
|
||||
model_name: "vllm-ascend/Qwen3-VL-8B-Instruct-W8A8"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 8192
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: false
|
||||
quantization: ascend
|
||||
|
||||
tasks:
|
||||
- name: "mmmu_val"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.52
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: 32
|
||||
21
tests/e2e/models/configs/Qwen3-VL-8B-Instruct.yaml
Normal file
21
tests/e2e/models/configs/Qwen3-VL-8B-Instruct.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "Qwen/Qwen3-VL-8B-Instruct"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 8192
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
|
||||
tasks:
|
||||
- name: "mmmu_val"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.55
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: true
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: 32
|
||||
@@ -1,4 +1,14 @@
|
||||
DeepSeek-V2-Lite.yaml
|
||||
Qwen3-8B-Base.yaml
|
||||
Qwen2.5-VL-7B-Instruct.yaml
|
||||
Qwen3-30B-A3B.yaml
|
||||
Qwen3-30B-A3B.yaml
|
||||
Qwen3-8B.yaml
|
||||
Qwen2-Audio-7B-Instruct.yaml
|
||||
Qwen3-VL-30B-A3B-Instruct.yaml
|
||||
Qwen3-VL-8B-Instruct.yaml
|
||||
Qwen3-Omni-30B-A3B-Instruct.yaml
|
||||
InternVL3_5-8B-hf.yaml
|
||||
ERNIE-4.5-21B-A3B-PT.yaml
|
||||
gemma-3-4b-it.yaml
|
||||
internlm3-8b-instruct.yaml
|
||||
Molmo-7B-D-0924.yaml
|
||||
llava-onevision-qwen2-0.5b-ov-hf.yaml
|
||||
Llama-3.2-3B-Instruct.yaml
|
||||
Qwen3-ASR-1.7B.yaml
|
||||
|
||||
24
tests/e2e/models/configs/gemma-3-4b-it.yaml
Normal file
24
tests/e2e/models/configs/gemma-3-4b-it.yaml
Normal file
@@ -0,0 +1,24 @@
|
||||
model_name: "LLM-Research/gemma-3-4b-it"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.7
|
||||
trust_remote_code: false
|
||||
enforce_eager: true
|
||||
|
||||
tasks:
|
||||
- name: "gsm8k"
|
||||
metrics:
|
||||
- name: "exact_match,strict-match"
|
||||
value: 0.59
|
||||
- name: "exact_match,flexible-extract"
|
||||
value: 0.59
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
21
tests/e2e/models/configs/internlm3-8b-instruct.yaml
Normal file
21
tests/e2e/models/configs/internlm3-8b-instruct.yaml
Normal file
@@ -0,0 +1,21 @@
|
||||
model_name: "Shanghai_AI_Laboratory/internlm3-8b-instruct"
|
||||
model_type: "vllm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: "bfloat16"
|
||||
max_model_len: 2048
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "ceval-valid"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.42
|
||||
|
||||
num_fewshot: 5
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
@@ -0,0 +1,21 @@
|
||||
model_name: "llava-hf/llava-onevision-qwen2-0.5b-ov-hf"
|
||||
model_type: "vllm-vlm"
|
||||
hardware: "Atlas A2 Series"
|
||||
|
||||
serve:
|
||||
tensor_parallel_size: 1
|
||||
dtype: auto
|
||||
max_model_len: 4096
|
||||
gpu_memory_utilization: 0.8
|
||||
trust_remote_code: true
|
||||
|
||||
tasks:
|
||||
- name: "ceval-valid"
|
||||
metrics:
|
||||
- name: "acc,none"
|
||||
value: 0.42
|
||||
|
||||
num_fewshot: 0
|
||||
apply_chat_template: false
|
||||
fewshot_as_multiturn: false
|
||||
batch_size: "auto"
|
||||
@@ -21,7 +21,7 @@ def pytest_addoption(parser):
|
||||
parser.addoption(
|
||||
"--config",
|
||||
action="store",
|
||||
default="./tests/e2e/models/configs/Qwen3-8B-Base.yaml",
|
||||
default="./tests/e2e/models/configs/Qwen3-8B.yaml",
|
||||
help="Path to the model config YAML file",
|
||||
)
|
||||
parser.addoption(
|
||||
@@ -55,16 +55,12 @@ def report_dir(pytestconfig):
|
||||
|
||||
def pytest_generate_tests(metafunc):
|
||||
if "config_filename" in metafunc.fixturenames:
|
||||
|
||||
if metafunc.config.getoption("--config-list-file"):
|
||||
rel_path = metafunc.config.getoption("--config-list-file")
|
||||
config_list_file = Path(rel_path).resolve()
|
||||
config_dir = config_list_file.parent
|
||||
with open(config_list_file, encoding="utf-8") as f:
|
||||
configs = [
|
||||
config_dir / line.strip() for line in f
|
||||
if line.strip() and not line.startswith("#")
|
||||
]
|
||||
configs = [config_dir / line.strip() for line in f if line.strip() and not line.startswith("#")]
|
||||
metafunc.parametrize("config_filename", configs)
|
||||
else:
|
||||
single_config = metafunc.config.getoption("--config")
|
||||
|
||||
@@ -1,30 +1,33 @@
|
||||
# {{ model_name }}
|
||||
|
||||
- **vLLM Version**: vLLM: {{ vllm_version }} ([{{ vllm_commit[:7] }}](https://github.com/vllm-project/vllm/commit/{{ vllm_commit }})), **vLLM Ascend Version**: {{ vllm_ascend_version }} ([{{ vllm_ascend_commit[:7] }}](https://github.com/vllm-project/vllm-ascend/commit/{{ vllm_ascend_commit }}))
|
||||
- **Software Environment**: **CANN**: {{ cann_version }}, **PyTorch**: {{ torch_version }}, **torch-npu**: {{ torch_npu_version }}
|
||||
- **Software Environment**: **CANN**: {{ cann_version }}, **PyTorch**: {{ torch_version }}, **TorchNPU**: {{ torch_npu_version }}
|
||||
- **Hardware Environment**: {{ hardware }}
|
||||
- **Parallel mode**: {{ parallel_mode }}
|
||||
- **Execution mode**: {{ execution_model }}
|
||||
|
||||
{% if show_command is not defined or show_command %}
|
||||
**Command**:
|
||||
|
||||
```bash
|
||||
export MODEL_ARGS={{ model_args }}
|
||||
lm_eval --model {{ model_type }} --model_args $MODEL_ARGS --tasks {{ datasets }} \
|
||||
{% if apply_chat_template is defined and (apply_chat_template|string|lower in ["true", "1"]) -%}
|
||||
lm_eval --model {{ model_type }} --model_args $MODEL_ARGS \
|
||||
--tasks {{ datasets }} \
|
||||
{%- if apply_chat_template is defined and (apply_chat_template|string|lower in ["true", "1"]) %}
|
||||
--apply_chat_template \
|
||||
{%- endif %}
|
||||
{% if fewshot_as_multiturn is defined and (fewshot_as_multiturn|string|lower in ["true", "1"]) -%}
|
||||
{%- if fewshot_as_multiturn is defined and (fewshot_as_multiturn|string|lower in ["true", "1"]) %}
|
||||
--fewshot_as_multiturn \
|
||||
{%- endif %}
|
||||
{% if num_fewshot is defined and num_fewshot != "N/A" -%}
|
||||
{%- if num_fewshot is defined and num_fewshot != "N/A" %}
|
||||
--num_fewshot {{ num_fewshot }} \
|
||||
{%- endif %}
|
||||
{% if limit is defined and limit != "N/A" -%}
|
||||
{%- if limit is defined and limit != "N/A" %}
|
||||
--limit {{ limit }} \
|
||||
{%- endif %}
|
||||
--batch_size {{ batch_size }}
|
||||
--batch_size {{ batch_size }}
|
||||
```
|
||||
{% endif %}
|
||||
|
||||
| Task | Metric | Value | Stderr |
|
||||
|-----------------------|-------------|----------:|-------:|
|
||||
|
||||
290
tests/e2e/models/test_asr_eval_correctness.py
Normal file
290
tests/e2e/models/test_asr_eval_correctness.py
Normal file
@@ -0,0 +1,290 @@
|
||||
import io
|
||||
import os
|
||||
import string
|
||||
from dataclasses import dataclass
|
||||
|
||||
import jiwer # type: ignore[import-untyped]
|
||||
import numpy as np
|
||||
import pytest
|
||||
import scipy.io.wavfile as wav_io # type: ignore[import-untyped]
|
||||
import soundfile as sf # type: ignore[import-untyped]
|
||||
import yaml
|
||||
from datasets import Audio
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
from modelscope.msdatasets import MsDataset # type: ignore[import-untyped]
|
||||
from vllm.utils.network_utils import get_open_port
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
|
||||
# Allow up to 10% relative deviation from the declared ground-truth WER.
|
||||
# ASR results have higher variance than classification tasks, so we use a
|
||||
# more generous tolerance than the 5% used in test_lm_eval_correctness.py.
|
||||
RTOL = 0.03
|
||||
|
||||
TEST_DIR = os.path.dirname(__file__)
|
||||
|
||||
_PUNCT_TABLE = str.maketrans("", "", string.punctuation)
|
||||
|
||||
|
||||
@dataclass
|
||||
class EnvConfig:
|
||||
vllm_version: str
|
||||
vllm_commit: str
|
||||
vllm_ascend_version: str
|
||||
vllm_ascend_commit: str
|
||||
cann_version: str
|
||||
torch_version: str
|
||||
torch_npu_version: str
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def env_config() -> EnvConfig:
|
||||
return EnvConfig(
|
||||
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
|
||||
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
|
||||
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
|
||||
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
|
||||
cann_version=os.getenv("CANN_VERSION", "unknown"),
|
||||
torch_version=os.getenv("TORCH_VERSION", "unknown"),
|
||||
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
|
||||
)
|
||||
|
||||
|
||||
def build_serve_args(eval_config: dict) -> list[str]:
|
||||
"""Convert the serve: section of the YAML into a vllm serve CLI args list.
|
||||
|
||||
Example — serve: {tensor_parallel_size: 2, dtype: auto} becomes:
|
||||
["--tensor-parallel-size", "2", "--dtype", "auto"]
|
||||
"""
|
||||
serve_cfg = eval_config.get("serve", {})
|
||||
flag_map = {
|
||||
"tensor_parallel_size": "--tensor-parallel-size",
|
||||
"dtype": "--dtype",
|
||||
"max_model_len": "--max-model-len",
|
||||
"gpu_memory_utilization": "--gpu-memory-utilization",
|
||||
"trust_remote_code": "--trust-remote-code",
|
||||
"enforce_eager": "--enforce-eager",
|
||||
"quantization": "--quantization",
|
||||
}
|
||||
args: list[str] = []
|
||||
for key, flag in flag_map.items():
|
||||
value = serve_cfg.get(key)
|
||||
if value is None:
|
||||
continue
|
||||
if isinstance(value, bool):
|
||||
if value:
|
||||
args.append(flag)
|
||||
else:
|
||||
args.extend([flag, str(value)])
|
||||
return args
|
||||
|
||||
|
||||
def audio_to_wav_bytes(audio_array: np.ndarray, sample_rate: int) -> bytes:
|
||||
"""Convert a numpy audio array to in-memory WAV bytes at the given sample rate."""
|
||||
buf = io.BytesIO()
|
||||
# Ensure int16 encoding for maximum API compatibility.
|
||||
if audio_array.dtype != np.int16:
|
||||
if np.issubdtype(audio_array.dtype, np.floating):
|
||||
audio_array = np.clip(audio_array, -1.0, 1.0)
|
||||
audio_array = (audio_array * 32767).astype(np.int16)
|
||||
else:
|
||||
audio_array = audio_array.astype(np.int16)
|
||||
wav_io.write(buf, sample_rate, audio_array)
|
||||
return buf.getvalue()
|
||||
|
||||
|
||||
def normalize_text(text: str) -> str:
|
||||
"""Normalize text for WER calculation: lowercase, strip punctuation, collapse whitespace."""
|
||||
text = text.lower()
|
||||
text = text.translate(_PUNCT_TABLE)
|
||||
text = " ".join(text.split())
|
||||
return text
|
||||
|
||||
|
||||
def transcribe_batch(client, model_name: str, audio_items: list[dict], language: str) -> list[str]:
|
||||
"""Call /v1/audio/transcriptions for a list of audio items.
|
||||
|
||||
Each item in audio_items must have keys: audio_array (np.ndarray), sample_rate (int).
|
||||
Returns the raw transcription strings in the same order.
|
||||
"""
|
||||
hypotheses: list[str] = []
|
||||
for item in audio_items:
|
||||
wav_bytes = audio_to_wav_bytes(item["audio_array"], item["sample_rate"])
|
||||
response = client.audio.transcriptions.create(
|
||||
model=model_name,
|
||||
file=("audio.wav", wav_bytes, "audio/wav"),
|
||||
language=language,
|
||||
)
|
||||
hypotheses.append(response.text)
|
||||
return hypotheses
|
||||
|
||||
|
||||
def generate_asr_report(
|
||||
eval_config: dict,
|
||||
report_data: dict,
|
||||
report_dir: str,
|
||||
env_config: EnvConfig,
|
||||
) -> None:
|
||||
"""Write a Markdown accuracy report using the same Jinja2 template as lm_eval tests."""
|
||||
env = Environment(loader=FileSystemLoader(TEST_DIR))
|
||||
template = env.get_template("report_template.md")
|
||||
|
||||
serve_cfg = eval_config.get("serve", {})
|
||||
tp_size = serve_cfg.get("tensor_parallel_size", 1)
|
||||
ep_enabled = serve_cfg.get("enable_expert_parallel", False)
|
||||
enforce_eager = serve_cfg.get("enforce_eager", False)
|
||||
|
||||
parallel_mode = f"TP{tp_size}"
|
||||
if ep_enabled:
|
||||
parallel_mode += " + EP"
|
||||
execution_model = "Eager" if enforce_eager else "ACLGraph"
|
||||
|
||||
model_args_str = ",".join(f"{k}={v}" for k, v in serve_cfg.items())
|
||||
|
||||
report_content = template.render(
|
||||
vllm_version=env_config.vllm_version,
|
||||
vllm_commit=env_config.vllm_commit,
|
||||
vllm_ascend_version=env_config.vllm_ascend_version,
|
||||
vllm_ascend_commit=env_config.vllm_ascend_commit,
|
||||
cann_version=env_config.cann_version,
|
||||
torch_version=env_config.torch_version,
|
||||
torch_npu_version=env_config.torch_npu_version,
|
||||
hardware=eval_config.get("hardware", "unknown"),
|
||||
model_name=eval_config["model_name"],
|
||||
model_args=f"'{model_args_str}'",
|
||||
model_type=eval_config.get("model_type", "vllm-asr"),
|
||||
datasets=",".join(t["name"] for t in eval_config["tasks"]),
|
||||
apply_chat_template=False,
|
||||
fewshot_as_multiturn=False,
|
||||
limit=eval_config.get("limit", "N/A"),
|
||||
batch_size=eval_config.get("batch_size", 8),
|
||||
num_fewshot="N/A",
|
||||
rows=report_data["rows"],
|
||||
parallel_mode=parallel_mode,
|
||||
execution_model=execution_model,
|
||||
show_command=False,
|
||||
)
|
||||
|
||||
report_path = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
|
||||
os.makedirs(os.path.dirname(report_path), exist_ok=True)
|
||||
with open(report_path, "w", encoding="utf-8") as f:
|
||||
f.write(report_content)
|
||||
|
||||
|
||||
def test_asr_eval_param(config_filename, tp_size, report_dir, env_config):
|
||||
"""Parametrised ASR accuracy test driven by a YAML config file.
|
||||
|
||||
Skips automatically when the config's model_type is not "vllm-asr".
|
||||
"""
|
||||
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
|
||||
|
||||
if eval_config.get("model_type", "vllm") != "vllm-asr":
|
||||
pytest.skip(f"Skipping non-ASR config (model_type={eval_config.get('model_type', 'vllm')})")
|
||||
|
||||
model_name: str = eval_config["model_name"]
|
||||
language: str = eval_config.get("language", "en")
|
||||
limit: int | None = eval_config.get("limit", None)
|
||||
batch_size: int = eval_config.get("batch_size", 8)
|
||||
|
||||
# Build serve args, letting --tp-size CLI flag override the YAML value.
|
||||
serve_args = build_serve_args(eval_config)
|
||||
if tp_size and tp_size != "1":
|
||||
# Drop any --tensor-parallel-size already in serve_args, then append
|
||||
# the CLI-supplied value so it takes precedence over the YAML setting.
|
||||
it = iter(serve_args)
|
||||
serve_args = [a for a in it if a != "--tensor-parallel-size" or not next(it, None)]
|
||||
serve_args += ["--tensor-parallel-size", str(tp_size)]
|
||||
|
||||
print(f"\nStarting vllm serve for {model_name}")
|
||||
print(f" serve args: {serve_args}")
|
||||
|
||||
success = True
|
||||
report_data: dict[str, list[dict]] = {"rows": []}
|
||||
|
||||
server_port = get_open_port()
|
||||
serve_args = serve_args + ["--port", str(server_port)]
|
||||
with RemoteOpenAIServer(model_name, serve_args, server_port=server_port, auto_port=False) as server:
|
||||
client = server.get_client()
|
||||
|
||||
for task in eval_config["tasks"]:
|
||||
task_name: str = task["name"]
|
||||
dataset_name: str = task["dataset"]
|
||||
split: str = task["split"]
|
||||
dataset_config_name: str | None = task.get("dataset_config")
|
||||
audio_col: str = task.get("audio_column", "audio")
|
||||
text_col: str = task.get("text_column", "text")
|
||||
|
||||
split_expr = f"{split}[:{limit}]" if limit is not None else split
|
||||
print(f"\nLoading dataset via modelscope: {dataset_name} / {dataset_config_name} ({split_expr})")
|
||||
ds = MsDataset.load(
|
||||
dataset_name,
|
||||
subset_name=dataset_config_name,
|
||||
split=split_expr,
|
||||
)
|
||||
if limit is not None:
|
||||
ds = ds.select(range(min(limit, len(ds))))
|
||||
|
||||
# Disable automatic audio decoding so we can use soundfile instead
|
||||
# of torchcodec (which requires CUDA libs unavailable on Ascend NPU).
|
||||
if hasattr(ds, "cast_column"):
|
||||
ds = ds.cast_column(audio_col, Audio(decode=False))
|
||||
|
||||
print(f" {len(ds)} samples to evaluate")
|
||||
|
||||
# Collect audio items and references in batches.
|
||||
all_hypotheses: list[str] = []
|
||||
all_references: list[str] = []
|
||||
|
||||
for batch_start in range(0, len(ds), batch_size):
|
||||
batch = ds.select(range(batch_start, min(batch_start + batch_size, len(ds))))
|
||||
audio_items = []
|
||||
for sample in batch:
|
||||
raw = sample[audio_col]
|
||||
if isinstance(raw, dict) and "bytes" in raw and raw["bytes"] is not None:
|
||||
audio_array, sample_rate = sf.read(io.BytesIO(raw["bytes"]))
|
||||
elif isinstance(raw, dict) and "path" in raw and raw["path"] is not None:
|
||||
audio_array, sample_rate = sf.read(raw["path"])
|
||||
else:
|
||||
# Already decoded (e.g. MsDataset with native decoding)
|
||||
audio_array = raw["array"]
|
||||
sample_rate = raw["sampling_rate"]
|
||||
audio_items.append({"audio_array": audio_array, "sample_rate": sample_rate})
|
||||
references = [sample[text_col] for sample in batch]
|
||||
|
||||
hypotheses = transcribe_batch(client, model_name, audio_items, language)
|
||||
all_hypotheses.extend(hypotheses)
|
||||
all_references.extend(references)
|
||||
|
||||
if (batch_start // batch_size + 1) % 5 == 0:
|
||||
print(f" processed {batch_start + len(batch)}/{len(ds)} samples …")
|
||||
|
||||
# Normalise both sides before WER calculation.
|
||||
norm_hypotheses = [normalize_text(h) for h in all_hypotheses]
|
||||
norm_references = [normalize_text(r) for r in all_references]
|
||||
|
||||
measured_wer = round(jiwer.wer(norm_references, norm_hypotheses), 4)
|
||||
print(f"\n{task_name} WER = {measured_wer:.4f}")
|
||||
|
||||
for metric in task["metrics"]:
|
||||
if metric["name"] != "wer":
|
||||
continue
|
||||
ground_truth = metric["value"]
|
||||
# Pass if measured WER is at or below the threshold (better is OK);
|
||||
# allow up to RTOL relative degradation above the threshold.
|
||||
task_success = measured_wer <= ground_truth * (1 + RTOL)
|
||||
success = success and task_success
|
||||
|
||||
status = "✅" if task_success else "❌"
|
||||
print(f"{task_name} | wer: ground_truth={ground_truth} | measured={measured_wer} | {status}")
|
||||
|
||||
report_data["rows"].append(
|
||||
{
|
||||
"task": task_name,
|
||||
"metric": "wer",
|
||||
"value": f"{status}{measured_wer}",
|
||||
"stderr": "N/A",
|
||||
}
|
||||
)
|
||||
|
||||
generate_asr_report(eval_config, report_data, report_dir, env_config)
|
||||
assert success, "One or more ASR tasks exceeded the WER tolerance. See output above."
|
||||
@@ -7,7 +7,7 @@ import pytest
|
||||
import yaml
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
|
||||
RTOL = 0.03
|
||||
RTOL = 0.05
|
||||
TEST_DIR = os.path.dirname(__file__)
|
||||
|
||||
|
||||
@@ -24,33 +24,39 @@ class EnvConfig:
|
||||
|
||||
@pytest.fixture
|
||||
def env_config() -> EnvConfig:
|
||||
return EnvConfig(vllm_version=os.getenv('VLLM_VERSION', 'unknown'),
|
||||
vllm_commit=os.getenv('VLLM_COMMIT', 'unknown'),
|
||||
vllm_ascend_version=os.getenv('VLLM_ASCEND_VERSION',
|
||||
'unknown'),
|
||||
vllm_ascend_commit=os.getenv('VLLM_ASCEND_COMMIT',
|
||||
'unknown'),
|
||||
cann_version=os.getenv('CANN_VERSION', 'unknown'),
|
||||
torch_version=os.getenv('TORCH_VERSION', 'unknown'),
|
||||
torch_npu_version=os.getenv('TORCH_NPU_VERSION',
|
||||
'unknown'))
|
||||
return EnvConfig(
|
||||
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
|
||||
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
|
||||
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
|
||||
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
|
||||
cann_version=os.getenv("CANN_VERSION", "unknown"),
|
||||
torch_version=os.getenv("TORCH_VERSION", "unknown"),
|
||||
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
|
||||
)
|
||||
|
||||
|
||||
def build_model_args(eval_config, tp_size):
|
||||
trust_remote_code = eval_config.get("trust_remote_code", False)
|
||||
max_model_len = eval_config.get("max_model_len", 4096)
|
||||
serve_cfg = eval_config.get("serve", {})
|
||||
trust_remote_code = serve_cfg.get("trust_remote_code", False)
|
||||
max_model_len = serve_cfg.get("max_model_len", 4096)
|
||||
dtype = serve_cfg.get("dtype", "auto")
|
||||
model_args = {
|
||||
"pretrained": eval_config["model_name"],
|
||||
"tensor_parallel_size": tp_size,
|
||||
"dtype": "auto",
|
||||
"dtype": dtype,
|
||||
"trust_remote_code": trust_remote_code,
|
||||
"max_model_len": max_model_len,
|
||||
}
|
||||
for s in [
|
||||
"max_images", "gpu_memory_utilization", "enable_expert_parallel",
|
||||
"tensor_parallel_size", "enforce_eager"
|
||||
"max_images",
|
||||
"gpu_memory_utilization",
|
||||
"enable_expert_parallel",
|
||||
"tensor_parallel_size",
|
||||
"enforce_eager",
|
||||
"enable_thinking",
|
||||
"quantization",
|
||||
]:
|
||||
val = eval_config.get(s, None)
|
||||
val = serve_cfg.get(s, None)
|
||||
if val is not None:
|
||||
model_args[s] = val
|
||||
|
||||
@@ -66,7 +72,7 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
|
||||
model_args = build_model_args(eval_config, tp_size)
|
||||
|
||||
parallel_mode = f"TP{model_args.get('tensor_parallel_size', 1)}"
|
||||
if model_args.get('enable_expert_parallel', False):
|
||||
if model_args.get("enable_expert_parallel", False):
|
||||
parallel_mode += " + EP"
|
||||
|
||||
execution_model = f"{'Eager' if model_args.get('enforce_eager', False) else 'ACLGraph'}"
|
||||
@@ -82,7 +88,7 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
|
||||
hardware=eval_config.get("hardware", "unknown"),
|
||||
model_name=eval_config["model_name"],
|
||||
model_args=f"'{','.join(f'{k}={v}' for k, v in model_args.items())}'",
|
||||
model_type=eval_config.get("model", "vllm"),
|
||||
model_type=eval_config.get("model_type", "vllm"),
|
||||
datasets=",".join([task["name"] for task in eval_config["tasks"]]),
|
||||
apply_chat_template=eval_config.get("apply_chat_template", True),
|
||||
fewshot_as_multiturn=eval_config.get("fewshot_as_multiturn", True),
|
||||
@@ -91,24 +97,27 @@ def generate_report(tp_size, eval_config, report_data, report_dir, env_config):
|
||||
num_fewshot=eval_config.get("num_fewshot", "N/A"),
|
||||
rows=report_data["rows"],
|
||||
parallel_mode=parallel_mode,
|
||||
execution_model=execution_model)
|
||||
execution_model=execution_model,
|
||||
)
|
||||
|
||||
report_output = os.path.join(
|
||||
report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
|
||||
report_output = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
|
||||
os.makedirs(os.path.dirname(report_output), exist_ok=True)
|
||||
with open(report_output, 'w', encoding='utf-8') as f:
|
||||
with open(report_output, "w", encoding="utf-8") as f:
|
||||
f.write(report_content)
|
||||
|
||||
|
||||
def test_lm_eval_correctness_param(config_filename, tp_size, report_dir,
|
||||
env_config):
|
||||
def test_lm_eval_correctness_param(config_filename, tp_size, report_dir, env_config):
|
||||
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
|
||||
|
||||
if eval_config.get("model_type", "vllm") == "vllm-asr":
|
||||
pytest.skip("Skipping ASR config, use test_asr_eval.py instead")
|
||||
|
||||
model_args = build_model_args(eval_config, tp_size)
|
||||
success = True
|
||||
report_data: dict[str, list[dict]] = {"rows": []}
|
||||
|
||||
eval_params = {
|
||||
"model": eval_config.get("model", "vllm"),
|
||||
"model": eval_config.get("model_type", "vllm"),
|
||||
"model_args": model_args,
|
||||
"tasks": [task["name"] for task in eval_config["tasks"]],
|
||||
"apply_chat_template": eval_config.get("apply_chat_template", True),
|
||||
@@ -133,25 +142,26 @@ def test_lm_eval_correctness_param(config_filename, tp_size, report_dir,
|
||||
metric_name = metric["name"]
|
||||
ground_truth = metric["value"]
|
||||
measured_value = round(task_result[metric_name], 4)
|
||||
task_success = bool(
|
||||
np.isclose(ground_truth, measured_value, rtol=RTOL))
|
||||
task_success = bool(np.isclose(ground_truth, measured_value, rtol=RTOL))
|
||||
success = success and task_success
|
||||
|
||||
print(f"{task_name} | {metric_name}: "
|
||||
f"ground_truth={ground_truth} | measured={measured_value} | "
|
||||
f"success={'✅' if task_success else '❌'}")
|
||||
print(
|
||||
f"{task_name} | {metric_name}: "
|
||||
f"ground_truth={ground_truth} | measured={measured_value} | "
|
||||
f"success={'✅' if task_success else '❌'}"
|
||||
)
|
||||
|
||||
report_data["rows"].append({
|
||||
"task":
|
||||
task_name,
|
||||
"metric":
|
||||
metric_name,
|
||||
"value":
|
||||
f"✅{measured_value}" if success else f"❌{measured_value}",
|
||||
"stderr":
|
||||
task_result[
|
||||
metric_name.replace(',', '_stderr,') if metric_name ==
|
||||
"acc,none" else metric_name.replace(',', '_stderr,')]
|
||||
})
|
||||
report_data["rows"].append(
|
||||
{
|
||||
"task": task_name,
|
||||
"metric": metric_name,
|
||||
"value": f"✅{measured_value}" if success else f"❌{measured_value}",
|
||||
"stderr": task_result[
|
||||
metric_name.replace(",", "_stderr,")
|
||||
if metric_name == "acc,none"
|
||||
else metric_name.replace(",", "_stderr,")
|
||||
],
|
||||
}
|
||||
)
|
||||
generate_report(tp_size, eval_config, report_data, report_dir, env_config)
|
||||
assert success
|
||||
|
||||
255
tests/e2e/models/test_rm_eval_correctness.py
Normal file
255
tests/e2e/models/test_rm_eval_correctness.py
Normal file
@@ -0,0 +1,255 @@
|
||||
import os
|
||||
from dataclasses import dataclass
|
||||
|
||||
import pytest
|
||||
import regex as re
|
||||
import yaml
|
||||
from jinja2 import Environment, FileSystemLoader
|
||||
from modelscope.msdatasets import MsDataset # type: ignore[import-untyped]
|
||||
|
||||
from tests.e2e.conftest import VllmRunner
|
||||
|
||||
# Allow up to 5 % relative degradation from the declared ground-truth accuracy.
|
||||
RTOL = 0.05
|
||||
|
||||
TEST_DIR = os.path.dirname(__file__)
|
||||
|
||||
# Default system prompt for Qwen2.5-Math-RM style models.
|
||||
_DEFAULT_SYSTEM_PROMPT = "Please reason step by step, and put your final answer within \\boxed{}."
|
||||
|
||||
|
||||
@dataclass
|
||||
class EnvConfig:
|
||||
vllm_version: str
|
||||
vllm_commit: str
|
||||
vllm_ascend_version: str
|
||||
vllm_ascend_commit: str
|
||||
cann_version: str
|
||||
torch_version: str
|
||||
torch_npu_version: str
|
||||
|
||||
|
||||
@pytest.fixture
|
||||
def env_config() -> EnvConfig:
|
||||
return EnvConfig(
|
||||
vllm_version=os.getenv("VLLM_VERSION", "unknown"),
|
||||
vllm_commit=os.getenv("VLLM_COMMIT", "unknown"),
|
||||
vllm_ascend_version=os.getenv("VLLM_ASCEND_VERSION", "unknown"),
|
||||
vllm_ascend_commit=os.getenv("VLLM_ASCEND_COMMIT", "unknown"),
|
||||
cann_version=os.getenv("CANN_VERSION", "unknown"),
|
||||
torch_version=os.getenv("TORCH_VERSION", "unknown"),
|
||||
torch_npu_version=os.getenv("TORCH_NPU_VERSION", "unknown"),
|
||||
)
|
||||
|
||||
|
||||
def format_rm_input(system_prompt: str, problem: str, solution: str) -> str:
|
||||
"""Format a (problem, solution) pair using the Qwen chat template."""
|
||||
return (
|
||||
f"<|im_start|>system\n{system_prompt}<|im_end|>\n"
|
||||
f"<|im_start|>user\n{problem}<|im_end|>\n"
|
||||
f"<|im_start|>assistant\n{solution}<|im_end|>"
|
||||
)
|
||||
|
||||
|
||||
def perturb_answer(solution: str) -> str:
|
||||
"""Create an obviously wrong solution for a GSM8K-style answer string.
|
||||
|
||||
GSM8K answers end with ``#### <number>``. We replace that number with
|
||||
``correct * 3 + 137`` so the final answer is clearly incorrect while the
|
||||
reasoning chain looks plausible.
|
||||
"""
|
||||
match = re.search(r"####\s*([\d,]+(?:\.\d+)?)", solution)
|
||||
if match:
|
||||
num_str = match.group(1).replace(",", "")
|
||||
try:
|
||||
correct_num = float(num_str)
|
||||
wrong_num = int(correct_num * 3 + 137)
|
||||
return solution[: match.start()] + f"#### {wrong_num}"
|
||||
except ValueError:
|
||||
pass
|
||||
# Fallback: append an unmistakably wrong sentinel answer.
|
||||
return solution + "\n#### -999999"
|
||||
|
||||
|
||||
def extract_reward_score(reward_output) -> float:
|
||||
"""Extract a scalar score from VllmRunner.reward() output for one sample.
|
||||
|
||||
VllmRunner.reward() returns list[list[float]] or list[Tensor]; for a reward
|
||||
model with a single output the inner list has one element. For a token-level
|
||||
reward model the output is a 2-D tensor [seq_len, 1]; in both cases we take
|
||||
the last element (final-step score).
|
||||
"""
|
||||
if isinstance(reward_output, (list, tuple)):
|
||||
return float(reward_output[-1])
|
||||
# Tensor (e.g. shape [seq_len, 1] from a token-level reward model)
|
||||
return float(reward_output.flatten()[-1].item())
|
||||
|
||||
|
||||
def generate_rm_report(
|
||||
eval_config: dict,
|
||||
report_data: dict,
|
||||
report_dir: str,
|
||||
env_config: EnvConfig,
|
||||
) -> None:
|
||||
"""Write a Markdown accuracy report using the shared Jinja2 template."""
|
||||
jinja_env = Environment(loader=FileSystemLoader(TEST_DIR))
|
||||
template = jinja_env.get_template("report_template.md")
|
||||
|
||||
serve_cfg = eval_config.get("serve", {})
|
||||
tp_size = serve_cfg.get("tensor_parallel_size", 1)
|
||||
ep_enabled = serve_cfg.get("enable_expert_parallel", False)
|
||||
enforce_eager = serve_cfg.get("enforce_eager", False)
|
||||
|
||||
parallel_mode = f"TP{tp_size}"
|
||||
if ep_enabled:
|
||||
parallel_mode += " + EP"
|
||||
execution_model = "Eager" if enforce_eager else "ACLGraph"
|
||||
model_args_str = ",".join(f"{k}={v}" for k, v in serve_cfg.items())
|
||||
|
||||
report_content = template.render(
|
||||
vllm_version=env_config.vllm_version,
|
||||
vllm_commit=env_config.vllm_commit,
|
||||
vllm_ascend_version=env_config.vllm_ascend_version,
|
||||
vllm_ascend_commit=env_config.vllm_ascend_commit,
|
||||
cann_version=env_config.cann_version,
|
||||
torch_version=env_config.torch_version,
|
||||
torch_npu_version=env_config.torch_npu_version,
|
||||
hardware=eval_config.get("hardware", "unknown"),
|
||||
model_name=eval_config["model_name"],
|
||||
model_args=f"'{model_args_str}'",
|
||||
model_type=eval_config.get("model_type", "vllm-rm"),
|
||||
datasets=",".join(t["name"] for t in eval_config["tasks"]),
|
||||
apply_chat_template=False,
|
||||
fewshot_as_multiturn=False,
|
||||
limit=eval_config.get("limit", "N/A"),
|
||||
batch_size=eval_config.get("batch_size", 4),
|
||||
num_fewshot="N/A",
|
||||
rows=report_data["rows"],
|
||||
parallel_mode=parallel_mode,
|
||||
execution_model=execution_model,
|
||||
)
|
||||
|
||||
report_path = os.path.join(report_dir, f"{os.path.basename(eval_config['model_name'])}.md")
|
||||
os.makedirs(os.path.dirname(report_path), exist_ok=True)
|
||||
with open(report_path, "w", encoding="utf-8") as f:
|
||||
f.write(report_content)
|
||||
|
||||
|
||||
def test_rm_eval_param(config_filename, tp_size, report_dir, env_config):
|
||||
"""Parametrised reward-model accuracy test driven by a YAML config file.
|
||||
|
||||
Skips automatically when the config's model_type is not "vllm-rm".
|
||||
"""
|
||||
eval_config = yaml.safe_load(config_filename.read_text(encoding="utf-8"))
|
||||
|
||||
if eval_config.get("model_type", "vllm") != "vllm-rm":
|
||||
pytest.skip(f"Skipping non-RM config (model_type={eval_config.get('model_type', 'vllm')})")
|
||||
|
||||
model_name: str = eval_config["model_name"]
|
||||
limit: int | None = eval_config.get("limit", None)
|
||||
batch_size: int = eval_config.get("batch_size", 4)
|
||||
system_prompt: str = eval_config.get("system_prompt", _DEFAULT_SYSTEM_PROMPT)
|
||||
serve_cfg: dict = eval_config.get("serve", {})
|
||||
|
||||
# CLI --tp-size takes precedence over the YAML tensor_parallel_size.
|
||||
effective_tp = int(tp_size) if (tp_size and tp_size != "1") else int(serve_cfg.get("tensor_parallel_size", 1))
|
||||
|
||||
runner_kwargs: dict = {
|
||||
k: v
|
||||
for k, v in {
|
||||
"runner": "pooling",
|
||||
"dtype": serve_cfg.get("dtype", "auto"),
|
||||
"tensor_parallel_size": effective_tp,
|
||||
"enforce_eager": serve_cfg.get("enforce_eager", False),
|
||||
"max_model_len": serve_cfg.get("max_model_len"),
|
||||
"gpu_memory_utilization": serve_cfg.get("gpu_memory_utilization"),
|
||||
}.items()
|
||||
if v is not None
|
||||
}
|
||||
|
||||
print(f"\nLoading reward model: {model_name}")
|
||||
print(f" VllmRunner kwargs: {runner_kwargs}")
|
||||
|
||||
success = True
|
||||
report_data: dict[str, list[dict]] = {"rows": []}
|
||||
|
||||
with VllmRunner(model_name, **runner_kwargs) as vllm_model:
|
||||
for task in eval_config["tasks"]:
|
||||
task_name: str = task["name"]
|
||||
dataset_name: str = task["dataset"]
|
||||
split: str = task["split"]
|
||||
dataset_config_name: str | None = task.get("dataset_config")
|
||||
task_type: str = task.get("task_type", "correctness")
|
||||
|
||||
# Column names for "correctness" tasks (e.g. GSM8K).
|
||||
problem_col: str = task.get("problem_column", "question")
|
||||
solution_col: str = task.get("solution_column", "answer")
|
||||
|
||||
# Column names for "pairwise" tasks (e.g. reward-bench).
|
||||
prompt_col: str = task.get("prompt_column", "prompt")
|
||||
chosen_col: str = task.get("chosen_column", "chosen")
|
||||
rejected_col: str = task.get("rejected_column", "rejected")
|
||||
|
||||
split_expr = f"{split}[:{limit}]" if limit is not None else split
|
||||
print(f"\nLoading dataset via ModelScope: {dataset_name} / {dataset_config_name} ({split_expr})")
|
||||
|
||||
# MsDataset may bypass the HF_HUB_OFFLINE lock; patch temporarily.
|
||||
ds = MsDataset.load(
|
||||
dataset_name,
|
||||
subset_name=dataset_config_name,
|
||||
split=split_expr,
|
||||
)
|
||||
print(f" {len(ds)} samples to evaluate (task_type={task_type})")
|
||||
|
||||
correct_count = 0
|
||||
total_count = 0
|
||||
|
||||
for batch_start in range(0, len(ds), batch_size):
|
||||
batch = ds.select(range(batch_start, min(batch_start + batch_size, len(ds))))
|
||||
|
||||
if task_type == "pairwise":
|
||||
positive_texts = [format_rm_input(system_prompt, s[prompt_col], s[chosen_col]) for s in batch]
|
||||
negative_texts = [format_rm_input(system_prompt, s[prompt_col], s[rejected_col]) for s in batch]
|
||||
else:
|
||||
positive_texts = [format_rm_input(system_prompt, s[problem_col], s[solution_col]) for s in batch]
|
||||
negative_texts = [
|
||||
format_rm_input(system_prompt, s[problem_col], perturb_answer(s[solution_col])) for s in batch
|
||||
]
|
||||
|
||||
pos_rewards = vllm_model.reward(positive_texts)
|
||||
neg_rewards = vllm_model.reward(negative_texts)
|
||||
|
||||
for pos_r, neg_r in zip(pos_rewards, neg_rewards):
|
||||
if extract_reward_score(pos_r) > extract_reward_score(neg_r):
|
||||
correct_count += 1
|
||||
total_count += 1
|
||||
|
||||
if (batch_start // batch_size + 1) % 5 == 0:
|
||||
print(f" processed {batch_start + len(batch)}/{len(ds)} samples …")
|
||||
|
||||
measured_accuracy = round(correct_count / total_count, 4) if total_count > 0 else 0.0
|
||||
print(f"\n{task_name} accuracy = {measured_accuracy:.4f}")
|
||||
|
||||
for metric in task["metrics"]:
|
||||
if metric["name"] != "accuracy":
|
||||
continue
|
||||
ground_truth = metric["value"]
|
||||
# Pass if measured accuracy meets or exceeds the threshold
|
||||
# (allow up to RTOL relative degradation).
|
||||
task_success = measured_accuracy >= ground_truth * (1 - RTOL)
|
||||
success = success and task_success
|
||||
|
||||
status = "✅" if task_success else "❌"
|
||||
print(f"{task_name} | accuracy: ground_truth={ground_truth} | measured={measured_accuracy} | {status}")
|
||||
|
||||
report_data["rows"].append(
|
||||
{
|
||||
"task": task_name,
|
||||
"metric": "accuracy",
|
||||
"value": f"{status}{measured_accuracy}",
|
||||
"stderr": "N/A",
|
||||
}
|
||||
)
|
||||
|
||||
generate_rm_report(eval_config, report_data, report_dir, env_config)
|
||||
assert success, "One or more RM tasks did not meet the accuracy threshold. See output above."
|
||||
@@ -0,0 +1,167 @@
|
||||
"""
|
||||
chunk_fwd_o correctness tests on Ascend 310P via torch.ops._C_ascend binding.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch_npu # noqa: F401
|
||||
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
CHUNK_SIZE = 64
|
||||
|
||||
|
||||
def npu_chunk_fwd_o(q, k, v, h, g, scale):
|
||||
enable_custom_op()
|
||||
return torch.ops._C_ascend.chunk_fwd_o(
|
||||
q,
|
||||
k,
|
||||
v,
|
||||
h,
|
||||
scale,
|
||||
g=g,
|
||||
g_gamma=None,
|
||||
cu_seqlens=None,
|
||||
chunk_indices=None,
|
||||
chunk_size=CHUNK_SIZE,
|
||||
transpose_state_layout=False,
|
||||
)
|
||||
|
||||
|
||||
def golden_chunk_fwd_o(q, k, v, h_state, g, scale):
|
||||
"""CPU fp32 reference.
|
||||
|
||||
Per chunk c (CS tokens starting at t0):
|
||||
attn = q[c] @ k[c].T [CS, CS]
|
||||
gate[i,j] = exp(min(0, g[j] - g[i])) * (j<=i) [CS, CS]
|
||||
attn_masked = attn * gate
|
||||
h_work = q[c] @ h_state[c] [CS, Dv]
|
||||
v_work = attn_masked @ v[c] [CS, Dv]
|
||||
o[c] = scale * (v_work + exp(g[c]) * h_work)
|
||||
"""
|
||||
q, k, v, g = q.float(), k.float(), v.float(), g.float()
|
||||
h_state = h_state.float()
|
||||
B, H_k, L, D_k = q.shape
|
||||
H_v, D_v = v.shape[1], v.shape[3]
|
||||
CS = CHUNK_SIZE
|
||||
NT = L // CS
|
||||
head_groups = H_v // H_k
|
||||
o = torch.zeros(B, H_v, L, D_v)
|
||||
for b in range(B):
|
||||
for hv in range(H_v):
|
||||
hk = hv // head_groups
|
||||
for c in range(NT):
|
||||
t0 = c * CS
|
||||
q_c = q[b, hk, t0 : t0 + CS]
|
||||
k_c = k[b, hk, t0 : t0 + CS]
|
||||
v_c = v[b, hv, t0 : t0 + CS]
|
||||
g_c = g[b, hv, t0 : t0 + CS]
|
||||
h_c = h_state[b, hv, c * D_k : (c + 1) * D_k]
|
||||
attn = q_c @ k_c.T
|
||||
g_row = g_c.unsqueeze(1)
|
||||
g_col = g_c.unsqueeze(0)
|
||||
gate = torch.exp(torch.clamp(g_col - g_row, max=0.0))
|
||||
causal = torch.tril(torch.ones(CS, CS))
|
||||
attn_masked = attn * gate * causal
|
||||
h_work = q_c @ h_c
|
||||
v_work = attn_masked @ v_c
|
||||
g_exp = torch.exp(g_c).unsqueeze(1)
|
||||
o[b, hv, t0 : t0 + CS] = scale * (v_work + g_exp * h_work)
|
||||
return o
|
||||
|
||||
|
||||
class TestChunkFwdO310:
|
||||
"""chunk_fwd_o kernel correctness on Ascend 310P."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"B,Hk,Hv,L,Dk,Dv",
|
||||
[
|
||||
(1, 2, 2, 128, 128, 128),
|
||||
(1, 4, 4, 256, 128, 128),
|
||||
],
|
||||
)
|
||||
def test_constant_inputs(self, B, Hk, Hv, L, Dk, Dv):
|
||||
"""Constant q=k=v, h=0, g=0 => analytically verifiable output."""
|
||||
scale = 1.0 / (Dk**0.5)
|
||||
NC = L // CHUNK_SIZE
|
||||
c = 0.01
|
||||
q = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
|
||||
k = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
|
||||
v = torch.full((B, Hv, L, Dv), c, dtype=torch.float16).npu()
|
||||
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16).npu()
|
||||
g = torch.zeros(B, Hv, L, dtype=torch.float32).npu()
|
||||
|
||||
o = npu_chunk_fwd_o(q, k, v, h, g, scale)
|
||||
oc = o.cpu().float()
|
||||
|
||||
assert torch.isnan(oc).sum() == 0, "output has NaN"
|
||||
assert torch.isinf(oc).sum() == 0, "output has Inf"
|
||||
|
||||
attn_val = c * c * Dk
|
||||
for i in range(min(CHUNK_SIZE, 8)):
|
||||
expected = scale * (i + 1) * attn_val * c
|
||||
actual = oc[0, 0, i, 0].item()
|
||||
rel_err = abs(actual - expected) / max(abs(expected), 1e-10)
|
||||
assert rel_err < 0.10, f"row {i}: actual={actual:.8f} expected={expected:.8f} rel_err={rel_err:.2f}"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"B,Hk,Hv,L,Dk,Dv",
|
||||
[
|
||||
(1, 2, 2, 128, 128, 128),
|
||||
(1, 4, 4, 256, 128, 128),
|
||||
],
|
||||
)
|
||||
def test_random_inputs_no_nan(self, B, Hk, Hv, L, Dk, Dv):
|
||||
"""Random small inputs: no NaN/Inf in output."""
|
||||
torch.manual_seed(42)
|
||||
scale = 1.0 / (Dk**0.5)
|
||||
NC = L // CHUNK_SIZE
|
||||
q = (torch.randn(B, Hk, L, Dk) * 0.01).half().npu()
|
||||
k = (torch.randn(B, Hk, L, Dk) * 0.01).half().npu()
|
||||
v = (torch.randn(B, Hv, L, Dv) * 0.01).half().npu()
|
||||
h = (torch.randn(B, Hv, NC * Dk, Dv) * 0.01).half().npu()
|
||||
g = torch.randn(B, Hv, L, dtype=torch.float32).npu() * 0.001
|
||||
|
||||
o = npu_chunk_fwd_o(q, k, v, h, g, scale)
|
||||
oc = o.cpu().float()
|
||||
|
||||
assert torch.isnan(oc).sum() == 0, "output has NaN"
|
||||
assert torch.isinf(oc).sum() == 0, "output has Inf"
|
||||
assert oc.abs().max() > 0, "output is all zeros"
|
||||
|
||||
def test_g_zero_reduces_to_standard_attention(self):
|
||||
"""g=0 => gate=1, so kernel = scale*(causal_attn@v + q@h)."""
|
||||
torch.manual_seed(123)
|
||||
B, Hk, Hv, L, Dk, Dv = 1, 2, 2, 128, 128, 128
|
||||
scale = 1.0 / (Dk**0.5)
|
||||
NC = L // CHUNK_SIZE
|
||||
q = (torch.randn(B, Hk, L, Dk) * 0.01).half()
|
||||
k = (torch.randn(B, Hk, L, Dk) * 0.01).half()
|
||||
v = (torch.randn(B, Hv, L, Dv) * 0.01).half()
|
||||
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16)
|
||||
g = torch.zeros(B, Hv, L, dtype=torch.float32)
|
||||
|
||||
o_npu = npu_chunk_fwd_o(q.npu(), k.npu(), v.npu(), h.npu(), g.npu(), scale)
|
||||
o_ref = golden_chunk_fwd_o(q, k, v, h, g, scale)
|
||||
|
||||
cos = torch.nn.functional.cosine_similarity(o_npu.cpu().float().flatten(), o_ref.flatten(), dim=0).item()
|
||||
assert cos > 0.999, f"cosine {cos:.4f} too low for g=0 h=0 case"
|
||||
|
||||
def test_chunk_boundary_independence(self):
|
||||
"""Each chunk should produce the same output for identical data."""
|
||||
B, Hk, Hv, L, Dk, Dv = 1, 2, 2, 128, 128, 128
|
||||
scale = 1.0 / (Dk**0.5)
|
||||
NC = L // CHUNK_SIZE
|
||||
c = 0.02
|
||||
q = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
|
||||
k = torch.full((B, Hk, L, Dk), c, dtype=torch.float16).npu()
|
||||
v = torch.full((B, Hv, L, Dv), c, dtype=torch.float16).npu()
|
||||
h = torch.zeros(B, Hv, NC * Dk, Dv, dtype=torch.float16).npu()
|
||||
g = torch.zeros(B, Hv, L, dtype=torch.float32).npu()
|
||||
|
||||
o = npu_chunk_fwd_o(q, k, v, h, g, scale).cpu().float()
|
||||
|
||||
chunk0 = o[0, 0, :CHUNK_SIZE, :]
|
||||
chunk1 = o[0, 0, CHUNK_SIZE:, :]
|
||||
cos = torch.nn.functional.cosine_similarity(chunk0.flatten(), chunk1.flatten(), dim=0).item()
|
||||
assert cos > 0.999, f"chunks differ: cosine={cos:.6f}"
|
||||
@@ -0,0 +1,160 @@
|
||||
"""
|
||||
chunk_gated_delta_rule_fwd_h correctness tests on Ascend 310P
|
||||
via torch.ops._C_ascend binding.
|
||||
"""
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch_npu # noqa: F401
|
||||
|
||||
from vllm_ascend.utils import enable_custom_op
|
||||
|
||||
CHUNK_SIZE = 64
|
||||
|
||||
|
||||
def npu_chunk_gdr_fwd_h(k, w, u, g, initial_state=None, chunk_size=64):
|
||||
enable_custom_op()
|
||||
return torch.ops._C_ascend.chunk_gated_delta_rule_fwd_h(
|
||||
k,
|
||||
w,
|
||||
u,
|
||||
g=g,
|
||||
initial_state=initial_state,
|
||||
output_final_state=False,
|
||||
chunk_size=chunk_size,
|
||||
save_new_value=True,
|
||||
)
|
||||
|
||||
|
||||
def cpu_reference(k, w, u, g, initial_state=None, chunk_size=64):
|
||||
"""CPU fp32 reference matching kernel semantics."""
|
||||
k, w, u, g = k.float(), w.float(), u.float(), g.float()
|
||||
B, Hg, T, K = k.shape
|
||||
HV, V = u.shape[1], u.shape[3]
|
||||
NT = T // chunk_size
|
||||
h = initial_state.float().clone() if initial_state is not None else torch.zeros(B, HV, K, V)
|
||||
h_chunks = [h.clone()]
|
||||
v_new = torch.zeros_like(u)
|
||||
|
||||
for c in range(NT):
|
||||
t0 = c * chunk_size
|
||||
W_chunk = w[:, :, t0 : t0 + chunk_size, :]
|
||||
ws = torch.einsum("bhik,bhkv->bhiv", W_chunk, h)
|
||||
g_chunk = g[:, :, t0 : t0 + chunk_size]
|
||||
v_update = torch.zeros(B, HV, chunk_size, V)
|
||||
for i in range(chunk_size):
|
||||
gi_cum = g_chunk[:, :, -1] - g_chunk[:, :, i]
|
||||
vn = u[:, :, t0 + i, :] - ws[:, :, i, :]
|
||||
v_new[:, :, t0 + i, :] = vn
|
||||
v_update[:, :, i, :] = gi_cum.unsqueeze(-1).exp() * vn
|
||||
K_chunk = k[:, :, t0 : t0 + chunk_size, :]
|
||||
h_work = torch.einsum("bhik,bhiv->bhkv", K_chunk, v_update)
|
||||
h = h * g_chunk[:, :, -1:].unsqueeze(-1).exp() + h_work
|
||||
h_chunks.append(h.clone())
|
||||
return h_chunks, v_new
|
||||
|
||||
|
||||
def cosine(a, b):
|
||||
a, b = a.flatten().double(), b.flatten().double()
|
||||
if a.norm() == 0 and b.norm() == 0:
|
||||
return 1.0
|
||||
if a.norm() == 0 or b.norm() == 0:
|
||||
return 0.0
|
||||
return torch.nn.functional.cosine_similarity(a.unsqueeze(0), b.unsqueeze(0)).item()
|
||||
|
||||
|
||||
class TestChunkGatedDeltaRuleFwdH310:
|
||||
"""chunk_gated_delta_rule_fwd_h kernel correctness on Ascend 310P."""
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"B,Hg,HV,T,K,V",
|
||||
[
|
||||
(1, 1, 1, 128, 128, 128),
|
||||
(1, 2, 2, 128, 128, 128),
|
||||
],
|
||||
)
|
||||
def test_h_state_correctness(self, B, Hg, HV, T, K, V):
|
||||
torch.manual_seed(42)
|
||||
DTYPE = torch.float16
|
||||
k = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
|
||||
w = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
|
||||
u = torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1
|
||||
g = (-torch.rand(B, HV, T) * 0.1).float()
|
||||
init = torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01
|
||||
|
||||
h_ref, _ = cpu_reference(k, w, u, g, init, CHUNK_SIZE)
|
||||
h_out, _, _ = npu_chunk_gdr_fwd_h(
|
||||
k.npu(),
|
||||
w.npu(),
|
||||
u.npu(),
|
||||
g.npu(),
|
||||
initial_state=init.npu(),
|
||||
chunk_size=CHUNK_SIZE,
|
||||
)
|
||||
h_npu = h_out.cpu().float()
|
||||
NT = T // CHUNK_SIZE
|
||||
|
||||
for c in range(min(NT + 1, h_npu.shape[2])):
|
||||
ref = h_ref[c].flatten()
|
||||
npu = h_npu[0, :, c].flatten()
|
||||
cos = cosine(npu, ref)
|
||||
assert cos >= 0.99, f"h[{c}] cos={cos:.6f} too low"
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"B,Hg,HV,T,K,V",
|
||||
[
|
||||
(1, 1, 1, 128, 128, 128),
|
||||
(1, 2, 2, 128, 128, 128),
|
||||
],
|
||||
)
|
||||
def test_v_new_correctness(self, B, Hg, HV, T, K, V):
|
||||
torch.manual_seed(42)
|
||||
DTYPE = torch.float16
|
||||
k = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
|
||||
w = torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1
|
||||
u = torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1
|
||||
g = (-torch.rand(B, HV, T) * 0.1).float()
|
||||
init = torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01
|
||||
|
||||
_, vn_ref = cpu_reference(k, w, u, g, init, CHUNK_SIZE)
|
||||
_, vn_out, _ = npu_chunk_gdr_fwd_h(
|
||||
k.npu(),
|
||||
w.npu(),
|
||||
u.npu(),
|
||||
g.npu(),
|
||||
initial_state=init.npu(),
|
||||
chunk_size=CHUNK_SIZE,
|
||||
)
|
||||
vn_npu = vn_out.cpu().float()
|
||||
NT = T // CHUNK_SIZE
|
||||
|
||||
for c in range(NT):
|
||||
t0, t1 = c * CHUNK_SIZE, (c + 1) * CHUNK_SIZE
|
||||
ref = vn_ref[:, :, t0:t1].flatten()
|
||||
npu = vn_npu[:, :, t0:t1].flatten()
|
||||
cos = cosine(npu, ref)
|
||||
assert cos >= 0.99, f"v_new chunk {c} cos={cos:.6f} too low"
|
||||
|
||||
def test_no_nan(self):
|
||||
torch.manual_seed(42)
|
||||
B, Hg, HV, T, K, V = 1, 1, 1, 128, 128, 128
|
||||
DTYPE = torch.float16
|
||||
k = (torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1).npu()
|
||||
w = (torch.randn(B, Hg, T, K, dtype=DTYPE) * 0.1).npu()
|
||||
u = (torch.randn(B, HV, T, V, dtype=DTYPE) * 0.1).npu()
|
||||
g = (-torch.rand(B, HV, T).float() * 0.1).npu()
|
||||
init = (torch.randn(B, HV, K, V, dtype=DTYPE) * 0.01).npu()
|
||||
|
||||
h_out, vn_out, _ = npu_chunk_gdr_fwd_h(
|
||||
k,
|
||||
w,
|
||||
u,
|
||||
g,
|
||||
initial_state=init,
|
||||
chunk_size=CHUNK_SIZE,
|
||||
)
|
||||
|
||||
assert torch.isnan(h_out.cpu()).sum() == 0, "h_out has NaN"
|
||||
assert torch.isnan(vn_out.cpu()).sum() == 0, "vn_out has NaN"
|
||||
assert torch.isinf(h_out.cpu()).sum() == 0, "h_out has Inf"
|
||||
assert torch.isinf(vn_out.cpu()).sum() == 0, "vn_out has Inf"
|
||||
@@ -0,0 +1,178 @@
|
||||
"""Test V310 kernel via ctypes API against golden CPU reference."""
|
||||
|
||||
import ctypes
|
||||
import os
|
||||
|
||||
import pytest
|
||||
import torch
|
||||
import torch_npu
|
||||
|
||||
torch_npu.npu.set_compile_mode(jit_compile=False)
|
||||
|
||||
_CANN = os.environ.get("ASCEND_HOME_PATH", "/usr/local/Ascend/ascend-toolkit/latest")
|
||||
_CUST = f"{_CANN}/opp/vendors/custom_transformer/op_api/lib"
|
||||
_LIB_PATHS = [_CUST, f"{_CANN}/lib64", f"{_CANN}/aarch64-linux/lib64"]
|
||||
|
||||
|
||||
def _find_lib(name, paths):
|
||||
for p in paths:
|
||||
full = os.path.join(p, name)
|
||||
if os.path.exists(full):
|
||||
return full
|
||||
return name
|
||||
|
||||
|
||||
_acl = ctypes.CDLL(_find_lib("libnnopbase.so", _LIB_PATHS))
|
||||
_opapi = ctypes.CDLL(_find_lib("libcust_opapi.so", _LIB_PATHS))
|
||||
|
||||
_acl.aclCreateTensor.restype = ctypes.c_void_p
|
||||
_acl.aclCreateTensor.argtypes = [
|
||||
ctypes.POINTER(ctypes.c_int64),
|
||||
ctypes.c_uint64,
|
||||
ctypes.c_int,
|
||||
ctypes.POINTER(ctypes.c_int64),
|
||||
ctypes.c_int64,
|
||||
ctypes.c_int,
|
||||
ctypes.POINTER(ctypes.c_int64),
|
||||
ctypes.c_uint64,
|
||||
ctypes.c_void_p,
|
||||
]
|
||||
_acl.aclDestroyTensor.argtypes = [ctypes.c_void_p]
|
||||
|
||||
_DTYPE_MAP = {torch.float16: 1, torch.float32: 0, torch.int32: 3}
|
||||
|
||||
|
||||
def mk(t):
|
||||
if t is None:
|
||||
return None
|
||||
shape, strides, ndim = list(t.shape), list(t.stride()), len(t.shape)
|
||||
return ctypes.c_void_p(
|
||||
_acl.aclCreateTensor(
|
||||
(ctypes.c_int64 * ndim)(*shape),
|
||||
ndim,
|
||||
_DTYPE_MAP[t.dtype],
|
||||
(ctypes.c_int64 * ndim)(*strides),
|
||||
ctypes.c_int64(0),
|
||||
2,
|
||||
(ctypes.c_int64 * ndim)(*shape),
|
||||
ndim,
|
||||
ctypes.c_void_p(t.data_ptr()),
|
||||
)
|
||||
)
|
||||
|
||||
|
||||
def call_v310(query, key, value, beta, state, seq_lens, indices, g, nat, scale):
|
||||
out = torch.empty_like(value)
|
||||
ws_size = ctypes.c_uint64(0)
|
||||
executor = ctypes.c_void_p(0)
|
||||
ret = _opapi.aclnnRecurrentGatedDeltaRuleV310GetWorkspaceSize(
|
||||
mk(query),
|
||||
mk(key),
|
||||
mk(value),
|
||||
mk(beta),
|
||||
mk(state),
|
||||
mk(seq_lens),
|
||||
mk(indices),
|
||||
mk(g),
|
||||
None,
|
||||
mk(nat),
|
||||
ctypes.c_float(scale),
|
||||
mk(out),
|
||||
ctypes.byref(ws_size),
|
||||
ctypes.byref(executor),
|
||||
)
|
||||
assert ret == 0, f"GetWorkspaceSize failed: {ret}"
|
||||
ws_ptr = ctypes.c_void_p(0)
|
||||
if ws_size.value > 0:
|
||||
ws = torch.empty(ws_size.value, dtype=torch.uint8, device=query.device)
|
||||
ws_ptr = ctypes.c_void_p(ws.data_ptr())
|
||||
stream = torch.npu.current_stream().npu_stream
|
||||
ret = _opapi.aclnnRecurrentGatedDeltaRuleV310(ws_ptr, ws_size, executor, ctypes.c_void_p(stream))
|
||||
assert ret == 0, f"Execute failed: {ret}"
|
||||
torch.npu.synchronize()
|
||||
return out
|
||||
|
||||
|
||||
def golden(query, key, value, state, beta, scale, seq_lens, indices, g, nat):
|
||||
k = key.float()
|
||||
q = query.float()
|
||||
v = value.float()
|
||||
S = state.clone().float()
|
||||
T, nv, Dv = v.shape
|
||||
nk = q.shape[1]
|
||||
g_f = torch.ones(T, nv) if g is None else g.float().exp()
|
||||
beta_f = beta.float()
|
||||
o = torch.empty_like(v, dtype=torch.float32)
|
||||
q = q * scale
|
||||
seq_start = 0
|
||||
for i in range(len(seq_lens)):
|
||||
init_idx = indices[seq_start + nat[i] - 1] if nat is not None else indices[seq_start]
|
||||
for head in range(nv):
|
||||
s = S[init_idx][head].clone()
|
||||
for t in range(seq_start, seq_start + seq_lens[i]):
|
||||
qi = q[t][head // (nv // nk)]
|
||||
ki = k[t][head // (nv // nk)]
|
||||
vi = v[t][head]
|
||||
s = s * g_f[t][head]
|
||||
x = (s * ki.unsqueeze(-2)).sum(dim=-1)
|
||||
y = (vi - x) * beta_f[t][head]
|
||||
s = s + y[:, None] * ki[None, :]
|
||||
S[indices[t]][head] = s
|
||||
o[t][head] = (s * qi.unsqueeze(-2)).sum(dim=-1)
|
||||
seq_start += seq_lens[i]
|
||||
return o, S
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
"batch_size,mtp,nk,nv,dk,dv,num_slots",
|
||||
[
|
||||
(1, 1, 8, 16, 128, 128, 444),
|
||||
(2, 2, 8, 16, 128, 128, 444),
|
||||
(4, 2, 4, 4, 64, 64, 32),
|
||||
],
|
||||
)
|
||||
def test_recurrent_gated_delta_rule_v310(batch_size, mtp, nk, nv, dk, dv, num_slots):
|
||||
torch.manual_seed(42)
|
||||
scale = dk**-0.5
|
||||
seq_lens = torch.ones(batch_size, dtype=torch.int32) * mtp
|
||||
T = int(seq_lens.sum())
|
||||
state = torch.rand(num_slots, nv, dv, dk, dtype=torch.float16)
|
||||
indices = torch.randperm(num_slots, dtype=torch.int32)[:T]
|
||||
nat = torch.ones(batch_size, dtype=torch.int32)
|
||||
query = torch.nn.functional.normalize(torch.randn(T, nk, dk), dim=-1).to(torch.float16)
|
||||
key = torch.nn.functional.normalize(torch.randn(T, nk, dk), dim=-1).to(torch.float16)
|
||||
value = torch.randn(T, nv, dv, dtype=torch.float16)
|
||||
beta = torch.rand(T, nv, dtype=torch.float16)
|
||||
g = torch.rand(T, nv, dtype=torch.float32)
|
||||
|
||||
out_gold, state_gold = golden(query, key, value, state, beta, scale, seq_lens, indices, g, nat)
|
||||
|
||||
state_npu = state.clone().npu()
|
||||
out_npu = call_v310(
|
||||
query.npu(),
|
||||
key.npu(),
|
||||
value.npu(),
|
||||
beta.npu(),
|
||||
state_npu,
|
||||
seq_lens.npu(),
|
||||
indices.npu(),
|
||||
g.npu(),
|
||||
nat.npu(),
|
||||
scale,
|
||||
)
|
||||
|
||||
touched = indices.long()
|
||||
torch.testing.assert_close(
|
||||
out_npu.float().cpu(),
|
||||
out_gold,
|
||||
rtol=3e-3,
|
||||
atol=2e-3,
|
||||
equal_nan=True,
|
||||
)
|
||||
torch.testing.assert_close(
|
||||
state_npu.float().cpu()[touched],
|
||||
state_gold.float()[touched],
|
||||
rtol=3e-3,
|
||||
atol=2e-3,
|
||||
equal_nan=True,
|
||||
)
|
||||
0
tests/e2e/nightly/multi_node/__init__.py
Normal file
0
tests/e2e/nightly/multi_node/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/external_dp/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/external_dp/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
"""External DP nightly test package."""
|
||||
@@ -0,0 +1,308 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-1M-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 0
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 1
|
||||
dp_rank_start: 1
|
||||
tp_size: 16
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "2048"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "128"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "60"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "128"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 2, "tp_size": 16}, "decode": {"dp_size": 2, "tp_size": 16}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":false,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_1M_1k_prefix99_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_1M_1k_prefix99:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1M-bs4-prefix99-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1024
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 11.93
|
||||
threshold: 0.95
|
||||
@@ -0,0 +1,324 @@
|
||||
test_name: "DeepSeek-V4-Pro-w4a8-prefix-cache-PD"
|
||||
model: "Eco-Tech/DeepSeek-V4-Pro-w4a8-mtp"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0, 1 ]
|
||||
decoder: [ 2, 3 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 0
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 8
|
||||
dp_rank_start: 8
|
||||
tp_size: 2
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "6000"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_BUFFSIZE: "1800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --block-size
|
||||
- "32"
|
||||
- --enforce-eager
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "204800"
|
||||
- --max-num-batched-tokens
|
||||
- "120"
|
||||
- --max-num-seqs
|
||||
- "30"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30800", "engine_id": "8", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 8}, "decode": {"dp_size": 16, "tp_size": 2}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
perf_TPOT50_128k_1_prefix_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 1
|
||||
batch_size: 4
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.95
|
||||
perf_TPOT50_128k_1_prefix90:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in128k-bs512-prefix90-ds
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 192
|
||||
max_out_len: 1024
|
||||
batch_size: 48
|
||||
request_rate: 1
|
||||
baseline: 869.13
|
||||
threshold: 0.95
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,176 @@
|
||||
test_name: "DeepSeek-V4-Flash-w8a8-PD-prefix"
|
||||
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 16
|
||||
dp_size_local: 16
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_RPC_TIMEOUT: "3600"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_PREFIX_CACHE_RETENTION_INTERVAL: "4096"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1500"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp"}'
|
||||
- --trust-remote-code
|
||||
- --block-size
|
||||
- "32"
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.9"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"enable_cpu_binding": true, "enable_dsa_cp": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_producer", "kv_port": "30000", "engine_id": "0", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "1048576"
|
||||
- --max-num-batched-tokens
|
||||
- "240"
|
||||
- --max-num-seqs
|
||||
- "60"
|
||||
- --async-scheduling
|
||||
- --block-size
|
||||
- "32"
|
||||
- --no-enable-prefix-caching
|
||||
- --no-disable-hybrid-kv-cache-manager
|
||||
- --model-loader-extra-config
|
||||
- '{"enable_multithread_load": true, "num_threads": 128}'
|
||||
- --trust-remote-code
|
||||
- --tokenizer-mode
|
||||
- "deepseek_v4"
|
||||
- --tool-call-parser
|
||||
- "deepseek_v4"
|
||||
- --enable-auto-tool-choice
|
||||
- --reasoning-parser
|
||||
- "deepseek_v4"
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3,"method": "mtp"}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeHybridConnector", "kv_role": "kv_consumer", "kv_port": "30100", "engine_id": "1", "kv_connector_extra_config": {"prefill": {"dp_size": 4, "tp_size": 4}, "decode": {"dp_size": 16, "tp_size": 1}}}'
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":true,"multistream_overlap_shared_expert":true,"recompute_scheduler_enable":true}'
|
||||
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
temperature: 1.0
|
||||
top_p: 1.0
|
||||
thinking: "true"
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
@@ -0,0 +1,335 @@
|
||||
test_name: "multi-node-glm-5.1-w8a8-ep-external-dp"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0, 1]
|
||||
decoder: [2, 3]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 2
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
- node_index: 3
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 8
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 4
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_2_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: "0"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_INTRA_PCIE_ENABLE: "1"
|
||||
HCCL_INTRA_ROCE_ENABLE: "0"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "131072"
|
||||
- --additional-config
|
||||
- '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "64"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.95"
|
||||
- --enforce-eager
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 2
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 3
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- ${PORT}
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --speculative-config
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "202752"
|
||||
- --max-num-batched-tokens
|
||||
- "32"
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
- --max-num-batched-tokens
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --max-num-seqs
|
||||
- "8"
|
||||
- --quantization
|
||||
- ascend
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --async-scheduling
|
||||
- --enable-auto-tool-choice
|
||||
- --tool-call-parser
|
||||
- glm47
|
||||
- --reasoning-parser
|
||||
- glm45
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1500
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,161 @@
|
||||
test_name: "Kimi-K2.6-W4A8-64k-1k-TPOT50-PD"
|
||||
model: "Eco-Tech/Kimi-K2.6-w4a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 4
|
||||
dp_rank_start: 0
|
||||
tp_size: 4
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
HCCL_EXEC_TIMEOUT: "204"
|
||||
HCCL_CONNECT_TIMEOUT: "120"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
VLLM_SERVER_DEV_MODE: "1"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "512"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "800"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --enable-expert-parallel
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "8192"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --enforce-eager
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.94"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 1}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_producer","kv_port": "30000","engine_id": "0","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --allowed-local-media-path
|
||||
- "/"
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --safetensors-load-strategy
|
||||
- 'prefetch'
|
||||
- --seed
|
||||
- "1024"
|
||||
- --max-model-len
|
||||
- "68000"
|
||||
- --max-num-batched-tokens
|
||||
- "256"
|
||||
- --max-num-seqs
|
||||
- "16"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.92"
|
||||
- --speculative-config
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 15}'
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true, "lmhead_tensor_parallel_size":16}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1","kv_role": "kv_consumer","kv_port": "30100","engine_id": "1","kv_connector_extra_config": {"use_ascend_direct": true,"prefill": {"dp_size": 2,"tp_size": 8},"decode": {"dp_size": 4,"tp_size": 4}}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in65536_bs1000_kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 60
|
||||
max_out_len: 1024
|
||||
batch_size: 15
|
||||
request_rate: 0.4
|
||||
baseline: 347.4475
|
||||
threshold: 0.97
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 93.33
|
||||
threshold: 10
|
||||
temperature: 1.0
|
||||
top_p: 1
|
||||
@@ -0,0 +1,197 @@
|
||||
test_name: "Minimax_m2.7_in3_5_tpot50"
|
||||
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [ 0 ]
|
||||
decoder: [ 1 ]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 8
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "${PORT}"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
|
||||
LD_LIBRARY_PATH: "/usr/local/Ascend/ascend-toolkit/latest/python/site-packages/mooncake:$LD_LIBRARY_PATH"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
PYTHONHASHSEED: "0"
|
||||
|
||||
|
||||
|
||||
env_prefill: &env_prefill
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
|
||||
env_decode: &env_decode
|
||||
<<: *env_common
|
||||
HCCL_BUFFSIZE: "2048"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_prefill
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --enforce-eager
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "55880",
|
||||
"engine_id": "0",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}} }'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_decode
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --enable-expert-parallel
|
||||
- --max-model-len
|
||||
- "199608"
|
||||
- --max-num-batched-tokens
|
||||
- "16384"
|
||||
- --max-num-seqs
|
||||
- "24"
|
||||
- --trust-remote-code
|
||||
- --gpu-memory-utilization
|
||||
- "0.8"
|
||||
- --quantization
|
||||
- "ascend"
|
||||
- --speculative-config
|
||||
- '{"method": "eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- --async-scheduling
|
||||
- --compilation-config
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- --additional-config
|
||||
- '{"recompute_scheduler_enable":true,"enable_cpu_binding": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}}'
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "56900",
|
||||
"engine_id": "1",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf_warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2
|
||||
max_out_len: 1
|
||||
batch_size: 2
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_Minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1024
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 717.5332
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
temperature: 1
|
||||
top_p: 1
|
||||
top_k: 40
|
||||
ignore_eos: false
|
||||
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
360
tests/e2e/nightly/multi_node/external_dp/config/template.md
Normal file
@@ -0,0 +1,360 @@
|
||||
# External DP Config Template
|
||||
|
||||
This document shows how to write YAML configs consumed by
|
||||
`tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py`.
|
||||
|
||||
`server_cmd_template` contains only the arguments after
|
||||
`vllm serve <model>`. The framework prepends `vllm serve` and the top-level
|
||||
`model` automatically.
|
||||
|
||||
Do not write `proxy_node_index`, `proxy_host`, `proxy_port`, `proxy_script`, or
|
||||
`dp_group` in YAML. The framework derives proxy metadata from `routing.type`,
|
||||
and roles are selected by `routing.groups`.
|
||||
|
||||
## Generic DP Template
|
||||
|
||||
Use this template for generic external data parallel serving. This mode uses
|
||||
`--data-parallel-rank`, so it is intended for MoE models. For dense models, use
|
||||
independent vLLM instances instead of external DP rank arguments.
|
||||
|
||||
```yaml
|
||||
test_name: "test Qwen3-30B-A3B generic external dp"
|
||||
model: "Qwen/Qwen3-30B-A3B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "generic_dp"
|
||||
groups:
|
||||
worker: [0, 1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 4
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 2
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs: &generic_env
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template: &generic_server_cmd
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --max-model-len
|
||||
- "4096"
|
||||
- --trust-remote-code
|
||||
- --enable-expert-parallel
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *generic_env
|
||||
server_cmd_template: *generic_server_cmd
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 4
|
||||
max_out_len: 16
|
||||
batch_size: 1
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Disaggregated Prefill Template
|
||||
|
||||
Use this template for PD disaggregation. `routing.groups` decides which config
|
||||
entries run as prefillers or decoders. The framework derives the PD proxy script
|
||||
from `routing.type`, so do not write `proxy_*` fields in YAML.
|
||||
|
||||
```yaml
|
||||
test_name: "test DeepSeek-V2-Lite-W8A8 external dp disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-V2-Lite-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
|
||||
# Optional for local debugging. In CI, cluster IPs are resolved from LWS DNS.
|
||||
# cluster_hosts:
|
||||
# - "172.22.0.xxx"
|
||||
# - "172.22.0.xxx"
|
||||
|
||||
routing:
|
||||
type: "disaggregated_prefill"
|
||||
groups:
|
||||
prefiller: [0]
|
||||
decoder: [1]
|
||||
|
||||
config:
|
||||
- node_index: 0
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_0_IP}"
|
||||
|
||||
- node_index: 1
|
||||
port_start: 7100
|
||||
dp_rpc_port: 12321
|
||||
dp_size: 2
|
||||
dp_size_local: 2
|
||||
dp_rank_start: 0
|
||||
tp_size: 1
|
||||
dp_address: "${NODE_1_IP}"
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "10"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
ASCEND_RT_VISIBLE_DEVICES: "${VISIBLE_DEVICES}"
|
||||
HCCL_BUFFSIZE: "256"
|
||||
SERVER_PORT: "${PORT}"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "0"
|
||||
|
||||
templates:
|
||||
- node_index: 0
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
- node_index: 1
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd_template:
|
||||
- --host
|
||||
- "0.0.0.0"
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
- --data-parallel-size
|
||||
- ${DP_SIZE}
|
||||
- --data-parallel-rank
|
||||
- ${DP_RANK}
|
||||
- --data-parallel-address
|
||||
- ${DP_ADDRESS}
|
||||
- --data-parallel-rpc-port
|
||||
- ${DP_RPC_PORT}
|
||||
- --tensor-parallel-size
|
||||
- ${TP_SIZE}
|
||||
- --trust-remote-code
|
||||
- --quantization
|
||||
- ascend
|
||||
- --enable-expert-parallel
|
||||
- --kv-transfer-config
|
||||
- '{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 1
|
||||
}
|
||||
}}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
max_out_len: 128
|
||||
batch_size: 4
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.1
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 48
|
||||
batch_size: 4
|
||||
baseline: 0
|
||||
threshold: 100
|
||||
```
|
||||
|
||||
## Field Notes
|
||||
|
||||
- `test_name`: Human-readable test name. It is also used when writing benchmark
|
||||
result metadata.
|
||||
- `model`: Model passed to `vllm serve <model>` and AISBench requests.
|
||||
- `num_nodes`: Number of config entries and templates expected.
|
||||
- `npu_per_node`: Device capacity validation for each node.
|
||||
- `cluster_hosts`: Optional local-debug IP list. Omit it in CI unless a test
|
||||
needs fixed hosts.
|
||||
- `routing.type`: Supported values are `generic_dp` and
|
||||
`disaggregated_prefill`.
|
||||
- `routing.groups`: Maps config indices to roles. `generic_dp` requires
|
||||
`worker`; `disaggregated_prefill` requires `prefiller` and `decoder`.
|
||||
- For `disaggregated_prefill`, use `kv_producer` for prefiller templates and
|
||||
`kv_consumer` for decoder templates.
|
||||
- `config[].dp_size`: Global DP size for this DP group.
|
||||
- `config[].dp_size_local`: Number of vLLM ranks started on this node.
|
||||
- `config[].dp_rank_start`: First global DP rank owned by this node.
|
||||
- `config[].dp_address`: DP master address. For one global DP group, use
|
||||
`${NODE_0_IP}` on all nodes. For PD disaggregation, use the prefiller master
|
||||
address for prefiller nodes and the decoder master address for decoder nodes.
|
||||
- `templates`: One template per config entry. The framework expands one command
|
||||
per local DP rank.
|
||||
|
||||
The framework injects distributed network envs at startup:
|
||||
|
||||
```text
|
||||
HCCL_IF_IP
|
||||
HCCL_SOCKET_IFNAME
|
||||
GLOO_SOCKET_IFNAME
|
||||
TP_SOCKET_IFNAME
|
||||
LOCAL_IP
|
||||
NIC_NAME
|
||||
MASTER_IP
|
||||
```
|
||||
|
||||
The framework also derives proxy metadata from `routing.type`:
|
||||
|
||||
```text
|
||||
generic_dp -> examples/external_online_dp/dp_load_balance_proxy_server.py
|
||||
disaggregated_prefill -> examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py
|
||||
```
|
||||
|
||||
The proxy runs on node 0, listens on `${NODE_0_IP}:1999`, and is used by node 0
|
||||
for benchmark requests.
|
||||
|
||||
## Template Variables
|
||||
|
||||
The following variables are available in `envs` and `server_cmd_template`:
|
||||
|
||||
```text
|
||||
${MODEL}
|
||||
${PORT_START}
|
||||
${PORT}
|
||||
${DP_SIZE}
|
||||
${DP_SIZE_LOCAL}
|
||||
${DP_RANK_START}
|
||||
${DP_RANK}
|
||||
${LOCAL_RANK}
|
||||
${TP_SIZE}
|
||||
${CP_SIZE}
|
||||
${SP_SIZE}
|
||||
${PP_SIZE}
|
||||
${DP_ADDRESS}
|
||||
${DP_RPC_PORT}
|
||||
${VISIBLE_DEVICES}
|
||||
${NODE_INDEX}
|
||||
${CONFIG_INDEX}
|
||||
${NODE_0_IP}, ${NODE_1_IP}, ...
|
||||
${LOCAL_IP}
|
||||
${MASTER_IP}
|
||||
${LWS_WORKER_INDEX}
|
||||
```
|
||||
|
||||
Command arguments can also reference rendered environment variables with
|
||||
shell-style `$VARNAME`, for example:
|
||||
|
||||
```yaml
|
||||
envs:
|
||||
SERVER_PORT: "${PORT}"
|
||||
server_cmd_template:
|
||||
- --port
|
||||
- $SERVER_PORT
|
||||
```
|
||||
|
||||
## Checks Before Running
|
||||
|
||||
- Keep `len(config) == num_nodes` and `len(templates) == num_nodes`.
|
||||
- Make sure each config index is assigned to exactly one routing group.
|
||||
- Ensure `dp_rank_start + dp_size_local <= dp_size`.
|
||||
- Ensure `dp_size_local * tp_size * cp_size * sp_size * pp_size <= npu_per_node`.
|
||||
- For `generic_dp` with `--data-parallel-rank`, use an MoE model and
|
||||
`--enable-expert-parallel`.
|
||||
- Set `--max-model-len` large enough for benchmark input tokens plus
|
||||
`max_out_len`.
|
||||
@@ -0,0 +1 @@
|
||||
"""External DP nightly test helpers."""
|
||||
@@ -0,0 +1,449 @@
|
||||
import logging
|
||||
import os
|
||||
from dataclasses import dataclass, field
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
load_yaml_mapping,
|
||||
resolve_cluster_ips,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
resolve_current_node_index as resolve_node_index,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
ROUTING_GENERIC_DP = "generic_dp"
|
||||
ROUTING_DISAGGREGATED_PREFILL = "disaggregated_prefill"
|
||||
PROXY_SCRIPT_BY_ROUTING_TYPE = {
|
||||
ROUTING_GENERIC_DP: "examples/external_online_dp/dp_load_balance_proxy_server.py",
|
||||
ROUTING_DISAGGREGATED_PREFILL: "examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
|
||||
}
|
||||
|
||||
CLUSTER_PLACEHOLDER_RE = re.compile(r"\$\{(NODE_(\d+)_IP|LOCAL_IP|MASTER_IP|LWS_WORKER_INDEX)\}")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RoutingConfig:
|
||||
"""Proxy routing metadata shared by all external DP ranks."""
|
||||
|
||||
type: str
|
||||
proxy_node_index: int
|
||||
proxy_host: str
|
||||
proxy_port: int
|
||||
proxy_script: str
|
||||
groups: dict[str, list[int]]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeInfo:
|
||||
"""Per-node external DP server topology loaded from one config entry."""
|
||||
|
||||
ip: str
|
||||
port_start: int
|
||||
dp_rpc_port: int
|
||||
dp_size: int
|
||||
dp_size_local: int
|
||||
dp_rank_start: int
|
||||
tp_size: int
|
||||
dp_address: str
|
||||
cp_size: int = 1
|
||||
sp_size: int = 1
|
||||
pp_size: int = 1
|
||||
|
||||
@property
|
||||
def devices_per_rank(self) -> int:
|
||||
return self.tp_size * self.cp_size * self.sp_size * self.pp_size
|
||||
|
||||
@property
|
||||
def devices_per_node(self) -> int:
|
||||
return self.dp_size_local * self.devices_per_rank
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeTemplate:
|
||||
"""Per-node env and argument template for launching vLLM servers."""
|
||||
|
||||
envs: dict[str, Any]
|
||||
server_cmd_template: list[str]
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class RankInfo:
|
||||
"""One concrete vLLM server rank expanded from a node config."""
|
||||
|
||||
node_index: int
|
||||
role: str
|
||||
local_rank: int
|
||||
dp_rank: int
|
||||
host: str
|
||||
port: int
|
||||
visible_devices: str
|
||||
dp_size: int
|
||||
dp_size_local: int
|
||||
tp_size: int
|
||||
cp_size: int
|
||||
sp_size: int
|
||||
pp_size: int
|
||||
dp_address: str
|
||||
dp_rpc_port: int
|
||||
port_start: int
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ExternalDPConfig:
|
||||
"""Top-level external DP test config after YAML anchors are merged."""
|
||||
|
||||
test_name: str
|
||||
model: str
|
||||
num_nodes: int
|
||||
npu_per_node: int
|
||||
cluster_hosts: list[str] | None
|
||||
cluster_ips: list[str]
|
||||
routing: RoutingConfig
|
||||
nodes: list[NodeInfo]
|
||||
launch_templates: list[NodeTemplate]
|
||||
benchmark_cases: list[dict[str, Any]] = field(default_factory=list)
|
||||
special_dependencies: dict[str, str] = field(default_factory=dict)
|
||||
|
||||
@property
|
||||
def is_disaggregated_prefill(self) -> bool:
|
||||
return self.routing.type == ROUTING_DISAGGREGATED_PREFILL
|
||||
|
||||
|
||||
def replace_cluster_placeholders(
|
||||
value: Any,
|
||||
*,
|
||||
cluster_ips: list[str],
|
||||
local_ip: str | None = None,
|
||||
current_node_index: int | None = None,
|
||||
) -> Any:
|
||||
if isinstance(value, dict):
|
||||
return {
|
||||
key: replace_cluster_placeholders(
|
||||
val,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=current_node_index,
|
||||
)
|
||||
for key, val in value.items()
|
||||
}
|
||||
if isinstance(value, list):
|
||||
return [
|
||||
replace_cluster_placeholders(
|
||||
item,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=current_node_index,
|
||||
)
|
||||
for item in value
|
||||
]
|
||||
if not isinstance(value, str):
|
||||
return value
|
||||
|
||||
def repl(match: re.Match[str]) -> str:
|
||||
token = match.group(1)
|
||||
node_index = match.group(2)
|
||||
if node_index is not None:
|
||||
idx = int(node_index)
|
||||
if idx >= len(cluster_ips):
|
||||
raise ValueError(f"Cluster placeholder ${{{token}}} is out of range")
|
||||
return cluster_ips[idx]
|
||||
if token == "MASTER_IP":
|
||||
return cluster_ips[0]
|
||||
if token == "LOCAL_IP":
|
||||
if local_ip is None:
|
||||
return match.group(0)
|
||||
return local_ip
|
||||
if token == "LWS_WORKER_INDEX":
|
||||
if current_node_index is None:
|
||||
return os.environ.get("LWS_WORKER_INDEX", match.group(0))
|
||||
return str(current_node_index)
|
||||
return match.group(0)
|
||||
|
||||
return CLUSTER_PLACEHOLDER_RE.sub(repl, value)
|
||||
|
||||
|
||||
def resolve_current_node_index(config: ExternalDPConfig) -> int:
|
||||
return resolve_node_index(config.cluster_ips)
|
||||
|
||||
|
||||
class ExternalDPConfigLoader:
|
||||
"""Load, normalize, and validate external DP YAML files."""
|
||||
|
||||
@classmethod
|
||||
def from_yaml(
|
||||
cls,
|
||||
yaml_path: str | None = None,
|
||||
*,
|
||||
cluster_ips: list[str] | None = None,
|
||||
) -> ExternalDPConfig:
|
||||
raw_config = cls._load_yaml(yaml_path)
|
||||
cls._validate_root(raw_config)
|
||||
|
||||
num_nodes = int(raw_config["num_nodes"])
|
||||
resolved_cluster_ips = cls._resolve_cluster_ips(raw_config, num_nodes, cluster_ips)
|
||||
|
||||
model = str(raw_config["model"])
|
||||
routing = cls._parse_routing(raw_config["routing"], resolved_cluster_ips)
|
||||
nodes = cls._parse_nodes(raw_config, resolved_cluster_ips)
|
||||
launch_templates = cls._parse_templates(raw_config)
|
||||
benchmark_cases = cls._parse_benchmarks(raw_config)
|
||||
|
||||
config = ExternalDPConfig(
|
||||
test_name=str(raw_config.get("test_name", "external_dp_test")),
|
||||
model=model,
|
||||
num_nodes=num_nodes,
|
||||
npu_per_node=int(raw_config["npu_per_node"]),
|
||||
cluster_hosts=raw_config.get("cluster_hosts"),
|
||||
cluster_ips=resolved_cluster_ips,
|
||||
routing=routing,
|
||||
nodes=nodes,
|
||||
launch_templates=launch_templates,
|
||||
benchmark_cases=benchmark_cases,
|
||||
special_dependencies=dict(raw_config.get("special_dependencies", {})),
|
||||
)
|
||||
cls._validate_config(config)
|
||||
return config
|
||||
|
||||
@staticmethod
|
||||
def _load_yaml(yaml_path: str | None) -> dict[str, Any]:
|
||||
default_config_name = "GLM5_1-W8A8-EP-external.yaml"
|
||||
default_config_base_path = "tests/e2e/nightly/multi_node/external_dp/config/"
|
||||
return load_yaml_mapping(
|
||||
yaml_path,
|
||||
default_name=default_config_name,
|
||||
default_base_path=default_config_base_path,
|
||||
description="external DP config",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_root(config: dict[str, Any]) -> None:
|
||||
required = ["model", "num_nodes", "npu_per_node", "routing", "config", "templates", "benchmarks"]
|
||||
missing = [key for key in required if key not in config]
|
||||
if missing:
|
||||
raise KeyError(f"Missing required external DP config fields: {missing}")
|
||||
if int(config["num_nodes"]) <= 0:
|
||||
raise ValueError("num_nodes must be greater than 0")
|
||||
|
||||
@staticmethod
|
||||
def _resolve_cluster_ips(
|
||||
raw_config: dict[str, Any],
|
||||
num_nodes: int,
|
||||
cluster_ips: list[str] | None,
|
||||
) -> list[str]:
|
||||
return resolve_cluster_ips(
|
||||
raw_config,
|
||||
num_nodes,
|
||||
cluster_ips,
|
||||
dns_log_message="Resolving external DP cluster IPs via LWS DNS",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _parse_routing(raw_routing: dict[str, Any], cluster_ips: list[str]) -> RoutingConfig:
|
||||
routing_type = str(raw_routing["type"])
|
||||
if routing_type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
|
||||
raise ValueError(f"Unsupported routing.type: {routing_type}")
|
||||
|
||||
proxy_node_index = 0
|
||||
proxy_port = 1999
|
||||
if proxy_node_index >= len(cluster_ips) or proxy_node_index < 0:
|
||||
raise ValueError("routing.proxy_node_index out of range")
|
||||
local_ip = cluster_ips[proxy_node_index]
|
||||
routing = replace_cluster_placeholders(
|
||||
raw_routing,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=local_ip,
|
||||
current_node_index=proxy_node_index,
|
||||
)
|
||||
return RoutingConfig(
|
||||
type=routing_type,
|
||||
proxy_node_index=proxy_node_index,
|
||||
proxy_host=local_ip,
|
||||
proxy_port=proxy_port,
|
||||
proxy_script=PROXY_SCRIPT_BY_ROUTING_TYPE[routing_type],
|
||||
groups={
|
||||
str(name): [int(index) for index in indices] for name, indices in routing.get("groups", {}).items()
|
||||
},
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _parse_nodes(raw_config: dict[str, Any], cluster_ips: list[str]) -> list[NodeInfo]:
|
||||
nodes: list[NodeInfo] = []
|
||||
for index, raw_node in enumerate(raw_config["config"]):
|
||||
raw_node_index = raw_node.get("node_index")
|
||||
if raw_node_index is not None and int(raw_node_index) != index:
|
||||
raise ValueError(f"config[{index}].node_index must equal {index}")
|
||||
node = replace_cluster_placeholders(
|
||||
raw_node,
|
||||
cluster_ips=cluster_ips,
|
||||
local_ip=cluster_ips[index],
|
||||
current_node_index=index,
|
||||
)
|
||||
nodes.append(
|
||||
NodeInfo(
|
||||
ip=cluster_ips[index],
|
||||
port_start=int(node["port_start"]),
|
||||
dp_rpc_port=int(node["dp_rpc_port"]),
|
||||
dp_size=int(node.get("dp_size", 1)),
|
||||
dp_size_local=int(node.get("dp_size_local", 1)),
|
||||
dp_rank_start=int(node.get("dp_rank_start", 0)),
|
||||
tp_size=int(node.get("tp_size", 1)),
|
||||
cp_size=int(node.get("cp_size", 1)),
|
||||
sp_size=int(node.get("sp_size", 1)),
|
||||
dp_address=str(node["dp_address"]),
|
||||
pp_size=int(node.get("pp_size", 1)),
|
||||
)
|
||||
)
|
||||
return nodes
|
||||
|
||||
@staticmethod
|
||||
def _parse_templates(raw_config: dict[str, Any]) -> list[NodeTemplate]:
|
||||
templates: list[NodeTemplate] = []
|
||||
for index, raw_template in enumerate(raw_config["templates"]):
|
||||
envs = raw_template.get("envs")
|
||||
server_cmd_template = raw_template.get("server_cmd_template")
|
||||
if envs is None or server_cmd_template is None:
|
||||
raise KeyError(f"templates[{index}] must contain envs and server_cmd_template")
|
||||
if not isinstance(server_cmd_template, list):
|
||||
raise TypeError(f"templates[{index}].server_cmd_template must be a list")
|
||||
templates.append(
|
||||
NodeTemplate(
|
||||
envs=dict(envs),
|
||||
server_cmd_template=[str(arg) for arg in server_cmd_template],
|
||||
)
|
||||
)
|
||||
return templates
|
||||
|
||||
@staticmethod
|
||||
def _parse_benchmarks(raw_config: dict[str, Any]) -> list[dict[str, Any]]:
|
||||
benchmark_cases: list[dict[str, Any]] = []
|
||||
for name, case in (raw_config.get("benchmarks") or {}).items():
|
||||
case_with_name = dict(case)
|
||||
case_with_name["case_name"] = name
|
||||
benchmark_cases.append(case_with_name)
|
||||
return benchmark_cases
|
||||
|
||||
@classmethod
|
||||
def _validate_config(cls, config: ExternalDPConfig) -> None:
|
||||
cls._validate_config_sizes(config)
|
||||
cls._validate_routing(config)
|
||||
cls._validate_node_parallel_config(config)
|
||||
|
||||
@staticmethod
|
||||
def _validate_config_sizes(config: ExternalDPConfig) -> None:
|
||||
if len(config.nodes) != config.num_nodes:
|
||||
raise AssertionError(f"config size ({len(config.nodes)}) != num_nodes ({config.num_nodes})")
|
||||
if len(config.launch_templates) != config.num_nodes:
|
||||
raise AssertionError(f"templates size ({len(config.launch_templates)}) != num_nodes ({config.num_nodes})")
|
||||
if config.cluster_hosts and len(config.cluster_hosts) != config.num_nodes:
|
||||
raise AssertionError("cluster_hosts size mismatch")
|
||||
|
||||
@staticmethod
|
||||
def _validate_routing(config: ExternalDPConfig) -> None:
|
||||
if config.routing.type not in PROXY_SCRIPT_BY_ROUTING_TYPE:
|
||||
raise ValueError(f"Unsupported routing.type: {config.routing.type}")
|
||||
|
||||
groups = config.routing.groups
|
||||
if config.routing.type == ROUTING_GENERIC_DP and not groups.get("worker"):
|
||||
raise ValueError("generic_dp routing requires routing.groups.worker")
|
||||
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL and (
|
||||
not groups.get("prefiller") or not groups.get("decoder")
|
||||
):
|
||||
raise ValueError("disaggregated_prefill routing requires prefiller and decoder groups")
|
||||
|
||||
seen_group_indices: dict[int, str] = {}
|
||||
for group_name, indices in groups.items():
|
||||
for index in indices:
|
||||
if index < 0 or index >= config.num_nodes:
|
||||
raise ValueError(f"routing.groups.{group_name} index out of range: {index}")
|
||||
if index in seen_group_indices:
|
||||
raise ValueError(f"node index {index} appears in both {seen_group_indices[index]} and {group_name}")
|
||||
seen_group_indices[index] = group_name
|
||||
|
||||
if config.routing.proxy_node_index < 0 or config.routing.proxy_node_index >= config.num_nodes:
|
||||
raise ValueError("routing.proxy_node_index out of range")
|
||||
|
||||
@staticmethod
|
||||
def _validate_node_parallel_config(config: ExternalDPConfig) -> None:
|
||||
for node_index, node in enumerate(config.nodes):
|
||||
parallel_sizes = {
|
||||
"dp_size": node.dp_size,
|
||||
"dp_size_local": node.dp_size_local,
|
||||
"tp_size": node.tp_size,
|
||||
"cp_size": node.cp_size,
|
||||
"sp_size": node.sp_size,
|
||||
"pp_size": node.pp_size,
|
||||
}
|
||||
invalid_sizes = {name: value for name, value in parallel_sizes.items() if value < 1}
|
||||
if invalid_sizes:
|
||||
raise ValueError(f"node {node_index} parallel sizes must be >= 1: {invalid_sizes}")
|
||||
if node.dp_rank_start < 0:
|
||||
raise ValueError(f"node {node_index} dp_rank_start must be >= 0")
|
||||
if node.devices_per_node > config.npu_per_node:
|
||||
raise ValueError(
|
||||
f"node {node_index} uses {node.devices_per_node} NPUs, but npu_per_node is {config.npu_per_node}"
|
||||
)
|
||||
if node.dp_rank_start + node.dp_size_local > node.dp_size:
|
||||
raise ValueError(f"node {node_index} dp rank range exceeds dp_size")
|
||||
|
||||
|
||||
class RankResolver:
|
||||
"""Expand node-level configs into concrete vLLM server ranks."""
|
||||
|
||||
def __init__(self, config: ExternalDPConfig):
|
||||
self.config = config
|
||||
|
||||
def resolve(self) -> list[RankInfo]:
|
||||
role_by_node_index = self._role_by_node_index()
|
||||
ranks: list[RankInfo] = []
|
||||
for node_index, node_info in enumerate(self.config.nodes):
|
||||
role = role_by_node_index[node_index]
|
||||
ranks.extend(self._expand_node(node_index, role, node_info))
|
||||
return ranks
|
||||
|
||||
def _role_by_node_index(self) -> dict[int, str]:
|
||||
role_by_index: dict[int, str] = {}
|
||||
for role, node_indices in self.config.routing.groups.items():
|
||||
for index in node_indices:
|
||||
role_by_index[index] = role
|
||||
|
||||
missing = [index for index in range(self.config.num_nodes) if index not in role_by_index]
|
||||
if missing:
|
||||
raise ValueError(f"routing.groups does not assign role for node indices: {missing}")
|
||||
return role_by_index
|
||||
|
||||
@staticmethod
|
||||
def _expand_node(node_index: int, role: str, node_info: NodeInfo) -> list[RankInfo]:
|
||||
ranks: list[RankInfo] = []
|
||||
for local_rank in range(node_info.dp_size_local):
|
||||
dp_rank = node_info.dp_rank_start + local_rank
|
||||
port = node_info.port_start + local_rank
|
||||
device_range = range(
|
||||
local_rank * node_info.devices_per_rank,
|
||||
(local_rank + 1) * node_info.devices_per_rank,
|
||||
)
|
||||
visible_devices = ",".join(str(device) for device in device_range)
|
||||
ranks.append(
|
||||
RankInfo(
|
||||
node_index=node_index,
|
||||
role=role,
|
||||
local_rank=local_rank,
|
||||
dp_rank=dp_rank,
|
||||
host=node_info.ip,
|
||||
port=port,
|
||||
visible_devices=visible_devices,
|
||||
dp_size=node_info.dp_size,
|
||||
dp_size_local=node_info.dp_size_local,
|
||||
tp_size=node_info.tp_size,
|
||||
cp_size=node_info.cp_size,
|
||||
sp_size=node_info.sp_size,
|
||||
pp_size=node_info.pp_size,
|
||||
dp_address=node_info.dp_address,
|
||||
dp_rpc_port=node_info.dp_rpc_port,
|
||||
port_start=node_info.port_start,
|
||||
)
|
||||
)
|
||||
return ranks
|
||||
435
tests/e2e/nightly/multi_node/external_dp/scripts/runtime.py
Normal file
435
tests/e2e/nightly/multi_node/external_dp/scripts/runtime.py
Normal file
@@ -0,0 +1,435 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import time
|
||||
from collections.abc import Iterable
|
||||
from dataclasses import dataclass
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ROUTING_DISAGGREGATED_PREFILL,
|
||||
ROUTING_GENERIC_DP,
|
||||
ExternalDPConfig,
|
||||
NodeTemplate,
|
||||
RankInfo,
|
||||
replace_cluster_placeholders,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
|
||||
format_server_cmd,
|
||||
is_http_ready,
|
||||
start_logged_process,
|
||||
terminate_process_tree,
|
||||
wait_http_ready,
|
||||
wait_http_unready,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import get_net_interface
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
SERVER_READY_TIMEOUT_SECONDS = 3600
|
||||
TEMPLATE_VAR_RE = re.compile(r"\$\{([A-Za-z_][A-Za-z0-9_]*)\}")
|
||||
ENV_VAR_RE = re.compile(r"(?<!\$)\$([A-Za-z_][A-Za-z0-9_]*)")
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class ServerCommand:
|
||||
"""Rendered command, env, and printable command line."""
|
||||
|
||||
cmd: list[str]
|
||||
env: dict[str, str]
|
||||
display_cmd: str
|
||||
|
||||
|
||||
RankProcess = tuple[subprocess.Popen, RankInfo, Path]
|
||||
|
||||
|
||||
class ServerCommandBuilder:
|
||||
"""Render rank templates into vLLM serve commands."""
|
||||
|
||||
def __init__(self, config: ExternalDPConfig):
|
||||
self.config = config
|
||||
|
||||
def build(self, rank: RankInfo, template: NodeTemplate) -> ServerCommand:
|
||||
variables = self._build_variables(rank)
|
||||
rendered_env = self._render_envs(template.envs, rank, variables)
|
||||
rendered_args = [
|
||||
self._render_string(
|
||||
arg,
|
||||
rank=rank,
|
||||
braced_variables=variables,
|
||||
unbraced_variables=rendered_env,
|
||||
allow_missing_unbraced=False,
|
||||
)
|
||||
for arg in template.server_cmd_template
|
||||
]
|
||||
cmd = ["vllm", "serve", self.config.model, *rendered_args]
|
||||
|
||||
env = {key: str(value) for key, value in rendered_env.items()}
|
||||
display_cmd = format_server_cmd(cmd, env)
|
||||
logger.info(
|
||||
"External DP server command node=%s rank=%s: %s",
|
||||
rank.node_index,
|
||||
rank.local_rank,
|
||||
display_cmd,
|
||||
)
|
||||
return ServerCommand(cmd=cmd, env=env, display_cmd=display_cmd)
|
||||
|
||||
def build_all(self, ranks: list[RankInfo]) -> list[ServerCommand]:
|
||||
return [self.build(rank, self.config.launch_templates[rank.node_index]) for rank in ranks]
|
||||
|
||||
def _build_variables(self, rank: RankInfo) -> dict[str, str]:
|
||||
return {
|
||||
"MODEL": self.config.model,
|
||||
"PORT_START": str(rank.port_start),
|
||||
"PORT": str(rank.port),
|
||||
"DP_SIZE": str(rank.dp_size),
|
||||
"DP_SIZE_LOCAL": str(rank.dp_size_local),
|
||||
"DP_RANK_START": str(rank.dp_rank - rank.local_rank),
|
||||
"DP_RANK": str(rank.dp_rank),
|
||||
"LOCAL_RANK": str(rank.local_rank),
|
||||
"TP_SIZE": str(rank.tp_size),
|
||||
"CP_SIZE": str(rank.cp_size),
|
||||
"SP_SIZE": str(rank.sp_size),
|
||||
"PP_SIZE": str(rank.pp_size),
|
||||
"DP_ADDRESS": rank.dp_address,
|
||||
"DP_RPC_PORT": str(rank.dp_rpc_port),
|
||||
"VISIBLE_DEVICES": rank.visible_devices,
|
||||
"NODE_INDEX": str(rank.node_index),
|
||||
"CONFIG_INDEX": str(rank.node_index),
|
||||
}
|
||||
|
||||
def _render_envs(
|
||||
self,
|
||||
envs: dict[str, Any],
|
||||
rank: RankInfo,
|
||||
variables: dict[str, str],
|
||||
) -> dict[str, str]:
|
||||
rendered_envs: dict[str, str] = {}
|
||||
for key, value in envs.items():
|
||||
if isinstance(value, str):
|
||||
value = self._render_string(
|
||||
value,
|
||||
rank=rank,
|
||||
braced_variables=variables,
|
||||
unbraced_variables={**os.environ, **rendered_envs},
|
||||
allow_missing_unbraced=True,
|
||||
)
|
||||
rendered_envs[str(key)] = str(value)
|
||||
return rendered_envs
|
||||
|
||||
def _render_string(
|
||||
self,
|
||||
value: str,
|
||||
*,
|
||||
rank: RankInfo,
|
||||
braced_variables: dict[str, str],
|
||||
unbraced_variables: dict[str, str],
|
||||
allow_missing_unbraced: bool,
|
||||
) -> str:
|
||||
value = replace_cluster_placeholders(
|
||||
value,
|
||||
cluster_ips=self.config.cluster_ips,
|
||||
local_ip=rank.host,
|
||||
current_node_index=rank.node_index,
|
||||
)
|
||||
value = self._render_variables(
|
||||
value,
|
||||
braced_variables,
|
||||
pattern=TEMPLATE_VAR_RE,
|
||||
allow_missing=False,
|
||||
)
|
||||
return self._render_variables(
|
||||
value,
|
||||
unbraced_variables,
|
||||
pattern=ENV_VAR_RE,
|
||||
allow_missing=allow_missing_unbraced,
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _render_variables(
|
||||
value: str,
|
||||
variables: dict[str, str],
|
||||
*,
|
||||
pattern: re.Pattern[str],
|
||||
allow_missing: bool,
|
||||
) -> str:
|
||||
def repl(match: re.Match[str]) -> str:
|
||||
key = match.group(1)
|
||||
if key not in variables:
|
||||
if allow_missing:
|
||||
return ""
|
||||
raise KeyError(f"Unknown external DP template variable: {key}")
|
||||
return variables[key]
|
||||
|
||||
return pattern.sub(repl, value)
|
||||
|
||||
|
||||
class ExternalDPServerManager:
|
||||
"""Start and stop the external DP ranks owned by the current node."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
current_node_index: int,
|
||||
log_root: Path,
|
||||
):
|
||||
self.config = config
|
||||
self.ranks = ranks
|
||||
self.current_node_index = current_node_index
|
||||
self.log_root = log_root
|
||||
self.command_builder = ServerCommandBuilder(config)
|
||||
self.dist_envs = build_dist_envs(
|
||||
config.cluster_ips[current_node_index],
|
||||
config.cluster_ips[0],
|
||||
)
|
||||
self.rank_processes: list[RankProcess] = []
|
||||
|
||||
def start_current_node(self) -> None:
|
||||
local_ranks = [rank for rank in self.ranks if rank.node_index == self.current_node_index]
|
||||
logger.info("Starting %d external DP ranks on node %d", len(local_ranks), self.current_node_index)
|
||||
try:
|
||||
for rank in local_ranks:
|
||||
template = self.config.launch_templates[rank.node_index]
|
||||
template = type(template)(
|
||||
envs={**template.envs, **self.dist_envs},
|
||||
server_cmd_template=template.server_cmd_template,
|
||||
)
|
||||
server_cmd = self.command_builder.build(rank, template)
|
||||
log_file = self._rank_log_file(rank)
|
||||
process = start_logged_process(server_cmd.cmd, server_cmd.env, log_file)
|
||||
self.rank_processes.append((process, rank, log_file))
|
||||
|
||||
wait_ranks_ready(
|
||||
local_ranks,
|
||||
timeout=SERVER_READY_TIMEOUT_SECONDS,
|
||||
rank_processes=self.rank_processes,
|
||||
)
|
||||
except Exception:
|
||||
self.cleanup()
|
||||
raise
|
||||
|
||||
def __enter__(self):
|
||||
self.start_current_node()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
self.cleanup()
|
||||
|
||||
def cleanup(self) -> None:
|
||||
for process, rank, _log_file in reversed(self.rank_processes):
|
||||
logger.info(
|
||||
"Stopping external DP rank node=%d rank=%d pid=%d",
|
||||
rank.node_index,
|
||||
rank.local_rank,
|
||||
process.pid,
|
||||
)
|
||||
terminate_process_tree(process.pid)
|
||||
self.rank_processes.clear()
|
||||
|
||||
def _rank_log_file(self, rank: RankInfo) -> Path:
|
||||
return self.log_root / f"node-{rank.node_index}" / f"rank-{rank.local_rank}.log"
|
||||
|
||||
|
||||
class ExternalDPProxyLauncher:
|
||||
"""Launch the external DP proxy on the configured proxy node."""
|
||||
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
current_node_index: int,
|
||||
log_root: Path,
|
||||
):
|
||||
self.config = config
|
||||
self.ranks = ranks
|
||||
self.current_node_index = current_node_index
|
||||
self.log_root = log_root
|
||||
self.pid: int | None = None
|
||||
|
||||
def start(self) -> None:
|
||||
if self.current_node_index != self.config.routing.proxy_node_index:
|
||||
logger.info("Current node is not proxy node, skip launching external DP proxy")
|
||||
return
|
||||
|
||||
cmd = build_proxy_server_cmd(self.config, self.ranks)
|
||||
log_file = self.log_root / f"node-{self.current_node_index}" / "proxy.log"
|
||||
process = start_logged_process(cmd, {}, log_file)
|
||||
self.pid = process.pid
|
||||
logger.info("External DP proxy launched: %s", proxy_server_health_url(self.config))
|
||||
|
||||
def wait_ready(self, timeout: int = 300) -> None:
|
||||
wait_http_ready(proxy_server_health_url(self.config), timeout=timeout)
|
||||
logger.info("External DP proxy ready: %s", proxy_server_health_url(self.config))
|
||||
|
||||
def __enter__(self):
|
||||
self.start()
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc_value, traceback):
|
||||
self.cleanup()
|
||||
|
||||
def cleanup(self) -> None:
|
||||
if self.pid is None:
|
||||
return
|
||||
logger.info("Stopping external DP proxy pid=%d", self.pid)
|
||||
terminate_process_tree(self.pid)
|
||||
self.pid = None
|
||||
|
||||
|
||||
def build_all_server_commands(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[ServerCommand]:
|
||||
return ServerCommandBuilder(config).build_all(ranks)
|
||||
|
||||
|
||||
def build_dist_envs(cur_ip: str, master_ip: str) -> dict[str, str]:
|
||||
nic_name = get_net_interface(cur_ip)
|
||||
return {
|
||||
"HCCL_IF_IP": cur_ip,
|
||||
"HCCL_SOCKET_IFNAME": nic_name,
|
||||
"GLOO_SOCKET_IFNAME": nic_name,
|
||||
"TP_SOCKET_IFNAME": nic_name,
|
||||
"LOCAL_IP": cur_ip,
|
||||
"NIC_NAME": nic_name,
|
||||
"MASTER_IP": master_ip,
|
||||
}
|
||||
|
||||
|
||||
def build_proxy_server_cmd(config: ExternalDPConfig, ranks: list[RankInfo]) -> list[str]:
|
||||
routing = config.routing
|
||||
cmd = [sys.executable, routing.proxy_script, "--host", routing.proxy_host, "--port", str(routing.proxy_port)]
|
||||
|
||||
if routing.type == ROUTING_GENERIC_DP:
|
||||
worker_ranks = [rank for rank in ranks if rank.role == "worker"]
|
||||
if not worker_ranks:
|
||||
raise ValueError("generic_dp proxy requires worker ranks")
|
||||
cmd.extend(["--dp-hosts", *[rank.host for rank in worker_ranks]])
|
||||
cmd.extend(["--dp-ports", *[str(rank.port) for rank in worker_ranks]])
|
||||
return cmd
|
||||
|
||||
if routing.type == ROUTING_DISAGGREGATED_PREFILL:
|
||||
prefiller_ranks = [rank for rank in ranks if rank.role == "prefiller"]
|
||||
decoder_ranks = [rank for rank in ranks if rank.role == "decoder"]
|
||||
if not prefiller_ranks or not decoder_ranks:
|
||||
raise ValueError("disaggregated_prefill proxy requires prefiller and decoder ranks")
|
||||
cmd.extend(["--prefiller-hosts", *[rank.host for rank in prefiller_ranks]])
|
||||
cmd.extend(["--prefiller-ports", *[str(rank.port) for rank in prefiller_ranks]])
|
||||
cmd.extend(["--decoder-hosts", *[rank.host for rank in decoder_ranks]])
|
||||
cmd.extend(["--decoder-ports", *[str(rank.port) for rank in decoder_ranks]])
|
||||
return cmd
|
||||
|
||||
raise ValueError(f"Unsupported routing.type: {routing.type}")
|
||||
|
||||
|
||||
def proxy_server_health_url(config: ExternalDPConfig) -> str:
|
||||
return f"http://{config.routing.proxy_host}:{config.routing.proxy_port}/healthcheck"
|
||||
|
||||
|
||||
def rank_health_url(rank: RankInfo) -> str:
|
||||
return f"http://{rank.host}:{rank.port}/health"
|
||||
|
||||
|
||||
def master_rank_health_url(ranks: list[RankInfo]) -> str:
|
||||
for rank in ranks:
|
||||
if rank.node_index == 0 and rank.local_rank == 0:
|
||||
return rank_health_url(rank)
|
||||
raise RuntimeError("External DP master rank was not found")
|
||||
|
||||
|
||||
def rank_label(rank: RankInfo) -> str:
|
||||
return f"node={rank.node_index} rank={rank.local_rank} role={rank.role} url={rank_health_url(rank)}"
|
||||
|
||||
|
||||
def format_http_status(label: str, url: str) -> str:
|
||||
status = "ready" if is_http_ready(url, timeout=1.0) else "waiting"
|
||||
return f"{label}={status} url={url}"
|
||||
|
||||
|
||||
def _format_rank_statuses(
|
||||
ranks: list[RankInfo],
|
||||
rank_ready: dict[RankInfo, bool],
|
||||
) -> str:
|
||||
parts = []
|
||||
for rank in ranks:
|
||||
status = "ready" if rank_ready[rank] else "waiting"
|
||||
parts.append(f" {rank_label(rank)} status={status}")
|
||||
return "\n".join(parts)
|
||||
|
||||
|
||||
def _raise_if_rank_process_exited(rank_processes: list[RankProcess] | None) -> None:
|
||||
if not rank_processes:
|
||||
return
|
||||
|
||||
exited = []
|
||||
for process, rank, log_file in rank_processes:
|
||||
returncode = process.poll()
|
||||
if returncode is not None:
|
||||
exited.append(f"{rank_label(rank)} pid={process.pid} returncode={returncode} log={log_file}")
|
||||
|
||||
if exited:
|
||||
raise RuntimeError("External DP rank process exited before ready: " + "; ".join(exited))
|
||||
|
||||
|
||||
def wait_ranks_ready(
|
||||
ranks: Iterable[RankInfo],
|
||||
timeout: int,
|
||||
rank_processes: list[RankProcess] | None = None,
|
||||
) -> None:
|
||||
ranks = list(ranks)
|
||||
rank_ready = {rank: False for rank in ranks}
|
||||
deadline = time.monotonic() + timeout
|
||||
last_log_time = 0.0
|
||||
|
||||
while True:
|
||||
_raise_if_rank_process_exited(rank_processes)
|
||||
|
||||
all_ready = True
|
||||
unhealthy_after_ready = []
|
||||
|
||||
for rank in ranks:
|
||||
is_ready = is_http_ready(rank_health_url(rank), timeout=1.0)
|
||||
if is_ready:
|
||||
if not rank_ready[rank]:
|
||||
logger.info("[READY] External DP rank %s", rank_label(rank))
|
||||
rank_ready[rank] = True
|
||||
continue
|
||||
|
||||
all_ready = False
|
||||
if rank_ready[rank]:
|
||||
unhealthy_after_ready.append(rank)
|
||||
|
||||
if unhealthy_after_ready:
|
||||
failed = "; ".join(rank_label(rank) for rank in unhealthy_after_ready)
|
||||
raise RuntimeError(f"External DP rank became unhealthy after ready: {failed}")
|
||||
|
||||
if all_ready:
|
||||
return
|
||||
|
||||
now = time.monotonic()
|
||||
if now - last_log_time >= 30:
|
||||
logger.info(
|
||||
"Polling external DP ranks: ready=%d/%d\n%s",
|
||||
sum(rank_ready.values()),
|
||||
len(ranks),
|
||||
_format_rank_statuses(ranks, rank_ready),
|
||||
)
|
||||
last_log_time = now
|
||||
|
||||
if now >= deadline:
|
||||
pending = [rank for rank in ranks if not rank_ready[rank]]
|
||||
pending_labels = "; ".join(rank_label(rank) for rank in pending)
|
||||
raise TimeoutError(f"Timed out waiting for external DP ranks ready: {pending_labels}")
|
||||
|
||||
time.sleep(5)
|
||||
|
||||
|
||||
def wait_master_rank_stopped(ranks: list[RankInfo], timeout: int) -> None:
|
||||
url = master_rank_health_url(ranks)
|
||||
wait_http_ready(url, timeout=SERVER_READY_TIMEOUT_SECONDS)
|
||||
logger.info("Hanging until master external DP rank stops: %s", url)
|
||||
wait_http_unready(url, timeout=timeout)
|
||||
@@ -0,0 +1,163 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
from collections.abc import Callable
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ExternalDPConfig,
|
||||
ExternalDPConfigLoader,
|
||||
RankResolver,
|
||||
resolve_current_node_index,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import (
|
||||
ExternalDPProxyLauncher,
|
||||
ExternalDPServerManager,
|
||||
build_all_server_commands,
|
||||
format_http_status,
|
||||
master_rank_health_url,
|
||||
proxy_server_health_url,
|
||||
wait_master_rank_stopped,
|
||||
wait_ranks_ready,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.utils import (
|
||||
collect_logs,
|
||||
write_benchmark_results_json,
|
||||
)
|
||||
from tools.aisbench import run_aisbench_cases
|
||||
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_LOG_ROOT = Path("/tmp/external_dp_logs")
|
||||
|
||||
|
||||
def _install_special_dependencies(config: ExternalDPConfig) -> None:
|
||||
for package, version in config.special_dependencies.items():
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"pip",
|
||||
"install",
|
||||
f"{package}=={version}",
|
||||
]
|
||||
subprocess.call(command)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def _heartbeat(
|
||||
task_name: str,
|
||||
*,
|
||||
interval: int = 30,
|
||||
status_fn: Callable[[], str] | None = None,
|
||||
):
|
||||
start_time = time.monotonic()
|
||||
stop_event = threading.Event()
|
||||
|
||||
def report_progress() -> None:
|
||||
while not stop_event.wait(interval):
|
||||
elapsed = int(time.monotonic() - start_time)
|
||||
status = ""
|
||||
if status_fn is not None:
|
||||
try:
|
||||
status = f" {status_fn()}"
|
||||
except Exception as exc: # pragma: no cover - diagnostic only
|
||||
status = f" status_error={exc!r}"
|
||||
logger.info("%s still running: elapsed=%ds%s", task_name, elapsed, status)
|
||||
|
||||
logger.info("%s started", task_name)
|
||||
thread = threading.Thread(target=report_progress, daemon=True)
|
||||
thread.start()
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
stop_event.set()
|
||||
thread.join(timeout=1)
|
||||
elapsed = int(time.monotonic() - start_time)
|
||||
logger.info("%s finished: elapsed=%ds", task_name, elapsed)
|
||||
|
||||
|
||||
def _format_benchmark_cases(config: ExternalDPConfig) -> str:
|
||||
names = [str(case.get("case_name", "<unnamed>")) for case in config.benchmark_cases]
|
||||
return ", ".join(names) if names else "<none>"
|
||||
|
||||
|
||||
def _archive_rank_logs(log_root: Path, current_node_index: int) -> None:
|
||||
log_prefix = os.environ.get("LOG_PREFIX")
|
||||
if not log_prefix:
|
||||
return
|
||||
node_log_dir = log_root / f"node-{current_node_index}"
|
||||
output_tar = Path(log_prefix) / f"node_{current_node_index}_external_dp_logs.tar.gz"
|
||||
collect_logs(node_log_dir, output_tar)
|
||||
|
||||
|
||||
def test_external_dp() -> None:
|
||||
config = ExternalDPConfigLoader.from_yaml()
|
||||
_install_special_dependencies(config)
|
||||
ranks = RankResolver(config).resolve()
|
||||
current_node_index = resolve_current_node_index(config)
|
||||
log_root = Path(os.environ.get("EXTERNAL_DP_LOG_DIR", str(DEFAULT_LOG_ROOT)))
|
||||
max_wait_seconds = int(os.environ.get("EXTERNAL_DP_MAX_WAIT_SECONDS", "3600"))
|
||||
is_master = current_node_index == 0
|
||||
|
||||
server_manager = ExternalDPServerManager(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
current_node_index=current_node_index,
|
||||
log_root=log_root,
|
||||
)
|
||||
proxy_launcher = ExternalDPProxyLauncher(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
current_node_index=current_node_index,
|
||||
log_root=log_root,
|
||||
)
|
||||
|
||||
try:
|
||||
with server_manager, proxy_launcher:
|
||||
if is_master:
|
||||
wait_ranks_ready(ranks, timeout=max_wait_seconds)
|
||||
proxy_launcher.wait_ready()
|
||||
target = f"http://{config.routing.proxy_host}:{config.routing.proxy_port}"
|
||||
logger.info(
|
||||
"Running AISBench cases: model=%s target=%s cases=[%s]",
|
||||
config.model,
|
||||
target,
|
||||
_format_benchmark_cases(config),
|
||||
)
|
||||
with _heartbeat(
|
||||
"Running AISBench",
|
||||
status_fn=lambda: format_http_status("proxy", proxy_server_health_url(config)),
|
||||
):
|
||||
results = run_aisbench_cases(
|
||||
model=config.model,
|
||||
port=config.routing.proxy_port,
|
||||
aisbench_cases=config.benchmark_cases,
|
||||
host_ip=config.routing.proxy_host,
|
||||
)
|
||||
logger.info("AISBench completed: results=%d", len(results or []))
|
||||
all_commands = build_all_server_commands(config, ranks)
|
||||
write_benchmark_results_json(
|
||||
config=config,
|
||||
ranks=ranks,
|
||||
commands=all_commands,
|
||||
results=results,
|
||||
)
|
||||
wait_ranks_ready(ranks, timeout=30)
|
||||
else:
|
||||
master_url = master_rank_health_url(ranks)
|
||||
with _heartbeat(
|
||||
"Waiting for master external DP rank to stop",
|
||||
status_fn=lambda: format_http_status("master", master_url),
|
||||
):
|
||||
wait_master_rank_stopped(ranks, timeout=max_wait_seconds)
|
||||
finally:
|
||||
_archive_rank_logs(log_root, current_node_index)
|
||||
236
tests/e2e/nightly/multi_node/external_dp/scripts/utils.py
Normal file
236
tests/e2e/nightly/multi_node/external_dp/scripts/utils.py
Normal file
@@ -0,0 +1,236 @@
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import signal
|
||||
import subprocess
|
||||
import tarfile
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
from typing import TYPE_CHECKING, Any
|
||||
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.external_dp_config import (
|
||||
ROUTING_DISAGGREGATED_PREFILL,
|
||||
ExternalDPConfig,
|
||||
RankInfo,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
|
||||
build_task_entry,
|
||||
extract_hardware,
|
||||
filter_environment,
|
||||
get_vllm_version,
|
||||
write_results_json,
|
||||
)
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
if TYPE_CHECKING:
|
||||
from tests.e2e.nightly.multi_node.external_dp.scripts.runtime import ServerCommand
|
||||
|
||||
SENSITIVE_ENV_TOKENS = ("TOKEN", "SECRET", "PASSWORD", "ACCESS_KEY")
|
||||
|
||||
|
||||
def format_server_cmd(cmd: list[str], env: dict[str, str] | None = None) -> str:
|
||||
env_parts: list[str] = []
|
||||
for key, value in sorted((env or {}).items()):
|
||||
display_value = "***" if any(token in key.upper() for token in SENSITIVE_ENV_TOKENS) else str(value)
|
||||
env_parts.append(f"{key}={shlex.quote(display_value)}")
|
||||
return " ".join([*env_parts, shlex.join(cmd)])
|
||||
|
||||
|
||||
def start_logged_process(cmd: list[str], env: dict[str, str], log_file: Path) -> subprocess.Popen:
|
||||
log_file.parent.mkdir(parents=True, exist_ok=True)
|
||||
merged_env = {**os.environ, **env}
|
||||
with log_file.open("ab") as f:
|
||||
f.write(f"Starting command: {format_server_cmd(cmd, env)}\n".encode())
|
||||
f.flush()
|
||||
return subprocess.Popen(
|
||||
cmd,
|
||||
stdout=f,
|
||||
stderr=subprocess.STDOUT,
|
||||
env=merged_env,
|
||||
start_new_session=True,
|
||||
)
|
||||
|
||||
|
||||
def terminate_process_tree(pid: int, timeout: int = 30) -> None:
|
||||
try:
|
||||
import psutil
|
||||
except ModuleNotFoundError:
|
||||
try:
|
||||
os.killpg(pid, signal.SIGTERM)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
os.kill(pid, 0)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
time.sleep(0.2)
|
||||
try:
|
||||
os.killpg(pid, signal.SIGKILL)
|
||||
except ProcessLookupError:
|
||||
return
|
||||
return
|
||||
|
||||
try:
|
||||
parent = psutil.Process(pid)
|
||||
except psutil.NoSuchProcess:
|
||||
return
|
||||
|
||||
children = parent.children(recursive=True)
|
||||
for process in children:
|
||||
process.terminate()
|
||||
parent.terminate()
|
||||
|
||||
gone, alive = psutil.wait_procs([parent, *children], timeout=timeout)
|
||||
del gone
|
||||
for process in alive:
|
||||
process.kill()
|
||||
|
||||
|
||||
def is_http_ready(url: str, timeout: float = 5.0) -> bool:
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=timeout) as response:
|
||||
return 200 <= response.status < 300
|
||||
except (urllib.error.URLError, TimeoutError, OSError):
|
||||
return False
|
||||
|
||||
|
||||
def wait_http_ready(url: str, timeout: int, interval: float = 2.0) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
last_error: Exception | None = None
|
||||
while time.monotonic() < deadline:
|
||||
try:
|
||||
with urllib.request.urlopen(url, timeout=5) as response:
|
||||
if 200 <= response.status < 300:
|
||||
return
|
||||
except (urllib.error.URLError, TimeoutError, OSError) as exc:
|
||||
last_error = exc
|
||||
time.sleep(interval)
|
||||
raise TimeoutError(f"Timed out waiting for HTTP ready: {url}; last_error={last_error}")
|
||||
|
||||
|
||||
def wait_http_unready(url: str, timeout: int, interval: float = 5.0) -> None:
|
||||
deadline = time.monotonic() + timeout
|
||||
while time.monotonic() < deadline:
|
||||
if not is_http_ready(url):
|
||||
return
|
||||
time.sleep(interval)
|
||||
raise TimeoutError(f"Timed out waiting for HTTP unready: {url}")
|
||||
|
||||
|
||||
def collect_logs(src_dir: Path, output_tar: Path) -> None:
|
||||
if not src_dir.exists():
|
||||
return
|
||||
output_tar.parent.mkdir(parents=True, exist_ok=True)
|
||||
with tarfile.open(output_tar, "w:gz") as tar:
|
||||
tar.add(src_dir, arcname=src_dir.name)
|
||||
|
||||
|
||||
def _common_command_envs(commands: list["ServerCommand"]) -> dict[str, str]:
|
||||
if not commands:
|
||||
return {}
|
||||
|
||||
common_keys = set(commands[0].env)
|
||||
for command in commands[1:]:
|
||||
common_keys.intersection_update(command.env)
|
||||
|
||||
common_envs: dict[str, str] = {}
|
||||
for key in sorted(common_keys):
|
||||
values = {command.env[key] for command in commands}
|
||||
if len(values) == 1:
|
||||
common_envs[key] = next(iter(values))
|
||||
return common_envs
|
||||
|
||||
|
||||
def _extract_dtype(config: ExternalDPConfig, commands: list["ServerCommand"]) -> str:
|
||||
has_w8a8 = "w8a8" in config.model.lower()
|
||||
has_quant_ascend = any("--quantization ascend" in command.display_cmd for command in commands)
|
||||
return "w8a8" if has_w8a8 and has_quant_ascend else "bf16"
|
||||
|
||||
|
||||
def _extract_features(commands: list["ServerCommand"]) -> list[str]:
|
||||
if not commands:
|
||||
return []
|
||||
features: list[str] = []
|
||||
command_args = [command.cmd for command in commands]
|
||||
command_displays = [" ".join(shlex.quote(arg) for arg in command.cmd) for command in commands]
|
||||
|
||||
if any("--async-scheduling" in cmd for cmd in command_args):
|
||||
features.append("async_scheduling")
|
||||
if any("--enable-expert-parallel" in cmd for cmd in command_args):
|
||||
features.append("expert_parallel")
|
||||
if any("--speculative-config" in cmd for cmd in command_args):
|
||||
features.append("speculative")
|
||||
if any("cudagraph_mode" in display for display in command_displays):
|
||||
features.append("aclgraph")
|
||||
|
||||
feature_envs = {
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
|
||||
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
|
||||
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
|
||||
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
|
||||
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
|
||||
}
|
||||
for env_key, feature_name in feature_envs.items():
|
||||
values = [str(command.env.get(env_key, "0")) for command in commands]
|
||||
if any(value not in ("0", "", "false", "False") for value in values):
|
||||
features.append(feature_name)
|
||||
return features
|
||||
|
||||
|
||||
def _build_serve_cmd(
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
) -> dict[str, Any]:
|
||||
entries: dict[str, str] = {}
|
||||
for rank, command in zip(ranks, commands):
|
||||
prefix = rank.role
|
||||
if config.routing.type == ROUTING_DISAGGREGATED_PREFILL:
|
||||
prefix = "prefill" if rank.role == "prefiller" else "decode"
|
||||
entries[f"{prefix}-node{rank.node_index}-rank{rank.local_rank}"] = command.display_cmd
|
||||
key = "external_dp_pd" if config.routing.type == ROUTING_DISAGGREGATED_PREFILL else "external_dp"
|
||||
return {key: entries}
|
||||
|
||||
|
||||
def build_benchmark_results(
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
results: list[Any],
|
||||
) -> dict[str, Any]:
|
||||
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
|
||||
tasks = [build_task_entry(key, case, result) for (key, case), result in zip(valid_items, results)]
|
||||
runner = os.environ.get("VLLM_CI_RUNNER", "")
|
||||
common_envs = _common_command_envs(commands)
|
||||
|
||||
return {
|
||||
"model_name": config.model,
|
||||
"hardware": extract_hardware(runner),
|
||||
"dtype": _extract_dtype(config, commands),
|
||||
"feature": _extract_features(commands),
|
||||
"vllm_version": get_vllm_version(),
|
||||
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
|
||||
"tasks": tasks,
|
||||
"serve_cmd": _build_serve_cmd(config, ranks, commands),
|
||||
"environment": filter_environment(common_envs),
|
||||
}
|
||||
|
||||
|
||||
def write_benchmark_results_json(
|
||||
*,
|
||||
config: ExternalDPConfig,
|
||||
ranks: list[RankInfo],
|
||||
commands: list["ServerCommand"],
|
||||
results: list[Any],
|
||||
output_dir: Path | None = None,
|
||||
) -> Path:
|
||||
output = build_benchmark_results(config=config, ranks=ranks, commands=commands, results=results)
|
||||
job_name = os.environ.get("BENCHMARK_JOB_NAME", "") or config.test_name.replace(" ", "-")
|
||||
return write_results_json(output, job_name=job_name, output_dir=output_dir)
|
||||
@@ -0,0 +1,196 @@
|
||||
test_name: "test DeepSeek-R1-W8A8 disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 10
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_DETERMINISTIC: True
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0, L2:0"
|
||||
DYNAMIC_EPLB: true
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 4
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 4
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"enable_prefill_optimizations":true,"enable_weight_nz_layout":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 32
|
||||
--data-parallel-size-local 16
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 1
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 28
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"multistream_overlap_shared_expert":true,"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--headless
|
||||
--data-parallel-size 32
|
||||
--data-parallel-size-local 16
|
||||
--data-parallel-start-rank 16
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 1
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 28
|
||||
--max-model-len 36864
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 32,
|
||||
"tp_size": 1
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"multistream_overlap_shared_expert":true,"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":2048,"algorithm_execution_interval":200}}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,114 @@
|
||||
test_name: "test DeepSeek-R1-W8A8-longseq disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 768
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_DETERMINISTIC: True
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_OP_RETRY_ENABLE: "L0:0, L1:0"
|
||||
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 1
|
||||
--decode-context-parallel-size 8
|
||||
--prefill-context-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 32
|
||||
--max-model-len 32768
|
||||
--max-num-batched-tokens 16384
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.85
|
||||
--enable-chunked-prefill
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-R1-0528-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 1
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 32
|
||||
--max-model-len 32768
|
||||
--max-num-batched-tokens 256
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.85
|
||||
--compilation_config '{"cudagraph_capture_sizes":[4,8,16,32],"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--enable-chunked-prefill
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
--additional-config '{"recompute_scheduler_enable":true}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
num_prompts: 360
|
||||
max_out_len: 4096
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,85 @@
|
||||
test_name: "test DeepSeek-V3.1-BF16 on A3"
|
||||
model: "unsloth/DeepSeek-V3.1-BF16"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 2048
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
OMP_NUM_THREADS: 1
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 1
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve unsloth/DeepSeek-V3.1-BF16
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--no-enable-prefix-caching
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 4096
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
|
||||
--additional_config '{"enable_multistream_moe": true}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve unsloth/DeepSeek-V3.1-BF16
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--no-enable-prefix-caching
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 4096
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--speculative-config '{"num_speculative_tokens": 1, "method":"mtp"}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[2, 4, 8, 16, 32]}'
|
||||
--additional_config '{"enable_multistream_moe": true}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 512
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
@@ -0,0 +1,127 @@
|
||||
test_name: "test DeepSeek-V3.2-W8A8 on A3"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
ASCEND_A3_EBA_ENABLE: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13399
|
||||
--tensor-parallel-size 8
|
||||
--quantization ascend
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 90000
|
||||
--max-num-batched-tokens 4096
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.85
|
||||
--trust-remote-code
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--data-parallel-rpc-port 13399
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--tensor-parallel-size 8
|
||||
--quantization ascend
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 90000
|
||||
--max-num-batched-tokens 4096
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.85
|
||||
--trust-remote-code
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--compilation-config '{"cudagraph_capture_sizes": [8, 16, 24, 32, 40, 48], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
benchmarks:
|
||||
perf_short_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 3000
|
||||
batch_size: 512
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 3000
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_short:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 3000
|
||||
batch_size: 256
|
||||
request_rate: 11.2
|
||||
baseline: 305.2903
|
||||
threshold: 0.97
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 128
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 80000
|
||||
batch_size: 32
|
||||
baseline: 57
|
||||
threshold: 10
|
||||
@@ -0,0 +1,268 @@
|
||||
test_name: "test DeepSeek-V3.2-W8A8-EP disaggregated_prefill"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 10
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: 360
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: 0
|
||||
ASCEND_AGGREGATE_ENABLE: 1
|
||||
ASCEND_TRANSPORT_PRINT: 1
|
||||
ACL_OP_INIT_MODE: 1
|
||||
ASCEND_A3_ENABLE: 1
|
||||
VLLM_MOONCAKE_ABORT_REQUEST_TIMEOUT: 480
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
HCCL_CONNECT_TIMEOUT: 1200
|
||||
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 16
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.90
|
||||
--enforce-eager
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 16
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.90
|
||||
--enforce-eager
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false, "enable_sfa_cp":false,"layer_sharding": ["q_b_proj", "o_proj"]}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 1100
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 42
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 14
|
||||
--gpu-memory-utilization 0.90
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 1100
|
||||
server_cmd: >
|
||||
vllm serve vllm-ascend/DeepSeek-V3.2-W8A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 2, "method":"deepseek_mtp"}'
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-model-len 133000
|
||||
--max-num-batched-tokens 42
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[3,6,9,12,15,18,21,24,27,30,33,36,39,42]}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 14
|
||||
--gpu-memory-utilization 0.90
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"enable_cpu_binding" : false,"recompute_scheduler_enable" : true}'
|
||||
--tokenizer-mode deepseek_v32
|
||||
--reasoning-parser deepseek_v3
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 16
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf_short_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1500
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1024
|
||||
batch_size: 1
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_short:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 128
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
perf_long:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in64000-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 16
|
||||
max_out_len: 1024
|
||||
batch_size: 4
|
||||
request_rate: 1
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 64
|
||||
baseline: 96.88
|
||||
threshold: 10
|
||||
@@ -0,0 +1,102 @@
|
||||
test_name: "multi-node-GLM-5.1-W8A8C8-MTP-A3_64k/128k"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8c8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_BUFFSIZE: "400"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
SERVER_PORT: 8077
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8c8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--data-parallel-rpc-port 12981
|
||||
--tensor-parallel-size 4
|
||||
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
|
||||
--seed 1024
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--enable-auto-tool-choice
|
||||
--max-num-seqs 6
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8c8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--data-parallel-rpc-port 12981
|
||||
--tensor-parallel-size 4
|
||||
--hf-overrides '{"use_index_cache": true, "index_topk_freq": 4}'
|
||||
--seed 1024
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--enable-auto-tool-choice
|
||||
--max-num-seqs 6
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--additional-config '{"enable_dsa_cp": true, "enable_sparse_sfa_c8": false, "enable_sparse_li_c8": true, "enable_balance_scheduling": true, "fuse_muls_add": true, "multistream_overlap_shared_expert": true}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp","enforce_eager":true}'
|
||||
benchmarks:
|
||||
perf_128k_warmup:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 1
|
||||
max_out_len: 1
|
||||
batch_size: 1
|
||||
request_rate: 0
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
perf_128k:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in131072-bs500-prefix90-glm51
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 256
|
||||
max_out_len: 1024
|
||||
batch_size: 64
|
||||
request_rate: 0
|
||||
baseline: 290.8308
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,84 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-A2"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 0
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: 3000
|
||||
VLLM_RPC_TIMEOUT: 600
|
||||
SERVER_PORT: 8078
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 64
|
||||
--max-model-len 38000
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 64
|
||||
--max-model-len 38000
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--no-enable-prefix-caching
|
||||
--additional-config '{"multistream_overlap_shared_expert": true, "ascend_compilation_config": {"fuse_qknorm_rope": false}}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY","cudagraph_capture_sizes": [1,4,8,12,16,20,24,28,32,36,48,60,72]}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 72
|
||||
max_out_len: 1500
|
||||
batch_size: 18
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,98 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-A3"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: 0
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "600"
|
||||
SERVER_PORT: 8080
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-rpc-port 13389
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--enable-chunked-prefill
|
||||
--enable-prefix-caching
|
||||
--async-scheduling
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 72348
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 12
|
||||
max_out_len: 1024
|
||||
batch_size: 3
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,240 @@
|
||||
test_name: "multi-node-GLM-5.1-w8a8-EP"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8"
|
||||
num_nodes: 4
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: 1024
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: 0
|
||||
ASCEND_AGGREGATE_ENABLE: 1
|
||||
ASCEND_TRANSPORT_PRINT: 1
|
||||
ACL_OP_INIT_MODE: 1
|
||||
ASCEND_A3_ENABLE: 1
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
HCCL_CONNECT_TIMEOUT: "1200"
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: 1
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.2.0"
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0, 1]
|
||||
decoder_host_index: [2, 3]
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 131072
|
||||
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--max-num-seqs 64
|
||||
--quantization ascend
|
||||
--gpu-memory-utilization 0.95
|
||||
--enforce-eager
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 131072
|
||||
--additional-config '{"ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--max-num-seqs 64
|
||||
--quantization ascend
|
||||
--gpu-memory-utilization 0.95
|
||||
--enforce-eager
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 10543
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 202752
|
||||
--max-num-batched-tokens 32
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
|
||||
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 8
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.1-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 8
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 4
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 10543
|
||||
--tensor-parallel-size 4
|
||||
--enable-expert-parallel
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
--profiler-config
|
||||
'{"profiler": "torch",
|
||||
"torch_profiler_dir": "./vllm_profile",
|
||||
"torch_profiler_with_stack": false}'
|
||||
--seed 1024
|
||||
--max-model-len 202752
|
||||
--max-num-batched-tokens 32
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 8, 12, 16,20,24,28, 32]}'
|
||||
--additional-config '{"recompute_scheduler_enable": true, "ascend_compilation_config": {"enable_npugraph_ex": true}}'
|
||||
--trust-remote-code
|
||||
--max-num-seqs 8
|
||||
--gpu-memory-utilization 0.92
|
||||
--quantization ascend
|
||||
--enable-auto-tool-choice
|
||||
--tool-call-parser glm47
|
||||
--reasoning-parser glm45
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"use_ascend_direct": true,
|
||||
"prefill": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 8,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 160
|
||||
max_out_len: 1500
|
||||
batch_size: 40
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,91 @@
|
||||
test_name: "multi-node-GLM-5.2-w8a8-A3"
|
||||
model: "Eco-Tech/GLM-5.2-w8a8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 200
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
VLLM_RPC_TIMEOUT: "600"
|
||||
SERVER_PORT: 8080
|
||||
|
||||
special_dependencies:
|
||||
transformers: "5.12.0"
|
||||
# TODO: need to identify why TP and mtp+1 divisibility rules break on dual-node case
|
||||
|
||||
deployment:
|
||||
- envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.2-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-rpc-port 13389
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/GLM-5.2-w8a8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--tensor-parallel-size 16
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-rpc-port 13389
|
||||
--headless
|
||||
--data-parallel-address $MASTER_IP
|
||||
--enable-expert-parallel
|
||||
--seed 1024
|
||||
--max-num-seqs 16
|
||||
--max-model-len 133120
|
||||
--max-num-batched-tokens 4096
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.95
|
||||
--quantization ascend
|
||||
--additional-config '{"multistream_overlap_shared_expert":true}'
|
||||
--compilation-config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"num_speculative_tokens": 3, "method": "deepseek_mtp", "enforce_eager":true}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 72348
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs100_glm
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 12
|
||||
max_out_len: 1024
|
||||
batch_size: 3
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,90 @@
|
||||
test_name: "test Kimi-K2.5-W4A8 A2 dual nodes"
|
||||
model: "Eco-Tech/Kimi-K2.5-W4A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 8
|
||||
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
HCCL_INTRA_PCIE_ENABLE: 1
|
||||
HCCL_INTRA_ROCE_ENABLE: 0
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
HCCL_BUFFSIZE: 512
|
||||
VLLM_ASCEND_ENABLE_MLAPO: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
SERVER_PORT: 8080
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/Kimi-K2.5-W4A8
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--quantization ascend
|
||||
--allowed-local-media-path /
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--seed 42
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 64
|
||||
--max-model-len 51200
|
||||
--max-num-batched-tokens 8192
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
--mm-processor-cache-gb 0
|
||||
--mm-encoder-tp-mode data
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve Eco-Tech/Kimi-K2.5-W4A8
|
||||
--host 0.0.0.0
|
||||
--headless
|
||||
--port $SERVER_PORT
|
||||
--quantization ascend
|
||||
--allowed-local-media-path /
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--seed 42
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 64
|
||||
--max-model-len 51200
|
||||
--max-num-batched-tokens 8192
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_capture_sizes":[16,32,128,160,256], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--speculative-config '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
--mm-processor-cache-gb 0
|
||||
--mm-encoder-tp-mode data
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 320
|
||||
max_out_len: 1500
|
||||
batch_size: 80
|
||||
trust_remote_code: True
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,74 @@
|
||||
test_name: "test Qwen3-235B-A22B multi-dp on A2"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 8
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 128
|
||||
--max-model-len 40960
|
||||
--max-num-batched-tokens 2048
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--headless
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 1
|
||||
--data-parallel-start-rank 1
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 128
|
||||
--max-model-len 40960
|
||||
--max-num-batched-tokens 2048
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 256
|
||||
request_rate: 4.8
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 256
|
||||
baseline: 96
|
||||
threshold: 10
|
||||
@@ -0,0 +1,77 @@
|
||||
test_name: "test Qwen3-235B-A22B multi-dp"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--headless
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 2
|
||||
--data-parallel-address $MASTER_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 512
|
||||
baseline: 95
|
||||
threshold: 3
|
||||
@@ -0,0 +1,93 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8 EPLB"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
DYNAMIC_EPLB: true
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":50,"algorithm_execution_interval":5}}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"eplb_config": {"dynamic_eplb":true,"expert_heat_collection_interval":600,"algorithm_execution_interval":50}}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,100 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8-longseq disaggregated_prefill"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
DYNAMIC_EPLB: true
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 1
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 2
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--seed 1024
|
||||
--enforce-eager
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"dynamic_eplb":true}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--decode-context-parallel-size 2
|
||||
--prefill-context-parallel-size 1
|
||||
--tensor-parallel-size 8
|
||||
--cp-kv-cache-interleave-size 128
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation_config '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 1,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
--additional-config
|
||||
'{"dynamic_eplb":true}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,89 @@
|
||||
test_name: "test Qwen3-235B-A22B-W8A8 disaggregated_prefill"
|
||||
model: "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
HCCL_OP_EXPANSION_MODE: AIV
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
NUMEXPR_MAX_THREADS: 128
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--quantization ascend
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "vllm-ascend/Qwen3-235B-A22B-W8A8"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--quantization ascend
|
||||
--max-num-seqs 16
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
@@ -0,0 +1,118 @@
|
||||
test_name: "test Qwen3-235B-A22B disaggregated_prefill"
|
||||
model: "Qwen/Qwen3-235B-A22B"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
HCCL_BUFFSIZE: 1024
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: 1
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: 2
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
SERVER_PORT: 8080
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--no-enable-prefix-caching
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-235B-A22B"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 4
|
||||
--data-parallel-start-rank 0
|
||||
--data-parallel-address $LOCAL_IP
|
||||
--data-parallel-rpc-port 13389
|
||||
--tensor-parallel-size 4
|
||||
--seed 1024
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--enable-expert-parallel
|
||||
--trust-remote-code
|
||||
--gpu-memory-utilization 0.9
|
||||
--no-enable-prefix-caching
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30100",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 700
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 7680
|
||||
batch_size: 512
|
||||
baseline: 97
|
||||
threshold: 10
|
||||
@@ -0,0 +1,108 @@
|
||||
test_name: "test Qwen3-VL-235B-A22B disaggregated_prefill"
|
||||
model: "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
num_nodes: 2
|
||||
npu_per_node: 16
|
||||
env_common: &env_common
|
||||
VLLM_USE_MODELSCOPE: true
|
||||
HCCL_BUFFSIZE: 1024
|
||||
SERVER_PORT: 8080
|
||||
OMP_PROC_BIND: false
|
||||
OMP_NUM_THREADS: 1
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
TASK_QUEUE_ENABLE: 1
|
||||
PYTORCH_NPU_ALLOC_CONF: expandable_segments:True
|
||||
|
||||
disaggregated_prefill:
|
||||
enabled: true
|
||||
prefiller_host_index: [0]
|
||||
decoder_host_index: [1]
|
||||
|
||||
deployment:
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 2
|
||||
--data-parallel-size-local 2
|
||||
--tensor-parallel-size 8
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_producer",
|
||||
"kv_port": "30000",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
-
|
||||
envs:
|
||||
<<: *env_common
|
||||
server_cmd: >
|
||||
vllm serve "Qwen/Qwen3-VL-235B-A22B-Instruct"
|
||||
--host 0.0.0.0
|
||||
--port $SERVER_PORT
|
||||
--data-parallel-size 4
|
||||
--data-parallel-size-local 4
|
||||
--tensor-parallel-size 4
|
||||
--seed 1024
|
||||
--enable-expert-parallel
|
||||
--max-num-seqs 32
|
||||
--max-model-len 8192
|
||||
--max-num-batched-tokens 8192
|
||||
--trust-remote-code
|
||||
--no-enable-prefix-caching
|
||||
--gpu-memory-utilization 0.9
|
||||
--compilation-config '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
--kv-transfer-config
|
||||
'{"kv_connector": "MooncakeConnectorV1",
|
||||
"kv_role": "kv_consumer",
|
||||
"kv_port": "30200",
|
||||
"kv_connector_extra_config": {
|
||||
"prefill": {
|
||||
"dp_size": 2,
|
||||
"tp_size": 8
|
||||
},
|
||||
"decode": {
|
||||
"dp_size": 4,
|
||||
"tp_size": 4
|
||||
}
|
||||
}
|
||||
}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/textvqa-perf-1080p
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
num_prompts: 2800
|
||||
max_out_len: 1500
|
||||
batch_size: 64
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/textvqa-lite
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: textvqa/textvqa_gen_base64
|
||||
max_out_len: 7680
|
||||
batch_size: 64
|
||||
baseline: 85
|
||||
threshold: 5
|
||||
@@ -0,0 +1,331 @@
|
||||
import logging
|
||||
import os
|
||||
import subprocess
|
||||
from dataclasses import dataclass
|
||||
from typing import Any
|
||||
|
||||
import regex as re
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
get_available_port,
|
||||
get_net_interface,
|
||||
load_yaml_mapping,
|
||||
resolve_cluster_ips,
|
||||
resolve_current_node_index,
|
||||
setup_logger,
|
||||
)
|
||||
|
||||
setup_logger()
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
|
||||
DEFAULT_SERVER_PORT = 8080
|
||||
|
||||
|
||||
@dataclass(frozen=True)
|
||||
class NodeInfo:
|
||||
index: int
|
||||
ip: str
|
||||
server_cmd: str
|
||||
envs: dict[str, Any] | None = None
|
||||
headless: bool = False
|
||||
|
||||
def __post_init__(self):
|
||||
if not self.ip:
|
||||
raise ValueError("NodeInfo.ip must not be empty")
|
||||
|
||||
def __str__(self) -> str:
|
||||
return f"NodeInfo(\n index={self.index},\n ip={self.ip},\n headless={self.headless},\n)"
|
||||
|
||||
|
||||
class DisaggregatedPrefillCfg:
|
||||
def __init__(self, raw_cfg: dict, num_nodes: int):
|
||||
self.prefiller_indices: list[int] = raw_cfg.get("prefiller_host_index", [])
|
||||
self.decoder_indices: list[int] = raw_cfg.get("decoder_host_index", [])
|
||||
|
||||
if not self.decoder_indices:
|
||||
raise RuntimeError("decoder_host_index must be provided")
|
||||
|
||||
self._validate(num_nodes)
|
||||
|
||||
self.decode_start_index = self.decoder_indices[0]
|
||||
self.num_prefillers = len(self.prefiller_indices)
|
||||
self.num_decoders = len(self.decoder_indices)
|
||||
|
||||
def _validate(self, num_nodes: int):
|
||||
overlap = set(self.prefiller_indices) & set(self.decoder_indices)
|
||||
if overlap:
|
||||
raise AssertionError(f"Prefiller and decoder overlap: {overlap}")
|
||||
|
||||
all_indices = self.prefiller_indices + self.decoder_indices
|
||||
if any(i >= num_nodes for i in all_indices):
|
||||
raise ValueError("Disaggregated prefill index out of range")
|
||||
|
||||
def is_prefiller(self, index: int) -> bool:
|
||||
return index in self.prefiller_indices
|
||||
|
||||
def is_decoder(self, index: int) -> bool:
|
||||
return index in self.decoder_indices
|
||||
|
||||
def master_ip_for_node(self, index: int, nodes: list[NodeInfo]) -> str:
|
||||
if self.is_prefiller(index):
|
||||
return nodes[0].ip
|
||||
return nodes[self.decode_start_index].ip
|
||||
|
||||
|
||||
class DistEnvBuilder:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
cur_node: NodeInfo,
|
||||
master_ip: str,
|
||||
):
|
||||
self.cur_ip = cur_node.ip
|
||||
self.nic_name = get_net_interface(self.cur_ip)
|
||||
self.master_ip = master_ip
|
||||
|
||||
self.base_envs = dict(cur_node.envs or {})
|
||||
|
||||
def build(self) -> dict:
|
||||
envs = dict(self.base_envs)
|
||||
|
||||
envs.update(
|
||||
{
|
||||
"HCCL_IF_IP": self.cur_ip,
|
||||
"HCCL_SOCKET_IFNAME": self.nic_name,
|
||||
"GLOO_SOCKET_IFNAME": self.nic_name,
|
||||
"TP_SOCKET_IFNAME": self.nic_name,
|
||||
"LOCAL_IP": self.cur_ip,
|
||||
"NIC_NAME": self.nic_name,
|
||||
"MASTER_IP": self.master_ip,
|
||||
}
|
||||
)
|
||||
|
||||
return {k: str(v) for k, v in envs.items()}
|
||||
|
||||
|
||||
class ProxyLauncher:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
nodes: list[NodeInfo],
|
||||
envs: dict,
|
||||
proxy_port: int,
|
||||
cur_index: int,
|
||||
disagg_cfg: DisaggregatedPrefillCfg | None = None,
|
||||
):
|
||||
self.nodes = nodes
|
||||
self.cfg = disagg_cfg
|
||||
self.server_port = envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
self.proxy_port = proxy_port
|
||||
self.proxy_script = envs.get(
|
||||
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
|
||||
"examples/disaggregated_prefill_v1/load_balance_proxy_server_example.py",
|
||||
)
|
||||
self.envs = envs
|
||||
self.is_master = cur_index == 0
|
||||
self.cur_ip = nodes[cur_index].ip
|
||||
self.process: subprocess.Popen[bytes] | None = None
|
||||
|
||||
def __enter__(self):
|
||||
if not self.is_master or self.cfg is None:
|
||||
logger.info("Not launching proxy on non-master node")
|
||||
return self
|
||||
prefiller_ips = [self.nodes[i].ip for i in self.cfg.prefiller_indices if not self.nodes[i].headless]
|
||||
decoder_ips = [self.nodes[i].ip for i in self.cfg.decoder_indices if not self.nodes[i].headless]
|
||||
|
||||
cmd = [
|
||||
"python",
|
||||
self.proxy_script,
|
||||
"--host",
|
||||
self.cur_ip,
|
||||
"--port",
|
||||
str(self.proxy_port),
|
||||
"--prefiller-hosts",
|
||||
*prefiller_ips,
|
||||
"--prefiller-ports",
|
||||
*[str(self.server_port)] * len(prefiller_ips),
|
||||
"--decoder-hosts",
|
||||
*decoder_ips,
|
||||
"--decoder-ports",
|
||||
*[str(self.server_port)] * len(decoder_ips),
|
||||
]
|
||||
|
||||
logger.info("Launching proxy: %s", " ".join(cmd))
|
||||
self.process = subprocess.Popen(cmd, env={**os.environ, **self.envs})
|
||||
return self
|
||||
|
||||
def __exit__(self, exc_type, exc, tb):
|
||||
if not self.process:
|
||||
return
|
||||
logger.info("Stopping proxy server...")
|
||||
self.process.terminate()
|
||||
try:
|
||||
self.process.wait(timeout=5)
|
||||
except subprocess.TimeoutExpired:
|
||||
self.process.kill()
|
||||
|
||||
|
||||
class MultiNodeConfig:
|
||||
def __init__(
|
||||
self,
|
||||
*,
|
||||
model: str,
|
||||
test_name: str,
|
||||
nodes: list[NodeInfo],
|
||||
npu_per_node: int,
|
||||
disaggregated_prefill: dict | None,
|
||||
benchmark_cases: list[dict],
|
||||
special_dependencies: dict,
|
||||
):
|
||||
self.model = model
|
||||
self.test_name = test_name
|
||||
self.nodes = nodes
|
||||
self.npu_per_node = npu_per_node
|
||||
self.benchmark_cases = benchmark_cases
|
||||
|
||||
self.cur_index = self._resolve_cur_index()
|
||||
self.cur_node = self.nodes[self.cur_index]
|
||||
self.special_dependencies = special_dependencies
|
||||
|
||||
self.disagg_cfg = DisaggregatedPrefillCfg(disaggregated_prefill, len(nodes)) if disaggregated_prefill else None
|
||||
|
||||
master_ip = (
|
||||
self.disagg_cfg.master_ip_for_node(self.cur_index, self.nodes) if self.disagg_cfg else self.nodes[0].ip
|
||||
)
|
||||
self.proxy_port = get_available_port()
|
||||
|
||||
self.envs = DistEnvBuilder(
|
||||
cur_node=self.cur_node,
|
||||
master_ip=master_ip,
|
||||
).build()
|
||||
logger.info("Node %d envs: %s", self.cur_index, self.envs)
|
||||
|
||||
self.server_cmd = self._expand_env(self.cur_node.server_cmd)
|
||||
|
||||
def _resolve_cur_index(self) -> int:
|
||||
return resolve_current_node_index([node.ip for node in self.nodes])
|
||||
|
||||
def _expand_env(self, cmd: str) -> str:
|
||||
pattern = re.compile(r"\$(\w+)|\$\{(\w+)\}")
|
||||
|
||||
def repl(m):
|
||||
key = m.group(1) or m.group(2)
|
||||
return self.envs.get(key, m.group(0))
|
||||
|
||||
return pattern.sub(repl, cmd)
|
||||
|
||||
@property
|
||||
def world_size(self) -> int:
|
||||
return len(self.nodes) * self.npu_per_node
|
||||
|
||||
@property
|
||||
def is_master(self) -> bool:
|
||||
return self.cur_index == 0
|
||||
|
||||
@property
|
||||
def server_port(self) -> int:
|
||||
return self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
|
||||
@property
|
||||
def master_ip(self) -> str:
|
||||
return self.nodes[0].ip
|
||||
|
||||
@property
|
||||
def benchmark_endpoint(self) -> tuple[str, int]:
|
||||
"""
|
||||
Endpoint used by benchmark clients.
|
||||
"""
|
||||
master_ip = self.nodes[0].ip
|
||||
server_port = self.envs.get("SERVER_PORT", DEFAULT_SERVER_PORT)
|
||||
if self.disagg_cfg:
|
||||
return master_ip, self.proxy_port
|
||||
return master_ip, server_port
|
||||
|
||||
|
||||
class MultiNodeConfigLoader:
|
||||
"""Load MultiNodeConfig from yaml file."""
|
||||
|
||||
DEFAULT_CONFIG_NAME = "DeepSeek-V3.yaml"
|
||||
|
||||
@classmethod
|
||||
def from_yaml(cls, yaml_path: str | None = None) -> MultiNodeConfig:
|
||||
config = cls._load_yaml(yaml_path)
|
||||
cls._validate_root(config)
|
||||
|
||||
nodes = cls._parse_nodes(config)
|
||||
benchmarks = cls._parse_benchmarks(config)
|
||||
|
||||
return MultiNodeConfig(
|
||||
model=config["model"],
|
||||
test_name=config.get("test_name", "untitled_test"),
|
||||
nodes=nodes,
|
||||
npu_per_node=config.get("npu_per_node", 16),
|
||||
disaggregated_prefill=config.get("disaggregated_prefill"),
|
||||
special_dependencies=config.get("special_dependencies", {}),
|
||||
benchmark_cases=list(benchmarks.values()),
|
||||
)
|
||||
|
||||
@classmethod
|
||||
def _load_yaml(cls, yaml_path: str | None) -> dict:
|
||||
return load_yaml_mapping(
|
||||
yaml_path,
|
||||
default_name=cls.DEFAULT_CONFIG_NAME,
|
||||
default_base_path=DEFAULT_CONFIG_BASE_PATH,
|
||||
description="config",
|
||||
)
|
||||
|
||||
@staticmethod
|
||||
def _validate_root(cfg: dict):
|
||||
required = ["model", "deployment", "num_nodes", "npu_per_node", "benchmarks"]
|
||||
missing = [k for k in required if k not in cfg]
|
||||
if missing:
|
||||
raise KeyError(f"Missing required config fields: {missing}")
|
||||
|
||||
@classmethod
|
||||
def _parse_nodes(cls, cfg: dict) -> list[NodeInfo]:
|
||||
num_nodes = cfg["num_nodes"]
|
||||
deployments = cfg["deployment"]
|
||||
|
||||
if len(deployments) != num_nodes:
|
||||
raise AssertionError(f"deployment size ({len(deployments)}) != num_nodes ({num_nodes})")
|
||||
|
||||
for idx, deploy in enumerate(deployments):
|
||||
if deploy.get("envs") is None:
|
||||
raise KeyError(f"deployment[{idx}].envs is required for multi-node configs")
|
||||
|
||||
cluster_ips = cls._resolve_cluster_ips(cfg, num_nodes)
|
||||
|
||||
nodes: list[NodeInfo] = []
|
||||
for idx, deploy in enumerate(deployments):
|
||||
cmd = deploy.get("server_cmd", "")
|
||||
envs = deploy["envs"]
|
||||
nodes.append(
|
||||
NodeInfo(
|
||||
index=idx,
|
||||
ip=cluster_ips[idx],
|
||||
server_cmd=cmd,
|
||||
envs=envs,
|
||||
headless="--headless" in cmd,
|
||||
)
|
||||
)
|
||||
return nodes
|
||||
|
||||
@staticmethod
|
||||
def _parse_benchmarks(cfg: dict) -> dict:
|
||||
benchmarks = cfg.get("benchmarks") or {}
|
||||
for name, case in benchmarks.items():
|
||||
case["case_name"] = name
|
||||
return benchmarks
|
||||
|
||||
@staticmethod
|
||||
def _resolve_cluster_ips(cfg: dict, num_nodes: int) -> list[str]:
|
||||
return resolve_cluster_ips(
|
||||
cfg,
|
||||
num_nodes,
|
||||
cluster_hosts_log_message=(
|
||||
"Using cluster_hosts from config. This typically indicates that your current environment is a "
|
||||
"non-Kubernetes environment."
|
||||
),
|
||||
dns_log_message="Resolving cluster IPs via DNS...",
|
||||
)
|
||||
@@ -0,0 +1,207 @@
|
||||
import json
|
||||
import logging
|
||||
import os
|
||||
import shlex
|
||||
import subprocess
|
||||
import sys
|
||||
from typing import Any
|
||||
|
||||
import pytest
|
||||
import vllm
|
||||
|
||||
from tests.e2e.conftest import RemoteOpenAIServer
|
||||
from tests.e2e.nightly.multi_node.internal_dp.scripts.multi_node_config import (
|
||||
MultiNodeConfig,
|
||||
MultiNodeConfigLoader,
|
||||
ProxyLauncher,
|
||||
)
|
||||
from tests.e2e.nightly.multi_node.scripts.benchmark_results import (
|
||||
build_task_entry,
|
||||
extract_hardware,
|
||||
filter_environment,
|
||||
write_results_json,
|
||||
)
|
||||
from tools.aisbench import run_aisbench_cases
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
_FEATURE_ENVS: dict[str, str] = {
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM": "flashcomm",
|
||||
"VLLM_ASCEND_ENABLE_FLASHCOMM1": "flashcomm1",
|
||||
"VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE": "topk_optimize",
|
||||
"VLLM_ASCEND_ENABLE_MATMUL_ALLREDUCE": "matmul_allreduce",
|
||||
"VLLM_ASCEND_ENABLE_MLAPO": "mlapo",
|
||||
"VLLM_ASCEND_ENABLE_FUSED_MC2": "fused_mc2",
|
||||
}
|
||||
|
||||
|
||||
def _extract_dtype(config: MultiNodeConfig) -> str:
|
||||
"""Determine weight dtype: w8a8 if model name contains 'w8a8' and any node uses --quantization ascend."""
|
||||
has_w8a8 = "w8a8" in config.model.lower()
|
||||
has_quant_ascend = any("--quantization ascend" in node.server_cmd for node in config.nodes)
|
||||
return "w8a8" if (has_w8a8 and has_quant_ascend) else "bf16"
|
||||
|
||||
|
||||
def _cmd_to_list(server_cmd: list[str] | str) -> list[str]:
|
||||
"""Normalize server_cmd to a list of argument strings."""
|
||||
if isinstance(server_cmd, str):
|
||||
try:
|
||||
return shlex.split(server_cmd)
|
||||
except ValueError:
|
||||
return server_cmd.split()
|
||||
return list(server_cmd)
|
||||
|
||||
|
||||
def _extract_server_cmd_value(cmd_list: list[str], flag: str) -> str | None:
|
||||
"""Return the value following `flag` in a command list, or None."""
|
||||
try:
|
||||
idx = cmd_list.index(flag)
|
||||
return cmd_list[idx + 1]
|
||||
except (ValueError, IndexError):
|
||||
return None
|
||||
|
||||
|
||||
def _parse_json_flag(cmd_list: list[str], flag: str) -> dict[str, Any]:
|
||||
"""Extract and JSON-parse the value following `flag` in a command list."""
|
||||
val = _extract_server_cmd_value(cmd_list, flag)
|
||||
if not val:
|
||||
return {}
|
||||
try:
|
||||
return json.loads(val)
|
||||
except (json.JSONDecodeError, ValueError):
|
||||
return {}
|
||||
|
||||
|
||||
def _extract_features(server_cmd: list[str] | str, envs: dict[str, Any]) -> list[str]:
|
||||
"""Extract enabled feature names from server_cmd and environment variables."""
|
||||
cmd_list = _cmd_to_list(server_cmd)
|
||||
features: list[str] = []
|
||||
|
||||
# Features from --additional-config JSON
|
||||
additional = _parse_json_flag(cmd_list, "--additional-config")
|
||||
if additional.get("enable_weight_nz_layout"):
|
||||
features.append("weight_nz_layout")
|
||||
wp = additional.get("weight_prefetch_config") or {}
|
||||
if isinstance(wp, dict) and wp.get("enabled"):
|
||||
features.append("weight_prefetch")
|
||||
tc = additional.get("torchair_graph_config") or {}
|
||||
if isinstance(tc, dict) and tc.get("enabled"):
|
||||
features.append("torchair_graph")
|
||||
asc = additional.get("ascend_scheduler_config") or {}
|
||||
if isinstance(asc, dict) and asc.get("enabled"):
|
||||
features.append("ascend_scheduler")
|
||||
|
||||
# Features from --compilation-config JSON
|
||||
compilation = _parse_json_flag(cmd_list, "--compilation-config")
|
||||
if compilation.get("cudagraph_mode"):
|
||||
features.append("aclgraph")
|
||||
|
||||
# Features from --speculative-config JSON
|
||||
speculative = _parse_json_flag(cmd_list, "--speculative-config")
|
||||
if speculative:
|
||||
features.append(speculative.get("method", "speculative"))
|
||||
|
||||
# Features from direct flags
|
||||
if "--enable-expert-parallel" in cmd_list:
|
||||
features.append("expert_parallel")
|
||||
|
||||
# Features from environment variables
|
||||
for env_key, feature_name in _FEATURE_ENVS.items():
|
||||
val = str(envs.get(env_key, "0"))
|
||||
if val not in ("0", "", "false", "False"):
|
||||
features.append(feature_name)
|
||||
if int(envs.get("VLLM_ASCEND_FLASHCOMM2_PARALLEL_SIZE", 0)) > 0:
|
||||
features.append("flashcomm2")
|
||||
|
||||
return features
|
||||
|
||||
|
||||
def _build_serve_cmd(config: MultiNodeConfig) -> dict[str, Any]:
|
||||
"""Build serve_cmd dict: pd format for disaggregated, dp format for multi-node."""
|
||||
if config.disagg_cfg:
|
||||
pd: dict[str, str] = {}
|
||||
for node in config.nodes:
|
||||
idx = node.index
|
||||
if config.disagg_cfg.is_prefiller(idx):
|
||||
n = config.disagg_cfg.prefiller_indices.index(idx)
|
||||
pd[f"prefill-{n}"] = node.server_cmd
|
||||
elif config.disagg_cfg.is_decoder(idx):
|
||||
n = config.disagg_cfg.decoder_indices.index(idx)
|
||||
pd[f"decode-{n}"] = node.server_cmd
|
||||
return {"pd": pd}
|
||||
return {"dp": {f"node{node.index}": node.server_cmd for node in config.nodes}}
|
||||
|
||||
|
||||
def _save_benchmark_results_json(config: MultiNodeConfig, results: list[Any]) -> None:
|
||||
"""Serialize acc & perf benchmark results to a JSON file under benchmark_results/."""
|
||||
runner = os.environ.get("VLLM_CI_RUNNER", "")
|
||||
|
||||
# Filter out None benchmark cases; results align with the non-None ones in order
|
||||
valid_items = [(case["case_name"], case) for case in config.benchmark_cases]
|
||||
|
||||
tasks = [build_task_entry(key, case_cfg, result) for (key, case_cfg), result in zip(valid_items, results)]
|
||||
|
||||
output: dict[str, Any] = {
|
||||
"model_name": config.model,
|
||||
"hardware": extract_hardware(runner),
|
||||
"dtype": _extract_dtype(config),
|
||||
"feature": _extract_features(config.nodes[0].server_cmd, config.envs),
|
||||
"vllm_version": vllm.__version__,
|
||||
"vllm_ascend_version": os.environ.get("VLLM_ASCEND_REF", ""),
|
||||
"tasks": tasks,
|
||||
"serve_cmd": _build_serve_cmd(config),
|
||||
"environment": filter_environment(config.envs),
|
||||
}
|
||||
|
||||
job_name = os.environ.get("BENCHMARK_JOB_NAME", "")
|
||||
write_results_json(output, job_name=job_name)
|
||||
|
||||
|
||||
@pytest.mark.asyncio
|
||||
async def test_multi_node() -> None:
|
||||
config = MultiNodeConfigLoader.from_yaml()
|
||||
if config.special_dependencies:
|
||||
for k, v in config.special_dependencies.items():
|
||||
command = [
|
||||
sys.executable,
|
||||
"-m",
|
||||
"pip",
|
||||
"install",
|
||||
f"{k}=={v}",
|
||||
]
|
||||
subprocess.call(command)
|
||||
|
||||
with (
|
||||
ProxyLauncher(
|
||||
nodes=config.nodes,
|
||||
disagg_cfg=config.disagg_cfg,
|
||||
envs=config.envs,
|
||||
proxy_port=config.proxy_port,
|
||||
cur_index=config.cur_index,
|
||||
) as proxy,
|
||||
RemoteOpenAIServer(
|
||||
model=config.model,
|
||||
vllm_serve_args=config.server_cmd,
|
||||
server_port=config.server_port,
|
||||
server_host=config.master_ip,
|
||||
env_dict=config.envs,
|
||||
auto_port=False,
|
||||
proxy_port=proxy.proxy_port,
|
||||
disaggregated_prefill=config.disagg_cfg,
|
||||
nodes_info=config.nodes,
|
||||
max_wait_seconds=2800,
|
||||
) as server,
|
||||
):
|
||||
host, port = config.benchmark_endpoint
|
||||
|
||||
if config.is_master:
|
||||
results = run_aisbench_cases(
|
||||
model=config.model,
|
||||
port=port,
|
||||
aisbench_cases=config.benchmark_cases,
|
||||
host_ip=host,
|
||||
)
|
||||
_save_benchmark_results_json(config, results)
|
||||
else:
|
||||
# We should keep listening on the master node's server url determining when to exit.
|
||||
server.hang_until_terminated(f"http://{host}:{config.server_port}/health")
|
||||
28
tests/e2e/nightly/multi_node/internal_dp/scripts/utils.py
Normal file
28
tests/e2e/nightly/multi_node/internal_dp/scripts/utils.py
Normal file
@@ -0,0 +1,28 @@
|
||||
import os
|
||||
|
||||
from tests.e2e.nightly.multi_node.scripts.utils import (
|
||||
get_all_ipv4,
|
||||
get_available_port,
|
||||
get_cluster_ips,
|
||||
get_net_interface,
|
||||
setup_logger,
|
||||
temp_env,
|
||||
)
|
||||
|
||||
DISAGGEGATED_PREFILL_PORT = 5333
|
||||
DEFAULT_CONFIG_BASE_PATH = "tests/e2e/nightly/multi_node/internal_dp/config/"
|
||||
CONFIG_BASE_PATH = os.getenv("CONFIG_BASE_PATH") or DEFAULT_CONFIG_BASE_PATH
|
||||
DEFAULT_SERVER_PORT = 8080
|
||||
|
||||
__all__ = [
|
||||
"CONFIG_BASE_PATH",
|
||||
"DEFAULT_CONFIG_BASE_PATH",
|
||||
"DEFAULT_SERVER_PORT",
|
||||
"DISAGGEGATED_PREFILL_PORT",
|
||||
"get_all_ipv4",
|
||||
"get_available_port",
|
||||
"get_cluster_ips",
|
||||
"get_net_interface",
|
||||
"setup_logger",
|
||||
"temp_env",
|
||||
]
|
||||
1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
1
tests/e2e/nightly/multi_node/scripts/__init__.py
Normal file
@@ -0,0 +1 @@
|
||||
|
||||
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
128
tests/e2e/nightly/multi_node/scripts/benchmark_results.py
Normal file
@@ -0,0 +1,128 @@
|
||||
import json
|
||||
import logging
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
PORT_ENV_KEYS = {"SERVER_PORT", "ENCODE_PORT", "PD_PORT", "PROXY_PORT"}
|
||||
INFRA_ENV_KEYS = {
|
||||
"HCCL_IF_IP",
|
||||
"HCCL_SOCKET_IFNAME",
|
||||
"GLOO_SOCKET_IFNAME",
|
||||
"TP_SOCKET_IFNAME",
|
||||
"LOCAL_IP",
|
||||
"NIC_NAME",
|
||||
"MASTER_IP",
|
||||
"DISAGGREGATED_PREFILL_PROXY_SCRIPT",
|
||||
}
|
||||
PERF_METRIC_RENAME: dict[str, str] = {
|
||||
"Benchmark Duration": "Benchmark_Duration(BD)",
|
||||
"Prefill Token Throughput": "Prefill_Token_Throughput(PTT)",
|
||||
"Input Token Throughput": "Input_Token_Throughput(ITT)",
|
||||
"Output Token Throughput": "Output_Token_Throughput(OTT)",
|
||||
"Total Token Throughput": "Total_Token_Throughput(TTT)",
|
||||
}
|
||||
|
||||
|
||||
def extract_hardware(runner: str) -> str:
|
||||
runner_lower = runner.lower()
|
||||
for label in ("a3", "a2"):
|
||||
if label in runner_lower:
|
||||
return label.upper()
|
||||
return runner
|
||||
|
||||
|
||||
def get_vllm_version() -> str:
|
||||
try:
|
||||
import vllm
|
||||
|
||||
return vllm.__version__
|
||||
except Exception:
|
||||
return ""
|
||||
|
||||
|
||||
def task_passed(case_config: dict[str, Any], result: Any) -> bool:
|
||||
if result == "":
|
||||
return False
|
||||
case_type = case_config.get("case_type")
|
||||
baseline = case_config.get("baseline")
|
||||
threshold = case_config.get("threshold")
|
||||
if baseline is None or threshold is None:
|
||||
return True
|
||||
if case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
return abs(float(result) - float(baseline)) <= float(threshold)
|
||||
if case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
throughput_str = result_json.get("Output Token Throughput", {}).get("total", "")
|
||||
try:
|
||||
throughput_val = float(throughput_str.replace("token/s", "").strip())
|
||||
return throughput_val >= float(threshold) * float(baseline)
|
||||
except (ValueError, AttributeError):
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
def build_task_entry(case_key: str, case_config: dict[str, Any], result: Any) -> dict[str, Any]:
|
||||
dataset_path = case_config.get("dataset_path", "")
|
||||
dataset_conf = case_config.get("dataset_conf", "")
|
||||
if dataset_path:
|
||||
task_name = dataset_path.split("/", 1)[-1]
|
||||
elif dataset_conf:
|
||||
task_name = dataset_conf.split("/")[0]
|
||||
else:
|
||||
task_name = case_key
|
||||
|
||||
case_type = case_config.get("case_type", "unknown")
|
||||
metrics: dict[str, float] = {}
|
||||
if result == "":
|
||||
pass
|
||||
elif case_type == "accuracy" and isinstance(result, (int, float)):
|
||||
metrics["accuracy"] = round(float(result), 4)
|
||||
elif case_type == "performance" and isinstance(result, list) and len(result) == 2:
|
||||
_, result_json = result
|
||||
for metric_name, metric_data in result_json.items():
|
||||
if not isinstance(metric_data, dict):
|
||||
continue
|
||||
total_str = metric_data.get("total", "")
|
||||
try:
|
||||
value = float(total_str.replace("token/s", "").replace("ms", "").replace("s", "").strip())
|
||||
metrics[PERF_METRIC_RENAME.get(metric_name, metric_name)] = round(value, 4)
|
||||
except (ValueError, AttributeError):
|
||||
pass
|
||||
|
||||
test_input_keys = ("num_prompts", "max_out_len", "batch_size", "request_rate")
|
||||
test_input = {key: case_config[key] for key in test_input_keys if key in case_config}
|
||||
|
||||
target: dict[str, Any] = {}
|
||||
if case_config.get("baseline") is not None:
|
||||
target["baseline"] = case_config["baseline"]
|
||||
if case_config.get("threshold") is not None:
|
||||
target["threshold"] = case_config["threshold"]
|
||||
|
||||
entry: dict[str, Any] = {"name": task_name, "metrics": metrics, "test_input": test_input}
|
||||
if target:
|
||||
entry["target"] = target
|
||||
entry["pass_fail"] = "pass" if task_passed(case_config, result) else "fail"
|
||||
return entry
|
||||
|
||||
|
||||
def filter_environment(envs: dict[str, Any]) -> dict[str, Any]:
|
||||
exclude = PORT_ENV_KEYS | INFRA_ENV_KEYS
|
||||
return {key: value for key, value in envs.items() if key not in exclude}
|
||||
|
||||
|
||||
def write_results_json(
|
||||
output: dict[str, Any],
|
||||
*,
|
||||
job_name: str,
|
||||
output_dir: Path | None = None,
|
||||
) -> Path:
|
||||
if output_dir is None:
|
||||
output_dir = Path("/root/.cache/benchmark_results") / job_name
|
||||
output_dir.mkdir(parents=True, exist_ok=True)
|
||||
output_path = output_dir / f"{job_name}.json"
|
||||
output_path.write_text(json.dumps(output, indent=2, ensure_ascii=False), encoding="utf-8")
|
||||
logger.info("Benchmark results saved to PVC at %s", output_path)
|
||||
print(f"Benchmark results saved to PVC at {output_path}")
|
||||
return output_path
|
||||
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
174
tests/e2e/nightly/multi_node/scripts/lws.yaml.jinja2
Normal file
@@ -0,0 +1,174 @@
|
||||
apiVersion: leaderworkerset.x-k8s.io/v1
|
||||
kind: LeaderWorkerSet
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
replicas: {{ replicas | default(1) }}
|
||||
leaderWorkerTemplate:
|
||||
size: {{ size | default(2) }}
|
||||
restartPolicy: None
|
||||
leaderTemplate:
|
||||
metadata:
|
||||
labels:
|
||||
role: leader
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-leader
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: VLLM_ASCEND_VERSION
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
ports:
|
||||
- containerPort: 8080
|
||||
# readinessProbe:
|
||||
# tcpSocket:
|
||||
# port: 8080
|
||||
# initialDelaySeconds: 15
|
||||
# periodSeconds: 10
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
workerTemplate:
|
||||
spec:
|
||||
tolerations:
|
||||
- key: "dedicated"
|
||||
operator: "Equal"
|
||||
value: "night"
|
||||
effect: "NoSchedule"
|
||||
containers:
|
||||
- name: vllm-worker
|
||||
imagePullPolicy: Always
|
||||
image: {{ image | default("swr.cn-southwest-2.myhuaweicloud.com/base_image/ascend-ci/vllm-ascend:nightly-a3") }}
|
||||
env:
|
||||
- name: CONFIG_YAML_PATH
|
||||
value: {{ config_file_path | default("DeepSeek-V3.yaml") }}
|
||||
- name: CONFIG_BASE_PATH
|
||||
value: "{{ config_base_path | default("") }}"
|
||||
- name: LOG_PREFIX
|
||||
value: {{ log_prefix | default("/root/.cache/ascend-logs") }}
|
||||
- name: WORKSPACE
|
||||
value: "/vllm-workspace"
|
||||
- name: FAIL_TAG
|
||||
value: {{ fail_tag | default("FAIL_TAG") }}
|
||||
- name: IS_PR_TEST
|
||||
value: "{{ is_pr_test | default("false") }}"
|
||||
- name: VLLM_ASCEND_REF
|
||||
value: {{ vllm_ascend_ref | default("main") }}
|
||||
- name: VLLM_ASCEND_REMOTE_URL
|
||||
value: {{ vllm_ascend_remote_url | default("https://github.com/vllm-project/vllm-ascend.git") }}
|
||||
- name: BENCHMARK_JOB_NAME
|
||||
value: {{ benchmark_job_name | default("") }}
|
||||
- name: VLLM_CI_RUNNER
|
||||
value: {{ runner | default("linux-aarch64-a3-0") }}
|
||||
- name: AOP_MULTI_ENABLED
|
||||
value: "{{ aop_multi_enabled }}"
|
||||
- name: GOOD_TABLE
|
||||
value: "{{ good_table | default("/root/.cache/vllm-ascend/nightly/good_table.csv") }}"
|
||||
command:
|
||||
- sh
|
||||
- -c
|
||||
- |
|
||||
bash /root/.cache/tests/run.sh
|
||||
resources:
|
||||
limits:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
memory: 512Gi
|
||||
ephemeral-storage: 100Gi
|
||||
requests:
|
||||
huawei.com/ascend-1980: {{ npu_per_node | default("16") }}
|
||||
ephemeral-storage: 100Gi
|
||||
cpu: 125
|
||||
volumeMounts:
|
||||
- mountPath: /root/.cache
|
||||
name: shared-volume
|
||||
- mountPath: /usr/local/Ascend/driver/tools
|
||||
name: driver-tools
|
||||
- mountPath: /dev/shm
|
||||
name: dshm
|
||||
volumes:
|
||||
- name: dshm
|
||||
emptyDir:
|
||||
medium: Memory
|
||||
sizeLimit: 512Gi
|
||||
- name: shared-volume
|
||||
persistentVolumeClaim:
|
||||
claimName: {{ pvc_name | default("nv-action-vllm-benchmarks-v2") }}
|
||||
- name: driver-tools
|
||||
hostPath:
|
||||
path: /usr/local/Ascend/driver/tools
|
||||
---
|
||||
apiVersion: v1
|
||||
kind: Service
|
||||
metadata:
|
||||
name: {{ lws_name | default("vllm") }}-leader
|
||||
namespace: vllm-project
|
||||
spec:
|
||||
ports:
|
||||
- name: http
|
||||
port: 8080
|
||||
protocol: TCP
|
||||
targetPort: 8080
|
||||
selector:
|
||||
leaderworkerset.sigs.k8s.io/name: {{ lws_name | default("vllm") }}
|
||||
role: leader
|
||||
type: ClusterIP
|
||||
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
466
tests/e2e/nightly/multi_node/scripts/run.sh
Normal file
@@ -0,0 +1,466 @@
|
||||
#!/bin/bash
|
||||
set -euo pipefail
|
||||
|
||||
# Color definitions
|
||||
GREEN="\033[0;32m"
|
||||
BLUE="\033[0;34m"
|
||||
YELLOW="\033[0;33m"
|
||||
RED="\033[0;31m"
|
||||
NC="\033[0m" # No Color
|
||||
|
||||
INTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/internal_dp/scripts/test_multi_node.py"
|
||||
EXTERNAL_DP_TEST_PATH="tests/e2e/nightly/multi_node/external_dp/scripts/test_external_dp.py"
|
||||
|
||||
if [ -z "${MULTI_NODE_TEST_PATH:-}" ]; then
|
||||
if [[ "${CONFIG_BASE_PATH:-}" == *"external_dp/config"* || "${CONFIG_YAML_PATH:-}" == *"external_dp/config"* ]]; then
|
||||
MULTI_NODE_TEST_PATH="$EXTERNAL_DP_TEST_PATH"
|
||||
else
|
||||
MULTI_NODE_TEST_PATH="$INTERNAL_DP_TEST_PATH"
|
||||
fi
|
||||
fi
|
||||
|
||||
# Configuration
|
||||
export LD_LIBRARY_PATH=/usr/local/Ascend/ascend-toolkit/latest/python/site-packages:$LD_LIBRARY_PATH
|
||||
export LD_LIBRARY_PATH=/usr/local/lib:$LD_LIBRARY_PATH
|
||||
# cann and atb environment setup
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
source /usr/local/Ascend/cann-9.1.0/share/info/ascendnpu-ir/bin/set_env.sh
|
||||
|
||||
set +eu
|
||||
source /usr/local/Ascend/nnal/atb/set_env.sh
|
||||
set -eu
|
||||
|
||||
# Home path for aisbench
|
||||
export BENCHMARK_HOME=${WORKSPACE}/vllm-ascend/benchmark
|
||||
|
||||
# Logging configurations
|
||||
export VLLM_LOGGING_LEVEL="INFO"
|
||||
# Reduce glog verbosity for mooncake
|
||||
export GLOG_minloglevel=1
|
||||
# Set transformers to offline mode to avoid downloading models during tests
|
||||
export HF_HUB_OFFLINE="1"
|
||||
# Default is 600s
|
||||
export VLLM_ENGINE_READY_TIMEOUT_S=1800
|
||||
|
||||
# Function to print section headers
|
||||
print_section() {
|
||||
echo -e "\n${BLUE}=== $1 ===${NC}"
|
||||
}
|
||||
|
||||
print_failure() {
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
# Function to print success messages
|
||||
print_success() {
|
||||
echo -e "${GREEN}✓ $1${NC}"
|
||||
}
|
||||
|
||||
# Function to print error messages and exit
|
||||
print_error() {
|
||||
echo -e "${RED}✗ ERROR: $1${NC}"
|
||||
exit 1
|
||||
}
|
||||
|
||||
show_vllm_info() {
|
||||
cd "$WORKSPACE"
|
||||
echo "Installed vLLM-related Python packages:"
|
||||
pip list | grep vllm || echo "No vllm packages found."
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM Git information"
|
||||
echo "============================"
|
||||
cd vllm
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm"
|
||||
fi
|
||||
cd ..
|
||||
|
||||
echo ""
|
||||
echo "============================"
|
||||
echo "vLLM-Ascend Git information"
|
||||
echo "============================"
|
||||
cd vllm-ascend
|
||||
if [ -d .git ]; then
|
||||
echo "Branch: $(git rev-parse --abbrev-ref HEAD)"
|
||||
echo "Commit hash: $(git rev-parse HEAD)"
|
||||
echo "Author: $(git log -1 --pretty=format:'%an <%ae>')"
|
||||
echo "Date: $(git log -1 --pretty=format:'%ad' --date=iso)"
|
||||
echo "Message: $(git log -1 --pretty=format:'%s')"
|
||||
echo "Tags: $(git tag --points-at HEAD || echo 'None')"
|
||||
echo "Remote: $(git remote -v | head -n1)"
|
||||
echo ""
|
||||
else
|
||||
echo "No .git directory found in vllm-ascend"
|
||||
fi
|
||||
cd ..
|
||||
}
|
||||
|
||||
check_npu_info() {
|
||||
echo "====> Check NPU info"
|
||||
npu-smi info
|
||||
cat "/usr/local/Ascend/ascend-toolkit/latest/$(uname -i)-linux/ascend_toolkit_install.info"
|
||||
}
|
||||
|
||||
check_and_config() {
|
||||
echo "====> Configure mirrors and git proxy"
|
||||
git config --global url."https://ghfast.top/https://github.com/".insteadOf "https://github.com/"
|
||||
pip config set global.index-url https://mirrors.tuna.tsinghua.edu.cn/pypi/web/simple
|
||||
export PIP_EXTRA_INDEX_URL="https://mirrors.huaweicloud.com/ascend/repos/pypi"
|
||||
}
|
||||
|
||||
install_extra_components() {
|
||||
echo "====> Installing extra components for DeepSeek-v3.2-exp-bf16"
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/CANN-custom_ops-sfa-linux.aarch64.run; then
|
||||
echo "Failed to download CANN-custom_ops-sfa-linux.aarch64.run"
|
||||
return 1
|
||||
fi
|
||||
chmod +x ./CANN-custom_ops-sfa-linux.aarch64.run
|
||||
./CANN-custom_ops-sfa-linux.aarch64.run --quiet
|
||||
|
||||
if ! wget -q https://vllm-ascend.obs.cn-north-4.myhuaweicloud.com/vllm-ascend/a3/custom_ops-1.0-cp311-cp311-linux_aarch64.whl; then
|
||||
echo "Failed to download custom_ops wheel"
|
||||
return 1
|
||||
fi
|
||||
pip install custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
|
||||
export ASCEND_CUSTOM_OPP_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize${ASCEND_CUSTOM_OPP_PATH:+:${ASCEND_CUSTOM_OPP_PATH}}"
|
||||
export LD_LIBRARY_PATH="/usr/local/Ascend/ascend-toolkit/latest/opp/vendors/customize/op_api/lib/${LD_LIBRARY_PATH:+:${LD_LIBRARY_PATH}}"
|
||||
source /usr/local/Ascend/ascend-toolkit/set_env.sh
|
||||
|
||||
rm -f CANN-custom_ops-sfa-linux.aarch64.run \
|
||||
custom_ops-1.0-cp311-cp311-linux_aarch64.whl
|
||||
echo "====> Extra components installation completed"
|
||||
}
|
||||
|
||||
checkout_src() {
|
||||
echo "====> Checkout source code"
|
||||
mkdir -p "$WORKSPACE"
|
||||
cd "$WORKSPACE"
|
||||
pip uninstall -y vllm-ascend || true
|
||||
cp -r "$WORKSPACE/vllm-ascend/benchmark" /tmp/aisbench-backup || true
|
||||
rm -rf "$WORKSPACE/vllm-ascend"
|
||||
|
||||
if [ ! -d "$WORKSPACE/vllm-ascend" ]; then
|
||||
echo "Cloning vllm-ascend from $VLLM_ASCEND_REMOTE_URL"
|
||||
git clone --depth 1 --recurse-submodules "$VLLM_ASCEND_REMOTE_URL" "$WORKSPACE/vllm-ascend"
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
PR_REF=$(git ls-remote origin 'refs/pull/*/head' | grep "^${VLLM_ASCEND_REF}" | awk '{print $2}' | head -1)
|
||||
if [ -n "$PR_REF" ]; then
|
||||
git fetch --depth 1 origin "$PR_REF"
|
||||
git checkout FETCH_HEAD
|
||||
else
|
||||
git fetch origin '+refs/pull/*/head:refs/remotes/pull/*' 2>/dev/null || true
|
||||
git checkout "$VLLM_ASCEND_REF"
|
||||
fi
|
||||
git submodule update --init --recursive
|
||||
fi
|
||||
}
|
||||
|
||||
install_vllm_ascend() {
|
||||
echo "====> Install vllm-ascend"
|
||||
pip install -r "$WORKSPACE/vllm-ascend/requirements-dev.txt"
|
||||
pip install -e "$WORKSPACE/vllm-ascend"
|
||||
}
|
||||
|
||||
install_aisbench() {
|
||||
echo "====> Install AISBench benchmark"
|
||||
|
||||
BENCH_DIR="$WORKSPACE/vllm-ascend/benchmark"
|
||||
|
||||
cp -r /tmp/aisbench-backup "$BENCH_DIR"
|
||||
|
||||
cd "$BENCH_DIR"
|
||||
pip install -e . \
|
||||
-r requirements/api.txt \
|
||||
-r requirements/extra.txt
|
||||
|
||||
python3 -m pip cache purge || echo "WARNING: pip cache purge failed, but proceeding..."
|
||||
|
||||
}
|
||||
|
||||
show_triton_ascend_info() {
|
||||
echo "====> Check triton ascend info"
|
||||
clang -v
|
||||
which bishengir-compile
|
||||
pip show triton-ascend
|
||||
}
|
||||
|
||||
kill_npu_processes() {
|
||||
pgrep python3 | xargs -r kill -9
|
||||
pgrep VLLM | xargs -r kill -9
|
||||
|
||||
sleep 4
|
||||
}
|
||||
|
||||
run_tests_with_log() {
|
||||
set +e
|
||||
kill_npu_processes
|
||||
mkdir -p "${LOG_PREFIX}"
|
||||
echo "====> Run pytest entry: $MULTI_NODE_TEST_PATH"
|
||||
local log_file="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-?}_pytest.log"
|
||||
pytest -sv --show-capture=no "$MULTI_NODE_TEST_PATH" 2>&1 | tee "$log_file"
|
||||
ret=$?
|
||||
echo "pytest exit code: ret=${ret}"
|
||||
set -e
|
||||
if [ "${LWS_WORKER_INDEX:-}" = "0" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
print_success "All tests passed!"
|
||||
touch "${LOG_PREFIX}/aop_done" 2>/dev/null
|
||||
else
|
||||
echo "Leader: waiting 10s for worker logs..."
|
||||
sleep 10
|
||||
if [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
set +e; aop_pipeline; set -e
|
||||
fi
|
||||
local done_file="${LOG_PREFIX}/aop_done"
|
||||
touch "$done_file"
|
||||
echo "Leader: notifying workers (${done_file})"
|
||||
echo -e "${RED}${FAIL_TAG:-test_failed} ✗ ERROR: Some tests failed${NC}"
|
||||
exit 1
|
||||
fi
|
||||
elif [ "${AOP_MULTI_ENABLED:-}" = "true" ]; then
|
||||
if [ $ret -eq 0 ]; then
|
||||
echo "Worker: test passed, waiting for leader..."
|
||||
local wait_timeout=30
|
||||
while [ $wait_timeout -gt 0 ] && [ ! -f "${LOG_PREFIX}/aop_done" ]; do
|
||||
sleep 1
|
||||
wait_timeout=$((wait_timeout - 1))
|
||||
done
|
||||
fi
|
||||
if [ ! -f "${LOG_PREFIX}/aop_done" ]; then
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
local release="${LOG_PREFIX}/aop_done"
|
||||
mkdir -p "$coord"
|
||||
touch "${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: signalling ready at ${coord}/worker_ready_${LWS_WORKER_INDEX}"
|
||||
echo "Worker: joining bisect as worker node (index ${LWS_WORKER_INDEX})..."
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--coord-dir "${coord}" \
|
||||
--release-file "${release}"
|
||||
while [ ! -f "$release" ]; do sleep 5; done
|
||||
echo "Worker: release signal received, exiting"
|
||||
exit 1
|
||||
else
|
||||
echo "Worker: leader finished successfully, exiting"
|
||||
fi
|
||||
fi
|
||||
}
|
||||
|
||||
# Run AOP decision pipeline on failure: classify → check age → bisect-or-exit
|
||||
# Same logic as _e2e_nightly_multi_node.yaml AOP hooks.
|
||||
aop_pipeline() {
|
||||
local rules="$WORKSPACE/vllm-ascend/tests/e2e/nightly/scripts/rules-env.txt"
|
||||
local table="${GOOD_TABLE:-}"
|
||||
# Strip branch prefix from BENCHMARK_JOB_NAME (e.g. "main-Qwen3.5-27B-w8a8-A2" → "Qwen3.5-27B-w8a8-A2")
|
||||
local case_name="${BENCHMARK_JOB_NAME#*-}"
|
||||
if [ -z "$case_name" ] || [ "$case_name" = "$BENCHMARK_JOB_NAME" ]; then
|
||||
case_name="${CONFIG_YAML_PATH%.yaml}"
|
||||
fi
|
||||
|
||||
echo "============================================"
|
||||
echo " AOP Pipeline (Pod) - START"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Case name : ${case_name}"
|
||||
echo " Rules file : ${rules}"
|
||||
echo " Table file : ${table}"
|
||||
echo " Log prefix : ${LOG_PREFIX}"
|
||||
echo " BENCHMARK_JOB_NAME: ${BENCHMARK_JOB_NAME:-}"
|
||||
echo "============================================"
|
||||
|
||||
# ---- Step 1: Classify ----
|
||||
echo ""
|
||||
echo "--- [1/3] Classify: scanning pod logs for env patterns ---"
|
||||
echo " Rules content:"
|
||||
if [ -f "$rules" ]; then
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | sed 's/^/ > /'
|
||||
else
|
||||
echo " (rules file not found)"
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo " Pod logs found:"
|
||||
local found_any=0
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
echo " - ${f} ($(wc -l < "$f") lines)"
|
||||
found_any=1
|
||||
fi
|
||||
done
|
||||
[ "$found_any" -eq 0 ] && echo " (no pod logs found)"
|
||||
|
||||
local env_count=0
|
||||
if [ -f "$rules" ]; then
|
||||
for f in "${LOG_PREFIX}/node_"*"_pytest.log"; do
|
||||
if [ -f "$f" ]; then
|
||||
local n
|
||||
n=$(grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -ciEf - "$f" 2>/dev/null || echo 0)
|
||||
n=${n%%[!0-9]*}
|
||||
echo " Scan ${f}: ${n} matches"
|
||||
env_count=$((env_count + n))
|
||||
if [ "$n" -gt 0 ]; then
|
||||
echo " Matched lines:"
|
||||
grep -vE '^[[:space:]]*(#|$)' "$rules" | grep -niEf - "$f" | head -5 | sed 's/^/ /'
|
||||
fi
|
||||
fi
|
||||
done
|
||||
fi
|
||||
echo " Classify result: env_count=${env_count}"
|
||||
|
||||
if [ "$found_any" -eq 0 ]; then
|
||||
echo " Decision: no pod logs → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (no logs) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
if [ "$env_count" -gt 0 ]; then
|
||||
echo " Decision: env_failure → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (env skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 2: Check age ----
|
||||
echo ""
|
||||
echo "--- [2/3] Check commit age ---"
|
||||
echo " Looking up: ${case_name}"
|
||||
local skip_age=0
|
||||
if [ ! -f "$table" ]; then
|
||||
echo " Table file not found: ${table}"
|
||||
echo " Decision: no table → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Only consider success rows
|
||||
local success_rows
|
||||
success_rows=$(grep "^${case_name}," "$table" | grep -F ',success,' || true)
|
||||
if [ -z "$success_rows" ]; then
|
||||
echo " No success row found for '${case_name}'"
|
||||
echo " Decision: no success entry → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# Pick most recent success row
|
||||
local best_date=""
|
||||
while IFS= read -r row; do
|
||||
local d
|
||||
d=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
|
||||
[ -z "$d" ] && continue
|
||||
if [ -z "$best_date" ] || [[ "$d" > "$best_date" ]]; then
|
||||
best_date="$d"
|
||||
fi
|
||||
done <<< "$success_rows"
|
||||
|
||||
if [ -z "$best_date" ]; then
|
||||
echo " No valid date in success rows"
|
||||
echo " Decision: no date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
echo " Matched row: $(grep -m1 "$best_date" <<< "$success_rows")"
|
||||
local last_ts now_ts age_days
|
||||
last_ts=$(date -d "$best_date" +%s 2>/dev/null || echo 0)
|
||||
if [ "$last_ts" = "0" ] || [ -z "$last_ts" ]; then
|
||||
echo " Date parse failed: ${best_date}"
|
||||
echo " Decision: invalid date → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
now_ts=$(date +%s)
|
||||
age_days=$(( (now_ts - last_ts) / 86400 ))
|
||||
echo " Last success: ${best_date} (${age_days} days ago, threshold: 3 days)"
|
||||
|
||||
if [ "$age_days" -gt 3 ]; then
|
||||
echo " Decision: old commit (> 3 days) → SKIP"
|
||||
echo "=== AOP Pipeline (Pod) - END (age skip) ==="
|
||||
return 1
|
||||
fi
|
||||
|
||||
# ---- Step 3: Bisect ----
|
||||
echo ""
|
||||
echo "--- [3/3] Run bisect ---"
|
||||
echo " Scene : multi_node"
|
||||
echo " Config : ${CONFIG_YAML_PATH}"
|
||||
echo " Bad commit : HEAD"
|
||||
echo " Name : ${case_name}"
|
||||
local coord="${COORD_DIR:-/root/.cache/nightly_bisect/coord}"
|
||||
echo " Coord dir : ${coord}"
|
||||
|
||||
# Wait for all workers to signal ready
|
||||
echo " Waiting for workers..."
|
||||
for i in $(seq 1 30); do
|
||||
local ready_count=0
|
||||
for f in "${coord}"/worker_ready_*; do
|
||||
[ -e "$f" ] && ready_count=$((ready_count + 1))
|
||||
done
|
||||
echo " [${i}/30] ready workers: ${ready_count}"
|
||||
if [ "$ready_count" -ge 1 ]; then break; fi
|
||||
sleep 2
|
||||
done
|
||||
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
local bisect_rc=0
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect \
|
||||
--scene multi_node \
|
||||
--config-yaml "${CONFIG_YAML_PATH}" \
|
||||
--bad-commit HEAD \
|
||||
--good-table "${table}" \
|
||||
--name "${case_name}" \
|
||||
--coord-dir "${coord}" || bisect_rc=$?
|
||||
echo " bisect completed (exit code: ${bisect_rc})"
|
||||
echo "=== AOP Pipeline (Pod) - END ==="
|
||||
return 1
|
||||
}
|
||||
|
||||
clear_logs() {
|
||||
print_section "Clearing logs from previous runs"
|
||||
rm -fr "$HOME/ascend/log" || true
|
||||
}
|
||||
|
||||
backup_ascend_logs() {
|
||||
if [ -n "${LOG_PREFIX:-}" ]; then
|
||||
local dest="${LOG_PREFIX}/node_${LWS_WORKER_INDEX:-unknown}_plogs"
|
||||
mkdir -p "$dest"
|
||||
cp -r /root/ascend/log/. "$dest/" 2>/dev/null || true
|
||||
echo "Ascend logs backed up to $dest"
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
trap backup_ascend_logs EXIT
|
||||
check_npu_info
|
||||
clear_logs
|
||||
check_and_config
|
||||
if [[ "$IS_PR_TEST" == "true" ]]; then
|
||||
checkout_src
|
||||
install_vllm_ascend
|
||||
install_aisbench
|
||||
fi
|
||||
show_vllm_info
|
||||
show_triton_ascend_info
|
||||
if [[ "$CONFIG_YAML_PATH" == *"DeepSeek-V3_2-Exp-bf16.yaml" ]]; then
|
||||
install_extra_components
|
||||
fi
|
||||
cd "$WORKSPACE/vllm-ascend"
|
||||
run_tests_with_log
|
||||
}
|
||||
|
||||
main "$@"
|
||||
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
183
tests/e2e/nightly/multi_node/scripts/utils.py
Normal file
@@ -0,0 +1,183 @@
|
||||
import logging
|
||||
import os
|
||||
import socket
|
||||
import time
|
||||
from contextlib import contextmanager
|
||||
from pathlib import Path
|
||||
from typing import Any
|
||||
|
||||
import yaml
|
||||
|
||||
logger = logging.getLogger(__name__)
|
||||
|
||||
|
||||
@contextmanager
|
||||
def temp_env(env_dict: dict[str, Any]):
|
||||
old_env = {}
|
||||
for key, value in env_dict.items():
|
||||
old_env[key] = os.environ.get(key)
|
||||
os.environ[key] = str(value)
|
||||
try:
|
||||
yield
|
||||
finally:
|
||||
for key, value in old_env.items():
|
||||
if value is None:
|
||||
os.environ.pop(key, None)
|
||||
else:
|
||||
os.environ[key] = value
|
||||
|
||||
|
||||
def setup_logger() -> None:
|
||||
logging.basicConfig(
|
||||
level=logging.INFO,
|
||||
format="[%(asctime)s] [%(levelname)s] %(message)s",
|
||||
datefmt="%Y-%m-%d %H:%M:%S",
|
||||
)
|
||||
|
||||
|
||||
def load_yaml_mapping(
|
||||
yaml_path: str | None,
|
||||
*,
|
||||
default_name: str,
|
||||
default_base_path: str,
|
||||
description: str,
|
||||
) -> dict[str, Any]:
|
||||
if not yaml_path:
|
||||
yaml_path = os.getenv("CONFIG_YAML_PATH", default_name)
|
||||
|
||||
path = Path(yaml_path)
|
||||
if not path.is_absolute() and not path.exists():
|
||||
base_path = os.getenv("CONFIG_BASE_PATH") or default_base_path
|
||||
path = Path(base_path) / yaml_path
|
||||
|
||||
logger.info("Loading %s yaml: %s", description, path)
|
||||
with path.open(encoding="utf-8") as f:
|
||||
data = yaml.safe_load(f)
|
||||
if not isinstance(data, dict):
|
||||
raise TypeError(f"{description} must be a mapping: {path}")
|
||||
return data
|
||||
|
||||
|
||||
def dns_resolver(retries: int = 240, base_delay: float = 0.5):
|
||||
def resolve(dns: str) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
return socket.gethostbyname(dns)
|
||||
except socket.gaierror:
|
||||
if attempt == retries - 1:
|
||||
raise
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError(f"Unable to resolve DNS: {dns}")
|
||||
|
||||
return resolve
|
||||
|
||||
|
||||
def get_cluster_dns_list(world_size: int) -> list[str]:
|
||||
if world_size < 1:
|
||||
raise ValueError(f"world_size must be >= 1, got {world_size}")
|
||||
|
||||
leader_dns = os.getenv("LWS_LEADER_ADDRESS")
|
||||
if not leader_dns:
|
||||
raise RuntimeError("environment variable LWS_LEADER_ADDRESS is not set")
|
||||
|
||||
parts = leader_dns.split(".")
|
||||
if len(parts) < 3:
|
||||
raise ValueError(f"invalid leader DNS format: {leader_dns}")
|
||||
|
||||
leader_name, group_name, namespace = parts[0], parts[1], parts[2]
|
||||
worker_dns_list = [f"{leader_name}-{idx}.{group_name}.{namespace}" for idx in range(1, world_size)]
|
||||
return [leader_dns, *worker_dns_list]
|
||||
|
||||
|
||||
def get_cluster_ips(world_size: int = 2) -> list[str]:
|
||||
resolver = dns_resolver()
|
||||
return [resolver(dns) for dns in get_cluster_dns_list(world_size)]
|
||||
|
||||
|
||||
def resolve_cluster_ips(
|
||||
raw_config: dict[str, Any],
|
||||
num_nodes: int,
|
||||
explicit_cluster_ips: list[str] | None = None,
|
||||
*,
|
||||
cluster_hosts_log_message: str | None = None,
|
||||
dns_log_message: str = "Resolving cluster IPs via DNS...",
|
||||
) -> list[str]:
|
||||
if explicit_cluster_ips is not None:
|
||||
if len(explicit_cluster_ips) != num_nodes:
|
||||
raise AssertionError("cluster_ips size mismatch")
|
||||
return explicit_cluster_ips
|
||||
|
||||
cluster_hosts = raw_config.get("cluster_hosts")
|
||||
if cluster_hosts:
|
||||
if cluster_hosts_log_message:
|
||||
logger.info(cluster_hosts_log_message)
|
||||
if len(cluster_hosts) != num_nodes:
|
||||
raise AssertionError("cluster_hosts size mismatch")
|
||||
return list(cluster_hosts)
|
||||
|
||||
logger.info(dns_log_message)
|
||||
return get_cluster_ips(num_nodes)
|
||||
|
||||
|
||||
def get_available_port(start_port: int = 6000, end_port: int = 7000) -> int:
|
||||
for port in range(start_port, end_port):
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as s:
|
||||
try:
|
||||
s.bind(("", port))
|
||||
return port
|
||||
except OSError:
|
||||
continue
|
||||
raise RuntimeError("No available port found")
|
||||
|
||||
|
||||
def get_cur_ip(retries: int = 20, base_delay: float = 0.5) -> str:
|
||||
delay = base_delay
|
||||
for attempt in range(retries):
|
||||
try:
|
||||
with socket.socket(socket.AF_INET, socket.SOCK_DGRAM) as s:
|
||||
s.connect(("8.8.8.8", 80))
|
||||
return s.getsockname()[0]
|
||||
except Exception:
|
||||
try:
|
||||
return socket.gethostbyname(socket.gethostname())
|
||||
except Exception:
|
||||
if attempt == retries - 1:
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
time.sleep(delay)
|
||||
delay = min(delay * 1.5, 5)
|
||||
raise RuntimeError("Failed to determine local IP address")
|
||||
|
||||
|
||||
def get_net_interface(ip: str | None = None) -> str:
|
||||
import psutil
|
||||
|
||||
if ip is None:
|
||||
ip = get_cur_ip()
|
||||
|
||||
for iface, addrs in psutil.net_if_addrs().items():
|
||||
for addr in addrs:
|
||||
if addr.family == socket.AF_INET and addr.address == ip:
|
||||
return iface
|
||||
raise RuntimeError(f"No network interface found for IP {ip}")
|
||||
|
||||
|
||||
def get_all_ipv4() -> list[str]:
|
||||
ipv4s = {"127.0.0.1"}
|
||||
hostname = socket.gethostname()
|
||||
for info in socket.getaddrinfo(hostname, None, family=socket.AF_INET):
|
||||
ipv4s.add(info[4][0])
|
||||
return list(ipv4s)
|
||||
|
||||
|
||||
def resolve_current_node_index(cluster_ips: list[str]) -> int:
|
||||
worker_index = os.environ.get("LWS_WORKER_INDEX")
|
||||
if worker_index:
|
||||
return int(worker_index)
|
||||
|
||||
local_ips = set(get_all_ipv4())
|
||||
for index, ip in enumerate(cluster_ips):
|
||||
if ip in local_ips:
|
||||
return index
|
||||
raise RuntimeError("Unable to determine current node index")
|
||||
59
tests/e2e/nightly/scripts/aop_capture.sh
Normal file
59
tests/e2e/nightly/scripts/aop_capture.sh
Normal file
@@ -0,0 +1,59 @@
|
||||
#!/bin/bash
|
||||
# ============================================================
|
||||
# aop_capture.sh - Capture test results from log files
|
||||
#
|
||||
# Called from _e2e_nightly_single_node.yaml [AOP] steps.
|
||||
# Writes outputs to $GITHUB_OUTPUT.
|
||||
#
|
||||
# Usage: aop_capture.sh <yaml_outcome> <pytest_outcome>
|
||||
# ============================================================
|
||||
set -euo pipefail
|
||||
|
||||
YAML_OUTCOME="$1"
|
||||
PYTEST_OUTCOME="$2"
|
||||
LOG_DIR="/tmp/test-logs"
|
||||
|
||||
echo "============================================"
|
||||
echo " Test Result Summary"
|
||||
echo " YAML-driven : ${YAML_OUTCOME:-skipped}"
|
||||
echo " Pytest-driven: ${PYTEST_OUTCOME:-skipped}"
|
||||
echo "============================================"
|
||||
|
||||
parse_log() {
|
||||
local log_file="$1"
|
||||
local prefix="$2"
|
||||
|
||||
if [ ! -f "$log_file" ]; then
|
||||
return 0
|
||||
fi
|
||||
|
||||
echo ""
|
||||
echo "--- ${prefix} tail (last 40 lines) ---"
|
||||
tail -n 40 "$log_file"
|
||||
echo "--- end ---"
|
||||
|
||||
local summary
|
||||
summary=$(grep -E '=+.*(passed|failed|error).*=+' "$log_file" | tail -1 || true)
|
||||
echo "${prefix}_summary=${summary}" >> "$GITHUB_OUTPUT"
|
||||
echo "${prefix}_summary: ${summary}"
|
||||
|
||||
local failures
|
||||
failures=$(grep -c 'FAILED' "$log_file" || true)
|
||||
echo "${prefix}_failures=${failures}" >> "$GITHUB_OUTPUT"
|
||||
}
|
||||
|
||||
parse_log "${LOG_DIR}/pytest-driven.log" "pytest"
|
||||
parse_log "${LOG_DIR}/yaml-test.log" "yaml"
|
||||
|
||||
# Final verdict + which test failed
|
||||
if [ "$YAML_OUTCOME" = "failure" ] || [ "$PYTEST_OUTCOME" = "failure" ]; then
|
||||
echo "result=failure" >> "$GITHUB_OUTPUT"
|
||||
FAILED=""
|
||||
[ "$YAML_OUTCOME" = "failure" ] && FAILED="${FAILED}yaml,"
|
||||
[ "$PYTEST_OUTCOME" = "failure" ] && FAILED="${FAILED}pytest,"
|
||||
echo "failed_test=${FAILED%,}" >> "$GITHUB_OUTPUT"
|
||||
elif [ "$YAML_OUTCOME" = "success" ] || [ "$PYTEST_OUTCOME" = "success" ]; then
|
||||
echo "result=success" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "result=skipped" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
60
tests/e2e/nightly/scripts/aop_classify.sh
Normal file
60
tests/e2e/nightly/scripts/aop_classify.sh
Normal file
@@ -0,0 +1,60 @@
|
||||
#!/bin/bash
|
||||
# ============================================================
|
||||
# aop_classify.sh - Check if failure is environmental
|
||||
#
|
||||
# Args: $1 = failed_test (from capture: "yaml", "pytest", "yaml,pytest")
|
||||
#
|
||||
# Only scans the log files that actually failed.
|
||||
# Writes failure_type to $GITHUB_OUTPUT.
|
||||
# ============================================================
|
||||
set -euo pipefail
|
||||
|
||||
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
||||
RULES="$SCRIPT_DIR/rules-env.txt"
|
||||
LOG_DIR="/tmp/test-logs"
|
||||
FAILED_TEST="${1:-yaml,pytest}"
|
||||
|
||||
check_log() {
|
||||
local log_file="$1"
|
||||
local label="$2"
|
||||
|
||||
if [ ! -f "$log_file" ] || [ ! -s "$log_file" ]; then
|
||||
echo " [$label] log empty/missing -> skipped"
|
||||
return 0
|
||||
fi
|
||||
|
||||
local count
|
||||
count=$(grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -ciEf - "$log_file" 2>/dev/null || echo 0)
|
||||
|
||||
echo " [$label] env patterns matched: ${count}"
|
||||
|
||||
if [ "$count" -gt 0 ]; then
|
||||
echo " [$label] --- matches ---"
|
||||
grep -vE '^[[:space:]]*(#|$)' "$RULES" | grep -niEf - "$log_file" | head -10
|
||||
return 1
|
||||
fi
|
||||
return 0
|
||||
}
|
||||
|
||||
echo "=== Failure Classification ==="
|
||||
echo "[DEBUG] rules file : ${RULES}"
|
||||
echo "[DEBUG] rules exists: $(test -f "$RULES" && echo yes || echo no)"
|
||||
echo "[DEBUG] rules lines : $(grep -c . "$RULES" 2>/dev/null || echo 0)"
|
||||
echo "[DEBUG] rules content:"
|
||||
cat -n "$RULES" 2>/dev/null || echo "(file not found)"
|
||||
|
||||
ENV_FOUND=0
|
||||
if [[ "$FAILED_TEST" == *pytest* ]]; then
|
||||
check_log "${LOG_DIR}/pytest-driven.log" "pytest-driven" || ENV_FOUND=1
|
||||
fi
|
||||
if [[ "$FAILED_TEST" == *yaml* ]]; then
|
||||
check_log "${LOG_DIR}/yaml-test.log" "yaml-test" || ENV_FOUND=1
|
||||
fi
|
||||
|
||||
if [ "$ENV_FOUND" -eq 1 ]; then
|
||||
echo "=== Result: env_failure ==="
|
||||
echo "failure_type=env_failure" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "=== Result: not_env_failure ==="
|
||||
echo "failure_type=not_env_failure" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
98
tests/e2e/nightly/scripts/aop_commit_age.sh
Normal file
98
tests/e2e/nightly/scripts/aop_commit_age.sh
Normal file
@@ -0,0 +1,98 @@
|
||||
#!/bin/bash
|
||||
# ============================================================
|
||||
# aop_commit_age.sh - Look up last successful time in good_table.csv
|
||||
#
|
||||
# CSV format (good_table.csv):
|
||||
# name,yaml/path,link,status,vLLM Git information,vLLM-Ascend Git information,time
|
||||
#
|
||||
# Finds rows matching config_name where status=success,
|
||||
# picks the most recent "time" column, and calculates age.
|
||||
#
|
||||
# Args: $1 = config_name
|
||||
# $2 = csv_path
|
||||
#
|
||||
# Writes to $GITHUB_OUTPUT:
|
||||
# commit_age_days - days since last success
|
||||
# is_old - true if > 3 days
|
||||
# last_status - status from table
|
||||
# last_date - date from table
|
||||
# ============================================================
|
||||
set -euo pipefail
|
||||
|
||||
CONFIG_NAME="${1:-}"
|
||||
CSV_PATH="${GOOD_TABLE:-$2}"
|
||||
|
||||
if [ -z "$CONFIG_NAME" ]; then
|
||||
echo "ERROR: no config name provided"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
echo ">>> Looking up config : ${CONFIG_NAME}"
|
||||
echo ">>> CSV path : ${CSV_PATH}"
|
||||
|
||||
if [ ! -f "$CSV_PATH" ]; then
|
||||
echo ">>> CSV not found → skip"
|
||||
echo "is_old=true" >> "$GITHUB_OUTPUT"
|
||||
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
|
||||
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Find matching rows, only consider success rows (match name column only)
|
||||
ROWS=$(grep "^${CONFIG_NAME}," "$CSV_PATH" | grep -F ',success,' || true)
|
||||
|
||||
if [ -z "$ROWS" ]; then
|
||||
echo ">>> No success row for '${CONFIG_NAME}' → skip"
|
||||
echo "is_old=true" >> "$GITHUB_OUTPUT"
|
||||
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
|
||||
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Pick most recent success row
|
||||
BEST_ROW=""
|
||||
BEST_DATE=""
|
||||
while IFS= read -r row; do
|
||||
date_str=$(echo "$row" | awk -F',' '{print $NF}' | xargs)
|
||||
[ -z "$date_str" ] && continue
|
||||
if [ -z "$BEST_DATE" ] || [[ "$date_str" > "$BEST_DATE" ]]; then
|
||||
BEST_ROW="$row"
|
||||
BEST_DATE="$date_str"
|
||||
fi
|
||||
done <<< "$ROWS"
|
||||
|
||||
if [ -z "$BEST_ROW" ]; then
|
||||
echo ">>> No valid date in success rows → skip"
|
||||
echo "is_old=true" >> "$GITHUB_OUTPUT"
|
||||
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
|
||||
echo "last_status=unknown" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
LAST_STATUS=$(echo "$BEST_ROW" | awk -F',' '{print $4}' | xargs)
|
||||
LAST_DATE="$BEST_DATE"
|
||||
|
||||
echo ">>> Matched row: ${BEST_ROW}"
|
||||
echo "last_status=${LAST_STATUS}" >> "$GITHUB_OUTPUT"
|
||||
echo "last_date=${LAST_DATE}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
LAST_TS=$(date -d "$LAST_DATE" +%s 2>/dev/null || true)
|
||||
if [ -z "$LAST_TS" ]; then
|
||||
echo ">>> Could not parse date: ${LAST_DATE} → skip"
|
||||
echo "is_old=true" >> "$GITHUB_OUTPUT"
|
||||
echo "commit_age_days=99" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
NOW=$(date +%s)
|
||||
AGE_DAYS=$(( (NOW - LAST_TS) / 86400 ))
|
||||
|
||||
echo "commit_age_days=${AGE_DAYS}" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$AGE_DAYS" -gt 3 ]; then
|
||||
echo "is_old=true" >> "$GITHUB_OUTPUT"
|
||||
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (> 3 days) → old"
|
||||
else
|
||||
echo "is_old=false" >> "$GITHUB_OUTPUT"
|
||||
echo ">>> ${CONFIG_NAME} last_status=${LAST_STATUS} date=${LAST_DATE} age=${AGE_DAYS}d (<= 3 days) → recent"
|
||||
fi
|
||||
93
tests/e2e/nightly/scripts/aop_process.sh
Normal file
93
tests/e2e/nightly/scripts/aop_process.sh
Normal file
@@ -0,0 +1,93 @@
|
||||
#!/bin/bash
|
||||
# ============================================================
|
||||
# aop_process.sh - Handle a recent real failure + auto bisect
|
||||
#
|
||||
# Args:
|
||||
# $1 failure_type
|
||||
# $2 commit_age_days
|
||||
# $3 runner
|
||||
# $4 tests
|
||||
# $5 config_file_path
|
||||
# $6 pytest_summary
|
||||
# $7 yaml_summary
|
||||
# $8 scene (single_node | multi_node)
|
||||
# $9 bad_commit (commit SHA, default HEAD)
|
||||
# $10 num_nodes (multi_node only)
|
||||
# $11 coord_dir (multi_node only)
|
||||
# $12 case_name (optional)
|
||||
# ============================================================
|
||||
set -euo pipefail
|
||||
|
||||
FT="${1:-unknown}"
|
||||
AGE="${2:-?}"
|
||||
RUNNER="${3:-?}"
|
||||
TESTS="${4:-}"
|
||||
CONFIG="${5:-}"
|
||||
PYTEST_SUMMARY="${6:-}"
|
||||
YAML_SUMMARY="${7:-}"
|
||||
SCENE="${8:-single_node}"
|
||||
BAD_COMMIT="${9:-HEAD}"
|
||||
NUM_NODES="${10:-}"
|
||||
COORD_DIR="${11:-}"
|
||||
NAME="${12:-}"
|
||||
|
||||
echo "================================================"
|
||||
echo " PROCESS - needs attention"
|
||||
echo " Failure type : ${FT}"
|
||||
echo " Commit age : ${AGE} days"
|
||||
echo " Runner : ${RUNNER}"
|
||||
echo " Tests : ${TESTS:-N/A}"
|
||||
echo " Config : ${CONFIG:-N/A}"
|
||||
echo " Scene : ${SCENE}"
|
||||
echo " Bad commit : ${BAD_COMMIT}"
|
||||
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
|
||||
echo " YAML : ${YAML_SUMMARY:-N/A}"
|
||||
echo "================================================"
|
||||
|
||||
echo "::group::Failed test details"
|
||||
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log /tmp/test-logs/multi-node.log; do
|
||||
if [ -f "$f" ]; then
|
||||
grep -A 10 'FAILED' "$f" || true
|
||||
fi
|
||||
done
|
||||
echo "::endgroup::"
|
||||
|
||||
# =====================================================
|
||||
# Auto bisect
|
||||
# =====================================================
|
||||
|
||||
# Extract case_name if not provided (single_node requires it)
|
||||
if [ -z "$NAME" ] && [ "$SCENE" = "single_node" ]; then
|
||||
if [ -n "$TESTS" ]; then
|
||||
# py-driven: tests/e2e/.../test_xxx.py → test_xxx
|
||||
NAME=$(basename "$TESTS" .py)
|
||||
elif [ -n "$CONFIG" ]; then
|
||||
# YAML-driven: Qwen3-32B-Int8.yaml → Qwen3-32B-Int8
|
||||
NAME=$(basename "$CONFIG" .yaml)
|
||||
fi
|
||||
|
||||
if [ -z "$NAME" ]; then
|
||||
echo "WARNING: could not extract case_name, bisect may fail"
|
||||
else
|
||||
echo "Extracted name: ${NAME}"
|
||||
fi
|
||||
fi
|
||||
|
||||
GOOD_TABLE="${GOOD_TABLE:-}"
|
||||
|
||||
BISECT_CMD=(
|
||||
python -m tests.e2e.nightly.bisect.auto_bisect
|
||||
--scene "${SCENE}"
|
||||
--bad-commit "${BAD_COMMIT}"
|
||||
--good-table "${GOOD_TABLE}"
|
||||
)
|
||||
|
||||
[ -n "$CONFIG" ] && BISECT_CMD+=(--config-yaml "$CONFIG")
|
||||
[ -n "$NAME" ] && BISECT_CMD+=(--name "$NAME")
|
||||
[ -n "$NUM_NODES" ] && BISECT_CMD+=(--num-nodes "$NUM_NODES")
|
||||
[ -n "$COORD_DIR" ] && BISECT_CMD+=(--coord-dir "$COORD_DIR")
|
||||
|
||||
echo ""
|
||||
echo "=== Running auto bisect ==="
|
||||
echo "${BISECT_CMD[@]}"
|
||||
"${BISECT_CMD[@]}"
|
||||
39
tests/e2e/nightly/scripts/aop_skip.sh
Normal file
39
tests/e2e/nightly/scripts/aop_skip.sh
Normal file
@@ -0,0 +1,39 @@
|
||||
#!/bin/bash
|
||||
# ============================================================
|
||||
# aop_skip.sh - Log skip reason and show failure details
|
||||
#
|
||||
# Args: failure_type last_status last_date age_days
|
||||
# pytest_summary yaml_summary
|
||||
# ============================================================
|
||||
set -euo pipefail
|
||||
|
||||
FT="${1:-unknown}"
|
||||
LAST_STATUS="${2:-?}"
|
||||
LAST_DATE="${3:-?}"
|
||||
AGE="${4:-?}"
|
||||
PYTEST_SUMMARY="${5:-}"
|
||||
YAML_SUMMARY="${6:-}"
|
||||
|
||||
case "$FT" in
|
||||
env_failure) REASON="environment issue" ;;
|
||||
*) REASON="last run > 3 days ago" ;;
|
||||
esac
|
||||
|
||||
echo "================================================"
|
||||
echo " SKIP - no further action"
|
||||
echo " Failure type : ${FT}"
|
||||
echo " Last status : ${LAST_STATUS}"
|
||||
echo " Last date : ${LAST_DATE}"
|
||||
echo " Age (days) : ${AGE}"
|
||||
echo " Reason : ${REASON}"
|
||||
echo " PyTest : ${PYTEST_SUMMARY:-N/A}"
|
||||
echo " YAML : ${YAML_SUMMARY:-N/A}"
|
||||
echo "================================================"
|
||||
|
||||
echo "::group::Failed test details"
|
||||
for f in /tmp/test-logs/pytest-driven.log /tmp/test-logs/yaml-test.log; do
|
||||
if [ -f "$f" ]; then
|
||||
grep -A 10 'FAILED' "$f" || true
|
||||
fi
|
||||
done
|
||||
echo "::endgroup::"
|
||||
6
tests/e2e/nightly/scripts/rules-env.txt
Normal file
6
tests/e2e/nightly/scripts/rules-env.txt
Normal file
@@ -0,0 +1,6 @@
|
||||
# Environment failure patterns (network / hardware / infra)
|
||||
# One regex per line. Lines starting with # are comments.
|
||||
# Used by aop_classify.sh via grep -Ef
|
||||
|
||||
RuntimeError: Timeout
|
||||
TimeoutError: Timed out waiting for engine core processes to start
|
||||
155
tests/e2e/nightly/scripts/update_good_table.py
Normal file
155
tests/e2e/nightly/scripts/update_good_table.py
Normal file
@@ -0,0 +1,155 @@
|
||||
#!/usr/bin/env python3
|
||||
#
|
||||
# Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved.
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
# This file is a part of the vllm-ascend project.
|
||||
#
|
||||
"""Update /root/.cache/vllm-ascend/main/nightly/good_table.csv with a
|
||||
successful test entry. Creates the file (with header) if it does not exist;
|
||||
replaces the existing row for the same test name if it does.
|
||||
|
||||
CSV columns:
|
||||
name, yaml/path, link, status,
|
||||
vLLM Git information, vLLM-Ascend Git information, time
|
||||
"""
|
||||
|
||||
import argparse
|
||||
import csv
|
||||
import os
|
||||
import subprocess
|
||||
from datetime import datetime, timedelta, timezone
|
||||
|
||||
HEADER = [
|
||||
"name",
|
||||
"yaml/path",
|
||||
"link",
|
||||
"status",
|
||||
"vLLM Git information",
|
||||
"vLLM-Ascend Git information",
|
||||
"time",
|
||||
]
|
||||
|
||||
|
||||
def git_head(repo_dir: str) -> str:
|
||||
try:
|
||||
return subprocess.check_output(
|
||||
["git", "rev-parse", "HEAD"],
|
||||
cwd=repo_dir,
|
||||
stderr=subprocess.DEVNULL,
|
||||
text=True,
|
||||
).strip()
|
||||
except Exception:
|
||||
return "N/A"
|
||||
|
||||
|
||||
def current_timestamp() -> str:
|
||||
tz = timezone(timedelta(hours=8))
|
||||
ts = datetime.now(tz).strftime("%Y-%m-%d %H:%M:%S %z")
|
||||
# Reformat +0800 → +08:00 to match existing CSV entries
|
||||
return ts[:-2] + ":" + ts[-2:]
|
||||
|
||||
|
||||
def load_rows(csv_path: str) -> list[list[str]]:
|
||||
if not os.path.isfile(csv_path):
|
||||
return []
|
||||
with open(csv_path, newline="", encoding="utf-8") as f:
|
||||
reader = csv.reader(f)
|
||||
rows = list(reader)
|
||||
# Drop the header row if present
|
||||
if rows and rows[0] == HEADER:
|
||||
rows = rows[1:]
|
||||
return rows
|
||||
|
||||
|
||||
def save_rows(csv_path: str, rows: list[list[str]]) -> None:
|
||||
os.makedirs(os.path.dirname(csv_path), exist_ok=True)
|
||||
with open(csv_path, "w", newline="", encoding="utf-8") as f:
|
||||
writer = csv.writer(f)
|
||||
writer.writerow(HEADER)
|
||||
writer.writerows(rows)
|
||||
|
||||
|
||||
_DEFAULT_SINGLE_NODE_CONFIG_BASE = "tests/e2e/nightly/single_node/models/configs"
|
||||
_DEFAULT_MULTI_NODE_CONFIG_BASES = (
|
||||
"tests/e2e/nightly/multi_node/internal_dp/config",
|
||||
"tests/e2e/nightly/multi_node/external_dp/config",
|
||||
)
|
||||
|
||||
|
||||
def resolve_test_path(
|
||||
test_path: str,
|
||||
config_base_path: str,
|
||||
scene: str = "single_node",
|
||||
repo_dir: str = ".",
|
||||
) -> str:
|
||||
"""Return the full relative path for the yaml/path CSV column.
|
||||
|
||||
Upper-level workflows pass config_file_path as a bare filename
|
||||
(e.g. ``Qwen3.5-27B-w8a8-A2.yaml``). When no directory component is
|
||||
present we prepend the config base path so the CSV matches the format
|
||||
used by the existing hand-curated good_table entries.
|
||||
"""
|
||||
if os.sep in test_path or "/" in test_path:
|
||||
return test_path
|
||||
if config_base_path.strip():
|
||||
return f"{config_base_path.strip()}/{test_path}"
|
||||
if scene == "multi_node":
|
||||
for base in _DEFAULT_MULTI_NODE_CONFIG_BASES:
|
||||
if os.path.isfile(os.path.join(repo_dir, base, test_path)):
|
||||
return f"{base}/{test_path}"
|
||||
return f"{_DEFAULT_MULTI_NODE_CONFIG_BASES[0]}/{test_path}"
|
||||
return f"{_DEFAULT_SINGLE_NODE_CONFIG_BASE}/{test_path}"
|
||||
|
||||
|
||||
def main() -> None:
|
||||
parser = argparse.ArgumentParser(description="Update good_table.csv on test success")
|
||||
parser.add_argument("--cache-csv", required=True)
|
||||
parser.add_argument("--test-name", required=True)
|
||||
parser.add_argument("--test-path", required=True)
|
||||
parser.add_argument("--config-base-path", default="")
|
||||
parser.add_argument("--scene", default="single_node", choices=["single_node", "multi_node"])
|
||||
parser.add_argument("--run-link", required=True)
|
||||
parser.add_argument("--vllm-dir", default="/vllm-workspace/vllm")
|
||||
parser.add_argument("--vllm-ascend-dir", default="/vllm-workspace/vllm-ascend")
|
||||
parser.add_argument("--vllm-ascend-version", default="")
|
||||
parser.add_argument("--vllm-version", default="")
|
||||
args = parser.parse_args()
|
||||
|
||||
vllm_hash = args.vllm_version.strip() or git_head(args.vllm_dir)
|
||||
vllm_ascend_hash = args.vllm_ascend_version.strip() or git_head(args.vllm_ascend_dir)
|
||||
timestamp = current_timestamp()
|
||||
test_path = resolve_test_path(args.test_path, args.config_base_path, args.scene, args.vllm_ascend_dir)
|
||||
|
||||
new_row = [
|
||||
args.test_name,
|
||||
test_path,
|
||||
args.run_link,
|
||||
"success",
|
||||
vllm_hash,
|
||||
vllm_ascend_hash,
|
||||
timestamp,
|
||||
]
|
||||
|
||||
is_new = not os.path.isfile(args.cache_csv)
|
||||
rows = load_rows(args.cache_csv)
|
||||
rows = [r for r in rows if r and r[0] != args.test_name]
|
||||
rows.append(new_row)
|
||||
save_rows(args.cache_csv, rows)
|
||||
|
||||
action = "Created" if is_new else "Updated"
|
||||
print(f">>> {action} {args.cache_csv}: name={args.test_name} status=success time={timestamp}")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,84 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "10"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "36864"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1, "method": "mtp"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
|
||||
_benchmarks_acc: &benchmarks_acc
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
_benchmarks_perf: &benchmarks_perf
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 400
|
||||
max_out_len: 1500
|
||||
batch_size: 1000
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-R1-0528-W8A8-single"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--enforce-eager"
|
||||
benchmarks:
|
||||
|
||||
- name: "DeepSeek-R1-0528-W8A8-aclgraph"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks_acc
|
||||
<<: *benchmarks_perf
|
||||
|
||||
@@ -0,0 +1,81 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V3.2-W8A8-DCP-replicated-indexer"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
envs:
|
||||
VLLM_ASCEND_ENABLE_NZ: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "20"
|
||||
HCCL_BUFFSIZE: "768"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_SERVER_DEV_MODE: "1"
|
||||
VLLM_WORKER_MULTIPROC_METHOD: "spawn"
|
||||
ASCEND_LAUNCH_BLOCKING: "0"
|
||||
ASCEND_ENABLE_USE_FABRIC_MEM: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "0"
|
||||
PYTHONHASHSEED: "0"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "10000"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "30000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
CPU_AFFINITY_CONF: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "1024"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--pipeline-parallel-size"
|
||||
- "1"
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--prefill-context-parallel-size"
|
||||
- "1"
|
||||
- "--decode-context-parallel-size"
|
||||
- "16"
|
||||
- "--cp-kv-cache-interleave-size"
|
||||
- "1"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.95"
|
||||
- "--api-server-count"
|
||||
- "1"
|
||||
- "--safetensors-load-strategy"
|
||||
- "prefetch"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY", "cudagraph_capture_sizes":[4, 16, 64, 128]}'
|
||||
- "--additional-config"
|
||||
- '{"enable_dsa_cp": true, "ascend_compilation_config":{"enable_npugraph_ex": true, "enable_static_kernel": false}, "multistream_overlap_shared_expert": true, "enable_mc2_hierarchy_comm": false, "enable_sparse_sfa_c8": true, "enable_sparse_li_c8": true, "enable_cpu_binding": true, "recompute_scheduler_enable": false}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method": "deepseek_mtp"}'
|
||||
test_content: []
|
||||
benchmarks:
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 8192
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 5
|
||||
@@ -0,0 +1,80 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V3.2-W8A8-TP8-DP2"
|
||||
model: "vllm-ascend/DeepSeek-V3.2-W8A8"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_USE_V1: "1"
|
||||
HCCL_BUFFSIZE: "256"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "67000"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "8"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--async-scheduling"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.95"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[1,2,4,8,16,24,32,40,48], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp"}'
|
||||
- "--reasoning-parser"
|
||||
- "deepseek_v3"
|
||||
- "--tokenizer_mode"
|
||||
- "deepseek_v32"
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 86.67
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
thinking: true
|
||||
threshold: 10
|
||||
|
||||
perf_2:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 16
|
||||
max_out_len: 1500
|
||||
batch_size: 4
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,80 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "DeepSeek-V4-Flash-W8A8-A3"
|
||||
model: "Eco-Tech/DeepSeek-V4-Flash-w8a8-mtp"
|
||||
special_dependencies:
|
||||
transformers: "5.9.0"
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "0"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
ASCEND_LAUNCH_BLOCKING: "0"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "3000"
|
||||
server_cmd:
|
||||
- "--enable-prefix-caching"
|
||||
- "--max-model-len"
|
||||
- "1048576"
|
||||
- "--max-num-batched-tokens"
|
||||
- "10240"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--max-num-seqs"
|
||||
- "64"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--enable-expert-parallel"
|
||||
- "--tokenizer-mode"
|
||||
- "deepseek_v4"
|
||||
- "--tool-call-parser"
|
||||
- "deepseek_v4"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--reasoning-parser"
|
||||
- "deepseek_v4"
|
||||
- "--safetensors-load-strategy"
|
||||
- "prefetch"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--api-server-count"
|
||||
- "1"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1,"method": "mtp","enforce_eager": true}'
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--block-size"
|
||||
- "128"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- "--async-scheduling"
|
||||
- "--additional-config"
|
||||
- '{"ascend_compilation_config":{"enable_npugraph_ex":true,"enable_static_kernel":false},"enable_cpu_binding":"true","enable_shared_expert_dp":true,"multistream_overlap_shared_expert":true}'
|
||||
benchmarks:
|
||||
acc-gpqa:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gpqa/gpqa_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 65536
|
||||
batch_size: 32
|
||||
baseline: 86.36
|
||||
threshold: 5
|
||||
thinking: true
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix0_in32768_bs1000_deepseek
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 64
|
||||
max_out_len: 1024
|
||||
batch_size: 16
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
66
tests/e2e/nightly/single_node/models/configs/GLM-4.7.yaml
Normal file
66
tests/e2e/nightly/single_node/models/configs/GLM-4.7.yaml
Normal file
@@ -0,0 +1,66 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
HCCL_BUFFSIZE: "512"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_ASCEND_ENABLE_TOPK_OPTIMIZE: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--enable-expert-parallel"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"mtp"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_shared_expert_dp": true, "ascend_fusion_config": {"fusion_ops_gmmswigluquant": false}}'
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 8
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "GLM-4.7-TP8-DP2-decodegraph"
|
||||
model: "Eco-Tech/GLM-4.7-W8A8-floatmtp"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes": [1,2,4,8,16,32,64,128,256,512], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,82 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "GLM-5.1-W8A8-PrefillMC2"
|
||||
model: "Eco-Tech/GLM-5.1-w8a8" #need update
|
||||
envs:
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_USE_V1: "1"
|
||||
HCCL_BUFFSIZE: "1800"
|
||||
ASCEND_AGGREGATE_ENABLE: "1"
|
||||
ASCEND_TRANSPORT_PRINT: "1"
|
||||
ACL_OP_INIT_MODE: "1"
|
||||
ASCEND_A3_ENABLE: "1"
|
||||
VLLM_NIXL_ABORT_REQUEST_TIMEOUT: "300000"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "10240"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "32"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--async-scheduling"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.94"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method":"deepseek_mtp", "enforce_eager": true}'
|
||||
- "--additional_config"
|
||||
- '{"enable_prefill_mc2": true}'
|
||||
- "--reasoning-parser"
|
||||
- "glm45"
|
||||
- "--tool-call-parser"
|
||||
- "glm47"
|
||||
|
||||
benchmarks:
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 8192
|
||||
batch_size: 32
|
||||
baseline: 96.88
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
thinking: true
|
||||
threshold: 5
|
||||
|
||||
perf_2:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 64
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,58 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-ep-weight-filter"
|
||||
- "--tool-call-parser"
|
||||
- "hy_v3"
|
||||
- "--reasoning-parser"
|
||||
- "hy_v3"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--max-model-len"
|
||||
- "32768"
|
||||
- "--max-num-seqs"
|
||||
- "8"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--speculative-config"
|
||||
- '{"method": "mtp", "num_speculative_tokens": 1}'
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc_gsm8k:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_4_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 8
|
||||
baseline: 93.07
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Hy3-preview-TP16-EP-MTP"
|
||||
model: "Tencent-Hunyuan/Hy3-preview"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,52 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Kimi-K2-Thinking-TP16-Case"
|
||||
model: "moonshotai/Kimi-K2-Thinking"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "16"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "12"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--trust-remote-code"
|
||||
- "--enable-expert-parallel"
|
||||
- "--no-enable-prefix-caching"
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 4096
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs400
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 256
|
||||
batch_size: 64
|
||||
trust_remote_code: true
|
||||
request_rate: 11.2
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
91
tests/e2e/nightly/single_node/models/configs/Kimi-K2.5.yaml
Normal file
91
tests/e2e/nightly/single_node/models/configs/Kimi-K2.5.yaml
Normal file
@@ -0,0 +1,91 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
HCCL_BUFFSIZE: "512"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_NZ: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-prefix-caching"
|
||||
- "--enable-chunked-prefill"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--data-parallel-size"
|
||||
- "4"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "133120"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--seed"
|
||||
- "42"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes":[4,8,12,16,32], "cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--speculative-config"
|
||||
- '{"method":"eagle3", "model":"lightseekorg/kimi-k2.5-eagle3", "num_speculative_tokens":3}'
|
||||
- "--additional-config"
|
||||
- '{"enable_shared_expert_dp":true}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--mm-encoder-tp-mode"
|
||||
- "data"
|
||||
|
||||
_benchmarks: &benchmarks
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 65536
|
||||
temperature: 0.0
|
||||
top_p: 1
|
||||
top_k: -1
|
||||
repetition_penalty: 1.0
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_prefix90_in131072_bs1000_kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 8
|
||||
max_out_len: 1024
|
||||
batch_size: 2
|
||||
trust_remote_code: true
|
||||
request_rate: 0
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "Kimi-K2.5-W4A8-Case"
|
||||
model: "Eco-Tech/Kimi-K2.5-W4A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
benchmarks:
|
||||
<<: *benchmarks
|
||||
@@ -0,0 +1,70 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
test_cases:
|
||||
- name: "Kimi-K2.6-W4A8-in3.5k-out1.5k-TPOT50-0-128-32"
|
||||
model: "Eco-Tech/Kimi-K2.6-w4a8"
|
||||
envs:
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_PROC_BIND: "false"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_BUFFSIZE: "800"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM_ASCEND_ENABLE_MLAPO: "1"
|
||||
VLLM_ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_ASCEND_ENABLE_FUSED_MC2: "1"
|
||||
DYNAMIC_EPLB: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--allowed-local-media-path"
|
||||
- "/"
|
||||
- "--trust-remote-code"
|
||||
- "--safetensors-load-strategy"
|
||||
- 'prefetch'
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--enable-expert-parallel"
|
||||
- "--max-num-seqs"
|
||||
- "24"
|
||||
- "--max-model-len"
|
||||
- "6144"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.85"
|
||||
- "--seed"
|
||||
- "42"
|
||||
- "--async-scheduling"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
- "--additional-config"
|
||||
- '{"eplb_config": {"dynamic_eplb": true},"ascend_compilation_config": {"enable_static_kernel": false}}'
|
||||
- "--profiler-config"
|
||||
- '{"profiler": "torch", "torch_profiler_dir": "./vllm_profile", "torch_profiler_with_stack": true}'
|
||||
- "--mm-processor-cache-gb"
|
||||
- "0"
|
||||
- "--mm-encoder-tp-mode"
|
||||
- "data"
|
||||
- "--speculative-config"
|
||||
- '{"method": "dflash","model": "z-lab/Kimi-K2.5-DFlash", "num_speculative_tokens": 7}'
|
||||
benchmarks:
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs4096-prefix0-kimi
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 128
|
||||
max_out_len: 1500
|
||||
batch_size: 32
|
||||
request_rate: 0
|
||||
baseline: 1433.4454
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,91 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
OMP_NUM_THREADS: "100"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
VLLM_RPC_TIMEOUT: "3600000"
|
||||
VLLM_EXECUTE_MODEL_TIMEOUT_SECONDS: "3600000"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "40960"
|
||||
- "--max-num-seqs"
|
||||
- "14"
|
||||
- "--trust-remote-code"
|
||||
|
||||
_benchmarks_gsm8k: &benchmarks_gsm8k
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gsm8k-lite
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 95
|
||||
threshold: 10
|
||||
|
||||
_benchmarks_aime: &benchmarks_aime
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2024
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2024/aime2024_gen_0_shot_chat_prompt
|
||||
max_out_len: 32768
|
||||
batch_size: 32
|
||||
baseline: 86.67
|
||||
threshold: 10
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp2"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 2, "method": "mtp"}'
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.92"
|
||||
benchmarks:
|
||||
<<: *benchmarks_gsm8k
|
||||
|
||||
- name: "MTPX-DeepSeek-R1-0528-W8A8-mtp3"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
<<: *envs
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-num-batched-tokens"
|
||||
- "2048"
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 3, "method": "mtp"}'
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_capture_sizes": [56], "cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
<<: *benchmarks_aime
|
||||
@@ -0,0 +1,88 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.5-w8a8"
|
||||
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "512"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
|
||||
HCCL_INTRA_PCIE_ENABLE: "1"
|
||||
HCCL_INTRA_ROCE_ENABLE: "0"
|
||||
OMP_PROC_BIND: "false"
|
||||
VLLM_TORCH_PROFILER_WITH_STACK: "0"
|
||||
VLLM_TORCH_PROFILER_DIR: "./profile"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--model-loader-extra-config"
|
||||
- '{"enable_multithread_load":true,"num_threads":16}'
|
||||
- "--speculative_config"
|
||||
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
|
||||
- "--enable-expert-parallel"
|
||||
- "--enable-chunked-prefill"
|
||||
- "--enable-prefix-caching"
|
||||
- "--max-num-seqs"
|
||||
- "100"
|
||||
- "--max-model-len"
|
||||
- "196608"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-num-batched-tokens"
|
||||
- "6144"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--tool-call-parser"
|
||||
- "minimax_m2"
|
||||
- "--reasoning-parser"
|
||||
- "minimax_m2_append_think"
|
||||
- "--enable-force-include-usage"
|
||||
- "--profiler-config"
|
||||
- '{"profiler":"torch","torch_profiler_dir":"./profile","torch_profiler_with_stack":false}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY","cudagraph_capture_sizes":[4,16,40,80,160,256,400]}'
|
||||
benchmarks:
|
||||
acc:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/gpqa
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: gsm8k/gpqa_gen_0_shot_str
|
||||
max_out_len: 131072
|
||||
batch_size: 64
|
||||
baseline: 83
|
||||
threshold: 5
|
||||
bos_token_id: 200019
|
||||
do_sample: true
|
||||
eos_token_id: 200020
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 40
|
||||
transformers_version: 4.46.1
|
||||
ignore_eos: false
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 360
|
||||
max_out_len: 1500
|
||||
batch_size: 120
|
||||
request_rate: 0
|
||||
baseline: 2042
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,73 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.5-w8a8"
|
||||
model: "Eco-Tech/MiniMax-M2.5-w8a8-QuaRot"
|
||||
envs:
|
||||
HCCL_BUFFSIZE: "512"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
VLLM-ASCEND_ENABLE_NZ: "1"
|
||||
VLLM_ASCEND_ENABLE_FLASHCOMM1: "1"
|
||||
VLLM-ASCEND_BALANCE_SCHEDULING: "1"
|
||||
VLLM_USE_MODELSCOPE: "true"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "1"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.85"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--no-enable-prefix-caching"
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding":true}'
|
||||
- "--speculative_config"
|
||||
- '{"method":"eagle3","model":"vllm-ascend/MiniMax-M2.5-eagle-model-0318","num_speculative_tokens":3}'
|
||||
- "--enable-expert-parallel"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--max-model-len"
|
||||
- "196608"
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode":"FULL_DECODE_ONLY"}'
|
||||
benchmarks:
|
||||
acc_aime2025:
|
||||
case_type: accuracy
|
||||
dataset_path: vllm-ascend/aime2025
|
||||
request_conf: vllm_api_general_chat
|
||||
dataset_conf: aime2025/aime2025_gen_0_shot_chat_prompt
|
||||
max_out_len: 131072
|
||||
batch_size: 32
|
||||
baseline: 90
|
||||
threshold: 10
|
||||
bos_token_id: 200019
|
||||
do_sample: true
|
||||
eos_token_id: 200020
|
||||
temperature: 1.0
|
||||
top_p: 0.95
|
||||
top_k: 40
|
||||
transformers_version: 4.46.1
|
||||
ignore_eos: false
|
||||
perf:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in3500-bs2800
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 512
|
||||
max_out_len: 1500
|
||||
batch_size: 128
|
||||
request_rate: 0
|
||||
baseline: 1116
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,85 @@
|
||||
# ==========================================
|
||||
# Shared Configurations
|
||||
# ==========================================
|
||||
|
||||
_envs: &envs
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
HCCL_BUFFSIZE: "1200"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
OMP_NUM_THREADS: "1"
|
||||
LD_PRELOAD: "/usr/lib/aarch64-linux-gnu/libjemalloc.so.2:$LD_PRELOAD"
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
|
||||
_server_cmd: &server_cmd
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--host"
|
||||
- "0.0.0.0"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--enable-expert-parallel"
|
||||
- "--async-scheduling"
|
||||
- "--max-num-seqs"
|
||||
- "128"
|
||||
- "--safetensors-load-strategy"
|
||||
- 'prefetch'
|
||||
- "--max-num-batched-tokens"
|
||||
- "16384"
|
||||
- "--trust-remote-code"
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--enable-auto-tool-choice"
|
||||
- "--tool-call-parser"
|
||||
- "minimax_m2"
|
||||
- "--speculative-config"
|
||||
- '{"method":"eagle3","model":"Eco-Tech/MiniMax-M2.7-eagle-model-short","num_speculative_tokens":3}'
|
||||
- "--compilation-config"
|
||||
- '{"cudagraph_mode": "FULL_DECODE_ONLY"}'
|
||||
- "--additional-config"
|
||||
- '{"enable_cpu_binding": true, "enable_npugraph_ex": true, "enable_static_kernel": true,"enable_fused_mc2":true,"weight_nz_mode":true,"enable_flashcomm1":true}'
|
||||
|
||||
_benchmarks_3500: &benchmarks_3500
|
||||
perf_50:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 760
|
||||
max_out_len: 1500
|
||||
batch_size: 190
|
||||
request_rate: 0
|
||||
baseline: 4573.02
|
||||
threshold: 0.97
|
||||
perf_20:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K_in3500_bs4000_minimax
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 192
|
||||
max_out_len: 1500
|
||||
batch_size: 48
|
||||
request_rate: 0
|
||||
baseline: 2229.147
|
||||
threshold: 0.97
|
||||
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "MiniMax-M2.7-3500"
|
||||
model: "vllm-ascend/MiniMax-M2.7-w8a8-QuaRot"
|
||||
envs:
|
||||
<<: *envs
|
||||
server_cmd: *server_cmd
|
||||
server_cmd_extra:
|
||||
- "--max-model-len"
|
||||
- "70000"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.8"
|
||||
- "--no-enable-prefix-caching"
|
||||
benchmarks:
|
||||
<<: *benchmarks_3500
|
||||
@@ -0,0 +1,78 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "prefix-cache-deepseek-r1-0528-w8a8"
|
||||
model: "vllm-ascend/DeepSeek-R1-0528-W8A8"
|
||||
envs:
|
||||
OMP_NUM_THREADS: "10"
|
||||
OMP_PROC_BIND: "false"
|
||||
HCCL_BUFFSIZE: "1024"
|
||||
PYTORCH_NPU_ALLOC_CONF: "expandable_segments:True"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
VLLM_ENGINE_READY_TIMEOUT_S: "7200"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--data-parallel-size"
|
||||
- "2"
|
||||
- "--tensor-parallel-size"
|
||||
- "8"
|
||||
- "--enable-expert-parallel"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--seed"
|
||||
- "1024"
|
||||
- "--max-model-len"
|
||||
- "5200"
|
||||
- "--max-num-batched-tokens"
|
||||
- "4096"
|
||||
- "--max-num-seqs"
|
||||
- "16"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
- "--speculative-config"
|
||||
- '{"num_speculative_tokens": 1, "method": "mtp"}'
|
||||
test_content:
|
||||
- "benchmark_comparisons"
|
||||
benchmark_comparisons_args:
|
||||
- metric: "TTFT"
|
||||
baseline: "prefix0"
|
||||
target: "prefix75"
|
||||
ratio: 0.5
|
||||
operator: "<"
|
||||
benchmarks:
|
||||
warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1024-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 1000
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
prefix0:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix0-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 18
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
prefix75:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix75-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 18
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
@@ -0,0 +1,70 @@
|
||||
# ==========================================
|
||||
# ACTUAL TEST CASES
|
||||
# ==========================================
|
||||
|
||||
test_cases:
|
||||
- name: "prefix-cache-qwen3-32b-w8a8"
|
||||
model: "vllm-ascend/Qwen3-32B-W8A8"
|
||||
envs:
|
||||
TASK_QUEUE_ENABLE: "1"
|
||||
HCCL_OP_EXPANSION_MODE: "AIV"
|
||||
SERVER_PORT: "DEFAULT_PORT"
|
||||
server_cmd:
|
||||
- "--quantization"
|
||||
- "ascend"
|
||||
- "--reasoning-parser"
|
||||
- "qwen3"
|
||||
- "--tensor-parallel-size"
|
||||
- "4"
|
||||
- "--port"
|
||||
- "$SERVER_PORT"
|
||||
- "--max-model-len"
|
||||
- "8192"
|
||||
- "--max-num-batched-tokens"
|
||||
- "8192"
|
||||
- "--max-num-seqs"
|
||||
- "256"
|
||||
- "--trust-remote-code"
|
||||
- "--gpu-memory-utilization"
|
||||
- "0.9"
|
||||
- "--additional-config"
|
||||
- '{"enable_weight_nz_layout": true}'
|
||||
test_content:
|
||||
- "benchmark_comparisons"
|
||||
benchmark_comparisons_args:
|
||||
- metric: "TTFT"
|
||||
baseline: "prefix0"
|
||||
target: "prefix75"
|
||||
ratio: 0.4
|
||||
operator: "<"
|
||||
benchmarks:
|
||||
warm_up:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/GSM8K-in1024-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 1000
|
||||
baseline: 0
|
||||
threshold: 0.97
|
||||
prefix0:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix0-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 48
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
prefix75:
|
||||
case_type: performance
|
||||
dataset_path: vllm-ascend/prefix75-in3500-bs210
|
||||
request_conf: vllm_api_stream_chat
|
||||
dataset_conf: gsm8k/gsm8k_gen_0_shot_cot_str_perf
|
||||
num_prompts: 210
|
||||
max_out_len: 1
|
||||
batch_size: 48
|
||||
baseline: 1
|
||||
threshold: 0.97
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user