[CI] refect e2e test (#4799)

### What this PR does / why we need it? This PR updates the CI configuration and adjusts a set of end-to-end (e2e) tests under tests/e2e/multicard, in order to refactor the test suite and ensure compatibility with current codebase and CI workflows. 1. tests/e2e/multicard/test_prefix_caching.py: change model to Qwen3-8B and rename the test case 2. tests/e2e/multicard/test_quantization.py: rename the test case 3. tests/e2e/multicard/test_qwen3_moe.py: remove duplicate test and rename test cases 4. tests/e2e/multicard/test_qwen3_next.py: rename test cases and change the W8A8 pruning model to the W8A8 model and remove the eager parameter 5. tests/e2e/multicard/test_shared_expert_dp.py: rename test case and remove the eager parameter 6. tests/e2e/multicard/test_single_request_aclgraph.py: rename test case and change Qwen3-30B to Qwen3-0.6B 7. tests/e2e/multicard/test_torchair_graph_mode.py: delete test cases about torchair - vLLM version: v0.12.0 - vLLM main: ad32e3e19c Signed-off-by: hfadzxy <starmoon_zhang@163.com>
2025-12-12 08:42:08 +08:00
parent a6ef3ac4e4
commit bfafe30953
8 changed files with 30 additions and 66 deletions
--- a/tests/e2e/multicard/test_shared_expert_dp.py
+++ b/tests/e2e/multicard/test_shared_expert_dp.py
@@ -7,13 +7,13 @@ from tests.e2e.conftest import VllmRunner
 from tests.e2e.model_utils import check_outputs_equal

 MODELS = [
-    "vllm-ascend/DeepSeek-V2-Lite",
+    "deepseek-ai/DeepSeek-V2-Lite",
 ]
 os.environ["VLLM_WORKER_MULTIPROC_METHOD"] = "spawn"


@pytest.mark.parametrize("model", MODELS)
-def test_models_with_enable_shared_expert_dp(model: str) -> None:
+def test_deepseek_v2_lite_enable_shared_expert_dp_tp2(model: str) -> None:

    if 'HCCL_OP_EXPANSION_MODE' in os.environ:
        del os.environ['HCCL_OP_EXPANSION_MODE']
@@ -51,7 +51,7 @@ def test_models_with_enable_shared_expert_dp(model: str) -> None:
            model,
            max_model_len=1024,
            tensor_parallel_size=2,
-            enforce_eager=False,
+            enable_expert_parallel=True,
            compilation_config={
                "cudagraph_capture_sizes": [1, 4, 8, 16],
                "cudagraph_mode": "FULL_DECODE_ONLY",