ACLgraph enable: Test cases revisions for all features (#3388)

### What this PR does / why we need it? This PR revise the test cases of various features on the warehouse which add the enablement of aclgraph to the test cases. ### Does this PR introduce _any_ user-facing change? no ### How was this patch tested? ut - vLLM version: v0.11.0rc3 - vLLM main: https://github.com/vllm-project/vllm/commit/v0.11.0 Signed-off-by: lilinsiman <lilinsiman@gmail.com>
2025-10-17 17:15:19 +08:00
parent bf87606932
commit 1b424fb7f1
17 changed files with 34 additions and 117 deletions
--- a/tests/e2e/multicard/test_offline_inference_distributed.py
+++ b/tests/e2e/multicard/test_offline_inference_distributed.py
@@ -52,7 +52,7 @@ def test_models_distributed_QwQ():
            dtype=dtype,
            tensor_parallel_size=2,
            distributed_executor_backend="mp",
-            enforce_eager=True,
+            enforce_eager=False,
    ) as vllm_model:
        vllm_model.generate_greedy(example_prompts, max_tokens)

@@ -163,11 +163,10 @@ def test_sp_for_qwen3_moe() -> None:
        vllm_model.generate(example_prompts, sampling_params)


-@pytest.mark.parametrize("enforce_eager", [True, False])
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE": "1"})
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_FLASHCOMM1": "1"})
-def test_models_distributed_Qwen_Dense_with_flashcomm_v1(model, enforce_eager):
+def test_models_distributed_Qwen_Dense_with_flashcomm_v1(model):
    example_prompts = [
        "Hello, my name is",
    ]
@@ -176,7 +175,7 @@ def test_models_distributed_Qwen_Dense_with_flashcomm_v1(model, enforce_eager):
    with VllmRunner(
            snapshot_download(model),
            max_model_len=8192,
-            enforce_eager=enforce_eager,
+            enforce_eager=False,
            dtype="auto",
            tensor_parallel_size=2,
            quantization="ascend",
@@ -184,12 +183,10 @@ def test_models_distributed_Qwen_Dense_with_flashcomm_v1(model, enforce_eager):
        vllm_model.generate_greedy(example_prompts, max_tokens)


-@pytest.mark.parametrize("enforce_eager", [True, False])
@pytest.mark.parametrize("model", QWEN_DENSE_MODELS)
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_DENSE_OPTIMIZE": "1"})
@patch.dict(os.environ, {"VLLM_ASCEND_ENABLE_PREFETCH_MLP": "1"})
-def test_models_distributed_Qwen_Dense_with_prefetch_mlp_weight(
-        model, enforce_eager):
+def test_models_distributed_Qwen_Dense_with_prefetch_mlp_weight(model):
    example_prompts = [
        "Hello, my name is",
    ]
@@ -198,7 +195,7 @@ def test_models_distributed_Qwen_Dense_with_prefetch_mlp_weight(
    with VllmRunner(
            snapshot_download(model),
            max_model_len=8192,
-            enforce_eager=enforce_eager,
+            enforce_eager=False,
            dtype="auto",
            tensor_parallel_size=2,
            quantization="ascend",