Refactor e2e CI (#2276)

Refactor E2E CI to make it clear and faster 1. remove some uesless e2e test 2. remove some uesless function 3. Make sure all test runs with VLLMRunner to avoid oom error 4. Make sure all ops test end with torch.empty_cache to avoid oom error 5. run the test one by one to avoid resource limit error - vLLM version: v0.10.1.1 - vLLM main: a344a5aa0a Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com>
2025-09-02 09:02:22 +08:00
parent 0df059f41a
commit fef18b60bc
41 changed files with 374 additions and 1757 deletions
--- a/tests/e2e/singlecard/ops/test_fused_moe.py
+++ b/tests/e2e/singlecard/ops/test_fused_moe.py
@@ -20,6 +20,7 @@
 Run `pytest tests/ops/test_fused_moe.py`.
 """

+import gc
 from unittest.mock import MagicMock, patch

 import pytest
@@ -173,7 +174,9 @@ def test_token_dispatcher_with_all_gather(
                               torch_output,
                               atol=4e-2,
                               rtol=1)
+    gc.collect()
    torch.npu.empty_cache()
+    torch.npu.reset_peak_memory_stats()


@pytest.mark.parametrize("m", [1, 33, 64])
@@ -247,6 +250,10 @@ def test_select_experts(
    assert topk_ids.dtype == torch.int32
    assert row_idx.shape == (m, topk)

+    gc.collect()
+    torch.npu.empty_cache()
+    torch.npu.reset_peak_memory_stats()
+

@pytest.mark.parametrize("device", DEVICE)
 def test_select_experts_invalid_scoring_func(device: str):
@@ -258,6 +265,9 @@ def test_select_experts_invalid_scoring_func(device: str):
                       use_grouped_topk=False,
                       renormalize=False,
                       scoring_func="invalid")
+    gc.collect()
+    torch.npu.empty_cache()
+    torch.npu.reset_peak_memory_stats()


@pytest.mark.parametrize("device", DEVICE)
@@ -269,3 +279,6 @@ def test_select_experts_missing_group_params(device: str):
                       use_grouped_topk=True,
                       renormalize=False,
                       scoring_func="softmax")
+    gc.collect()
+    torch.npu.empty_cache()
+    torch.npu.reset_peak_memory_stats()