Refactor e2e CI (#2276)

Refactor E2E CI to make it clear and faster 1. remove some uesless e2e test 2. remove some uesless function 3. Make sure all test runs with VLLMRunner to avoid oom error 4. Make sure all ops test end with torch.empty_cache to avoid oom error 5. run the test one by one to avoid resource limit error - vLLM version: v0.10.1.1 - vLLM main: a344a5aa0a Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com>
2025-09-02 09:02:22 +08:00
parent 0df059f41a
commit fef18b60bc
41 changed files with 374 additions and 1757 deletions
--- a/tests/e2e/singlecard/spec_decode_v1/test_v1_spec_decode.py
+++ b/tests/e2e/singlecard/spec_decode_v1/test_v1_spec_decode.py
@@ -7,6 +7,8 @@ from typing import Any
 import pytest
 from vllm import LLM, SamplingParams

+from tests.e2e.conftest import VllmRunner
+

@pytest.fixture
 def test_prompts():
@@ -72,19 +74,16 @@ def test_ngram_correctness(
    ref_llm = LLM(model=model_name, max_model_len=1024, enforce_eager=True)
    ref_outputs = ref_llm.chat(test_prompts, sampling_config)
    del ref_llm
-
-    spec_llm = LLM(
-        model=model_name,
-        speculative_config={
-            "method": "ngram",
-            "prompt_lookup_max": 5,
-            "prompt_lookup_min": 3,
-            "num_speculative_tokens": 3,
-        },
-        max_model_len=1024,
-        enforce_eager=True,
-    )
-    spec_outputs = spec_llm.chat(test_prompts, sampling_config)
+    with VllmRunner(model_name,
+                    speculative_config={
+                        "method": "ngram",
+                        "prompt_lookup_max": 5,
+                        "prompt_lookup_min": 3,
+                        "num_speculative_tokens": 3,
+                    },
+                    max_model_len=1024,
+                    enforce_eager=True) as runner:
+        spec_outputs = runner.model.chat(test_prompts, sampling_config)
    matches = 0
    misses = 0
    for ref_output, spec_output in zip(ref_outputs, spec_outputs):
@@ -98,7 +97,6 @@ def test_ngram_correctness(
    # Heuristic: expect at least 70% of the prompts to match exactly
    # Upon failure, inspect the outputs to check for inaccuracy.
    assert matches > int(0.7 * len(ref_outputs))
-    del spec_llm


@pytest.mark.skipif(True, reason="oom in CI, fix me")
@@ -121,23 +119,24 @@ def test_eagle_correctness(
    del ref_llm

    spec_model_name = eagle3_model_name() if use_eagle3 else eagle_model_name()
-    spec_llm = LLM(
-        model=model_name,
-        trust_remote_code=True,
-        enable_chunked_prefill=True,
-        max_num_seqs=1,
-        max_num_batched_tokens=2048,
-        gpu_memory_utilization=0.6,
-        speculative_config={
-            "method": "eagle3" if use_eagle3 else "eagle",
-            "model": spec_model_name,
-            "num_speculative_tokens": 2,
-            "max_model_len": 128,
-        },
-        max_model_len=128,
-        enforce_eager=True,
-    )
-    spec_outputs = spec_llm.chat(test_prompts, sampling_config)
+    with VllmRunner(
+            model_name,
+            trust_remote_code=True,
+            enable_chunked_prefill=True,
+            max_num_seqs=1,
+            max_num_batched_tokens=2048,
+            gpu_memory_utilization=0.6,
+            speculative_config={
+                "method": "eagle3" if use_eagle3 else "eagle",
+                "model": spec_model_name,
+                "num_speculative_tokens": 2,
+                "max_model_len": 128,
+            },
+            max_model_len=128,
+            enforce_eager=True,
+    ) as runner:
+        spec_outputs = runner.model.chat(test_prompts, sampling_config)
+
    matches = 0
    misses = 0
    for ref_output, spec_output in zip(ref_outputs, spec_outputs):
@@ -151,4 +150,3 @@ def test_eagle_correctness(
    # Heuristic: expect at least 66% of the prompts to match exactly
    # Upon failure, inspect the outputs to check for inaccuracy.
    assert matches > int(0.66 * len(ref_outputs))
-    del spec_llm