Drop ascend scheduler (#4623)

It's safe to drop ascend scheduler now. The related test and doc has been removed already - vLLM version: v0.12.0 - vLLM main: ad32e3e19c Signed-off-by: wangxiyuan <wangxiyuan1007@gmail.com>
2025-12-05 09:03:45 +08:00
parent 00b4fb80de
commit ea54388e19
12 changed files with 34 additions and 767 deletions
--- a/vllm_ascend/platform.py
+++ b/vllm_ascend/platform.py
@@ -30,6 +30,7 @@ from vllm_ascend.ascend_config import (check_ascend_config, get_ascend_config,
                                       init_ascend_config)
 from vllm_ascend.torchair.utils import (check_torchair_cache_exist,
                                        delete_torchair_cache_file)
+from vllm_ascend.utils import refresh_block_size

 # isort: off
 from vllm_ascend.utils import (
@@ -160,7 +161,6 @@ class NPUPlatform(Platform):
        model_config = vllm_config.model_config
        parallel_config = vllm_config.parallel_config
        cache_config = vllm_config.cache_config
-        ascend_scheduler_config = ascend_config.ascend_scheduler_config
        ascend_compilation_config = ascend_config.ascend_compilation_config
        if ascend_compilation_config:
            vllm_config.additional_config.setdefault(
@@ -307,38 +307,13 @@ class NPUPlatform(Platform):
            else:
                parallel_config.worker_cls = "vllm_ascend.worker.worker_v1.NPUWorker"

-        if cache_config:
-            if cache_config.block_size is None:
-                cache_config.block_size = 128
-
-            if cache_config.enable_prefix_caching or \
-                not ascend_scheduler_config.enabled or \
-                getattr(ascend_scheduler_config, "enable_chunked_prefill", False):
-                logger.warning(
-                    "If chunked prefill or prefix caching is enabled, block size must be set to 128."
-                )
-                origin_block_size = cache_config.block_size
-                cache_config.block_size = 128
-                # TODO(MengqingCao): Remove the model_type check, after resolving the hidden error in get_kv_cache_groups.
-                if model_config and model_config.hf_config.model_type == "qwen3_next":
-                    logger.warning(
-                        "When running qwen3-next model, block_size needs to be restored to its original value."
-                    )
-                    cache_config.block_size = origin_block_size
+        refresh_block_size(vllm_config)

        # Activate custom ops for v1, except on 310P
        if get_ascend_device_type() != AscendDeviceType._310P:
            compilation_config.custom_ops = ["all"]

-        # If ascend_scheduler_config is enabled,
-        # extents original scheduler_config to use AscendScheduler.
-        if ascend_config.ascend_scheduler_config.enabled:
-            from vllm_ascend.core.schedule_config import AscendSchedulerConfig
-            ascend_scheduler_config = AscendSchedulerConfig.initialize_from_config(
-                vllm_config.scheduler_config,
-                ascend_config.ascend_scheduler_config)
-            vllm_config.scheduler_config = ascend_scheduler_config
-        elif ascend_config.recompute_scheduler_enable:
+        if ascend_config.recompute_scheduler_enable:
            from vllm_ascend.core.recompute_schedule_config import \
                RecomputeSchedulerConfig
            recompute_scheduler_config = RecomputeSchedulerConfig.initialize_from_config(