Main2main upgrade to vllm 0317 afternoon (#7409)

### What this PR does / why we need it? 1.fix "TypeError: get_attn_backend() remove variable": [Refactor `check_and_update_config`](https://github.com/vllm-project/vllm/pull/35122) 2.fix [Rename `compile_ranges_split_points` to `compile_ranges_endpoints`](https://github.com/vllm-project/vllm/pull/36027) 3.fix "RuntimeError: device_allocator not a DeviceAllocator":[Replace memory related torch.cuda APIs"](https://github.com/vllm-project/vllm/pull/37031) 4.fix [Support multiple KV groups in OffloadingSpec ](https://github.com/vllm-project/vllm/pull/36610) removed self.offloaded_block_size and changed self.gpu_block_size from a scalar to a tuple of per-group block sizes, adding block_size_factor. 5.fix [Consolidate SupportsEagle](https://github.com/vllm-project/vllm/pull/36063) renamed get_eagle3_aux_hidden_state_layers() to get_eagle3_default_aux_hidden_state_layers() and added a supports_eagle3() guard before calling it. ### Does this PR introduce _any_ user-facing change? NA ### How was this patch tested? E2E - vLLM version: v0.17.0 - vLLM main: 8a680463fa --------- Signed-off-by: leo-pony <nengjunma@outlook.com> Co-authored-by: Claude Code <noreply@anthropic.com>
2026-03-18 23:24:27 +08:00
parent 305820f1a9
commit 8b79d4de52
13 changed files with 125 additions and 41 deletions
--- a/vllm_ascend/ascend_config.py
+++ b/vllm_ascend/ascend_config.py
@@ -181,30 +181,47 @@ class AscendConfig:
                stacklevel=2,
            )

+    @staticmethod
+    def _get_compile_ranges(compilation_config):
+        from vllm_ascend.utils import vllm_version_is
+
+        if vllm_version_is("0.17.0"):
+            return compilation_config.compile_ranges_split_points
+        else:
+            return compilation_config.compile_ranges_endpoints
+
+    @staticmethod
+    def _set_compile_ranges(compilation_config, value):
+        from vllm_ascend.utils import vllm_version_is
+
+        if vllm_version_is("0.17.0"):
+            compilation_config.compile_ranges_split_points = value
+        else:
+            compilation_config.compile_ranges_endpoints = value
+
    def update_compile_ranges_split_points(self):
        vllm_config = self.vllm_config
        if self.ascend_compilation_config.enable_npugraph_ex:
            if self.ascend_compilation_config.fuse_allreduce_rms:
                from vllm_ascend.compilation.passes.allreduce_rmsnorm_fusion_pass import ALLREDUCE_NORM_FUSE_THRESHOLD

-                new_compile_ranges_split_points = vllm_config.compilation_config.compile_ranges_split_points
+                new_compile_ranges_split_points = self._get_compile_ranges(vllm_config.compilation_config)
                new_compile_ranges_split_points.append(ALLREDUCE_NORM_FUSE_THRESHOLD)
                new_compile_ranges_split_points = sorted(new_compile_ranges_split_points)
-                vllm_config.compilation_config.compile_ranges_split_points = new_compile_ranges_split_points
+                self._set_compile_ranges(vllm_config.compilation_config, new_compile_ranges_split_points)
                logger.debug(
                    "set compile_ranges_split_points to "
                    "{new_compile_ranges_split_points} for matmul and allreduce fusion"
                )

        else:
-            new_compile_ranges_split_points = vllm_config.compilation_config.compile_ranges_split_points
+            new_compile_ranges_split_points = self._get_compile_ranges(vllm_config.compilation_config)
            if vllm_config.additional_config.get("ascend_compilation_config", {}).get("fuse_allreduce_rms", True):
                from vllm_ascend.compilation.passes.allreduce_rmsnorm_fusion_pass import ALLREDUCE_NORM_FUSE_THRESHOLD

-                new_compile_ranges_split_points = vllm_config.compilation_config.compile_ranges_split_points
                new_compile_ranges_split_points.append(ALLREDUCE_NORM_FUSE_THRESHOLD)
                new_compile_ranges_split_points = sorted(new_compile_ranges_split_points)
-                vllm_config.compilation_config.compile_ranges_split_points = new_compile_ranges_split_points
+                self._set_compile_ranges(vllm_config.compilation_config, new_compile_ranges_split_points)
                logger.debug(
                    "set compile_ranges_split_points to "
                    "{new_compile_ranges_split_points} for matmul and allreduce fusion"
@@ -218,9 +235,9 @@ class AscendConfig:
                sp_threshold = get_sp_threshold(vllm_config)
                new_compile_ranges_split_points.append(sp_threshold)
                logger.debug(f"add {sp_threshold} to compile_ranges_split_points for sequence parallelism")
-            if len(new_compile_ranges_split_points) > len(vllm_config.compilation_config.compile_ranges_split_points):
+            if len(new_compile_ranges_split_points) > len(self._get_compile_ranges(vllm_config.compilation_config)):
                new_compile_ranges_split_points = sorted(new_compile_ranges_split_points)
-                vllm_config.compilation_config.compile_ranges_split_points = new_compile_ranges_split_points
+                self._set_compile_ranges(vllm_config.compilation_config, new_compile_ranges_split_points)


 class FinegrainedTPConfig: