[Bugfix][LoRA] Fix the issue when enable LoRA + tp + fully_sharded_loras (#6650)

### What this PR does / why we need it? Fix the issue #6143 . ### Does this PR introduce _any_ user-facing change? Allow to start the server with "--enable-lora && --fully-sharded-loras && --tensor_parallel_size 2". ### How was this patch tested? pytest -sv tests/e2e/multicard/2-cards/test_llama32_lora_tp2.py - vLLM version: v0.15.0 - vLLM main: d7e17aaacd --------- Signed-off-by: paulyu12 <507435917@qq.com> Co-authored-by: wangxiyuan <wangxiyuan1007@gmail.com>
2026-03-11 15:43:15 +08:00
parent a7f91fce71
commit 830f39dd70
4 changed files with 113 additions and 9 deletions
--- a/vllm_ascend/lora/utils.py
+++ b/vllm_ascend/lora/utils.py
@@ -4,13 +4,18 @@ from transformers import PretrainedConfig
 from vllm.config import LoRAConfig
 from vllm.lora.layers import (
    ColumnParallelLinearWithLoRA,
+    ColumnParallelLinearWithShardedLoRA,
    MergedColumnParallelLinearWithLoRA,
+    MergedColumnParallelLinearWithShardedLoRA,
    MergedQKVParallelLinearWithLoRA,
+    MergedQKVParallelLinearWithShardedLoRA,
    QKVParallelLinearWithLoRA,
+    QKVParallelLinearWithShardedLoRA,
    RowParallelLinearWithLoRA,
+    RowParallelLinearWithShardedLoRA,
    VocabParallelEmbeddingWithLoRA,
 )
-from vllm.lora.layers.utils import _not_fully_sharded_can_replace
+from vllm.lora.layers.utils import _fully_sharded_can_replace, _not_fully_sharded_can_replace

 from vllm_ascend.ops.linear import (
    AscendColumnParallelLinear,
@@ -23,6 +28,7 @@ from vllm_ascend.ops.vocab_parallel_embedding import AscendVocabParallelEmbeddin

 class AscendColumnParallelLinearWithLoRA(ColumnParallelLinearWithLoRA):
    @classmethod
+    @_not_fully_sharded_can_replace
    def can_replace_layer(
        cls,
        source_layer: nn.Module,
@@ -35,6 +41,7 @@ class AscendColumnParallelLinearWithLoRA(ColumnParallelLinearWithLoRA):

 class AscendMergedColumnParallelLinearWithLoRA(MergedColumnParallelLinearWithLoRA):
    @classmethod
+    @_not_fully_sharded_can_replace
    def can_replace_layer(
        cls,
        source_layer: nn.Module,
@@ -47,6 +54,7 @@ class AscendMergedColumnParallelLinearWithLoRA(MergedColumnParallelLinearWithLoR

 class AscendRowParallelLinearWithLoRA(RowParallelLinearWithLoRA):
    @classmethod
+    @_not_fully_sharded_can_replace
    def can_replace_layer(
        cls,
        source_layer: nn.Module,
@@ -95,6 +103,71 @@ class AscendMergedQKVParallelLinearWithLoRA(MergedQKVParallelLinearWithLoRA):
        return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 3


+class AscendColumnParallelLinearWithShardedLoRA(ColumnParallelLinearWithShardedLoRA):
+    @classmethod
+    @_fully_sharded_can_replace
+    def can_replace_layer(
+        cls,
+        source_layer: nn.Module,
+        lora_config: LoRAConfig,
+        packed_modules_list: list,
+        model_config: PretrainedConfig | None = None,
+    ) -> bool:
+        return type(source_layer) is AscendColumnParallelLinear
+
+
+class AscendMergedColumnParallelLinearWithShardedLoRA(MergedColumnParallelLinearWithShardedLoRA):
+    @classmethod
+    @_fully_sharded_can_replace
+    def can_replace_layer(
+        cls,
+        source_layer: nn.Module,
+        lora_config: LoRAConfig,
+        packed_modules_list: list,
+        model_config: PretrainedConfig | None = None,
+    ) -> bool:
+        return type(source_layer) is AscendMergedColumnParallelLinear
+
+
+class AscendMergedQKVParallelLinearWithShardedLoRA(MergedQKVParallelLinearWithShardedLoRA):
+    @classmethod
+    @_fully_sharded_can_replace
+    def can_replace_layer(
+        cls,
+        source_layer: nn.Module,
+        lora_config: LoRAConfig,
+        packed_modules_list: list,
+        model_config: PretrainedConfig | None = None,
+    ) -> bool:
+        return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 3
+
+
+class AscendQKVParallelLinearWithShardedLoRA(QKVParallelLinearWithShardedLoRA):
+    @classmethod
+    @_fully_sharded_can_replace
+    def can_replace_layer(
+        cls,
+        source_layer: nn.Module,
+        lora_config: LoRAConfig,
+        packed_modules_list: list,
+        model_config: PretrainedConfig | None = None,
+    ) -> bool:
+        return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 1
+
+
+class AscendRowParallelLinearWithShardedLoRA(RowParallelLinearWithShardedLoRA):
+    @classmethod
+    @_fully_sharded_can_replace
+    def can_replace_layer(
+        cls,
+        source_layer: nn.Module,
+        lora_config: LoRAConfig,
+        packed_modules_list: list,
+        model_config: PretrainedConfig | None = None,
+    ) -> bool:
+        return type(source_layer) is AscendRowParallelLinear
+
+
 def refresh_all_lora_classes():
    vllm.lora.utils._all_lora_classes.add(AscendColumnParallelLinearWithLoRA)
    vllm.lora.utils._all_lora_classes.add(AscendMergedColumnParallelLinearWithLoRA)
@@ -102,3 +175,8 @@ def refresh_all_lora_classes():
    vllm.lora.utils._all_lora_classes.add(AscendVocabParallelEmbeddingWithLoRA)
    vllm.lora.utils._all_lora_classes.add(AscendQKVParallelLinearWithLoRA)
    vllm.lora.utils._all_lora_classes.add(AscendMergedQKVParallelLinearWithLoRA)
+    vllm.lora.utils._all_lora_classes.add(AscendColumnParallelLinearWithShardedLoRA)
+    vllm.lora.utils._all_lora_classes.add(AscendMergedColumnParallelLinearWithShardedLoRA)
+    vllm.lora.utils._all_lora_classes.add(AscendMergedQKVParallelLinearWithShardedLoRA)
+    vllm.lora.utils._all_lora_classes.add(AscendQKVParallelLinearWithShardedLoRA)
+    vllm.lora.utils._all_lora_classes.add(AscendRowParallelLinearWithShardedLoRA)