init v0.23.0

Signed-off-by: Sun Ruoxi <sunruoxi@4paradigm.com>
This commit is contained in:
2026-08-27 15:11:51 +08:00
parent b582a8e7d1
commit 7f8a1b1f7a
2849 changed files with 712887 additions and 22001 deletions

View File

@@ -0,0 +1,222 @@
#
# Copyright (c) 2026 Huawei Technologies Co., Ltd. All Rights Reserved.
#
# Licensed under the Apache License, Version 2.0 (the "License");
# you may not use this file except in compliance with the License.
# You may obtain a copy of the License at
#
# http://www.apache.org/licenses/LICENSE-2.0
#
# Unless required by applicable law or agreed to in writing, software
# distributed under the License is distributed on an "AS IS" BASIS,
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
# See the License for the specific language governing permissions and
# limitations under the License.
#
"""Ascend MoE-LoRA wrapper (v1).
Design (see plan in conversation history):
- Inherits weight allocation / set_lora / slice helpers from upstream
FusedMoEWithLoRA. Only the injection mechanism differs: upstream wraps
Triton modular kernel internals (`TritonExperts.activation` / `moe_sum`),
which do not exist on Ascend. We instead wrap the per-layer
`quant_method.apply` and, inside it, temporarily swap the active
`MoECommMethod._apply_mlp` so the LoRA delta is added on permuted
activations between the grouped GMMs.
- Per-layer ownership is critical: `_MoECommMethods` is a module-level
singleton shared by all 48 MoE layers. If we wrapped `_apply_mlp` at
init time, layer N+1 would compose on top of layer N's wrapper and
every forward would stack all layers' LoRA deltas. We bracket the swap
inside `apply_wrapper` so only the active layer is in effect.
- v1 deliberately limits scope to: unquant + AllGather + TP-only +
no shared experts + no FusedMC2 + no dynamic EPLB. These are the exact
conditions under which `Qwen3-30B-A3B-Thinking-2507` runs cleanly with
TP=4 EP=1 on 4×64GB. Other paths assert early so users get a clear
error rather than silently wrong outputs.
"""
from __future__ import annotations
import torch
from torch import nn
from vllm import envs
from vllm.distributed.parallel_state import (
get_tensor_model_parallel_rank,
get_tensor_model_parallel_world_size,
)
from vllm.lora.layers.base import BaseLayerWithLoRA
from vllm.lora.layers.fused_moe import FusedMoE3DWithLoRA, FusedMoEWithLoRA
from vllm.lora.layers.utils import _get_lora_device
import vllm_ascend.envs as envs_ascend
def _assert_ascend_moe_lora_supported(base_layer: nn.Module) -> None:
if getattr(base_layer, "use_ep", False):
raise AssertionError(
"Ascend MoE LoRA v1 does not support expert parallelism. "
"Launch with `--enable-expert-parallel=false` and use TP only "
"(e.g. TP=4 for Qwen3-30B-A3B on 4x64GB)."
)
if getattr(base_layer, "dynamic_eplb", False):
raise AssertionError(
"Ascend MoE LoRA v1 is incompatible with dynamic EPLB "
"(expert migration would break the per-expert LoRA layout)."
)
if int(envs_ascend.VLLM_ASCEND_ENABLE_FUSED_MC2) != 0:
raise AssertionError(
"Ascend MoE LoRA v1 cannot patch FusedMC2 path "
"(dispatch_ffn_combine is a single fused C++ op). "
"Set VLLM_ASCEND_ENABLE_FUSED_MC2=0."
)
if getattr(base_layer, "_shared_experts", None) is not None:
raise AssertionError(
"Ascend MoE LoRA v1 does not wrap the shared_experts path "
"(it runs outside quant_method.apply). The target model "
"Qwen3-30B-A3B-Thinking-2507 has no shared experts; models "
"like DeepSeek-V3 are not yet supported."
)
if getattr(base_layer, "multistream_overlap_gate", False):
raise AssertionError(
"multistream_overlap_gate=True interleaves quant_method.apply "
"calls on multiple streams; the MoE LoRA path has not been "
"validated under this overlap. Disable it for MoE LoRA."
)
def _recover_moe_lora_routing(lora_context, expanded_row_idx, topk_ids):
"""Recover per-permuted-row (expert_id, lora_slot) for the dispatched rows.
npu_moe_init_routing semantics (verified empirically): ``expanded_row_idx``
is indexed by the ORIGINAL flat (token, k) position and gives where that
pair landed in the expert-sorted array -- not the reverse. So recovering
"which (token, k) pair does sorted row i hold" needs the inverse permutation
of ``expanded``, not a direct gather by it. ``argsort`` output shape ==
input shape (value-independent), so this stays graph-capturable -- no
``.item()``/data-dependent host sync.
"""
top_k = lora_context.top_k
expanded = torch.abs(expanded_row_idx)
inv_perm = torch.argsort(expanded)
expert_per_row = topk_ids.reshape(-1)[inv_perm].to(torch.long)
# token_lora_indices is a 1D LongTensor sized to max_num_batched_tokens
# (host-known constant). Clamping defensively to the last index is a no-op
# in normal operation but keeps the gather graph-safe.
orig_token = inv_perm // top_k
token_lora_indices = lora_context.punica_wrapper.token_lora_indices
orig_token = orig_token.clamp_(max=token_lora_indices.numel() - 1)
lora_per_row = token_lora_indices[orig_token]
return expert_per_row, lora_per_row
def moe_lora_apply_w13(lora_context, *, gate_up_out, hidden_states, expanded_row_idx, topk_ids):
"""Add the w13 LoRA delta into ``gate_up_out`` (in place), before activation.
Called from ``unquant_apply_mlp`` right after the base gate_up GMM. Returns
the recovered per-row routing so the w2 delta can reuse it.
"""
routing = _recover_moe_lora_routing(lora_context, expanded_row_idx, topk_ids)
expert_per_row, lora_per_row = routing
lora_context.punica_wrapper.add_lora_fused_moe(
y=gate_up_out,
x=hidden_states,
lora_a_stacked=lora_context.w13_lora_a_stacked,
lora_b_stacked=lora_context.w13_lora_b_stacked,
expert_ids=expert_per_row,
adapter_enabled=lora_context.adapter_enabled,
token_lora_mapping=lora_per_row,
)
return routing
def moe_lora_apply_w2(lora_context, *, down_out, silu_out, lora_routing):
"""Add the w2 LoRA delta into ``down_out`` (in place), after the down GMM.
Reuses the per-row routing computed by ``moe_lora_apply_w13``; ``silu_out``
is the activation output that fed the base down GMM.
"""
expert_per_row, lora_per_row = lora_routing
lora_context.punica_wrapper.add_lora_fused_moe(
y=down_out,
x=silu_out,
lora_a_stacked=lora_context.w2_lora_a_stacked,
lora_b_stacked=lora_context.w2_lora_b_stacked,
expert_ids=expert_per_row,
adapter_enabled=lora_context.adapter_enabled,
token_lora_mapping=lora_per_row,
)
class AscendFusedMoEWithLoRA(FusedMoEWithLoRA):
"""Ascend-native MoE-LoRA wrapper.
Reuses upstream weight allocation, set_lora, reset_lora, and slicing.
Instead of the GPU modular-kernel injection, it publishes a per-layer
``MoELoRAContext`` onto the base layer (``_ascend_moe_lora_context``).
The Ascend unquant MoE path threads that context through
``MoEFusedExpertsInput`` -> ``MoEMlpComputeInput`` and applies the LoRA
delta natively inside ``unquant_apply_mlp`` (see
``moe_lora_apply_w13`` / ``moe_lora_apply_w2`` below) -- no runtime
monkey-patch of ``comm._apply_mlp``.
"""
def __init__(self, base_layer: nn.Module) -> None:
# Skip FusedMoEWithLoRA.__init__: it immediately asserts Triton
# internals and calls _inject_lora_into_fused_moe which is GPU-only.
BaseLayerWithLoRA.__init__(self)
self.base_layer = base_layer
_assert_ascend_moe_lora_supported(base_layer)
self.tp_size = get_tensor_model_parallel_world_size()
self.tp_rank = get_tensor_model_parallel_rank()
self.device = _get_lora_device(base_layer)
self._enable_aux_cuda_stream = envs.VLLM_LORA_ENABLE_DUAL_STREAM
self.moe_config = base_layer.moe_config
self._w13_slices = 2 if base_layer.moe_config.is_act_and_mul else 1
# ------------------------------------------------------------------
# Mapping
# ------------------------------------------------------------------
def set_mapping(self, punica_wrapper):
# Upstream FusedMoEWithLoRA.set_mapping (vllm v0.22.0+) chains into
# ``self._moe_kernel.fused_experts.set_lora_context(...)``, but
# ``_moe_kernel`` is only set by the GPU modular-kernel path that we
# deliberately skip in __init__. We instead build the per-layer
# MoELoRAContext (now that punica_wrapper is available) and publish it
# on the module that ``AscendUnquantizedFusedMoEMethod.apply`` reads via
# ``getattr(layer, "_ascend_moe_lora_context", None)`` -- the base layer
# itself on 0.23.0, but ``base_layer.routed_experts`` on main (there the
# runner *is* the layer and it calls apply with ``layer=routed_experts``).
# The context holds stable references (the in-place-updated LoRA stacks,
# adapter_enabled and the punica wrapper), so building it once here is
# sufficient.
BaseLayerWithLoRA.set_mapping(self, punica_wrapper)
self.base_layer.set_lora_context(self._build_lora_context())
class AscendFusedMoE3DWithLoRA(AscendFusedMoEWithLoRA, FusedMoE3DWithLoRA):
"""For checkpoints that already fuse w1+w3 into a 3D weight (single slice)."""
def __init__(self, base_layer: nn.Module) -> None:
AscendFusedMoEWithLoRA.__init__(self, base_layer)
# Override: 3D MoE LoRA uses a single w13 slice.
self._w13_slices = 1
# ----------------------------------------------------------------------
# Upstream compatibility shim: vllm/lora/model_manager.py:create_dummy_lora
# branches on `module.__class__.__name__ == "FusedMoEWithLoRA"` (and the
# 3D variant). Without this override, our subclasses would skip the
# pack_moe path and hit the generic pack() fallback, which produces a
# flat list of N_experts * 3 sub-LoRAs -- `set_lora` then fails with
# "too many values to unpack (expected 3)".
#
# Overriding only __name__ keeps the actual class object distinct (so
# isinstance / type identity / debugging are unaffected) but lets the
# upstream string compare hit our objects.
# ----------------------------------------------------------------------
AscendFusedMoEWithLoRA.__name__ = "FusedMoEWithLoRA"
AscendFusedMoE3DWithLoRA.__name__ = "FusedMoE3DWithLoRA"

View File

@@ -16,11 +16,13 @@
import torch
def bgmv_shrink(inputs: torch.Tensor,
lora_a_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
scaling: float = 1.0):
def bgmv_shrink(
inputs: torch.Tensor,
lora_a_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
scaling: float = 1.0,
):
return torch.ops._C_ascend.bgmv_shrink(
inputs,
lora_a_weights,
@@ -30,11 +32,13 @@ def bgmv_shrink(inputs: torch.Tensor,
)
def bgmv_expand(inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
add_inputs: bool = True):
def bgmv_expand(
inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
add_inputs: bool = True,
):
return torch.ops._C_ascend.bgmv_expand(
inputs,
lora_b_weights,
@@ -45,16 +49,18 @@ def bgmv_expand(inputs: torch.Tensor,
)
def bgmv_expand_slice(inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
slice_offset: int,
slice_size: int,
add_inputs: bool = True):
return torch.ops._C_ascend.bgmv_expand(inputs, lora_b_weights,
lora_indices_tensor, output_tensor,
slice_offset, slice_size)
def bgmv_expand_slice(
inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
slice_offset: int,
slice_size: int,
add_inputs: bool = True,
):
return torch.ops._C_ascend.bgmv_expand(
inputs, lora_b_weights, lora_indices_tensor, output_tensor, slice_offset, slice_size
)
def sgmv_shrink(
@@ -69,21 +75,23 @@ def sgmv_shrink(
token_nums: int,
scaling: float,
):
return torch.ops._C_ascend.sgmv_shrink(inputs, lora_a_weights,
lora_indices_tensor, seq_len_tensor,
output_tensor, scaling)
return torch.ops._C_ascend.sgmv_shrink(
inputs, lora_a_weights, lora_indices_tensor, seq_len_tensor, output_tensor, scaling
)
def sgmv_expand(inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
b_seq_start_loc: torch.Tensor,
seq_len_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
batches: int,
max_seq_length: int,
token_nums: int,
add_inputs: bool = False):
def sgmv_expand(
inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
b_seq_start_loc: torch.Tensor,
seq_len_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
batches: int,
max_seq_length: int,
token_nums: int,
add_inputs: bool = False,
):
return torch.ops._C_ascend.sgmv_expand(
inputs,
lora_b_weights,
@@ -95,19 +103,20 @@ def sgmv_expand(inputs: torch.Tensor,
)
def sgmv_expand_slice(inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
b_seq_start_loc: torch.Tensor,
seq_len_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
batches: int,
max_seq_length: int,
token_nums: int,
slice_offset: int,
slice_size: int,
add_inputs: bool = False):
return torch.ops._C_ascend.sgmv_expand(inputs, lora_b_weights,
lora_indices_tensor, seq_len_tensor,
output_tensor, slice_offset,
slice_size)
def sgmv_expand_slice(
inputs: torch.Tensor,
lora_b_weights: torch.Tensor,
output_tensor: torch.Tensor,
b_seq_start_loc: torch.Tensor,
seq_len_tensor: torch.Tensor,
lora_indices_tensor: torch.Tensor,
batches: int,
max_seq_length: int,
token_nums: int,
slice_offset: int,
slice_size: int,
add_inputs: bool = False,
):
return torch.ops._C_ascend.sgmv_expand(
inputs, lora_b_weights, lora_indices_tensor, seq_len_tensor, output_tensor, slice_offset, slice_size
)

View File

@@ -1,23 +1,12 @@
# SPDX-License-Identifier: Apache-2.0
from typing import Callable, Optional, Tuple, Union
from collections.abc import Callable
import torch
from vllm_ascend.utils import is_310p
if is_310p():
from vllm.lora.ops.torch_ops import (bgmv_expand, bgmv_expand_slice,
bgmv_shrink, sgmv_expand,
sgmv_expand_slice, sgmv_shrink)
else:
from vllm_ascend.lora.lora_ops import (bgmv_expand, bgmv_expand_slice,
bgmv_shrink, sgmv_expand,
sgmv_expand_slice, sgmv_shrink)
from vllm.lora.punica_wrapper.punica_base import PunicaWrapperBase
from vllm_ascend.lora.utils import refresh_all_lora_classes
from vllm_ascend.utils import AscendDeviceType, get_ascend_device_type
# The platforms that are compatible with the PyTorch-native implementation can
@@ -29,11 +18,36 @@ class PunicaWrapperNPU(PunicaWrapperBase):
Multi-LoRA, and to provide the interface for the pytorch punica ops.
"""
def __init__(self, max_num_batched_tokens: int, max_batches: int,
device: Union[torch.device, str], **kwargs):
PunicaWrapperBase.__init__(self, max_num_batched_tokens, max_batches,
device)
def __init__(self, max_num_batched_tokens: int, max_batches: int, device: torch.device | str, **kwargs):
PunicaWrapperBase.__init__(self, max_num_batched_tokens, max_batches, device)
refresh_all_lora_classes()
self.lora_config = kwargs.get("lora_config")
if get_ascend_device_type() == AscendDeviceType._310P or (
self.lora_config is not None and self.lora_config.max_lora_rank >= 128
):
from vllm.lora.ops.torch_ops import (
bgmv_expand,
bgmv_expand_slice,
bgmv_shrink,
sgmv_expand,
sgmv_expand_slice,
sgmv_shrink,
)
else:
from vllm_ascend.lora.lora_ops import (
bgmv_expand,
bgmv_expand_slice,
bgmv_shrink,
sgmv_expand,
sgmv_expand_slice,
sgmv_shrink,
)
self.bgmv_expand = bgmv_expand
self.bgmv_expand_slice = bgmv_expand_slice
self.bgmv_shrink = bgmv_shrink
self.sgmv_expand = sgmv_expand
self.sgmv_expand_slice = sgmv_expand_slice
self.sgmv_shrink = sgmv_shrink
def _shrink_prefill(
self,
@@ -42,10 +56,10 @@ class PunicaWrapperNPU(PunicaWrapperBase):
w_t_all: torch.Tensor,
scale: float,
):
#No LoRA request, so return directly
# No LoRA request, so return directly
if self.no_lora:
return
sgmv_shrink(
self.sgmv_shrink(
x,
w_t_all,
y,
@@ -60,7 +74,7 @@ class PunicaWrapperNPU(PunicaWrapperBase):
w_t_all: torch.Tensor,
scale: float,
):
bgmv_shrink(x, w_t_all, y, self.token_lora_indices, scale)
self.bgmv_shrink(x, w_t_all, y, self._get_token_lora_indices(x), scale)
def _expand_prefill(
self,
@@ -69,10 +83,10 @@ class PunicaWrapperNPU(PunicaWrapperBase):
w_t_all: torch.Tensor,
add_inputs: bool,
):
#No LoRA request, so return directly
# No LoRA request, so return directly
if self.no_lora:
return
sgmv_expand(
self.sgmv_expand(
x,
w_t_all,
y,
@@ -87,7 +101,7 @@ class PunicaWrapperNPU(PunicaWrapperBase):
w_t_all: torch.Tensor,
add_inputs: bool,
):
bgmv_expand(x, w_t_all, y, self.token_lora_indices, add_inputs)
self.bgmv_expand(x, w_t_all, y, self._get_token_lora_indices(x), add_inputs)
def _expand_slice_prefill(
self,
@@ -98,10 +112,10 @@ class PunicaWrapperNPU(PunicaWrapperBase):
y_slice_size: int,
add_inputs: bool,
):
#No LoRA request, so return directly
# No LoRA request, so return directly
if self.no_lora:
return
sgmv_expand_slice(
self.sgmv_expand_slice(
x,
w_t_all,
y,
@@ -120,8 +134,18 @@ class PunicaWrapperNPU(PunicaWrapperBase):
y_slice_size: int,
add_inputs: bool,
):
bgmv_expand_slice(x, w_t_all, y, self.token_lora_indices, y_offset,
y_slice_size, add_inputs)
self.bgmv_expand_slice(
x,
w_t_all,
y,
self._get_token_lora_indices(x),
y_offset,
y_slice_size,
add_inputs,
)
def _get_token_lora_indices(self, x: torch.Tensor) -> torch.Tensor:
return torch.narrow(self._token_lora_indices, 0, 0, x.size(0))
def _apply_expand(
self,
@@ -138,13 +162,10 @@ class PunicaWrapperNPU(PunicaWrapperBase):
GEMM of lora'b.
"""
expand_slice_fun: Callable = (self._expand_slice_prefill
if self.is_prefill else
self._expand_slice_decode)
expand_slice_fun: Callable = self._expand_slice_prefill if self.is_prefill else self._expand_slice_decode
expand_slice_fun(y, x, w_t_all, y_offset, y_slice_size, add_inputs)
def _apply_shrink(self, y: torch.Tensor, x: torch.Tensor,
w_t_all: torch.Tensor, scale: float):
def _apply_shrink(self, y: torch.Tensor, x: torch.Tensor, w_t_all: torch.Tensor, scale: float):
"""
Perform the ` y+=x@w_t_all` computation, which is suitable for the
GEMM of lora'a.
@@ -155,14 +176,18 @@ class PunicaWrapperNPU(PunicaWrapperBase):
"""
y_org = y
y = y.view(-1, y.shape[-1])
shrink_fun: Callable = (self._shrink_prefill
if self.is_prefill else self._shrink_decode)
shrink_fun: Callable = self._shrink_prefill if self.is_prefill else self._shrink_decode
shrink_fun(y, x, w_t_all, scale)
y = y.view_as(y_org)
def add_shrink(self, y: Union[Tuple[torch.Tensor, ...], torch.Tensor],
x: torch.Tensor, lora_a_stacked: Tuple[torch.Tensor, ...],
scale: float, **kwargs):
def add_shrink(
self,
y: tuple[torch.Tensor, ...] | torch.Tensor,
x: torch.Tensor,
lora_a_stacked: tuple[torch.Tensor, ...],
scale: float,
**kwargs,
):
"""
Performs GEMM for multiple slices of lora_a.
When `is_prefill is` true, it indicates that it is currently the
@@ -184,43 +209,38 @@ class PunicaWrapperNPU(PunicaWrapperBase):
x = x.view(-1, x.shape[-1])
# TODO fuse these kernels
for slice_idx in range(len(lora_a_stacked)):
self._apply_shrink(y[slice_idx], x, lora_a_stacked[slice_idx],
scale)
self._apply_shrink(y[slice_idx], x, lora_a_stacked[slice_idx], scale)
def add_expand(self,
y: torch.Tensor,
x: Union[Tuple[torch.Tensor, ...], torch.Tensor],
lora_b_stacked: Tuple[torch.Tensor, ...],
lora_bias_stacked: Optional[Tuple[torch.Tensor, ...]],
output_slices: Tuple[int, ...],
offset_start: int = 0,
add_inputs=True,
**kwargs) -> None:
def add_expand(
self,
y: torch.Tensor,
x: tuple[torch.Tensor, ...] | torch.Tensor,
lora_b_stacked: tuple[torch.Tensor, ...],
output_slices: tuple[int, ...],
offset_start: int = 0,
add_inputs=True,
**kwargs,
) -> None:
"""
Performs GEMM and bias addition for multiple slices of lora_b.
Semantics:
for i in range(len(lora_b_stacked)):
slice = output_slices[i]
y[:, offset:offset+slice] += x[i] @ lora_b_stacked[i] +
lora_bias_stacked[i]
y[:, offset:offset+slice] += x[i] @ lora_b_stacked[i]
offset += slice
Args:
y (torch.Tensor): Output tensor.
x (Union[Tuple[torch.Tensor, ...], torch.Tensor]): Input tensors
lora_b_stacked (Tuple[torch.Tensor, ...]): lora_b's weight
lora_bias_stacked (Optional[Tuple[torch.Tensor, ...]]):
bias's weight
output_slices (Tuple[int, ...]): Every slice's size
offset_start (int): The starting position of y, defaults to 0
add_inputs (bool): Defaults to True.
"""
y_org = y
y = y.view(-1, y.shape[-1])
offset_left = offset_start
if lora_bias_stacked is not None:
self._apply_bias(self.token_lora_indices, y, output_slices,
lora_bias_stacked)
for slice_idx in range(len(lora_b_stacked)):
self._apply_expand(
y,
@@ -233,12 +253,9 @@ class PunicaWrapperNPU(PunicaWrapperBase):
offset_left += output_slices[slice_idx]
y = y.view_as(y_org)
def add_lora_embedding(self,
y: torch.Tensor,
x: torch.Tensor,
lora_b_stacked: torch.Tensor,
add_inputs: bool = True,
**kwargs) -> None:
def add_lora_embedding(
self, y: torch.Tensor, x: torch.Tensor, lora_b_stacked: torch.Tensor, add_inputs: bool = True, **kwargs
) -> None:
"""
Applies lora specifically for VocabParallelEmbeddingWithLoRA.
@@ -253,30 +270,31 @@ class PunicaWrapperNPU(PunicaWrapperBase):
"""
# Embedding layer only need expand op
expand_fun: Callable = (self._expand_prefill
if self.is_prefill else self._expand_decode)
expand_fun: Callable = self._expand_prefill if self.is_prefill else self._expand_decode
x = x.to(torch.float32)
expand_fun(y, x, lora_b_stacked, add_inputs)
def add_lora_linear(self,
y: torch.Tensor,
x: torch.Tensor,
lora_a_stacked: Tuple[torch.Tensor, ...],
lora_b_stacked: Tuple[torch.Tensor, ...],
lora_bias_stacked: Optional[Tuple[torch.Tensor, ...]],
scale: float,
output_slices: Tuple[int, ...],
*,
buffer: Optional[Tuple[torch.Tensor, ...]] = None,
**kwargs) -> None:
def add_lora_linear(
self,
y: torch.Tensor,
x: torch.Tensor,
lora_a_stacked: tuple[torch.Tensor, ...],
lora_b_stacked: tuple[torch.Tensor, ...],
scale: float,
output_slices: tuple[int, ...],
*,
buffer: tuple[torch.Tensor, ...] | None = None,
**kwargs,
) -> None:
"""
Applicable to linear-related lora.
Semantics:
for i in range(len(lora_a_stacked)):
y[i] += (
x[i].unsqueeze(0)
@ lora_a_stacked[indices[i], layer_idx, :, :]
@ lora_b_stacked[indices[i], layer_idx, :, :]
x[i].unsqueeze(0) @ lora_a_stacked[
indices[i], layer_idx, :, :] @ lora_b_stacked[
indices[i], layer_idx, :, :]
* scale
).squeeze(0)+lora_bias_stacked[i]
@@ -292,37 +310,116 @@ class PunicaWrapperNPU(PunicaWrapperBase):
"""
assert len(lora_a_stacked) == len(lora_b_stacked) == len(output_slices)
if lora_bias_stacked is not None:
assert len(lora_bias_stacked) == len(output_slices)
y = self._apply_bias(self.token_lora_indices, y, output_slices,
lora_bias_stacked)
if buffer is None:
r = lora_b_stacked[0].size(-1)
# We set the buffer to be float32 by default, consistent with the
# triton op
buffer = tuple(
torch.zeros(
(x.size(0), r), dtype=torch.float32, device=x.device)
for _ in range(len(output_slices)))
torch.zeros((x.size(0), r), dtype=torch.float32, device=x.device) for _ in range(len(output_slices))
)
self.add_shrink(buffer, x, lora_a_stacked, scale, **kwargs)
self.add_expand(y,
buffer,
lora_b_stacked,
None,
output_slices,
add_inputs=True,
**kwargs)
self.add_expand(y, buffer, lora_b_stacked, output_slices, add_inputs=True, **kwargs)
def add_lora_logits(self,
y: torch.Tensor,
x: torch.Tensor,
lora_a_stacked: torch.Tensor,
lora_b_stacked: torch.Tensor,
scale,
*,
buffer: Optional[torch.Tensor] = None,
**kwargs) -> None:
def add_lora_fused_moe(
self,
y: torch.Tensor,
x: torch.Tensor,
lora_a_stacked: tuple[torch.Tensor, ...],
lora_b_stacked: tuple[torch.Tensor, ...],
*,
topk_weights: torch.Tensor | None = None,
sorted_token_ids: torch.Tensor | None = None,
expert_ids: torch.Tensor,
num_tokens_post_padded: torch.Tensor | None = None,
max_lora_rank: int = 0,
top_k_num: int = 1,
shrink_config=None,
expand_config=None,
adapter_enabled: torch.Tensor,
mul_routed_weight: bool = False,
fully_sharded: bool = False,
offset: int = 0,
token_lora_mapping: torch.Tensor | None = None,
) -> None:
"""
Ascend-native fused MoE LoRA (v2): static-shape per-row gather via the
same bgmv_shrink/bgmv_expand AscendC kernels (csrc/kernels/bgmv_*.cpp)
used by the dense Linear LoRA layers, instead of grouping rows by a
data-dependent ``torch.unique`` over active LoRA ids. The previous
``torch.unique``/``nonzero`` version produced output whose *shape*
depended on tensor values, which ACL Graph capture cannot record
(it failed with an `aclnnUnique2` error as soon as `enforce_eager`
was turned off) -- every tensor below has a shape that depends only
on input shapes, never on values, so this stays graph-capturable.
Rows are already one-token-per-row (top_k_num=1). Each row needs the
LoRA slot for (lora_id, expert_id), so we fold both into a single
gather index into a ``[max_loras * num_experts, ...]`` view of the
existing per-(lora, expert) weight stacks:
combined_idx[row] = lora_id[row] * num_experts + expert_id[row]
or -1 when the row has no active adapter, mirroring the -1 sentinel
``PunicaWrapperBase.token_lora_indices`` already uses. bgmv_shrink/
bgmv_expand skip any row whose index is negative (leaving the
zero-initialized shrink buffer / unmodified ``y`` in place), so
inactive rows get a zero delta for free -- no Python-level branching
needed.
"""
del sorted_token_ids, num_tokens_post_padded, max_lora_rank
del shrink_config, expand_config, fully_sharded
assert top_k_num == 1, "Ascend MoE LoRA v1 expects pre-expanded rows (top_k_num=1)."
if token_lora_mapping is None:
token_lora_mapping = self.token_lora_indices
x2d = x.view(-1, x.shape[-1])
y2d = y.view(-1, y.shape[-1])
expert_idx = expert_ids.view(-1).to(torch.long)
num_experts = lora_a_stacked[0].shape[1]
lora_idx_safe = token_lora_mapping.clamp(min=0)
enabled = (token_lora_mapping >= 0) & adapter_enabled[lora_idx_safe].bool()
combined_idx = torch.where(
enabled,
lora_idx_safe * num_experts + expert_idx,
torch.full_like(token_lora_mapping, -1),
).contiguous()
# bgmv_shrink writes fp32 (its Y_T); bgmv_expand reads fp32 (its X_T),
# so the shrink buffer is fp32.
rank = lora_a_stacked[0].shape[-2]
shrink_out = torch.zeros((x2d.shape[0], rank), dtype=torch.float32, device=x2d.device)
cur_offset = offset
for slice_idx in range(len(lora_a_stacked)):
# lora_a_stacked[s]/lora_b_stacked[s]: [max_loras, num_experts, rank, *].
# Flattening the leading two dims turns "gather by (lora, expert)"
# into "the plain per-row gather" to reuse bgmv_shrink/bgmv_expand.
a = lora_a_stacked[slice_idx]
b = lora_b_stacked[slice_idx]
out_size = b.shape[-2]
a_flat = a.view(-1, rank, a.shape[-1])
b_flat = b.view(-1, out_size, rank)
self.bgmv_shrink(x2d, a_flat, shrink_out, combined_idx, 1.0)
delta = shrink_out
if mul_routed_weight and topk_weights is not None:
delta = shrink_out * topk_weights.view(-1, 1)
self.bgmv_expand_slice(delta, b_flat, y2d, combined_idx, cur_offset, out_size, add_inputs=True)
cur_offset += out_size
def add_lora_logits(
self,
y: torch.Tensor,
x: torch.Tensor,
lora_a_stacked: torch.Tensor,
lora_b_stacked: torch.Tensor,
scale,
*,
buffer: torch.Tensor | None = None,
**kwargs,
) -> None:
"""
Applies lora specifically for LogitsProcessorWithLoRA.
@@ -344,13 +441,11 @@ class PunicaWrapperNPU(PunicaWrapperBase):
r = lora_b_stacked.size(-1)
if buffer is None:
buffer = torch.zeros((x.size(0), r),
dtype=torch.float32,
device=x.device)
buffer = torch.zeros((x.size(0), r), dtype=torch.float32, device=x.device)
indices = self.sampler_indices
indices = torch.narrow(self._sampler_indices, 0, 0, x.size(0))
bgmv_shrink(x, lora_a_stacked, buffer, indices, scale)
bgmv_expand(buffer, lora_b_stacked, y, indices, add_inputs=True)
self.bgmv_shrink(x, lora_a_stacked, buffer, indices, scale)
self.bgmv_expand(buffer, lora_b_stacked, y, indices, add_inputs=True)
y = y.view_as(y_org)

View File

@@ -1,91 +1,25 @@
from typing import Optional
import vllm
from torch import nn
from transformers import PretrainedConfig
from vllm.config import LoRAConfig
from vllm.lora.layers import (ColumnParallelLinearWithLoRA,
MergedColumnParallelLinearWithLoRA,
MergedQKVParallelLinearWithLoRA,
QKVParallelLinearWithLoRA,
RowParallelLinearWithLoRA,
VocabParallelEmbeddingWithLoRA)
from vllm.lora.layers.utils import _not_fully_sharded_can_replace
from vllm.lora.layers import (
MergedQKVParallelLinearWithLoRA,
MergedQKVParallelLinearWithShardedLoRA,
QKVParallelLinearWithLoRA,
QKVParallelLinearWithShardedLoRA,
)
from vllm.lora.layers.utils import _fully_sharded_can_replace, _not_fully_sharded_can_replace
from vllm_ascend.ops.linear import (AscendColumnParallelLinear,
AscendMergedColumnParallelLinear,
AscendQKVParallelLinear,
AscendRowParallelLinear)
from vllm_ascend.ops.vocab_parallel_embedding import \
AscendVocabParallelEmbedding
class AscendColumnParallelLinearWithLoRA(ColumnParallelLinearWithLoRA):
@classmethod
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: Optional[PretrainedConfig],
) -> bool:
return type(source_layer) is AscendColumnParallelLinear
class AscendMergedColumnParallelLinearWithLoRA(
MergedColumnParallelLinearWithLoRA):
@classmethod
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: Optional[PretrainedConfig],
) -> bool:
return type(source_layer) is AscendMergedColumnParallelLinear
class AscendRowParallelLinearWithLoRA(RowParallelLinearWithLoRA):
@classmethod
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: Optional[PretrainedConfig],
) -> bool:
return type(source_layer) is AscendRowParallelLinear
class AscendVocabParallelEmbeddingWithLoRA(VocabParallelEmbeddingWithLoRA):
@classmethod
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: Optional[PretrainedConfig],
) -> bool:
return type(source_layer) is AscendVocabParallelEmbedding
from vllm_ascend.lora.fused_moe import (
AscendFusedMoE3DWithLoRA,
AscendFusedMoEWithLoRA,
)
from vllm_ascend.ops.linear import (
AscendQKVParallelLinear,
)
class AscendQKVParallelLinearWithLoRA(QKVParallelLinearWithLoRA):
@classmethod
@_not_fully_sharded_can_replace
def can_replace_layer(cls, source_layer: nn.Module,
lora_config: LoRAConfig, packed_modules_list: list,
model_config: Optional[PretrainedConfig]) -> bool:
return type(source_layer) is AscendQKVParallelLinear and len(
packed_modules_list) == 1
class AscendMergedQKVParallelLinearWithLoRA(MergedQKVParallelLinearWithLoRA):
@classmethod
@_not_fully_sharded_can_replace
def can_replace_layer(
@@ -93,18 +27,62 @@ class AscendMergedQKVParallelLinearWithLoRA(MergedQKVParallelLinearWithLoRA):
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: Optional[PretrainedConfig],
model_config: PretrainedConfig | None,
) -> bool:
return (type(source_layer) is AscendQKVParallelLinear
and len(packed_modules_list) == 3)
return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 1
class AscendMergedQKVParallelLinearWithLoRA(MergedQKVParallelLinearWithLoRA):
@classmethod
@_not_fully_sharded_can_replace
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: PretrainedConfig | None,
) -> bool:
return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 3
class AscendMergedQKVParallelLinearWithShardedLoRA(MergedQKVParallelLinearWithShardedLoRA):
@classmethod
@_fully_sharded_can_replace
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: PretrainedConfig | None = None,
) -> bool:
return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 3
class AscendQKVParallelLinearWithShardedLoRA(QKVParallelLinearWithShardedLoRA):
@classmethod
@_fully_sharded_can_replace
def can_replace_layer(
cls,
source_layer: nn.Module,
lora_config: LoRAConfig,
packed_modules_list: list,
model_config: PretrainedConfig | None = None,
) -> bool:
return type(source_layer) is AscendQKVParallelLinear and len(packed_modules_list) == 1
def refresh_all_lora_classes():
vllm.lora.utils._all_lora_classes.add(AscendColumnParallelLinearWithLoRA)
vllm.lora.utils._all_lora_classes.add(
AscendMergedColumnParallelLinearWithLoRA)
vllm.lora.utils._all_lora_classes.add(AscendRowParallelLinearWithLoRA)
vllm.lora.utils._all_lora_classes.add(AscendVocabParallelEmbeddingWithLoRA)
vllm.lora.utils._all_lora_classes.add(AscendQKVParallelLinearWithLoRA)
vllm.lora.utils._all_lora_classes.add(
AscendMergedQKVParallelLinearWithLoRA)
ascend_classes = (
AscendQKVParallelLinearWithLoRA,
AscendMergedQKVParallelLinearWithLoRA,
AscendMergedQKVParallelLinearWithShardedLoRA,
AscendQKVParallelLinearWithShardedLoRA,
AscendFusedMoEWithLoRA,
AscendFusedMoE3DWithLoRA,
)
# vLLM #35077 changed _all_lora_classes from set to ordered tuple.
# Append the Ascend classes in a deterministic order.
vllm.lora.utils._all_lora_classes = (
*ascend_classes,
*vllm.lora.utils._all_lora_classes,
)