# # Copyright (c) 2025 Huawei Technologies Co., Ltd. All Rights Reserved. # # Licensed under the Apache License, Version 2.0 (the "License"); # you may not use this file except in compliance with the License. # You may obtain a copy of the License at # # http://www.apache.org/licenses/LICENSE-2.0 # # Unless required by applicable law or agreed to in writing, software # distributed under the License is distributed on an "AS IS" BASIS, # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. # See the License for the specific language governing permissions and # limitations under the License. # This file is a part of the vllm-ascend project. # from __future__ import annotations from typing import Any import torch import torch_npu from vllm.model_executor.layers.rotary_embedding import MRotaryEmbedding from vllm.model_executor.layers.rotary_embedding.common import ApplyRotaryEmb from vllm.model_executor.layers.rotary_embedding.mrope import apply_interleaved_rope from vllm_ascend.ops.rotary_embedding import AscendRotaryEmbedding, get_cos_and_sin_slice, update_cos_sin # Filled once per model forward in NPUModelRunner310._model_forward; read by every MRoPE layer. _mrope_cos_slice: torch.Tensor | None = None _mrope_sin_slice: torch.Tensor | None = None def _apply_rotary_mrope_torch( q_rot: torch.Tensor, k_rot: torch.Tensor, cos: torch.Tensor, sin: torch.Tensor, is_neox_style: bool, ) -> tuple[torch.Tensor, torch.Tensor]: """PyTorch path aligned with vLLM MRotaryEmbedding.forward_native -> ApplyRotaryEmb.""" half = cos.shape[-1] // 2 cos_h = cos[0, :, 0, :half].contiguous() sin_h = sin[0, :, 0, :half].contiguous() q_out = ApplyRotaryEmb.forward_static(q_rot[0], cos_h, sin_h, is_neox_style) k_out = ApplyRotaryEmb.forward_static(k_rot[0], cos_h, sin_h, is_neox_style) return q_out.unsqueeze(0), k_out.unsqueeze(0) def merge_mrope_cos_sin_for_apply( cos: torch.Tensor, sin: torch.Tensor, mrope_section: list[int], mrope_interleaved: bool, ) -> tuple[torch.Tensor, torch.Tensor]: if mrope_interleaved: return ( apply_interleaved_rope(cos, mrope_section), apply_interleaved_rope(sin, mrope_section), ) return ( torch.cat([m[i] for i, m in enumerate(cos.split(mrope_section, dim=-1))], dim=-1), torch.cat([m[i] for i, m in enumerate(sin.split(mrope_section, dim=-1))], dim=-1), ) def set_mrope_apply_rotary_slices( cos_sin_cache: torch.Tensor, positions: torch.Tensor, *, mrope_section: list[int] | None = None, mrope_interleaved: bool = False, capacity_tokens: int = 0, ) -> None: """Build cos/sin views for `npu_apply_rotary_pos_emb` from positions; must run once per forward before layers.""" global _mrope_cos_slice global _mrope_sin_slice assert positions.ndim in (1, 2), "M-RoPE positions must be [num_tokens] or [3, num_tokens]." cos_sin = cos_sin_cache[positions] cos, sin = cos_sin.chunk(2, dim=-1) if positions.ndim == 2: assert positions.shape[0] == 3, "MRoPE expects positions [3, num_tokens] (T/H/W)." assert mrope_section is not None cos, sin = merge_mrope_cos_sin_for_apply( cos, sin, list(mrope_section), mrope_interleaved, ) # `npu_apply_rotary_pos_emb` follows ApplyRotaryPosEmbV2 semantics: # q_embed = q * cos + rotate(q) * sin, where cos/sin have full rotary dim. # MRoPE merge above gives half-dim cos/sin, so expand to full dim here. cos = torch.cat((cos, cos), dim=-1) sin = torch.cat((sin, sin), dim=-1) num_tokens = positions.shape[-1] cos_view = cos.contiguous().view(1, num_tokens, 1, -1) sin_view = sin.contiguous().view(1, num_tokens, 1, -1) # Keep stable storage across forwards for graph replay. if _mrope_cos_slice is None or _mrope_sin_slice is None: capacity = capacity_tokens if capacity_tokens is not None else num_tokens if capacity < num_tokens: capacity = num_tokens _mrope_cos_slice = torch.empty( (1, capacity, 1, cos_view.shape[-1]), dtype=cos_view.dtype, device=cos_view.device, ) _mrope_sin_slice = torch.empty( (1, capacity, 1, sin_view.shape[-1]), dtype=sin_view.dtype, device=sin_view.device, ) _mrope_cos_slice[:, :num_tokens].copy_(cos_view) _mrope_sin_slice[:, :num_tokens].copy_(sin_view) def _rope_forward_oot( self, positions: torch.Tensor, query: torch.Tensor, key: torch.Tensor, is_neox_style: bool, offsets: torch.Tensor | None = None, ) -> tuple[torch.Tensor, torch.Tensor]: query_shape, key_shape = query.shape, key.shape if self.cos_sin_cache.device != query.device: self.cos_sin_cache = self.cos_sin_cache.to(query.device) if self.cos_sin_cache.dtype != query.dtype: self.cos_sin_cache = self.cos_sin_cache.to(query.dtype) # This flag should set to True when doing drafting. if getattr(self, "_is_drafting_update_enabled", False): update_cos_sin(positions) cos, sin = get_cos_and_sin_slice() if offsets is not None: raise NotImplementedError("Batched rotary embedding is currently not supported on NPU.") rotary_mode = "half" if is_neox_style else "interleave" if self.head_size == 128 and self.cos_sin_cache.shape[-1] == 128: query = query.contiguous().view(1, query.shape[0], -1, self.head_size) key = key.contiguous().view(1, key.shape[0], -1, self.head_size) query, key = torch_npu.npu_apply_rotary_pos_emb(query, key, cos, sin, rotary_mode=rotary_mode) elif self.rotary_dim < self.head_size: num_tokens = query.shape[0] query = query.view(num_tokens, -1, self.head_size) key = key.view(num_tokens, -1, self.head_size) q_rot = query[..., : self.rotary_dim] q_pass = query[..., self.rotary_dim :] k_rot = key[..., : self.rotary_dim] k_pass = key[..., self.rotary_dim :] if self.rotary_dim == 64: q_rot = q_rot.contiguous().view(1, num_tokens, -1, self.rotary_dim) k_rot = k_rot.contiguous().view(1, num_tokens, -1, self.rotary_dim) q_rot, k_rot = torch_npu.npu_apply_rotary_pos_emb(q_rot, k_rot, cos, sin, rotary_mode=rotary_mode) else: q_rot = q_rot.contiguous().view(num_tokens, -1) k_rot = k_rot.contiguous().view(num_tokens, -1) torch_npu._npu_rotary_embedding( positions, q_rot, k_rot, self.rotary_dim, self.cos_sin_cache, is_neox_style, ) q_rot = q_rot.view(num_tokens, -1, self.rotary_dim) k_rot = k_rot.view(num_tokens, -1, self.rotary_dim) query = torch.cat((q_rot, q_pass), dim=-1).reshape(query_shape) key = torch.cat((k_rot, k_pass), dim=-1).reshape(key_shape) else: query = query.contiguous().view(query.shape[0], -1) key = key.contiguous().view(key.shape[0], -1) torch_npu._npu_rotary_embedding( positions, query, key, self.head_size, self.cos_sin_cache, is_neox_style, ) return query.view(query_shape), key.view(key_shape) class AscendMRotaryEmbedding310(MRotaryEmbedding): def forward_oot( self, positions: torch.Tensor, query: torch.Tensor, key: torch.Tensor, ): query_shape, key_shape = query.shape, key.shape # MRoPE T/H/W layout is handled in `merge_mrope_cos_sin_for_apply` (mrope_interleaved). # Here `rotary_mode` matches vLLM ApplyRotaryEmb: half = neox chunk, interleave = GPT-J pairs. rotary_mode = "half" if self.is_neox_style else "interleave" num_tokens = query.shape[0] if _mrope_cos_slice is None or _mrope_sin_slice is None: raise RuntimeError( "MRoPE cos/sin slices are not initialized. Call set_mrope_apply_rotary_slices before forward." ) cos, sin = _mrope_cos_slice[:, :num_tokens], _mrope_sin_slice[:, :num_tokens] is_partial_rope = self.rotary_dim < self.head_size if is_partial_rope: query = query.view(num_tokens, -1, self.head_size) key = key.view(num_tokens, -1, self.head_size) q_pass = query[..., self.rotary_dim :] k_pass = key[..., self.rotary_dim :] q_rot = query[..., : self.rotary_dim].contiguous().view(1, num_tokens, -1, self.rotary_dim) k_rot = key[..., : self.rotary_dim].contiguous().view(1, num_tokens, -1, self.rotary_dim) else: q_rot = query.contiguous().view(1, num_tokens, -1, self.head_size) k_rot = key.contiguous().view(1, num_tokens, -1, self.head_size) # `npu_apply_rotary_pos_emb` only supports rotary_dim 64 or 128. use_npu_apply = self.rotary_dim in (64, 128) if use_npu_apply: q_rot, k_rot = torch_npu.npu_apply_rotary_pos_emb(q_rot, k_rot, cos, sin, rotary_mode=rotary_mode) else: q_rot, k_rot = _apply_rotary_mrope_torch(q_rot, k_rot, cos, sin, self.is_neox_style) if is_partial_rope: q_rot = q_rot.view(num_tokens, -1, self.rotary_dim) k_rot = k_rot.view(num_tokens, -1, self.rotary_dim) query = torch.cat((q_rot, q_pass), dim=-1).reshape(query_shape) key = torch.cat((k_rot, k_pass), dim=-1).reshape(key_shape) else: query = q_rot.view(query_shape) key = k_rot.view(key_shape) return query, key def prepare_mrope_cos_sin_slices_from_runner(runner: Any, positions: torch.Tensor) -> None: """Resolve MRoPE embedding from the runner and populate `_mrope_cos_slice` / `_mrope_sin_slice`.""" emb = getattr(runner, "_mrope_embedding", None) if emb is None: emb = next(module for module in runner.model.modules() if isinstance(module, AscendMRotaryEmbedding310)) runner._mrope_embedding = emb assert isinstance(emb, AscendMRotaryEmbedding310) set_mrope_apply_rotary_slices( emb.cos_sin_cache, positions, mrope_section=emb.mrope_section, mrope_interleaved=emb.mrope_interleaved, capacity_tokens=runner.max_num_tokens, ) class AscendRotaryEmbedding310(AscendRotaryEmbedding): _is_drafting_update_enabled: bool = False @classmethod def set_rope_position_flag_310p(cls, state: bool): cls._is_drafting_update_enabled = state def forward_oot( self, positions: torch.Tensor, query: torch.Tensor, key: torch.Tensor, offsets: torch.Tensor | None = None, is_neox_style_override: bool | None = None, ): is_neox_style = self.is_neox_style if is_neox_style_override is not None: is_neox_style = is_neox_style_override return _rope_forward_oot(self, positions, query, key, is_neox_style, offsets)