[MoE] [Refactor] Combine common_fused_moe and fused_moe (#3176)

### What this PR does / why we need it? 1. Move additional functionalities from fused_moe.py to common_fused_moe.py and remove fused_moe.py 2. Remove unnecessary custom classes from qwen3_moe.py, and it will be completely removed after we release vllm-ascend v0.11.0 ### Does this PR introduce _any_ user-facing change? No ### How was this patch tested? Qwen3-30B-A3B/Qwen3-30B-A3B-W8A8/DeepSeek-V3-W4A8-Pruing/deepseek-mtp/pangu-pro-moe-pruing: 1. Enable/Disable EP 3. Aclgraph & eager 4. SP - vLLM version: v0.11.0 --------- Signed-off-by: Pr0Wh1teGivee <calvin_zhu0210@outlook.com> Co-authored-by: weijinqian0 <12153182+weijinqian0@users.noreply.github.com>
2025-10-09 14:12:46 +08:00
parent a36e3da78e
commit 94dd832815
17 changed files with 175 additions and 1110 deletions
--- a/tests/ut/models/conftest.py
+++ b/tests/ut/models/conftest.py
@@ -96,7 +96,7 @@ def mock_distributed():
            patch("vllm_ascend.models.deepseek_v2.get_pp_group", return_value=pp_group), \
            patch("vllm_ascend.models.deepseek_v2.get_pp_group",
                  return_value=Mock(is_first_rank=False, is_last_rank=False)), \
-            patch("vllm_ascend.ops.fused_moe.get_current_vllm_config", return_value=mock_vllm_config), \
+            patch("vllm_ascend.ops.common_fused_moe.get_current_vllm_config", return_value=mock_vllm_config), \
            patch("vllm_ascend.ops.moe.token_dispatcher.torch.distributed.get_rank", return_value=0), \
            patch("vllm_ascend.ops.moe.token_dispatcher.get_ascend_soc_version", return_value=None), \
            patch.dict("vllm.distributed.parallel_state.__dict__", _TP=tp_group, _EP=ep_group, _DP=dp_group,
--- a/tests/ut/models/test_qwen3_moe.py
+++ b/tests/ut/models/test_qwen3_moe.py
@@ -1,68 +0,0 @@
-#
-# Licensed under the Apache License, Version 2.0 (the "License");
-# you may not use this file except in compliance with the License.
-# You may obtain a copy of the License at
-#
-#     http://www.apache.org/licenses/LICENSE-2.0
-#
-# Unless required by applicable law or agreed to in writing, software
-# distributed under the License is distributed on an "AS IS" BASIS,
-# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
-# See the License for the specific language governing permissions and
-# limitations under the License.
-# This file is a part of the vllm-ascend project.
-#
-import math
-import unittest
-
-import torch
-
-from vllm_ascend.torchair.models.qwen3_moe import CustomQwen3MoeAttention
-
-
-class DummyRMSNorm:
-
-    def __init__(self, dim: int, eps: float = 1e-6):
-        self.dim = dim
-        self.eps = eps
-
-    def __call__(self, x):
-        mean_sq = x.pow(2).mean(dim=-1, keepdim=True)
-        denom = (mean_sq + self.eps).sqrt()
-        return x / denom
-
-
-class TestCustomQwen3MoeAttention(unittest.TestCase):
-
-    def setUp(self):
-        self.batch = 2
-        self.seq_len = 3
-        self.q_size = 8
-        self.kv_size = 8
-        self.head_dim = 4
-        self.rms_eps = 1e-6
-
-        total_dim = self.q_size + 2 * self.kv_size
-
-        self.qkv = torch.arange(self.batch * self.seq_len * total_dim,
-                                dtype=torch.float32).reshape(
-                                    self.batch, self.seq_len, total_dim)
-
-    def test_constant_input_normalization(self):
-        ones_qkv = torch.ones((1, 1, self.q_size + 2 * self.kv_size),
-                              dtype=torch.float32)
-
-        q_norm = DummyRMSNorm(self.head_dim, self.rms_eps)
-        k_norm = DummyRMSNorm(self.head_dim, self.rms_eps)
-        q, k, v = CustomQwen3MoeAttention.normalize_qkv(
-            ones_qkv, self.q_size, self.kv_size, self.head_dim, q_norm, k_norm)
-
-        norm_val = 1.0 / math.sqrt(1.0 + self.rms_eps)
-
-        expected_q = torch.full((1, 1, self.q_size), norm_val)
-        expected_k = torch.full((1, 1, self.kv_size), norm_val)
-        expected_v = torch.ones((1, 1, self.kv_size), dtype=torch.float32)
-
-        self.assertTrue(torch.allclose(q, expected_q, atol=1e-6))
-        self.assertTrue(torch.allclose(k, expected_k, atol=1e-6))
-        self.assertTrue(torch.equal(v, expected_v))