data: port complete MoE + xllm layer call chains from upstream repos
MoE call chain from ds_vllm (vllm-project/vllm latest): ex_engine/moe/ — 20 files, 8736 lines - modular_kernel.py (1630 lines) — base classes for modular MoE - experts/fused_batched_moe.py (972 lines) — NaiveBatchedExperts - prepare_finalize/batched.py (171 lines) — token grouping by expert - topk_weight_and_reduce.py (176 lines) — scatter-add finalize - fused_moe.py (1740 lines) — main fused_moe dispatch - config.py (1407 lines) — FusedMoEQuantConfig - activation.py, utils.py, layer.py, etc. xllm layer code (jd-opensource/xllm): ex_engine/xllm_layers/ — 39 files, 5859 lines - ilu/fused_moe.cpp (797 lines) — production ixformer 7-step MoE pipeline - ilu/attention.cpp (189 lines) — paged_attention + flash_attn bridge - npu_torch/qwen3_gated_delta_net_base.cpp (576 lines) — GDN reference - common/rms_norm.cpp, rotary_embedding.cpp, activation.cpp, dense_mlp.cpp xllm ILU kernels — synced 10 files to upstream (diffs from prior edits) These are reference implementations, NOT hand-written. Source repos: vllm-project/vllm, jd-opensource/xllm
This commit is contained in:
@@ -122,32 +122,17 @@ def apply_moe_activation(
|
||||
|
||||
# Activations with gated multiplication (gate × activation(up))
|
||||
if activation == MoEActivation.SILU:
|
||||
# BI-V100: torch.ops._C.silu_and_mul not available
|
||||
# Use corex_attn_head_rms_norm pattern: try C++ first, fallback to PyTorch
|
||||
d = output.size(-1)
|
||||
gate = input[..., :d]
|
||||
up = input[..., d:]
|
||||
output.copy_(F.silu(gate) * up)
|
||||
torch.ops._C.silu_and_mul(output, input)
|
||||
elif activation == MoEActivation.GELU:
|
||||
d = output.size(-1)
|
||||
gate = input[..., :d]
|
||||
up = input[..., d:]
|
||||
output.copy_(F.gelu(gate) * up)
|
||||
torch.ops._C.gelu_and_mul(output, input)
|
||||
elif activation == MoEActivation.GELU_TANH:
|
||||
d = output.size(-1)
|
||||
gate = input[..., :d]
|
||||
up = input[..., d:]
|
||||
output.copy_(F.gelu(gate, approximate="tanh") * up)
|
||||
torch.ops._C.gelu_tanh_and_mul(output, input)
|
||||
elif activation == MoEActivation.SWIGLUOAI:
|
||||
d = output.size(-1)
|
||||
gate = input[..., :d]
|
||||
up = input[..., d:]
|
||||
output.copy_(F.silu(gate) * up)
|
||||
torch.ops._C.swigluoai_and_mul(output, input)
|
||||
elif activation == MoEActivation.SWIGLUSTEP:
|
||||
d = output.size(-1)
|
||||
gate = input[..., :d]
|
||||
up = input[..., d:]
|
||||
output.copy_(F.silu(gate) * up)
|
||||
from vllm.model_executor.layers.activation import swiglustep_and_mul_triton
|
||||
|
||||
swiglustep_and_mul_triton(output, input)
|
||||
|
||||
# Activations without gated multiplication
|
||||
elif activation == MoEActivation.SILU_NO_MUL:
|
||||
|
||||
Reference in New Issue
Block a user