MoE call chain from ds_vllm (vllm-project/vllm latest): ex_engine/moe/ — 20 files, 8736 lines - modular_kernel.py (1630 lines) — base classes for modular MoE - experts/fused_batched_moe.py (972 lines) — NaiveBatchedExperts - prepare_finalize/batched.py (171 lines) — token grouping by expert - topk_weight_and_reduce.py (176 lines) — scatter-add finalize - fused_moe.py (1740 lines) — main fused_moe dispatch - config.py (1407 lines) — FusedMoEQuantConfig - activation.py, utils.py, layer.py, etc. xllm layer code (jd-opensource/xllm): ex_engine/xllm_layers/ — 39 files, 5859 lines - ilu/fused_moe.cpp (797 lines) — production ixformer 7-step MoE pipeline - ilu/attention.cpp (189 lines) — paged_attention + flash_attn bridge - npu_torch/qwen3_gated_delta_net_base.cpp (576 lines) — GDN reference - common/rms_norm.cpp, rotary_embedding.cpp, activation.cpp, dense_mlp.cpp xllm ILU kernels — synced 10 files to upstream (diffs from prior edits) These are reference implementations, NOT hand-written. Source repos: vllm-project/vllm, jd-opensource/xllm
30 lines
1.1 KiB
Python
30 lines
1.1 KiB
Python
# SPDX-License-Identifier: Apache-2.0
|
|
# SPDX-FileCopyrightText: Copyright contributors to the vLLM project
|
|
|
|
from vllm.model_executor.layers.fused_moe.prepare_finalize.batched import (
|
|
BatchedPrepareAndFinalize,
|
|
)
|
|
from vllm.model_executor.layers.fused_moe.prepare_finalize.naive_dp_ep import (
|
|
MoEPrepareAndFinalizeNaiveDPEPModular,
|
|
MoEPrepareAndFinalizeNaiveDPEPMonolithic,
|
|
make_moe_prepare_and_finalize_naive_dp_ep,
|
|
)
|
|
from vllm.model_executor.layers.fused_moe.prepare_finalize.no_dp_ep import (
|
|
MoEPrepareAndFinalizeNoDPEPModular,
|
|
MoEPrepareAndFinalizeNoDPEPMonolithic,
|
|
make_moe_prepare_and_finalize_no_dp_ep,
|
|
)
|
|
|
|
__all__ = [
|
|
"BatchedPrepareAndFinalize",
|
|
"MoEPrepareAndFinalizeNaiveDPEPMonolithic",
|
|
"MoEPrepareAndFinalizeNaiveDPEPModular",
|
|
"make_moe_prepare_and_finalize_naive_dp_ep",
|
|
"MoEPrepareAndFinalizeNoDPEPMonolithic",
|
|
"MoEPrepareAndFinalizeNoDPEPModular",
|
|
"make_moe_prepare_and_finalize_no_dp_ep",
|
|
# deepep_ht, deepep_ll, and flashinfer_a2a are not
|
|
# imported here as they have optional dependencies (deep_ep, flashinfer).
|
|
# Import them directly from their modules as needed.
|
|
]
|