[feat] baseline4 profiling
This commit is contained in:
@@ -35,15 +35,25 @@ command:
|
|||||||
env:
|
env:
|
||||||
- name: VLLM_ENGINE_ITERATION_TIMEOUT_S
|
- name: VLLM_ENGINE_ITERATION_TIMEOUT_S
|
||||||
value: 3600
|
value: 3600
|
||||||
|
# --- MoE kernel selection ---
|
||||||
- name: BI100_MOE_COREX_DIRECT_ROUTED
|
- name: BI100_MOE_COREX_DIRECT_ROUTED
|
||||||
value: 1
|
value: '1'
|
||||||
|
- name: BI100_MOE_COREX_TOPK_SOFTMAX
|
||||||
|
value: '1'
|
||||||
|
# --- GDN kernel selection ---
|
||||||
- name: BI100_GDN_COREX_PACKED_DECODE
|
- name: BI100_GDN_COREX_PACKED_DECODE
|
||||||
value: 1
|
value: '1'
|
||||||
|
- name: BI100_GDN_COMBINED_QK_NORM
|
||||||
|
value: '1'
|
||||||
|
# --- Hybrid KV/GDN cache ---
|
||||||
- name: BI100_HYBRID_KV_ACCOUNTING
|
- name: BI100_HYBRID_KV_ACCOUNTING
|
||||||
value: full_attention
|
value: full_attention
|
||||||
- name: BI100_GDN_CACHE_POLICY
|
- name: BI100_GDN_CACHE_POLICY
|
||||||
value: admission64
|
value: admission64
|
||||||
- name: BI100_GDN_RESTORE_MODE
|
- name: BI100_GDN_RESTORE_MODE
|
||||||
value: hybrid64
|
value: hybrid64
|
||||||
- name: BI100_MOE_COREX_TOPK_SOFTMAX
|
# --- Profiling (event mode: no sync overhead) ---
|
||||||
|
- name: BI100_PROFILE
|
||||||
value: '1'
|
value: '1'
|
||||||
|
- name: BI100_PROFILE_MODE
|
||||||
|
value: event
|
||||||
@@ -1704,8 +1704,13 @@ class Qwen3_5MoeSparseBlock(nn.Module):
|
|||||||
# Tier 0: Full fused MoE via ix_moe_bridge (xllm 7-step pipeline)
|
# Tier 0: Full fused MoE via ix_moe_bridge (xllm 7-step pipeline)
|
||||||
# topk → gen_idx → expand → group_gemm → silu → group_gemm → combine
|
# topk → gen_idx → expand → group_gemm → silu → group_gemm → combine
|
||||||
# Source: xllm/core/layers/ilu/fused_moe.cpp
|
# Source: xllm/core/layers/ilu/fused_moe.cpp
|
||||||
|
# NOTE: Only use for prefill (T>1). For decode (T=1), group_gemm
|
||||||
|
# does 8× M=1 GEMMs that are completely memory-bound (<5% GPU util).
|
||||||
|
# The Tier 1 T=1 path below uses corex_moe_direct_routed or
|
||||||
|
# corex_batched_gemm.moe_decode_fused, which are purpose-built
|
||||||
|
# fused kernels for single-token MoE dispatch.
|
||||||
# ---------------------------------------------------------------
|
# ---------------------------------------------------------------
|
||||||
if _USE_IX_FUSED_MOE:
|
if _USE_IX_FUSED_MOE and hidden_states.shape[0] > 1:
|
||||||
w13 = self.experts.w13_weight # (E, 2*I, H)
|
w13 = self.experts.w13_weight # (E, 2*I, H)
|
||||||
w2 = self.experts.w2_weight # (E, H, I)
|
w2 = self.experts.w2_weight # (E, H, I)
|
||||||
return _ix_fused_moe.fused_moe_forward(
|
return _ix_fused_moe.fused_moe_forward(
|
||||||
@@ -1720,7 +1725,7 @@ class Qwen3_5MoeSparseBlock(nn.Module):
|
|||||||
# No physical transpose, no weight gather copy
|
# No physical transpose, no weight gather copy
|
||||||
# Source: ds_vllm/vllm/.../experts/fused_batched_moe.py
|
# Source: ds_vllm/vllm/.../experts/fused_batched_moe.py
|
||||||
# ---------------------------------------------------------------
|
# ---------------------------------------------------------------
|
||||||
if _USE_NAIVE_BATCHED_MOE:
|
if _USE_NAIVE_BATCHED_MOE and hidden_states.shape[0] > 1:
|
||||||
w13 = self.experts.w13_weight # (E, 2*I, H)
|
w13 = self.experts.w13_weight # (E, 2*I, H)
|
||||||
w2 = self.experts.w2_weight # (E, H, I)
|
w2 = self.experts.w2_weight # (E, H, I)
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user