From a0d76bc06e790537c1002d88530a475624ec4fbe Mon Sep 17 00:00:00 2001 From: Claude Date: Mon, 10 Aug 2026 07:41:53 +0000 Subject: [PATCH] =?UTF-8?q?fix:=20remove=20ix=5Fmoe=5Fbridge=20=E2=80=94?= =?UTF-8?q?=20nm=20-D=20confirms=20libixformer.so=20has=20NO=20MoE=20symbo?= =?UTF-8?q?ls?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit 真机探测确认: nm -D libixformer.so | grep topk_softmax → 空 ixf_F dir() → 无 vllm_moe_topk_softmax ixf_F dir() → 无 vllm_invoke_fused_moe_kernel ixf_F dir() → 无 vllm_moe_align_block_size _ixformer_torch.so symbols → 仅 cuinfer_gemm 系列, 无 MoE 结论: base 镜像的 MoE 路径: fused_moe.py → _custom_ops.topk_softmax → ixf_F.vllm_moe_topk_softmax → AttributeError → qwen3_5.py 捕获 → fallback to Python expert loop (这是唯一能工作的路径) 修改: 1. _custom_ops.py topk_softmax: 直接 PyTorch softmax+topk, 不尝试 ixf_F (消除 ERROR 日志) 2. 移除 ix_moe_bridge 加载逻辑 (libixformer.so 没有 MoE 符号, 链接会失败) 3. 移除 patch_ops.sh ix_moe_bridge JIT 编译步骤 comp 168 的 0 分根因不是 MoE fallback (所有参赛者都 fallback), 而是我们的自定义 qwen3_5.py 导致 GDN NaN 99.98% + OOM. 上一个 commit 已修复: 条件部署 qwen3_5.py + max_model_len=80000. --- qwen3_6_scripts/_custom_ops.py | 111 ++++----------------------------- qwen3_6_scripts/patch_ops.sh | 32 +--------- 2 files changed, 14 insertions(+), 129 deletions(-) diff --git a/qwen3_6_scripts/_custom_ops.py b/qwen3_6_scripts/_custom_ops.py index 1d183dff..1733d849 100644 --- a/qwen3_6_scripts/_custom_ops.py +++ b/qwen3_6_scripts/_custom_ops.py @@ -18,72 +18,6 @@ logger = init_logger(__name__) supports_moe_ops = True -# ============================================================================ -# EX Engine: ix_moe_bridge — JIT-compiled C++ bridge to ixformer::infer MoE ops -# This is the ONLY way to call topk_softmax, group_gemm, etc. on BI-V100 -# because ixformer.functions Python binding doesn't expose them. -# ============================================================================ -_ix_moe_bridge = None - -def _load_moe_bridge(): - """Load ix_moe_bridge via torch.utils.cpp_extension JIT compile.""" - import os, glob - bridge = None - - # Try 1: pre-compiled .so from ex_engine build - search_paths = [ - '/workspace/ex_engine/build', - os.path.join(os.path.dirname(__file__), '..', 'model_executor', 'models', 'ex_engine'), - '/usr/local/corex/lib/python3/dist-packages/ex_engine', - ] - for sp in search_paths: - so_files = glob.glob(os.path.join(sp, 'ix_moe_bridge*.so')) - if so_files: - try: - import importlib.util - spec = importlib.util.spec_from_file_location('ix_moe_bridge', so_files[0]) - bridge = importlib.util.module_from_spec(spec) - spec.loader.exec_module(bridge) - logger.info(f"[EX] Loaded ix_moe_bridge from {so_files[0]}") - return bridge - except Exception as e: - logger.warning(f"[EX] Failed to load pre-built bridge {so_files[0]}: {e}") - - # Try 2: JIT compile ix_moe_bridge.cpp against libixformer.so - cpp_search = [ - '/workspace/ex_engine/csrc/ix_moe_bridge.cpp', - os.path.join(os.path.dirname(__file__), 'ix_moe_bridge.cpp'), - os.path.join(os.path.dirname(__file__), '..', 'model_executor', 'models', 'ex_engine', 'csrc', 'ix_moe_bridge.cpp'), - ] - cpp_file = None - for p in cpp_search: - if os.path.isfile(p): - cpp_file = p - break - - if cpp_file: - try: - from torch.utils.cpp_extension import load - bridge = load( - name='ix_moe_bridge', - sources=[cpp_file], - extra_include_paths=['/usr/local/corex/include'], - extra_ldflags=[ - '-L/usr/local/corex/lib64', - '-L/usr/local/corex/lib64/python3/dist-packages/ixformer', - '-lixformer', - '-Wl,-rpath,/usr/local/corex/lib64/python3/dist-packages/ixformer', - ], - verbose=False, - ) - logger.info(f"[EX] JIT compiled ix_moe_bridge from {cpp_file}") - return bridge - except Exception as e: - logger.warning(f"[EX] JIT compile failed for {cpp_file}: {e}") - - logger.warning("[EX] ix_moe_bridge NOT available — topk_softmax will use PyTorch path") - return None - if TYPE_CHECKING: def register_fake(fn): @@ -896,39 +830,20 @@ def invoke_fused_moe_kernel( def topk_softmax(topk_weights: torch.Tensor, topk_ids: torch.Tensor, token_expert_indicies: torch.Tensor, gating_output: float) -> None: - # EX Engine: algorithm factor replacement for topk_softmax. - # ixformer::infer::topk_softmax is in libixformer.so (C++ level) - # but NOT exposed via ixformer.functions Python binding. - # We call it via ix_moe_bridge (pybind11 JIT-compiled against libixformer.so). - # NO FALLBACK — if bridge fails, raise immediately to catch integration bugs. - global _ix_moe_bridge - if _ix_moe_bridge is None: - _ix_moe_bridge = _load_moe_bridge() - if _ix_moe_bridge is not None: - # Bridge available — call ixformer::infer::topk_softmax via C++ - if isinstance(gating_output, torch.Tensor): - gating_output = gating_output.float().contiguous() - topk = topk_weights.shape[1] - tw, ti = _ix_moe_bridge.topk_softmax(gating_output, topk, False) - topk_weights.copy_(tw.to(topk_weights.dtype)) - topk_ids.copy_(ti.to(topk_ids.dtype)) - token_expert_indicies.copy_( - torch.arange(topk, device=topk_ids.device, dtype=topk_ids.dtype) - .unsqueeze(0).expand_as(topk_ids)) + # BI-V100 base image: ixf_F.vllm_moe_topk_softmax does NOT exist. + # libixformer.so has NO topk_softmax symbol (verified via nm -D). + # PyTorch implementation — silent, no ERROR log spam. + if isinstance(gating_output, torch.Tensor): + probs = torch.softmax(gating_output.float(), dim=-1) else: - # Bridge not loaded — use PyTorch (for build environments without GPU) - # In production this path should NOT be hit - if isinstance(gating_output, torch.Tensor): - probs = torch.softmax(gating_output.float(), dim=-1) - else: - probs = torch.softmax(gating_output, dim=-1) - topk = topk_weights.shape[1] - tw, ti = torch.topk(probs, topk, dim=-1) - topk_weights.copy_(tw) - topk_ids.copy_(ti.to(topk_ids.dtype)) - token_expert_indicies.copy_( - torch.arange(topk, device=topk_ids.device, dtype=topk_ids.dtype) - .unsqueeze(0).expand_as(topk_ids)) + probs = torch.softmax(gating_output, dim=-1) + topk = topk_weights.shape[1] + tw, ti = torch.topk(probs, topk, dim=-1) + topk_weights.copy_(tw.to(topk_weights.dtype)) + topk_ids.copy_(ti.to(topk_ids.dtype)) + token_expert_indicies.copy_( + torch.arange(topk, device=topk_ids.device, dtype=topk_ids.dtype) + .unsqueeze(0).expand_as(topk_ids)) if supports_moe_ops and hasattr(torch.ops._moe_C, "marlin_gemm_moe"): diff --git a/qwen3_6_scripts/patch_ops.sh b/qwen3_6_scripts/patch_ops.sh index 7514010d..8b22c900 100755 --- a/qwen3_6_scripts/patch_ops.sh +++ b/qwen3_6_scripts/patch_ops.sh @@ -260,37 +260,7 @@ if [ -d "$EX_ENGINE_SRC/python" ]; then echo "[patch_ops] EX Engine Python package deployed to $EX_PY_DST" fi -# 7a. JIT compile ix_moe_bridge.cpp → .so (bridge to ixformer::infer C++ API) -# This is CRITICAL: topk_softmax, group_gemm, etc. are ONLY accessible via C++ -IX_MOE_BRIDGE_CPP="/workspace/ex_engine/csrc/ix_moe_bridge.cpp" -if [ -f "$IX_MOE_BRIDGE_CPP" ]; then - echo "[patch_ops] Pre-compiling ix_moe_bridge.cpp (ixformer C++ bridge)..." - python3 -c " -import torch -from torch.utils.cpp_extension import load -try: - bridge = load( - name='ix_moe_bridge', - sources=['$IX_MOE_BRIDGE_CPP'], - extra_include_paths=['/usr/local/corex/include'], - extra_ldflags=[ - '-L/usr/local/corex/lib64', - '-L/usr/local/corex/lib64/python3/dist-packages/ixformer', - '-lixformer', - '-Wl,-rpath,/usr/local/corex/lib64/python3/dist-packages/ixformer', - ], - verbose=True, - ) - print('[patch_ops] ix_moe_bridge compiled successfully') - # Test basic function availability - print(f'[patch_ops] Bridge functions: {[x for x in dir(bridge) if not x.startswith(\"_\")]}') -except Exception as e: - print(f'[patch_ops] WARNING: ix_moe_bridge compile failed: {e}') - print('[patch_ops] topk_softmax will use PyTorch fallback') -" 2>&1 || echo "[patch_ops] WARNING: ix_moe_bridge pre-compile step failed" -fi - -# 7b. Precompile MoE topk_softmax CUDA kernel (.cu → .so) +# 7. Precompile MoE topk_softmax CUDA kernel (.cu → .so) # This replaces the missing ixf_F.vllm_moe_topk_softmax with our own CUDA kernel MOE_TOPK_CU="/workspace/ex_engine/csrc/moe_topk_softmax_v3.cu" if [ -f "$MOE_TOPK_CU" ]; then