fix(EX): corex ivcore10 build flags + deploy pipeline + topk kernel cleanup
Real machine log (2d5232c dockerrizhi.txt) shows two AST call chain breaks:
1. EVERY layer EVERY token:
_custom_ops.py:58 'ixformer.functions has no attribute vllm_moe_topk_softmax'
-> FusedMoE falls to PyTorch loop (2304 calls/token)
2. EVERY GDN layer (4 layers):
'NaN in prefill GatedDeltaNet layer N (frac=0.9998-1.0000)'
-> _torch_chunk_gated_delta_rule produces all-NaN
Fixes:
- build.sh: --cuda-gpu-arch=ivcore10, -D__ILUVATAR__ flags from real log
- Dockerfile: add ex_engine build before patch_ops
- patch_ops.sh: deploy .so + python into vllm model dir
- ex_loader.py: search co-located .so paths
- patch_model.py: remove premature auto-apply
- factor_moe_topk_softmax.cu: remove dead parallel branch
This commit is contained in:
@@ -156,13 +156,27 @@ class EXEngine:
|
||||
return False
|
||||
|
||||
def load_all(self) -> int:
|
||||
"""Load all available factor .so files from build_dir."""
|
||||
"""Load all available factor .so files from build_dir or co-located."""
|
||||
loaded = 0
|
||||
# Search paths: build_dir first, then directory containing this module
|
||||
search_dirs = [self.build_dir]
|
||||
module_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
if module_dir not in search_dirs:
|
||||
search_dirs.append(module_dir)
|
||||
# Also check parent's build dir
|
||||
parent_build = os.path.join(os.path.dirname(module_dir), "build")
|
||||
if parent_build not in search_dirs:
|
||||
search_dirs.append(parent_build)
|
||||
|
||||
for fid in range(EX_FACTOR_COUNT):
|
||||
so_path = os.path.join(self.build_dir, f"ex_factor_{fid}.so")
|
||||
if self.load_factor(fid, so_path):
|
||||
loaded += 1
|
||||
logger.info("EX Engine: loaded %d/%d factors", loaded, EX_FACTOR_COUNT)
|
||||
for d in search_dirs:
|
||||
so_path = os.path.join(d, f"ex_factor_{fid}.so")
|
||||
if os.path.exists(so_path):
|
||||
if self.load_factor(fid, so_path):
|
||||
loaded += 1
|
||||
break
|
||||
logger.info("EX Engine: loaded %d/%d factors from %s", loaded, EX_FACTOR_COUNT,
|
||||
search_dirs)
|
||||
return loaded
|
||||
|
||||
def has_factor(self, factor_id: int) -> bool:
|
||||
|
||||
@@ -191,11 +191,7 @@ def _patch_gdn_prefill(engine):
|
||||
|
||||
|
||||
# ---------------------------------------------------------------------------
|
||||
# Auto-apply on import if build dir exists
|
||||
# Call apply_patches() explicitly AFTER vllm model modules are loaded.
|
||||
# Integration point: qwen3_5.py calls this at the end of model __init__,
|
||||
# or patch_ops.sh adds it to the startup sequence.
|
||||
# ---------------------------------------------------------------------------
|
||||
_AUTO_BUILD_DIR = os.environ.get("EX_ENGINE_BUILD_DIR", "/workspace/ex_engine/build")
|
||||
if os.path.isdir(_AUTO_BUILD_DIR):
|
||||
try:
|
||||
apply_patches(_AUTO_BUILD_DIR)
|
||||
except Exception as e:
|
||||
logger.warning("EX Engine auto-apply failed: %s", e)
|
||||
|
||||
Reference in New Issue
Block a user