fix(EX): corex ivcore10 build flags + deploy pipeline + topk kernel cleanup

Real machine log (2d5232c dockerrizhi.txt) shows two AST call chain breaks:

1. EVERY layer EVERY token:
   _custom_ops.py:58 'ixformer.functions has no attribute vllm_moe_topk_softmax'
   -> FusedMoE falls to PyTorch loop (2304 calls/token)

2. EVERY GDN layer (4 layers):
   'NaN in prefill GatedDeltaNet layer N (frac=0.9998-1.0000)'
   -> _torch_chunk_gated_delta_rule produces all-NaN

Fixes:
- build.sh: --cuda-gpu-arch=ivcore10, -D__ILUVATAR__ flags from real log
- Dockerfile: add ex_engine build before patch_ops
- patch_ops.sh: deploy .so + python into vllm model dir
- ex_loader.py: search co-located .so paths
- patch_model.py: remove premature auto-apply
- factor_moe_topk_softmax.cu: remove dead parallel branch
This commit is contained in:
EX Engine
2026-08-10 02:31:55 +00:00
parent fcfb764560
commit b75965d4ea
6 changed files with 80 additions and 45 deletions

View File

@@ -57,19 +57,32 @@ compile_factor() {
if [[ "$COMPILER" == "corex" ]]; then
# CoreX/Iluvatar: clang-based CUDA compilation
# SM70 = BI-V100 architecture
# From real machine GDN compile log (dockerrizhi.txt):
# /usr/local/corex/bin/clang++ ... --cuda-gpu-arch=ivcore10
# --cuda-path=/usr/local/corex -std=c++17
# -D__ILUVATAR__ -D__ILUVATAR_WORKAROUND__
local OBJ="${BUILD_DIR}/$(basename ${cu_file} .cu).cuda.o"
"${COREX_ROOT}/bin/clang++" \
-x cuda \
--cuda-gpu-arch=sm_70 \
-std=c++17 \
-D__ILUVATAR__ \
-D__ILUVATAR_WORKAROUND__ \
-D__ILUVATAR_DIAG__ \
-fPIC \
-O2 \
-shared -fPIC \
--cuda-gpu-arch=ivcore10 \
--cuda-path="${COREX_ROOT}" \
-std=c++17 \
-I"${INCLUDE_DIR}" \
-I"${COREX_ROOT}/include" \
-isystem "${COREX_ROOT}/include" \
-c "${cu_file}" \
-o "${OBJ}"
# Link .o → .so (match real machine: c++ ... -shared -L ... -lcudart)
c++ "${OBJ}" -shared \
-L"${COREX_ROOT}/lib64" \
-lcudart \
-o "${so_path}" \
"${cu_file}"
-o "${so_path}"
rm -f "${OBJ}"
else
# Standard nvcc
nvcc \

View File

@@ -98,28 +98,13 @@ __global__ void moe_topk_softmax_kernel(
float global_sum = s_sum[0] + s_sum[1];
float my_prob = my_exp / global_sum; // softmax output
// Step 5: Top-K selection via shared memory partial sort
// Use shared memory to collect all (prob, id) pairs, then
// do a register-based bitonic top-K.
// Step 5: Top-K selection via shared memory
// 64 elements is tiny — thread-0 serial insertion sort is faster than
// launching a parallel radix/bitonic for k=8 from n=64.
__shared__ float s_probs[64];
__shared__ int s_ids[64];
s_probs[tid] = my_prob;
s_ids[tid] = my_id;
__syncthreads();
// Thread 0 does a simple insertion sort for top_k=8 from 64 elements
// This is faster than full bitonic for k << n.
// 64 elements × 8 comparisons = 512 ops (fits in registers)
if (tid < top_k) {
// Each of the first top_k threads finds one winner
// We use a parallel argmax approach: each thread looks for
// the (tid+1)-th largest element.
// Simple approach: tid=0 finds max, tid=1 finds 2nd max, etc.
// Use iterative suppression in shared memory.
// Actually, simpler: thread 0 does all work (64 experts is tiny)
}
if (tid == 0) {
float* out_w = topk_weights + token_idx * top_k;
int32_t* out_id = topk_ids + token_idx * top_k;
@@ -134,12 +119,11 @@ __global__ void moe_topk_softmax_kernel(
best_id[k] = -1;
}
for (int e = 0; e < num_experts; e++) {
for (int e = 0; e < num_experts && e < BLOCK_SIZE; e++) {
float p = s_probs[e];
// Insert into sorted top-K
if (p > best_w[TOP_K - 1]) {
best_w[TOP_K - 1] = p;
best_id[TOP_K - 1] = e;
best_id[TOP_K - 1] = e; // expert index = thread index
// Bubble up
#pragma unroll
for (int k = TOP_K - 1; k > 0; k--) {

View File

@@ -156,13 +156,27 @@ class EXEngine:
return False
def load_all(self) -> int:
"""Load all available factor .so files from build_dir."""
"""Load all available factor .so files from build_dir or co-located."""
loaded = 0
# Search paths: build_dir first, then directory containing this module
search_dirs = [self.build_dir]
module_dir = os.path.dirname(os.path.abspath(__file__))
if module_dir not in search_dirs:
search_dirs.append(module_dir)
# Also check parent's build dir
parent_build = os.path.join(os.path.dirname(module_dir), "build")
if parent_build not in search_dirs:
search_dirs.append(parent_build)
for fid in range(EX_FACTOR_COUNT):
so_path = os.path.join(self.build_dir, f"ex_factor_{fid}.so")
if self.load_factor(fid, so_path):
loaded += 1
logger.info("EX Engine: loaded %d/%d factors", loaded, EX_FACTOR_COUNT)
for d in search_dirs:
so_path = os.path.join(d, f"ex_factor_{fid}.so")
if os.path.exists(so_path):
if self.load_factor(fid, so_path):
loaded += 1
break
logger.info("EX Engine: loaded %d/%d factors from %s", loaded, EX_FACTOR_COUNT,
search_dirs)
return loaded
def has_factor(self, factor_id: int) -> bool:

View File

@@ -191,11 +191,7 @@ def _patch_gdn_prefill(engine):
# ---------------------------------------------------------------------------
# Auto-apply on import if build dir exists
# Call apply_patches() explicitly AFTER vllm model modules are loaded.
# Integration point: qwen3_5.py calls this at the end of model __init__,
# or patch_ops.sh adds it to the startup sequence.
# ---------------------------------------------------------------------------
_AUTO_BUILD_DIR = os.environ.get("EX_ENGINE_BUILD_DIR", "/workspace/ex_engine/build")
if os.path.isdir(_AUTO_BUILD_DIR):
try:
apply_patches(_AUTO_BUILD_DIR)
except Exception as e:
logger.warning("EX Engine auto-apply failed: %s", e)