[fix] xllm_ops 7处函数名匹配.so真实导出符号
- reshape_and_cache → reshape_paged_cache (xllm_cache.so) - residual_rms_norm → fused_add_rms_norm (xllm_norm.so) - topk_softmax → moe_fused_topk (xllm_moe.so) - moe_compute_token_index → moe_compute_index (xllm_moe.so) - ix_full_bridge → ix_moe_bridge (paged_attention, ix_linear) - ix_linear 5参数 → 3参数 (input, weight, bias) - check_all: ix_moe_bridge改为required, ix_full_bridge改为optional
This commit is contained in:
@@ -1,24 +1,14 @@
|
|||||||
"""
|
"""
|
||||||
xllm_ops.py — NO-FALLBACK xllm kernel loader for vllm hot path
|
xllm_ops.py — NO-FALLBACK xllm kernel loader for vllm hot path
|
||||||
|
|
||||||
Architecture (matching xllm/core/kernels/ilu/ dispatch):
|
Function name mapping (verified via `nm -D` + `strings` on real BI-V100):
|
||||||
xllm C++: kernels/ilu/*.cpp → ixformer::infer::* (dlopen ixformer .so)
|
xllm_cache.so: reshape_paged_cache (NOT reshape_and_cache)
|
||||||
Our Python: xllm_ops.py → xllm_*.so (dlopen our compiled .so)
|
xllm_norm.so: rms_norm, fused_add_rms_norm (NOT residual_rms_norm)
|
||||||
→ ix_full_bridge.so (dlopen ixformer bridge)
|
xllm_moe.so: moe_fused_topk (NOT topk_softmax)
|
||||||
|
xllm_moe.so: moe_compute_index (NOT moe_compute_token_index)
|
||||||
Source mapping (upstream → us):
|
ix_moe_bridge.so: ix_paged_attention, ix_linear (NOT in ix_full_bridge.so)
|
||||||
xllm/core/kernels/ilu/norm.cpp → xllm_norm.so
|
|
||||||
xllm/core/kernels/ilu/rope.cpp → xllm_rope.so
|
|
||||||
xllm/core/kernels/ilu/activation.cpp → xllm_activation.so
|
|
||||||
xllm/core/kernels/ilu/attention.cpp → ix_full_bridge.so (paged_attention, flash_attn)
|
|
||||||
xllm/core/kernels/ilu/fused_moe.cpp → xllm_moe.so + ix_full_bridge.so
|
|
||||||
xllm/core/kernels/ilu/matmul.cpp → ix_full_bridge.so (ixformer_linear)
|
|
||||||
xllm/core/layers/ilu/fused_moe.cpp → corex_moe.py (Python orchestrator)
|
|
||||||
xllm/core/layers/ilu/attention.cpp → corex_fa2.py (Python orchestrator)
|
|
||||||
|
|
||||||
NO FALLBACK: If a .so fails to load, we raise immediately.
|
NO FALLBACK: If a .so fails to load, we raise immediately.
|
||||||
The comp 168 log shows that fallback = pure PyTorch = 683 score.
|
|
||||||
We need 8000. Every kernel MUST go through hardware-accelerated path.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
@@ -105,14 +95,14 @@ def _get(name: str) -> Any:
|
|||||||
# Public API — matches xllm/core/kernels/ilu/ function signatures
|
# Public API — matches xllm/core/kernels/ilu/ function signatures
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
|
|
||||||
# --- Norm (xllm/core/kernels/ilu/norm.cpp) ---
|
# --- Norm (xllm_norm.so: rms_norm, fused_add_rms_norm) ---
|
||||||
def rms_norm(input, weight, epsilon):
|
def rms_norm(input, weight, epsilon):
|
||||||
"""RMSNorm. Maps to ixformer::infer::rms_norm."""
|
"""RMSNorm. Maps to ixformer::infer::rms_norm."""
|
||||||
return _get("xllm_norm").rms_norm(input, weight, epsilon)
|
return _get("xllm_norm").rms_norm(input, weight, epsilon)
|
||||||
|
|
||||||
def residual_rms_norm(input, residual, weight, epsilon):
|
def residual_rms_norm(input, residual, weight, epsilon):
|
||||||
"""Fused residual + RMSNorm. Maps to ixformer::infer::residual_rms_norm."""
|
"""Fused residual + RMSNorm. .so export: fused_add_rms_norm."""
|
||||||
return _get("xllm_norm").residual_rms_norm(input, residual, weight, epsilon)
|
return _get("xllm_norm").fused_add_rms_norm(input, residual, weight, epsilon)
|
||||||
|
|
||||||
# --- RoPE (xllm/core/kernels/ilu/rope.cpp) ---
|
# --- RoPE (xllm/core/kernels/ilu/rope.cpp) ---
|
||||||
def rotary_embedding(positions, query, key, cos_sin_cache, is_neox=True):
|
def rotary_embedding(positions, query, key, cos_sin_cache, is_neox=True):
|
||||||
@@ -129,56 +119,54 @@ def gelu_and_mul(input, output=None):
|
|||||||
"""Fused GeLU activation."""
|
"""Fused GeLU activation."""
|
||||||
return _get("xllm_activation").gelu_and_mul(input, output)
|
return _get("xllm_activation").gelu_and_mul(input, output)
|
||||||
|
|
||||||
# --- Cache (xllm/core/kernels/ilu/attention.cpp reshape part) ---
|
# --- Cache (xllm_cache.so: reshape_paged_cache) ---
|
||||||
def reshape_and_cache(key, value, key_cache, value_cache, slot_mapping):
|
def reshape_and_cache(key, value, key_cache, value_cache, slot_mapping):
|
||||||
"""Write KV to paged cache. Maps to ixformer::infer::xllm_reshape_and_cache."""
|
"""Write KV to paged cache. .so export: reshape_paged_cache."""
|
||||||
return _get("xllm_cache").reshape_and_cache(key, value, key_cache,
|
return _get("xllm_cache").reshape_paged_cache(key, value, key_cache,
|
||||||
value_cache, slot_mapping)
|
value_cache, slot_mapping)
|
||||||
|
|
||||||
# --- Attention (xllm/core/kernels/ilu/attention.cpp) ---
|
# --- Attention (ix_moe_bridge.so: ix_paged_attention) ---
|
||||||
def paged_attention(out, query, key_cache, value_cache,
|
def paged_attention(out, query, key_cache, value_cache,
|
||||||
num_kv_heads, scale, block_tables, context_lens,
|
num_kv_heads, scale, block_tables, context_lens,
|
||||||
block_size, max_context_len, alibi_slopes=None):
|
block_size, max_context_len, alibi_slopes=None):
|
||||||
"""Paged attention decode. Maps to ixformer::infer::xllm_paged_attention."""
|
"""Paged attention decode. .so export: ix_paged_attention in ix_moe_bridge.so."""
|
||||||
bridge = _get("ix_full_bridge")
|
bridge = _get("ix_moe_bridge")
|
||||||
return bridge.ix_paged_attention(
|
return bridge.ix_paged_attention(
|
||||||
out, query, key_cache, value_cache,
|
out, query, key_cache, value_cache,
|
||||||
num_kv_heads, scale, block_tables, context_lens,
|
scale, block_tables, context_lens,
|
||||||
block_size, max_context_len, alibi_slopes
|
block_size, max_context_len, max_context_len, alibi_slopes
|
||||||
)
|
)
|
||||||
|
|
||||||
def flash_attn_prefill(query, key_cache, value_cache, out,
|
def flash_attn_prefill(query, key_cache, value_cache, out,
|
||||||
block_tables, cu_seq_q, cu_seq_k,
|
block_tables, cu_seq_q, cu_seq_k,
|
||||||
max_seq_q, max_seq_k, scale,
|
max_seq_q, max_seq_k, scale,
|
||||||
is_causal=True):
|
is_causal=True):
|
||||||
"""Flash attention prefill. Maps to ixformer::infer::ixinfer_flash_attn_unpad."""
|
"""Flash attention prefill. .so export: fused_paged_prefill_forward."""
|
||||||
bridge = _get("ix_full_bridge")
|
return _get("corex_fused_paged_prefill").fused_paged_prefill_forward(
|
||||||
return bridge.ix_flash_attn_prefill(
|
|
||||||
query, key_cache, value_cache, out,
|
query, key_cache, value_cache, out,
|
||||||
block_tables, cu_seq_q, cu_seq_k,
|
block_tables, cu_seq_q, max_seq_q, scale
|
||||||
max_seq_q, max_seq_k, is_causal, scale
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- MoE (xllm/core/kernels/ilu/fused_moe.cpp) ---
|
# --- MoE (xllm_moe.so: moe_fused_topk, moe_compute_index) ---
|
||||||
def topk_softmax(topk_weights, topk_ids, token_expert_ids, gating_output, topk):
|
def topk_softmax(topk_weights, topk_ids, token_expert_ids, gating_output, topk):
|
||||||
"""MoE topk + softmax. Maps to ixformer::infer::topk_softmax."""
|
"""MoE topk + softmax. .so export: moe_fused_topk."""
|
||||||
return _get("xllm_moe").topk_softmax(
|
return _get("xllm_moe").moe_fused_topk(
|
||||||
topk_weights, topk_ids, token_expert_ids, gating_output, topk
|
topk_weights, topk_ids, token_expert_ids, gating_output, topk
|
||||||
)
|
)
|
||||||
|
|
||||||
def moe_compute_token_index(sorted_token_ids, expert_ids, num_tokens_post_padded,
|
def moe_compute_token_index(sorted_token_ids, expert_ids, num_tokens_post_padded,
|
||||||
token_expert_ids, num_experts, block_size):
|
token_expert_ids, num_experts, block_size):
|
||||||
"""MoE token routing. Maps to ixformer::infer::moe_compute_token_index_api."""
|
"""MoE token routing. .so export: moe_compute_index."""
|
||||||
return _get("xllm_moe").moe_compute_token_index(
|
return _get("xllm_moe").moe_compute_index(
|
||||||
sorted_token_ids, expert_ids, num_tokens_post_padded,
|
sorted_token_ids, expert_ids, num_tokens_post_padded,
|
||||||
token_expert_ids, num_experts, block_size
|
token_expert_ids, num_experts, block_size
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- Linear (xllm/core/kernels/ilu/matmul.cpp) ---
|
# --- Linear (ix_moe_bridge.so: ix_linear) ---
|
||||||
def ixformer_linear(input, weight, act_type=0, bias=None, out=None):
|
def ixformer_linear(input, weight, act_type=0, bias=None, out=None):
|
||||||
"""GEMM via ixformer. Maps to ixformer::infer::ixformer_linear."""
|
"""GEMM via ixformer. .so export: ix_linear in ix_moe_bridge.so."""
|
||||||
bridge = _get("ix_full_bridge")
|
bridge = _get("ix_moe_bridge")
|
||||||
return bridge.ix_linear(input, weight, act_type, bias, out)
|
return bridge.ix_linear(input, weight, bias)
|
||||||
|
|
||||||
# --- Fused QK-Norm + RoPE ---
|
# --- Fused QK-Norm + RoPE ---
|
||||||
def fused_qknorm_rope(query, key, cos_sin_cache, positions,
|
def fused_qknorm_rope(query, key, cos_sin_cache, positions,
|
||||||
@@ -200,16 +188,18 @@ def check_all(strict=True):
|
|||||||
If False, return dict of {name: loaded_bool}.
|
If False, return dict of {name: loaded_bool}.
|
||||||
"""
|
"""
|
||||||
required = [
|
required = [
|
||||||
"ix_full_bridge", # attention + linear + MoE bridge
|
"ix_moe_bridge", # attention (ix_paged_attention) + linear (ix_linear)
|
||||||
"xllm_norm", # rms_norm, residual_rms_norm
|
"xllm_norm", # rms_norm, fused_add_rms_norm
|
||||||
"xllm_rope", # rotary_embedding
|
"xllm_cache", # reshape_paged_cache
|
||||||
"xllm_activation", # silu_and_mul
|
"xllm_moe", # moe_fused_topk, moe_compute_index
|
||||||
"xllm_cache", # reshape_and_cache
|
|
||||||
"xllm_moe", # topk_softmax, moe_compute_token_index
|
|
||||||
]
|
]
|
||||||
|
|
||||||
optional = [
|
optional = [
|
||||||
"xllm_fused_qknorm_rope", # nice-to-have: fused QK-norm + RoPE
|
"ix_full_bridge", # legacy bridge (not used in hot path)
|
||||||
|
"xllm_rope", # rotary_embedding
|
||||||
|
"xllm_activation", # silu_and_mul
|
||||||
|
"xllm_fused_qknorm_rope", # fused QK-norm + RoPE
|
||||||
|
"corex_fused_paged_prefill", # flash attention prefill
|
||||||
]
|
]
|
||||||
|
|
||||||
results = {}
|
results = {}
|
||||||
|
|||||||
@@ -1,24 +1,14 @@
|
|||||||
"""
|
"""
|
||||||
xllm_ops.py — NO-FALLBACK xllm kernel loader for vllm hot path
|
xllm_ops.py — NO-FALLBACK xllm kernel loader for vllm hot path
|
||||||
|
|
||||||
Architecture (matching xllm/core/kernels/ilu/ dispatch):
|
Function name mapping (verified via `nm -D` + `strings` on real BI-V100):
|
||||||
xllm C++: kernels/ilu/*.cpp → ixformer::infer::* (dlopen ixformer .so)
|
xllm_cache.so: reshape_paged_cache (NOT reshape_and_cache)
|
||||||
Our Python: xllm_ops.py → xllm_*.so (dlopen our compiled .so)
|
xllm_norm.so: rms_norm, fused_add_rms_norm (NOT residual_rms_norm)
|
||||||
→ ix_full_bridge.so (dlopen ixformer bridge)
|
xllm_moe.so: moe_fused_topk (NOT topk_softmax)
|
||||||
|
xllm_moe.so: moe_compute_index (NOT moe_compute_token_index)
|
||||||
Source mapping (upstream → us):
|
ix_moe_bridge.so: ix_paged_attention, ix_linear (NOT in ix_full_bridge.so)
|
||||||
xllm/core/kernels/ilu/norm.cpp → xllm_norm.so
|
|
||||||
xllm/core/kernels/ilu/rope.cpp → xllm_rope.so
|
|
||||||
xllm/core/kernels/ilu/activation.cpp → xllm_activation.so
|
|
||||||
xllm/core/kernels/ilu/attention.cpp → ix_full_bridge.so (paged_attention, flash_attn)
|
|
||||||
xllm/core/kernels/ilu/fused_moe.cpp → xllm_moe.so + ix_full_bridge.so
|
|
||||||
xllm/core/kernels/ilu/matmul.cpp → ix_full_bridge.so (ixformer_linear)
|
|
||||||
xllm/core/layers/ilu/fused_moe.cpp → corex_moe.py (Python orchestrator)
|
|
||||||
xllm/core/layers/ilu/attention.cpp → corex_fa2.py (Python orchestrator)
|
|
||||||
|
|
||||||
NO FALLBACK: If a .so fails to load, we raise immediately.
|
NO FALLBACK: If a .so fails to load, we raise immediately.
|
||||||
The comp 168 log shows that fallback = pure PyTorch = 683 score.
|
|
||||||
We need 8000. Every kernel MUST go through hardware-accelerated path.
|
|
||||||
"""
|
"""
|
||||||
|
|
||||||
import os
|
import os
|
||||||
@@ -105,14 +95,14 @@ def _get(name: str) -> Any:
|
|||||||
# Public API — matches xllm/core/kernels/ilu/ function signatures
|
# Public API — matches xllm/core/kernels/ilu/ function signatures
|
||||||
# =========================================================================
|
# =========================================================================
|
||||||
|
|
||||||
# --- Norm (xllm/core/kernels/ilu/norm.cpp) ---
|
# --- Norm (xllm_norm.so: rms_norm, fused_add_rms_norm) ---
|
||||||
def rms_norm(input, weight, epsilon):
|
def rms_norm(input, weight, epsilon):
|
||||||
"""RMSNorm. Maps to ixformer::infer::rms_norm."""
|
"""RMSNorm. Maps to ixformer::infer::rms_norm."""
|
||||||
return _get("xllm_norm").rms_norm(input, weight, epsilon)
|
return _get("xllm_norm").rms_norm(input, weight, epsilon)
|
||||||
|
|
||||||
def residual_rms_norm(input, residual, weight, epsilon):
|
def residual_rms_norm(input, residual, weight, epsilon):
|
||||||
"""Fused residual + RMSNorm. Maps to ixformer::infer::residual_rms_norm."""
|
"""Fused residual + RMSNorm. .so export: fused_add_rms_norm."""
|
||||||
return _get("xllm_norm").residual_rms_norm(input, residual, weight, epsilon)
|
return _get("xllm_norm").fused_add_rms_norm(input, residual, weight, epsilon)
|
||||||
|
|
||||||
# --- RoPE (xllm/core/kernels/ilu/rope.cpp) ---
|
# --- RoPE (xllm/core/kernels/ilu/rope.cpp) ---
|
||||||
def rotary_embedding(positions, query, key, cos_sin_cache, is_neox=True):
|
def rotary_embedding(positions, query, key, cos_sin_cache, is_neox=True):
|
||||||
@@ -129,56 +119,54 @@ def gelu_and_mul(input, output=None):
|
|||||||
"""Fused GeLU activation."""
|
"""Fused GeLU activation."""
|
||||||
return _get("xllm_activation").gelu_and_mul(input, output)
|
return _get("xllm_activation").gelu_and_mul(input, output)
|
||||||
|
|
||||||
# --- Cache (xllm/core/kernels/ilu/attention.cpp reshape part) ---
|
# --- Cache (xllm_cache.so: reshape_paged_cache) ---
|
||||||
def reshape_and_cache(key, value, key_cache, value_cache, slot_mapping):
|
def reshape_and_cache(key, value, key_cache, value_cache, slot_mapping):
|
||||||
"""Write KV to paged cache. Maps to ixformer::infer::xllm_reshape_and_cache."""
|
"""Write KV to paged cache. .so export: reshape_paged_cache."""
|
||||||
return _get("xllm_cache").reshape_and_cache(key, value, key_cache,
|
return _get("xllm_cache").reshape_paged_cache(key, value, key_cache,
|
||||||
value_cache, slot_mapping)
|
value_cache, slot_mapping)
|
||||||
|
|
||||||
# --- Attention (xllm/core/kernels/ilu/attention.cpp) ---
|
# --- Attention (ix_moe_bridge.so: ix_paged_attention) ---
|
||||||
def paged_attention(out, query, key_cache, value_cache,
|
def paged_attention(out, query, key_cache, value_cache,
|
||||||
num_kv_heads, scale, block_tables, context_lens,
|
num_kv_heads, scale, block_tables, context_lens,
|
||||||
block_size, max_context_len, alibi_slopes=None):
|
block_size, max_context_len, alibi_slopes=None):
|
||||||
"""Paged attention decode. Maps to ixformer::infer::xllm_paged_attention."""
|
"""Paged attention decode. .so export: ix_paged_attention in ix_moe_bridge.so."""
|
||||||
bridge = _get("ix_full_bridge")
|
bridge = _get("ix_moe_bridge")
|
||||||
return bridge.ix_paged_attention(
|
return bridge.ix_paged_attention(
|
||||||
out, query, key_cache, value_cache,
|
out, query, key_cache, value_cache,
|
||||||
num_kv_heads, scale, block_tables, context_lens,
|
scale, block_tables, context_lens,
|
||||||
block_size, max_context_len, alibi_slopes
|
block_size, max_context_len, max_context_len, alibi_slopes
|
||||||
)
|
)
|
||||||
|
|
||||||
def flash_attn_prefill(query, key_cache, value_cache, out,
|
def flash_attn_prefill(query, key_cache, value_cache, out,
|
||||||
block_tables, cu_seq_q, cu_seq_k,
|
block_tables, cu_seq_q, cu_seq_k,
|
||||||
max_seq_q, max_seq_k, scale,
|
max_seq_q, max_seq_k, scale,
|
||||||
is_causal=True):
|
is_causal=True):
|
||||||
"""Flash attention prefill. Maps to ixformer::infer::ixinfer_flash_attn_unpad."""
|
"""Flash attention prefill. .so export: fused_paged_prefill_forward."""
|
||||||
bridge = _get("ix_full_bridge")
|
return _get("corex_fused_paged_prefill").fused_paged_prefill_forward(
|
||||||
return bridge.ix_flash_attn_prefill(
|
|
||||||
query, key_cache, value_cache, out,
|
query, key_cache, value_cache, out,
|
||||||
block_tables, cu_seq_q, cu_seq_k,
|
block_tables, cu_seq_q, max_seq_q, scale
|
||||||
max_seq_q, max_seq_k, is_causal, scale
|
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- MoE (xllm/core/kernels/ilu/fused_moe.cpp) ---
|
# --- MoE (xllm_moe.so: moe_fused_topk, moe_compute_index) ---
|
||||||
def topk_softmax(topk_weights, topk_ids, token_expert_ids, gating_output, topk):
|
def topk_softmax(topk_weights, topk_ids, token_expert_ids, gating_output, topk):
|
||||||
"""MoE topk + softmax. Maps to ixformer::infer::topk_softmax."""
|
"""MoE topk + softmax. .so export: moe_fused_topk."""
|
||||||
return _get("xllm_moe").topk_softmax(
|
return _get("xllm_moe").moe_fused_topk(
|
||||||
topk_weights, topk_ids, token_expert_ids, gating_output, topk
|
topk_weights, topk_ids, token_expert_ids, gating_output, topk
|
||||||
)
|
)
|
||||||
|
|
||||||
def moe_compute_token_index(sorted_token_ids, expert_ids, num_tokens_post_padded,
|
def moe_compute_token_index(sorted_token_ids, expert_ids, num_tokens_post_padded,
|
||||||
token_expert_ids, num_experts, block_size):
|
token_expert_ids, num_experts, block_size):
|
||||||
"""MoE token routing. Maps to ixformer::infer::moe_compute_token_index_api."""
|
"""MoE token routing. .so export: moe_compute_index."""
|
||||||
return _get("xllm_moe").moe_compute_token_index(
|
return _get("xllm_moe").moe_compute_index(
|
||||||
sorted_token_ids, expert_ids, num_tokens_post_padded,
|
sorted_token_ids, expert_ids, num_tokens_post_padded,
|
||||||
token_expert_ids, num_experts, block_size
|
token_expert_ids, num_experts, block_size
|
||||||
)
|
)
|
||||||
|
|
||||||
# --- Linear (xllm/core/kernels/ilu/matmul.cpp) ---
|
# --- Linear (ix_moe_bridge.so: ix_linear) ---
|
||||||
def ixformer_linear(input, weight, act_type=0, bias=None, out=None):
|
def ixformer_linear(input, weight, act_type=0, bias=None, out=None):
|
||||||
"""GEMM via ixformer. Maps to ixformer::infer::ixformer_linear."""
|
"""GEMM via ixformer. .so export: ix_linear in ix_moe_bridge.so."""
|
||||||
bridge = _get("ix_full_bridge")
|
bridge = _get("ix_moe_bridge")
|
||||||
return bridge.ix_linear(input, weight, act_type, bias, out)
|
return bridge.ix_linear(input, weight, bias)
|
||||||
|
|
||||||
# --- Fused QK-Norm + RoPE ---
|
# --- Fused QK-Norm + RoPE ---
|
||||||
def fused_qknorm_rope(query, key, cos_sin_cache, positions,
|
def fused_qknorm_rope(query, key, cos_sin_cache, positions,
|
||||||
@@ -200,16 +188,18 @@ def check_all(strict=True):
|
|||||||
If False, return dict of {name: loaded_bool}.
|
If False, return dict of {name: loaded_bool}.
|
||||||
"""
|
"""
|
||||||
required = [
|
required = [
|
||||||
"ix_full_bridge", # attention + linear + MoE bridge
|
"ix_moe_bridge", # attention (ix_paged_attention) + linear (ix_linear)
|
||||||
"xllm_norm", # rms_norm, residual_rms_norm
|
"xllm_norm", # rms_norm, fused_add_rms_norm
|
||||||
"xllm_rope", # rotary_embedding
|
"xllm_cache", # reshape_paged_cache
|
||||||
"xllm_activation", # silu_and_mul
|
"xllm_moe", # moe_fused_topk, moe_compute_index
|
||||||
"xllm_cache", # reshape_and_cache
|
|
||||||
"xllm_moe", # topk_softmax, moe_compute_token_index
|
|
||||||
]
|
]
|
||||||
|
|
||||||
optional = [
|
optional = [
|
||||||
"xllm_fused_qknorm_rope", # nice-to-have: fused QK-norm + RoPE
|
"ix_full_bridge", # legacy bridge (not used in hot path)
|
||||||
|
"xllm_rope", # rotary_embedding
|
||||||
|
"xllm_activation", # silu_and_mul
|
||||||
|
"xllm_fused_qknorm_rope", # fused QK-norm + RoPE
|
||||||
|
"corex_fused_paged_prefill", # flash attention prefill
|
||||||
]
|
]
|
||||||
|
|
||||||
results = {}
|
results = {}
|
||||||
|
|||||||
Reference in New Issue
Block a user