Sources cloned and tree'd (no --depth):
- jd-opensource/xllm: ILU kernels, CUDA kernels, MoE kernels
- NVIDIA/cccl: CUB tuning/dispatch headers (block-level primitives)
- fla-org/flash-linear-attention: Triton GDN kernels
- NVIDIA/cutlass: grouped GEMM reference (read, not copied)
- Dao-AILab/flash-attention: attention kernel reference (SM80+, read only)
New CUDA kernels (from xllm, SM-agnostic, portable to BI-V100):
ex_engine/xllm_kernels/cuda/activation.cu (188 lines) — silu_and_mul, gelu
ex_engine/xllm_kernels/cuda/norm.cu (600 lines) — rms_norm, fused_add_rms_norm
ex_engine/xllm_kernels/cuda/rope.cu (258 lines) — rotary_embedding
ex_engine/xllm_kernels/cuda/block_copy.cu (209 lines) — copy_blocks, swap_blocks
ex_engine/xllm_kernels/cuda/reshape_paged_cache.cu (101 lines) — KV cache ops
ex_engine/xllm_kernels/cuda/headers/ (5 headers for compilation)
ILU bridge kernel sources (from xllm, verified SAME as upstream):
ex_engine/xllm_kernels/ilu/ (10 files, 925 lines total)
— activation.cpp, attention.cpp, fused_moe.cpp, group_gemm.cpp,
matmul.cpp, norm.cpp, rope.cpp, ilu_ops_api.h, ixformer.h, utils.h
FLA Triton GDN kernels (for GatedDeltaNet without SM90+ FlashQLA):
ex_engine/fla_kernels/gated_delta_rule/ (7 files, 2370 lines)
— chunk_fwd.py (428), chunk.py (487), wy_fast.py (409),
fused_recurrent.py (392), naive.py (161), gate.py (380)
CCCL sync (12 tuning + 14 dispatch headers updated from NVIDIA/cccl):
cccl_upstream/cub/cub/device/dispatch/tuning/ — 12 changed files synced
cccl_upstream/cub/cub/device/dispatch/ — 14 changed dispatch files synced
Compilation targets for real machine (ivcore10):
1. CUDA kernels: --cuda-gpu-arch=ivcore10 via corex clang/16
2. ILU bridges: torch.utils.cpp_extension linking ixformer .so
3. FLA kernels: Triton JIT (if Triton works on BI-V100)
63 lines
3.0 KiB
C++
63 lines
3.0 KiB
C++
/* Copyright 2025-2026 The xLLM Authors.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
https://github.com/jd-opensource/xllm/blob/main/LICENSE
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
==============================================================================*/
|
|
#pragma once
|
|
namespace xllm::kernel::ilu {
|
|
#undef check_tensor_contiguous
|
|
#define check_tensor_contiguous(x, type) \
|
|
TORCH_CHECK(x.scalar_type() == type); \
|
|
TORCH_CHECK(x.is_cuda()); \
|
|
TORCH_CHECK(x.is_contiguous());
|
|
|
|
#undef check_tensor_half_bf_float
|
|
#define check_tensor_half_bf_float(x) \
|
|
TORCH_CHECK(x.scalar_type() == at::ScalarType::Half || \
|
|
x.scalar_type() == at::ScalarType::Float || \
|
|
x.scalar_type() == at::ScalarType::BFloat16); \
|
|
TORCH_CHECK(x.is_cuda());
|
|
|
|
// from torchCheckMsgImpl
|
|
inline const char* ixformer_check_msg_impl(const char* msg) { return msg; }
|
|
// // If there is just 1 user-provided C-string argument, use it.
|
|
|
|
#define IXFORMER_CHECK_MSG(cond, type, ...) \
|
|
(ixformer_check_msg_impl( \
|
|
"Expected " #cond \
|
|
" to be true, but got false. " \
|
|
"(Could this error message be improved? If so, " \
|
|
"please report an enhancement request to ixformer.)", \
|
|
##__VA_ARGS__))
|
|
|
|
#define IXFORMER_CHECK(cond, ...) \
|
|
{ \
|
|
if (!(cond)) { \
|
|
std::cerr << __FILE__ << " (" << __LINE__ << ")" \
|
|
<< "-" << __FUNCTION__ << " : " \
|
|
<< IXFORMER_CHECK_MSG(cond, "", ##__VA_ARGS__) << std::endl; \
|
|
throw std::runtime_error("IXFORMER_CHECK ERROR"); \
|
|
} \
|
|
}
|
|
|
|
#undef CUINFER_CHECK
|
|
#define CUINFER_CHECK(func) \
|
|
do { \
|
|
cuinferStatus_t status = (func); \
|
|
if (status != CUINFER_STATUS_SUCCESS) { \
|
|
std::cerr << "Error in file " << __FILE__ << " on line " << __LINE__ \
|
|
<< ": " << cuinferGetErrorString(status) << std::endl; \
|
|
throw std::runtime_error("CUINFER_CHECK ERROR"); \
|
|
} \
|
|
} while (0)
|
|
|
|
} // namespace xllm::kernel::ilu
|