Sources cloned and tree'd (no --depth):
- jd-opensource/xllm: ILU kernels, CUDA kernels, MoE kernels
- NVIDIA/cccl: CUB tuning/dispatch headers (block-level primitives)
- fla-org/flash-linear-attention: Triton GDN kernels
- NVIDIA/cutlass: grouped GEMM reference (read, not copied)
- Dao-AILab/flash-attention: attention kernel reference (SM80+, read only)
New CUDA kernels (from xllm, SM-agnostic, portable to BI-V100):
ex_engine/xllm_kernels/cuda/activation.cu (188 lines) — silu_and_mul, gelu
ex_engine/xllm_kernels/cuda/norm.cu (600 lines) — rms_norm, fused_add_rms_norm
ex_engine/xllm_kernels/cuda/rope.cu (258 lines) — rotary_embedding
ex_engine/xllm_kernels/cuda/block_copy.cu (209 lines) — copy_blocks, swap_blocks
ex_engine/xllm_kernels/cuda/reshape_paged_cache.cu (101 lines) — KV cache ops
ex_engine/xllm_kernels/cuda/headers/ (5 headers for compilation)
ILU bridge kernel sources (from xllm, verified SAME as upstream):
ex_engine/xllm_kernels/ilu/ (10 files, 925 lines total)
— activation.cpp, attention.cpp, fused_moe.cpp, group_gemm.cpp,
matmul.cpp, norm.cpp, rope.cpp, ilu_ops_api.h, ixformer.h, utils.h
FLA Triton GDN kernels (for GatedDeltaNet without SM90+ FlashQLA):
ex_engine/fla_kernels/gated_delta_rule/ (7 files, 2370 lines)
— chunk_fwd.py (428), chunk.py (487), wy_fast.py (409),
fused_recurrent.py (392), naive.py (161), gate.py (380)
CCCL sync (12 tuning + 14 dispatch headers updated from NVIDIA/cccl):
cccl_upstream/cub/cub/device/dispatch/tuning/ — 12 changed files synced
cccl_upstream/cub/cub/device/dispatch/ — 14 changed dispatch files synced
Compilation targets for real machine (ivcore10):
1. CUDA kernels: --cuda-gpu-arch=ivcore10 via corex clang/16
2. ILU bridges: torch.utils.cpp_extension linking ixformer .so
3. FLA kernels: Triton JIT (if Triton works on BI-V100)
117 lines
3.3 KiB
Plaintext
117 lines
3.3 KiB
Plaintext
/* Copyright 2025-2026 The xLLM Authors. All Rights Reserved.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
https://github.com/jd-opensource/xllm/blob/main/LICENSE
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
==============================================================================*/
|
|
|
|
#pragma once
|
|
|
|
#if defined(USE_DCU)
|
|
#include <hip/amd_detail/amd_hip_bf16.h>
|
|
|
|
#include <hipcub/hipcub.hpp>
|
|
|
|
namespace cub = hipcub;
|
|
#else
|
|
#include <cub/cub.cuh>
|
|
#if CUB_VERSION >= 200800
|
|
#include <cuda/functional>
|
|
#endif
|
|
#endif
|
|
|
|
namespace xllm::kernel::cuda {
|
|
#if !defined(USE_DCU)
|
|
using BFloat16Type = __nv_bfloat16;
|
|
|
|
#define WARP_SIZE 32
|
|
#define XLLM_KERNEL_ATTR(MAX_THREADS)
|
|
#else
|
|
using BFloat16Type = hip_bfloat16;
|
|
|
|
#define WARP_SIZE 64
|
|
#define XLLM_KERNEL_ATTR(MAX_THREADS) __launch_bounds__(MAX_THREADS, 1)
|
|
#endif
|
|
#define MAX(a, b) ((a) > (b) ? (a) : (b))
|
|
#define MIN(a, b) ((a) < (b) ? (a) : (b))
|
|
|
|
// Aligned array type
|
|
template <typename T,
|
|
// Number of elements in the array
|
|
int N,
|
|
// Alignment requirement in bytes
|
|
int Alignment = sizeof(T) * N>
|
|
class alignas(Alignment) AlignedArray {
|
|
T data[N];
|
|
};
|
|
|
|
#define XLLM_SHFL_XOR_SYNC(mask, var, lane_mask) \
|
|
__shfl_xor_sync((mask), (var), (lane_mask))
|
|
#define XLLM_SHFL_XOR_SYNC_WIDTH(mask, var, lane_mask, width) \
|
|
__shfl_xor_sync((mask), (var), (lane_mask), (width))
|
|
|
|
template <typename T>
|
|
__device__ __forceinline__ T xllm_ldg(const T* ptr) {
|
|
#if defined(USE_DCU)
|
|
return *ptr;
|
|
#else
|
|
return __ldg(ptr);
|
|
#endif
|
|
}
|
|
|
|
// Define reduction operators based on CUB version.
|
|
#if defined(USE_DCU)
|
|
using MaxReduceOp = hipcub::Max;
|
|
using MinReduceOp = hipcub::Min;
|
|
#elif CUB_VERSION >= 200800
|
|
using MaxReduceOp = ::cuda::maximum<>;
|
|
using MinReduceOp = ::cuda::minimum<>;
|
|
#else
|
|
using MaxReduceOp = cub::Max;
|
|
using MinReduceOp = cub::Min;
|
|
#endif
|
|
|
|
template <typename T>
|
|
__device__ float convert_to_float(T x) {
|
|
if constexpr (std::is_same_v<T, __half>) {
|
|
return __half2float(x);
|
|
#if defined(USE_DCU)
|
|
} else if constexpr (std::is_same_v<T, hip_bfloat16>) {
|
|
return __bfloat162float(reinterpret_cast<const __hip_bfloat16&>(x));
|
|
#else
|
|
} else if constexpr (std::is_same_v<T, __nv_bfloat16>) {
|
|
return __bfloat162float(x);
|
|
#endif
|
|
|
|
} else if constexpr (std::is_same_v<T, float>) {
|
|
return x;
|
|
} else {
|
|
return static_cast<float>(x);
|
|
}
|
|
}
|
|
|
|
// Constructs some constants needed to partition the work across threads at
|
|
// compile time.
|
|
template <typename T, int EXPERTS, int BYTES_PER_LDG>
|
|
struct TopkConstants {
|
|
static constexpr int ELTS_PER_LDG = BYTES_PER_LDG / sizeof(T);
|
|
static_assert(EXPERTS / (ELTS_PER_LDG * WARP_SIZE) == 0 ||
|
|
EXPERTS % (ELTS_PER_LDG * WARP_SIZE) == 0,
|
|
"");
|
|
static constexpr int VECs_PER_THREAD =
|
|
MAX(1, EXPERTS / (ELTS_PER_LDG * WARP_SIZE));
|
|
static constexpr int VPT = VECs_PER_THREAD * ELTS_PER_LDG;
|
|
static constexpr int THREADS_PER_ROW = EXPERTS / VPT;
|
|
static constexpr int ROWS_PER_WARP = WARP_SIZE / THREADS_PER_ROW;
|
|
};
|
|
|
|
} // namespace xllm::kernel::cuda
|