Sources cloned and tree'd (no --depth):
- jd-opensource/xllm: ILU kernels, CUDA kernels, MoE kernels
- NVIDIA/cccl: CUB tuning/dispatch headers (block-level primitives)
- fla-org/flash-linear-attention: Triton GDN kernels
- NVIDIA/cutlass: grouped GEMM reference (read, not copied)
- Dao-AILab/flash-attention: attention kernel reference (SM80+, read only)
New CUDA kernels (from xllm, SM-agnostic, portable to BI-V100):
ex_engine/xllm_kernels/cuda/activation.cu (188 lines) — silu_and_mul, gelu
ex_engine/xllm_kernels/cuda/norm.cu (600 lines) — rms_norm, fused_add_rms_norm
ex_engine/xllm_kernels/cuda/rope.cu (258 lines) — rotary_embedding
ex_engine/xllm_kernels/cuda/block_copy.cu (209 lines) — copy_blocks, swap_blocks
ex_engine/xllm_kernels/cuda/reshape_paged_cache.cu (101 lines) — KV cache ops
ex_engine/xllm_kernels/cuda/headers/ (5 headers for compilation)
ILU bridge kernel sources (from xllm, verified SAME as upstream):
ex_engine/xllm_kernels/ilu/ (10 files, 925 lines total)
— activation.cpp, attention.cpp, fused_moe.cpp, group_gemm.cpp,
matmul.cpp, norm.cpp, rope.cpp, ilu_ops_api.h, ixformer.h, utils.h
FLA Triton GDN kernels (for GatedDeltaNet without SM90+ FlashQLA):
ex_engine/fla_kernels/gated_delta_rule/ (7 files, 2370 lines)
— chunk_fwd.py (428), chunk.py (487), wy_fast.py (409),
fused_recurrent.py (392), naive.py (161), gate.py (380)
CCCL sync (12 tuning + 14 dispatch headers updated from NVIDIA/cccl):
cccl_upstream/cub/cub/device/dispatch/tuning/ — 12 changed files synced
cccl_upstream/cub/cub/device/dispatch/ — 14 changed dispatch files synced
Compilation targets for real machine (ivcore10):
1. CUDA kernels: --cuda-gpu-arch=ivcore10 via corex clang/16
2. ILU bridges: torch.utils.cpp_extension linking ixformer .so
3. FLA kernels: Triton JIT (if Triton works on BI-V100)
164 lines
5.6 KiB
C++
164 lines
5.6 KiB
C++
/* Copyright 2025-2026 The xLLM Authors.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
https://github.com/jd-opensource/xllm/blob/main/LICENSE
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
==============================================================================*/
|
|
|
|
#pragma once
|
|
|
|
#include <ATen/DynamicLibrary.h>
|
|
#if defined(USE_DCU)
|
|
#include <c10/hip/HIPGuard.h>
|
|
#else
|
|
#include <c10/cuda/CUDAGuard.h>
|
|
#endif
|
|
#include <glog/logging.h>
|
|
#include <torch/torch.h>
|
|
#if !defined(USE_DCU)
|
|
#include <tvm/ffi/container/array.h>
|
|
#include <tvm/ffi/container/tensor.h>
|
|
#include <tvm/ffi/extra/c_env_api.h>
|
|
#include <tvm/ffi/extra/module.h>
|
|
#include <tvm/ffi/optional.h>
|
|
#endif
|
|
|
|
#include <string>
|
|
#include <tuple>
|
|
#include <type_traits>
|
|
#include <unordered_map>
|
|
|
|
#if defined(__CUDACC__) || defined(_NVHPC_CUDA) || defined(__HIPCC__)
|
|
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
|
|
#define DEVICE_INLINE __device__ __forceinline__
|
|
#define HOST_INLINE __host__ __forceinline__
|
|
#else
|
|
#define HOST_DEVICE_INLINE inline
|
|
#define DEVICE_INLINE inline
|
|
#define HOST_INLINE inline
|
|
#endif
|
|
|
|
#if !defined(USE_DCU)
|
|
namespace ffi = tvm::ffi;
|
|
#endif
|
|
|
|
namespace xllm::kernel::cuda {
|
|
|
|
template <typename T>
|
|
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
|
|
ceil_div(T a, T b) {
|
|
return (a + b - 1) / b;
|
|
}
|
|
|
|
enum class ActivationType : int8_t {
|
|
GELU = 0,
|
|
RELU = 1,
|
|
SILU = 2,
|
|
SWIGLU = 3,
|
|
GEGLU = 4,
|
|
SWIGLU_BIAS = 5,
|
|
RELU2 = 6,
|
|
IDENTITY = 7,
|
|
INVALID_TYPE = 8
|
|
};
|
|
|
|
// torch tensor is only on cpu
|
|
torch::Tensor get_cache_buffer(const int32_t seq_len,
|
|
const torch::Device& device);
|
|
|
|
// NOLINTBEGIN(cppcoreguidelines-macro-usage)
|
|
#define DISPATCH_CASE_FLOATING_TYPES(...) \
|
|
AT_DISPATCH_CASE(at::ScalarType::Float, __VA_ARGS__) \
|
|
AT_DISPATCH_CASE(at::ScalarType::Half, __VA_ARGS__) \
|
|
AT_DISPATCH_CASE(at::ScalarType::BFloat16, __VA_ARGS__)
|
|
#define DISPATCH_FLOATING_TYPES(TYPE, NAME, ...) \
|
|
AT_DISPATCH_SWITCH(TYPE, NAME, DISPATCH_CASE_FLOATING_TYPES(__VA_ARGS__))
|
|
#define DISPATCH_CASE_HALF_TYPES(...) \
|
|
AT_DISPATCH_CASE(at::ScalarType::Half, __VA_ARGS__) \
|
|
AT_DISPATCH_CASE(at::ScalarType::BFloat16, __VA_ARGS__)
|
|
#define DISPATCH_HALF_TYPES(TYPE, NAME, ...) \
|
|
AT_DISPATCH_SWITCH(TYPE, NAME, DISPATCH_CASE_HALF_TYPES(__VA_ARGS__))
|
|
// NOLINTEND(cppcoreguidelines-macro-usage)
|
|
|
|
bool should_use_tensor_core(torch::ScalarType kv_cache_dtype,
|
|
int64_t num_attention_heads,
|
|
int64_t num_kv_heads);
|
|
|
|
bool support_pdl();
|
|
|
|
std::string path_to_uri_so_lib(const std::string& uri);
|
|
|
|
std::string determine_attention_backend(int64_t pos_encoding_mode,
|
|
bool use_fp16_qk_reduction,
|
|
bool use_custom_mask);
|
|
|
|
std::string get_batch_prefill_uri(const std::string& backend,
|
|
torch::ScalarType dtype_q,
|
|
torch::ScalarType dtype_kv,
|
|
torch::ScalarType dtype_o,
|
|
torch::ScalarType dtype_idx,
|
|
int64_t head_dim_qk,
|
|
int64_t head_dim_vo,
|
|
int64_t pos_encoding_mode,
|
|
bool use_sliding_window,
|
|
bool use_logits_soft_cap,
|
|
bool use_fp16_qk_reduction);
|
|
|
|
std::string get_batch_decode_uri(torch::ScalarType dtype_q,
|
|
torch::ScalarType dtype_kv,
|
|
torch::ScalarType dtype_o,
|
|
torch::ScalarType dtype_idx,
|
|
int64_t head_dim_qk,
|
|
int64_t head_dim_vo,
|
|
int64_t pos_encoding_mode,
|
|
bool use_sliding_window,
|
|
bool use_logits_soft_cap);
|
|
|
|
std::tuple<torch::Tensor, double> split_scale_param(const torch::Tensor& scale);
|
|
|
|
#if !defined(USE_DCU)
|
|
DLDataType to_dl_data_type(torch::ScalarType scalar_type);
|
|
|
|
// below are tvm-ffi related functions
|
|
ffi::Tensor to_ffi_tensor(const torch::Tensor& torch_tensor);
|
|
|
|
ffi::Optional<ffi::Tensor> to_ffi_optional_tensor(
|
|
const std::optional<torch::Tensor>& optional);
|
|
|
|
ffi::Array<ffi::Tensor> to_ffi_array_tensors(
|
|
const std::vector<torch::Tensor>& torch_tensors);
|
|
|
|
ffi::Optional<ffi::Array<ffi::Tensor>> to_ffi_optional_array_tensors(
|
|
const std::optional<std::vector<torch::Tensor>>& optional);
|
|
|
|
ffi::Module get_module(const std::string& uri);
|
|
|
|
ffi::Function get_function(const std::string& uri,
|
|
const std::string& func_name);
|
|
|
|
inline void bind_tvmffi_stream_to_current_torch_stream(
|
|
const torch::Device& device) {
|
|
const auto cur = c10::cuda::getCurrentCUDAStream(device.index());
|
|
// DLPack device type for CUDA is 2 (kDLCUDA).
|
|
void* original_stream = nullptr;
|
|
const int rc = TVMFFIEnvSetStream(
|
|
/*device_type=*/2,
|
|
/*device_id=*/device.index(),
|
|
reinterpret_cast<void*>(cur.stream()),
|
|
&original_stream);
|
|
if (rc != 0) {
|
|
LOG(WARNING) << "[tvmffi.stream] failed to set stream, rc=" << rc
|
|
<< " dev=" << device.index();
|
|
}
|
|
}
|
|
#endif // !defined(USE_DCU)
|
|
} // namespace xllm::kernel::cuda
|