Files
project_6/ex_engine/xllm_kernels/cuda/headers/utils.h
claude 8d75652949 feat: import CUDA kernels from xllm/CCCL/FLA upstream repos
Sources cloned and tree'd (no --depth):
  - jd-opensource/xllm: ILU kernels, CUDA kernels, MoE kernels
  - NVIDIA/cccl: CUB tuning/dispatch headers (block-level primitives)
  - fla-org/flash-linear-attention: Triton GDN kernels
  - NVIDIA/cutlass: grouped GEMM reference (read, not copied)
  - Dao-AILab/flash-attention: attention kernel reference (SM80+, read only)

New CUDA kernels (from xllm, SM-agnostic, portable to BI-V100):
  ex_engine/xllm_kernels/cuda/activation.cu    (188 lines) — silu_and_mul, gelu
  ex_engine/xllm_kernels/cuda/norm.cu          (600 lines) — rms_norm, fused_add_rms_norm
  ex_engine/xllm_kernels/cuda/rope.cu          (258 lines) — rotary_embedding
  ex_engine/xllm_kernels/cuda/block_copy.cu    (209 lines) — copy_blocks, swap_blocks
  ex_engine/xllm_kernels/cuda/reshape_paged_cache.cu (101 lines) — KV cache ops
  ex_engine/xllm_kernels/cuda/headers/         (5 headers for compilation)

ILU bridge kernel sources (from xllm, verified SAME as upstream):
  ex_engine/xllm_kernels/ilu/    (10 files, 925 lines total)
  — activation.cpp, attention.cpp, fused_moe.cpp, group_gemm.cpp,
    matmul.cpp, norm.cpp, rope.cpp, ilu_ops_api.h, ixformer.h, utils.h

FLA Triton GDN kernels (for GatedDeltaNet without SM90+ FlashQLA):
  ex_engine/fla_kernels/gated_delta_rule/  (7 files, 2370 lines)
  — chunk_fwd.py (428), chunk.py (487), wy_fast.py (409),
    fused_recurrent.py (392), naive.py (161), gate.py (380)

CCCL sync (12 tuning + 14 dispatch headers updated from NVIDIA/cccl):
  cccl_upstream/cub/cub/device/dispatch/tuning/ — 12 changed files synced
  cccl_upstream/cub/cub/device/dispatch/ — 14 changed dispatch files synced

Compilation targets for real machine (ivcore10):
  1. CUDA kernels: --cuda-gpu-arch=ivcore10 via corex clang/16
  2. ILU bridges: torch.utils.cpp_extension linking ixformer .so
  3. FLA kernels: Triton JIT (if Triton works on BI-V100)
2026-08-14 07:48:52 +00:00

164 lines
5.6 KiB
C++

/* Copyright 2025-2026 The xLLM Authors.
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
https://github.com/jd-opensource/xllm/blob/main/LICENSE
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
==============================================================================*/
#pragma once
#include <ATen/DynamicLibrary.h>
#if defined(USE_DCU)
#include <c10/hip/HIPGuard.h>
#else
#include <c10/cuda/CUDAGuard.h>
#endif
#include <glog/logging.h>
#include <torch/torch.h>
#if !defined(USE_DCU)
#include <tvm/ffi/container/array.h>
#include <tvm/ffi/container/tensor.h>
#include <tvm/ffi/extra/c_env_api.h>
#include <tvm/ffi/extra/module.h>
#include <tvm/ffi/optional.h>
#endif
#include <string>
#include <tuple>
#include <type_traits>
#include <unordered_map>
#if defined(__CUDACC__) || defined(_NVHPC_CUDA) || defined(__HIPCC__)
#define HOST_DEVICE_INLINE __host__ __device__ __forceinline__
#define DEVICE_INLINE __device__ __forceinline__
#define HOST_INLINE __host__ __forceinline__
#else
#define HOST_DEVICE_INLINE inline
#define DEVICE_INLINE inline
#define HOST_INLINE inline
#endif
#if !defined(USE_DCU)
namespace ffi = tvm::ffi;
#endif
namespace xllm::kernel::cuda {
template <typename T>
HOST_DEVICE_INLINE constexpr std::enable_if_t<std::is_integral_v<T>, T>
ceil_div(T a, T b) {
return (a + b - 1) / b;
}
enum class ActivationType : int8_t {
GELU = 0,
RELU = 1,
SILU = 2,
SWIGLU = 3,
GEGLU = 4,
SWIGLU_BIAS = 5,
RELU2 = 6,
IDENTITY = 7,
INVALID_TYPE = 8
};
// torch tensor is only on cpu
torch::Tensor get_cache_buffer(const int32_t seq_len,
const torch::Device& device);
// NOLINTBEGIN(cppcoreguidelines-macro-usage)
#define DISPATCH_CASE_FLOATING_TYPES(...) \
AT_DISPATCH_CASE(at::ScalarType::Float, __VA_ARGS__) \
AT_DISPATCH_CASE(at::ScalarType::Half, __VA_ARGS__) \
AT_DISPATCH_CASE(at::ScalarType::BFloat16, __VA_ARGS__)
#define DISPATCH_FLOATING_TYPES(TYPE, NAME, ...) \
AT_DISPATCH_SWITCH(TYPE, NAME, DISPATCH_CASE_FLOATING_TYPES(__VA_ARGS__))
#define DISPATCH_CASE_HALF_TYPES(...) \
AT_DISPATCH_CASE(at::ScalarType::Half, __VA_ARGS__) \
AT_DISPATCH_CASE(at::ScalarType::BFloat16, __VA_ARGS__)
#define DISPATCH_HALF_TYPES(TYPE, NAME, ...) \
AT_DISPATCH_SWITCH(TYPE, NAME, DISPATCH_CASE_HALF_TYPES(__VA_ARGS__))
// NOLINTEND(cppcoreguidelines-macro-usage)
bool should_use_tensor_core(torch::ScalarType kv_cache_dtype,
int64_t num_attention_heads,
int64_t num_kv_heads);
bool support_pdl();
std::string path_to_uri_so_lib(const std::string& uri);
std::string determine_attention_backend(int64_t pos_encoding_mode,
bool use_fp16_qk_reduction,
bool use_custom_mask);
std::string get_batch_prefill_uri(const std::string& backend,
torch::ScalarType dtype_q,
torch::ScalarType dtype_kv,
torch::ScalarType dtype_o,
torch::ScalarType dtype_idx,
int64_t head_dim_qk,
int64_t head_dim_vo,
int64_t pos_encoding_mode,
bool use_sliding_window,
bool use_logits_soft_cap,
bool use_fp16_qk_reduction);
std::string get_batch_decode_uri(torch::ScalarType dtype_q,
torch::ScalarType dtype_kv,
torch::ScalarType dtype_o,
torch::ScalarType dtype_idx,
int64_t head_dim_qk,
int64_t head_dim_vo,
int64_t pos_encoding_mode,
bool use_sliding_window,
bool use_logits_soft_cap);
std::tuple<torch::Tensor, double> split_scale_param(const torch::Tensor& scale);
#if !defined(USE_DCU)
DLDataType to_dl_data_type(torch::ScalarType scalar_type);
// below are tvm-ffi related functions
ffi::Tensor to_ffi_tensor(const torch::Tensor& torch_tensor);
ffi::Optional<ffi::Tensor> to_ffi_optional_tensor(
const std::optional<torch::Tensor>& optional);
ffi::Array<ffi::Tensor> to_ffi_array_tensors(
const std::vector<torch::Tensor>& torch_tensors);
ffi::Optional<ffi::Array<ffi::Tensor>> to_ffi_optional_array_tensors(
const std::optional<std::vector<torch::Tensor>>& optional);
ffi::Module get_module(const std::string& uri);
ffi::Function get_function(const std::string& uri,
const std::string& func_name);
inline void bind_tvmffi_stream_to_current_torch_stream(
const torch::Device& device) {
const auto cur = c10::cuda::getCurrentCUDAStream(device.index());
// DLPack device type for CUDA is 2 (kDLCUDA).
void* original_stream = nullptr;
const int rc = TVMFFIEnvSetStream(
/*device_type=*/2,
/*device_id=*/device.index(),
reinterpret_cast<void*>(cur.stream()),
&original_stream);
if (rc != 0) {
LOG(WARNING) << "[tvmffi.stream] failed to set stream, rc=" << rc
<< " dev=" << device.index();
}
}
#endif // !defined(USE_DCU)
} // namespace xllm::kernel::cuda