Sources (Apache 2.0, cloned 2026-08-09): - Deep-Spark/xllm: Iluvatar's official C++ inference engine - Deep-Spark/vllm: Iluvatar's vllm fork Key files for our EX Engine development: MoE topk_softmax (fixes 2304 calls/token PyTorch fallback): - xllm/kernels/cuda/moe/moe_topk_softmax_kernels.cuh CUB-based fused softmax+topk, power-of-2 expert count optimized For 64 experts: topk_gating_softmax<T,VPT=2,64,WARPS=4,BYTES=4> - xllm/kernels/ilu/ixformer.h Official ixformer C++ API: topk_softmax(), paged_attention(), etc. - xllm/kernels/ilu/fused_moe.cpp How xllm calls ixformer::infer::topk_softmax() - ds_vllm/csrc/moe/topk_softmax_kernels.cu vllm-native topk_softmax (TensorRT-LLM derived, 874 lines) GatedDeltaNet (fixes NaN in 4 GDN layers): - xllm/layers/npu_torch/qwen3_gated_delta_net_base.cpp fp32 state accumulation, proper recurrent update Complete FusedMoE pipeline reference: - xllm/layers/ilu/fused_moe.cpp gate -> topk -> expand -> gemm1 -> act -> gemm2 -> combine
80 lines
2.6 KiB
Plaintext
80 lines
2.6 KiB
Plaintext
/* Copyright 2025 The xLLM Authors. All Rights Reserved.
|
|
|
|
Licensed under the Apache License, Version 2.0 (the "License");
|
|
you may not use this file except in compliance with the License.
|
|
You may obtain a copy of the License at
|
|
|
|
https://github.com/jd-opensource/xllm/blob/main/LICENSE
|
|
|
|
Unless required by applicable law or agreed to in writing, software
|
|
distributed under the License is distributed on an "AS IS" BASIS,
|
|
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
See the License for the specific language governing permissions and
|
|
limitations under the License.
|
|
==============================================================================*/
|
|
|
|
#pragma once
|
|
|
|
#include <cub/cub.cuh>
|
|
|
|
namespace xllm::kernel::cuda {
|
|
|
|
#define WARP_SIZE 32
|
|
|
|
#define MAX(a, b) ((a) > (b) ? (a) : (b))
|
|
#define MIN(a, b) ((a) < (b) ? (a) : (b))
|
|
|
|
// Aligned array type
|
|
template <typename T,
|
|
// Number of elements in the array
|
|
int N,
|
|
// Alignment requirement in bytes
|
|
int Alignment = sizeof(T) * N>
|
|
class alignas(Alignment) AlignedArray {
|
|
T data[N];
|
|
};
|
|
|
|
#define XLLM_SHFL_XOR_SYNC(mask, var, lane_mask) \
|
|
__shfl_xor_sync((mask), (var), (lane_mask))
|
|
#define XLLM_SHFL_XOR_SYNC_WIDTH(mask, var, lane_mask, width) \
|
|
__shfl_xor_sync((mask), (var), (lane_mask), (width))
|
|
|
|
// Define reduction operators based on CUDA version
|
|
// CUDA 13 (12.9+) deprecated cub::Max/Min in favor of cuda::maximum/minimum
|
|
#if CUDA_VERSION >= 12090
|
|
using MaxReduceOp = ::cuda::maximum<>;
|
|
using MinReduceOp = ::cuda::minimum<>;
|
|
#else
|
|
using MaxReduceOp = cub::Max;
|
|
using MinReduceOp = cub::Min;
|
|
#endif
|
|
|
|
template <typename T>
|
|
__device__ float convert_to_float(T x) {
|
|
if constexpr (std::is_same_v<T, __half>) {
|
|
return __half2float(x);
|
|
} else if constexpr (std::is_same_v<T, __nv_bfloat16>) {
|
|
return __bfloat162float(x);
|
|
} else if constexpr (std::is_same_v<T, float>) {
|
|
return x;
|
|
} else {
|
|
return static_cast<float>(x);
|
|
}
|
|
}
|
|
|
|
// Constructs some constants needed to partition the work across threads at
|
|
// compile time.
|
|
template <typename T, int EXPERTS, int BYTES_PER_LDG>
|
|
struct TopkConstants {
|
|
static constexpr int ELTS_PER_LDG = BYTES_PER_LDG / sizeof(T);
|
|
static_assert(EXPERTS / (ELTS_PER_LDG * WARP_SIZE) == 0 ||
|
|
EXPERTS % (ELTS_PER_LDG * WARP_SIZE) == 0,
|
|
"");
|
|
static constexpr int VECs_PER_THREAD =
|
|
MAX(1, EXPERTS / (ELTS_PER_LDG * WARP_SIZE));
|
|
static constexpr int VPT = VECs_PER_THREAD * ELTS_PER_LDG;
|
|
static constexpr int THREADS_PER_ROW = EXPERTS / VPT;
|
|
static constexpr int ROWS_PER_WARP = WARP_SIZE / THREADS_PER_ROW;
|
|
};
|
|
|
|
} // namespace xllm::kernel::cuda |