[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,319 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___WARP_LANE_MASK_H
#define _CUDA___WARP_LANE_MASK_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_CUDA_COMPILATION()
# include <cuda/__ptx/instructions/get_sreg.h>
# include <cuda/std/cstdint>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
//! @brief A class representing a lane mask in a warp.
class lane_mask
{
::cuda::std::uint32_t __value_;
public:
//! @brief Constructs a lane mask object from a 32-bit unsigned integer.
//!
//! @param __v The value to initialize the lane mask with. Defaults to 0.
//!
//! @post `value() == __v`
_CCCL_DEVICE_API explicit constexpr lane_mask(::cuda::std::uint32_t __v = 0) noexcept
: __value_{__v}
{}
//! @brief Returns the value of the lane mask as a 32-bit unsigned integer.
//!
//! @return The value of the lane mask.
[[nodiscard]] _CCCL_DEVICE_API constexpr ::cuda::std::uint32_t value() const noexcept
{
return __value_;
}
//! @brief Converts the lane mask to a 32-bit unsigned integer.
//!
//! This operator allows explicit conversion of the lane mask to a 32-bit unsigned integer.
//!
//! @return The value of the lane mask as a 32-bit unsigned integer.
_CCCL_DEVICE_API explicit constexpr operator ::cuda::std::uint32_t() const noexcept
{
return __value_;
}
//! @brief Returns a lane mask object with no lane bits set.
//!
//! @return A lane mask with no lane bits set.
[[nodiscard]] _CCCL_DEVICE_API static constexpr lane_mask none() noexcept
{
return lane_mask{};
}
//! @brief Returns a lane mask object with all lane bits set.
//!
//! @return A lane mask with all lane bits set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits,
[[nodiscard]] _CCCL_DEVICE_API static constexpr lane_mask all() noexcept
{
return lane_mask{0xffffffff};
}
//! @brief Returns a lane mask object with all currently active lane bits set.
//!
//! This function returns a lane_mask object equivalent to calling `lane_mask{::__activemask()}`.
//!
//! @return A lane mask with all active lane bits set.
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_active() noexcept
{
return lane_mask{::__activemask()};
}
//! @brief Returns a lane mask object with the current lane bit set.
//!
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_eq PTX special register.
//!
//! @return A lane mask with the current lane bit set.
[[nodiscard]] _CCCL_DEVICE_API static lane_mask this_lane() noexcept
{
return lane_mask{::cuda::ptx::get_sreg_lanemask_eq()};
}
//! @brief Returns a lane mask object with all lanes less than the current lane set.
//!
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_lt PTX special register.
//!
//! @return A lane mask with all lanes less than the current lane set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits,
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_less() noexcept
{
return lane_mask{::cuda::ptx::get_sreg_lanemask_lt()};
}
//! @brief Returns a lane mask object with all lanes equal to or less than the current lane set.
//!
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_le PTX special register.
//!
//! @return A lane mask with all lanes equal to or less than the current lane set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits,
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_less_equal() noexcept
{
return lane_mask{::cuda::ptx::get_sreg_lanemask_le()};
}
//! @brief Returns a lane mask object with all lanes greater than the current lane set.
//!
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_gt PTX special register.
//!
//! @return A lane mask with all lanes greater than the current lane set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits,
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_greater() noexcept
{
return lane_mask{::cuda::ptx::get_sreg_lanemask_gt()};
}
//! @brief Returns a lane mask object with all lanes greater than or equal to the current lane set.
//!
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_ge PTX special register.
//!
//! @return A lane mask with all lanes greater than or equal to the current lane set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits,
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_greater_equal() noexcept
{
return lane_mask{::cuda::ptx::get_sreg_lanemask_ge()};
}
//! @brief Returns a lane mask object with all lanes not equal to the current lane set.
//!
//! This function is equivalent to constructing a lane_mask object with a negated value of %%lanemask_eq PTX special
//! register.
//!
//! @return A lane mask with all lanes not equal to the current lane set.
//!
//! @note This function may return a mask with 1s set even on inactive lane bits.
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_not_equal() noexcept
{
return lane_mask{~::cuda::ptx::get_sreg_lanemask_eq()};
}
//! @brief Bitwise AND operator for lane_mask.
//!
//! @param __lhs The left-hand side lane_mask.
//! @param __rhs The right-hand side lane_mask.
//!
//! @return A new lane_mask object representing the bitwise AND of the two lane_masks.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator&(lane_mask __lhs, lane_mask __rhs) noexcept
{
return lane_mask{__lhs.__value_ & __rhs.__value_};
}
//! @brief Bitwise AND assignment operator for lane_mask.
//!
//! @param __v The lane_mask to AND with the current lane_mask.
//!
//! @return A reference to the current lane_mask after the AND operation.
_CCCL_DEVICE_API constexpr lane_mask& operator&=(lane_mask __v) noexcept
{
return *this = *this & __v;
}
//! @brief Bitwise OR operator for lane_mask.
//!
//! @param __lhs The left-hand side lane_mask.
//! @param __rhs The right-hand side lane_mask.
//!
//! @return A new lane_mask object representing the bitwise OR of the two lane_masks.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator|(lane_mask __lhs, lane_mask __rhs) noexcept
{
return lane_mask{__lhs.__value_ | __rhs.__value_};
}
//! @brief Bitwise OR assignment operator for lane_mask.
//!
//! @param __v The lane_mask to OR with the current lane_mask.
//!
//! @return A reference to the current lane_mask after the OR operation.
_CCCL_DEVICE_API constexpr lane_mask& operator|=(lane_mask __v) noexcept
{
return *this = *this | __v;
}
//! @brief Bitwise XOR operator for lane_mask.
//!
//! @param __lhs The left-hand side lane_mask.
//! @param __rhs The right-hand side lane_mask.
//!
//! @return A new lane_mask object representing the bitwise XOR of the two lane_masks.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator^(lane_mask __lhs, lane_mask __rhs) noexcept
{
return lane_mask{__lhs.__value_ ^ __rhs.__value_};
}
//! @brief Bitwise XOR assignment operator for lane_mask.
//!
//! @param __v The lane_mask to XOR with the current lane_mask.
//!
//! @return A reference to the current lane_mask after the XOR operation.
_CCCL_DEVICE_API constexpr lane_mask& operator^=(lane_mask __v) noexcept
{
return *this = *this ^ __v;
}
//! @brief Left shift operator for lane_mask.
//!
//! @param __mask The lane_mask to shift.
//! @param __shift The number of bits to shift left.
//!
//! @return A new lane_mask object representing the left-shifted lane_mask.
//!
//! @pre `__shift` must be in the range [0, 32).
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator<<(lane_mask __mask, int __shift) noexcept
{
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
return lane_mask{__mask.__value_ << __shift};
}
//! @brief Left shift assignment operator for lane_mask.
//!
//! @param __shift The number of bits to shift left.
//!
//! @return A reference to the current lane_mask after the left shift operation.
//!
//! @pre `__shift` must be in the range [0, 32).
_CCCL_DEVICE_API constexpr lane_mask& operator<<=(int __shift) noexcept
{
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
return *this = *this << __shift;
}
//! @brief Right shift operator for lane_mask.
//!
//! @param __mask The lane_mask to shift.
//! @param __shift The number of bits to shift right.
//!
//! @return A new lane_mask object representing the right-shifted lane_mask.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator>>(lane_mask __mask, int __shift) noexcept
{
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
return lane_mask{__mask.__value_ >> __shift};
}
//! @brief Right shift assignment operator for lane_mask.
//!
//! @param __shift The number of bits to shift right.
//!
//! @return A reference to the current lane_mask after the right shift operation.
//!
//! @pre `__shift` must be in the range [0, 32).
_CCCL_DEVICE_API constexpr lane_mask& operator>>=(int __shift) noexcept
{
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
return *this = *this >> __shift;
}
//! @brief Bitwise NOT operator for lane_mask.
//!
//! @param __mask The lane_mask to negate.
//!
//! @return A new lane_mask object representing the negated lane_mask.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator~(lane_mask __mask) noexcept
{
return lane_mask{~__mask.__value_};
}
//! @brief Equality operator for lane_mask.
//!
//! @param __lhs The left-hand side lane_mask.
//! @param __rhs The right-hand side lane_mask.
//!
//! @return `true` if the two lane_masks are equal, `false` otherwise.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr bool operator==(lane_mask __lhs, lane_mask __rhs) noexcept
{
return __lhs.__value_ == __rhs.__value_;
}
//! @brief Inequality operator for lane_mask.
//!
//! @param __lhs The left-hand side lane_mask.
//! @param __rhs The right-hand side lane_mask.
//!
//! @return `true` if the two lane_masks are not equal, `false` otherwise.
[[nodiscard]] _CCCL_DEVICE_API friend constexpr bool operator!=(lane_mask __lhs, lane_mask __rhs) noexcept
{
return !(__lhs == __rhs);
}
};
_CCCL_END_NAMESPACE_CUDA_DEVICE
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_CUDA_COMPILATION()
#endif // _CUDA___WARP_LANE_MASK_H

View File

@@ -0,0 +1,87 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___WARP_WARP_MATCH_H
#define _CUDA___WARP_WARP_MATCH_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_CUDA_COMPILATION()
# include <cuda/__cmath/ceil_div.h>
# include <cuda/__type_traits/is_bitwise_comparable.h>
# include <cuda/__type_traits/is_trivially_copyable.h>
# include <cuda/__warp/lane_mask.h>
# include <cuda/std/__cstring/memcpy.h>
# include <cuda/std/__memory/addressof.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/cstdint>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
extern "C" _CCCL_DEVICE void __cuda__match_all_sync_is_not_supported_before_SM_70__();
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API bool
warp_match_all(const _Tp& __data, const lane_mask __lane_mask = lane_mask::all()) noexcept
{
static_assert(is_trivially_copyable_v<_Tp>, "data must be trivially copyable");
_CCCL_ASSERT(__lane_mask != lane_mask::none(), "lane_mask must be non-zero");
if constexpr (::cuda::std::is_same_v<_Tp, bool>)
{
const auto __mask = ::__ballot_sync(__lane_mask.value(), __data);
return (__mask == __lane_mask.value() || __mask == 0);
}
else
{
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Tp), sizeof(::cuda::std::uint32_t));
::cuda::std::uint32_t __array[__ratio]{};
# if defined(_CCCL_BUILTIN_CLEAR_PADDING)
auto __data_copy = __data;
_CCCL_BUILTIN_CLEAR_PADDING(&__data_copy);
const auto __data_ptr = ::cuda::std::addressof(__data_copy);
# else // ^^^ _CCCL_BUILTIN_CLEAR_PADDING ^^^ / vvv !_CCCL_BUILTIN_CLEAR_PADDING vvv
static_assert(is_bitwise_comparable_v<_Tp>, "data must be bitwise comparable");
const auto __data_ptr = ::cuda::std::addressof(__data);
# endif // _CCCL_BUILTIN_CLEAR_PADDING
::cuda::std::memcpy(__array, __data_ptr, sizeof(_Tp));
bool __ret = true;
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
int __pred = false;
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_70,
(::__match_all_sync(__lane_mask.value(), __array[i], &__pred);),
(::cuda::device::__cuda__match_all_sync_is_not_supported_before_SM_70__();));
__ret = __ret && __pred;
}
return __ret;
}
}
_CCCL_END_NAMESPACE_CUDA_DEVICE
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_CUDA_COMPILATION()
#endif // _CUDA___WARP_WARP_MATCH_H

View File

@@ -0,0 +1,97 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___WARP_WARP_MATCH_ANY_H
#define _CUDA___WARP_WARP_MATCH_ANY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_CUDA_COMPILATION()
# include <cuda/__cmath/ceil_div.h>
# include <cuda/__type_traits/is_bitwise_comparable.h>
# include <cuda/__type_traits/is_trivially_copyable.h>
# include <cuda/__warp/lane_mask.h>
# include <cuda/std/__cstring/memcpy.h>
# include <cuda/std/__memory/addressof.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/cstdint>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
extern "C" _CCCL_DEVICE void __cuda__match_any_sync_is_not_supported_before_SM_70__();
//! @brief Returns the mask of lanes with the same bitwise value as the calling lane.
//!
//! @param[in] __data The data to compare across lanes.
//! @param[in] __lane_mask The mask of participating lanes.
//!
//! @return A lane mask containing lanes in `__lane_mask` whose `__data` matches the calling lane's data.
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API lane_mask
warp_match_any(const _Tp& __data, const lane_mask __lane_mask = lane_mask::all()) noexcept
{
static_assert(is_trivially_copyable_v<_Tp>, "data must be trivially copyable");
_CCCL_ASSERT(__lane_mask != lane_mask::none(), "lane_mask must be non-zero");
if constexpr (::cuda::std::is_same_v<_Tp, bool>)
{
auto __mask = ::__ballot_sync(__lane_mask.value(), __data);
if (!__data)
{
__mask = (~__mask) & __lane_mask.value();
}
return lane_mask{__mask};
}
else
{
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Tp), sizeof(::cuda::std::uint32_t));
::cuda::std::uint32_t __array[__ratio]{};
# if defined(_CCCL_BUILTIN_CLEAR_PADDING)
auto __data_copy = __data;
_CCCL_BUILTIN_CLEAR_PADDING(&__data_copy);
const auto __data_ptr = ::cuda::std::addressof(__data_copy);
# else // ^^^ _CCCL_BUILTIN_CLEAR_PADDING ^^^ / vvv !_CCCL_BUILTIN_CLEAR_PADDING vvv
static_assert(is_bitwise_comparable_v<_Tp>, "data must be bitwise comparable");
const auto __data_ptr = ::cuda::std::addressof(__data);
# endif // _CCCL_BUILTIN_CLEAR_PADDING
::cuda::std::memcpy(__array, __data_ptr, sizeof(_Tp));
lane_mask __ret = __lane_mask;
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
::cuda::std::uint32_t __match_any_result = 0;
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_70,
(__match_any_result = ::__match_any_sync(__lane_mask.value(), __array[i]);),
(::cuda::device::__cuda__match_any_sync_is_not_supported_before_SM_70__();));
__ret &= lane_mask{__match_any_result};
}
return __ret;
}
}
_CCCL_END_NAMESPACE_CUDA_DEVICE
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_CUDA_COMPILATION()
#endif // _CUDA___WARP_WARP_MATCH_ANY_H

View File

@@ -0,0 +1,249 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___WARP_WARP_SHUFFLE_H
#define _CUDA___WARP_WARP_SHUFFLE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_CUDA_COMPILATION()
# if __cccl_ptx_isa >= 600
# include <cuda/__cmath/ceil_div.h>
# include <cuda/__cmath/pow2.h>
# include <cuda/__ptx/instructions/get_sreg.h>
# include <cuda/__ptx/instructions/shfl_sync.h>
# include <cuda/std/__memory/addressof.h>
# include <cuda/std/__type_traits/enable_if.h>
# include <cuda/std/__type_traits/integral_constant.h>
# include <cuda/std/__type_traits/is_default_constructible.h>
# include <cuda/std/__type_traits/is_pointer.h>
# include <cuda/std/cstdint>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
template <typename _Tp>
struct warp_shuffle_result
{
_Tp data;
bool pred;
template <typename _Up = _Tp>
[[nodiscard]] _CCCL_DEVICE_API operator ::cuda::std::enable_if_t<!::cuda::std::is_array_v<_Up>, _Up>() const
{
return data;
}
};
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_idx(
const _Tp& __data, int __src_lane, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
{
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
constexpr auto __warp_size = 32u;
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
"non-void pointers are not allowed to prevent bug-prone code");
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
"_Width must be a power of 2 and less or equal to the warp size");
if constexpr (_Width == 1)
{
return warp_shuffle_result<_Up>{__data, true};
}
else
{
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
bool __pred;
uint32_t __array[__ratio];
::cuda::std::memcpy(
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
__array[i] = ::cuda::ptx::shfl_sync_idx(__array[i], __pred, __src_lane, __clamp_segmask, __lane_mask);
}
warp_shuffle_result<_Up> __result;
__result.pred = __pred;
::cuda::std::memcpy(
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
return __result;
}
}
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
warp_shuffle_idx(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
{
return ::cuda::device::warp_shuffle_idx(__data, __src_lane, 0xFFFFFFFF, __width);
}
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Tp> warp_shuffle_up(
const _Tp& __data, int __delta, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
{
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
constexpr auto __warp_size = 32u;
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
"non-void pointers are not allowed to prevent bug-prone code");
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
"_Width must be a power of 2 and less or equal to the warp size");
if constexpr (_Width == 1)
{
_CCCL_ASSERT(__delta == 0, "delta must be 0 when Width == 1");
return warp_shuffle_result<_Up>{__data, true};
}
else
{
_CCCL_ASSERT(__delta >= 0 && __delta < _Width, "delta must be in the range [0, _Width)");
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
auto __clamp_segmask = (__warp_size - _Width) << 8;
bool __pred;
uint32_t __array[__ratio];
::cuda::std::memcpy(
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
__array[i] = ::cuda::ptx::shfl_sync_up(__array[i], __pred, __delta, __clamp_segmask, __lane_mask);
}
warp_shuffle_result<_Up> __result;
__result.pred = __pred;
::cuda::std::memcpy(
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
return __result;
}
}
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
warp_shuffle_up(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
{
return ::cuda::device::warp_shuffle_up(__data, __src_lane, 0xFFFFFFFF, __width);
}
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_down(
const _Tp& __data, int __delta, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
{
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
constexpr auto __warp_size = 32u;
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
"non-void pointers are not allowed to prevent bug-prone code");
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
"_Width must be a power of 2 and less or equal to the warp size");
if constexpr (_Width == 1)
{
_CCCL_ASSERT(__delta == 0, "__delta must be 0 when Width == 1");
return warp_shuffle_result<_Up>{__data, true};
}
else
{
_CCCL_ASSERT(__delta >= 0 && __delta < _Width, "__delta must be in the range [0, _Width)");
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
bool __pred;
uint32_t __array[__ratio];
::cuda::std::memcpy(
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
__array[i] = ::cuda::ptx::shfl_sync_down(__array[i], __pred, __delta, __clamp_segmask, __lane_mask);
}
warp_shuffle_result<_Up> __result;
__result.pred = __pred;
::cuda::std::memcpy(
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
return __result;
}
}
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Tp>
warp_shuffle_down(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
{
return ::cuda::device::warp_shuffle_down(__data, __src_lane, 0xFFFFFFFF, __width);
}
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_xor(
const _Tp& __data, int __xor_mask, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
{
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
constexpr auto __warp_size = 32u;
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
"non-void pointers are not allowed to prevent bug-prone code");
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
"_Width must be a power of 2 and less or equal to the warp size");
NV_IF_TARGET(NV_PROVIDES_SM_70,
([[maybe_unused]] int __pred1; _CCCL_ASSERT(::__match_all_sync(::__activemask(), __xor_mask, &__pred1),
"all active lanes must have the same delta");))
if constexpr (_Width == 1)
{
_CCCL_ASSERT(__xor_mask == 0, "delta must be 0 when Width == 1");
return warp_shuffle_result<_Up>{__data, true};
}
else
{
_CCCL_ASSERT(__xor_mask >= 1 && __xor_mask < _Width, "delta must be in the range [1, _Width)");
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
bool __pred;
uint32_t __array[__ratio];
::cuda::std::memcpy(
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
_CCCL_PRAGMA_UNROLL_FULL()
for (int i = 0; i < __ratio; ++i)
{
__array[i] = ::cuda::ptx::shfl_sync_bfly(__array[i], __pred, __xor_mask, __clamp_segmask, __lane_mask);
}
warp_shuffle_result<_Up> __result;
__result.pred = __pred;
::cuda::std::memcpy(
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
return __result;
}
}
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
warp_shuffle_xor(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
{
return ::cuda::device::warp_shuffle_xor(__data, __src_lane, 0xFFFFFFFF, __width);
}
_CCCL_END_NAMESPACE_CUDA_DEVICE
# include <cuda/std/__cccl/epilogue.h>
# endif // __cccl_ptx_isa >= 600
#endif // _CCCL_CUDA_COMPILATION()
#endif // _CUDA___WARP_WARP_SHUFFLE_H