[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
319
cccl_upstream/libcudacxx/include/cuda/__warp/lane_mask.h
Normal file
319
cccl_upstream/libcudacxx/include/cuda/__warp/lane_mask.h
Normal file
@@ -0,0 +1,319 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___WARP_LANE_MASK_H
|
||||
#define _CUDA___WARP_LANE_MASK_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# include <cuda/__ptx/instructions/get_sreg.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
//! @brief A class representing a lane mask in a warp.
|
||||
class lane_mask
|
||||
{
|
||||
::cuda::std::uint32_t __value_;
|
||||
|
||||
public:
|
||||
//! @brief Constructs a lane mask object from a 32-bit unsigned integer.
|
||||
//!
|
||||
//! @param __v The value to initialize the lane mask with. Defaults to 0.
|
||||
//!
|
||||
//! @post `value() == __v`
|
||||
_CCCL_DEVICE_API explicit constexpr lane_mask(::cuda::std::uint32_t __v = 0) noexcept
|
||||
: __value_{__v}
|
||||
{}
|
||||
|
||||
//! @brief Returns the value of the lane mask as a 32-bit unsigned integer.
|
||||
//!
|
||||
//! @return The value of the lane mask.
|
||||
[[nodiscard]] _CCCL_DEVICE_API constexpr ::cuda::std::uint32_t value() const noexcept
|
||||
{
|
||||
return __value_;
|
||||
}
|
||||
|
||||
//! @brief Converts the lane mask to a 32-bit unsigned integer.
|
||||
//!
|
||||
//! This operator allows explicit conversion of the lane mask to a 32-bit unsigned integer.
|
||||
//!
|
||||
//! @return The value of the lane mask as a 32-bit unsigned integer.
|
||||
_CCCL_DEVICE_API explicit constexpr operator ::cuda::std::uint32_t() const noexcept
|
||||
{
|
||||
return __value_;
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with no lane bits set.
|
||||
//!
|
||||
//! @return A lane mask with no lane bits set.
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr lane_mask none() noexcept
|
||||
{
|
||||
return lane_mask{};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lane bits set.
|
||||
//!
|
||||
//! @return A lane mask with all lane bits set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits,
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr lane_mask all() noexcept
|
||||
{
|
||||
return lane_mask{0xffffffff};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all currently active lane bits set.
|
||||
//!
|
||||
//! This function returns a lane_mask object equivalent to calling `lane_mask{::__activemask()}`.
|
||||
//!
|
||||
//! @return A lane mask with all active lane bits set.
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_active() noexcept
|
||||
{
|
||||
return lane_mask{::__activemask()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with the current lane bit set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_eq PTX special register.
|
||||
//!
|
||||
//! @return A lane mask with the current lane bit set.
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask this_lane() noexcept
|
||||
{
|
||||
return lane_mask{::cuda::ptx::get_sreg_lanemask_eq()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lanes less than the current lane set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_lt PTX special register.
|
||||
//!
|
||||
//! @return A lane mask with all lanes less than the current lane set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits,
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_less() noexcept
|
||||
{
|
||||
return lane_mask{::cuda::ptx::get_sreg_lanemask_lt()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lanes equal to or less than the current lane set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_le PTX special register.
|
||||
//!
|
||||
//! @return A lane mask with all lanes equal to or less than the current lane set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits,
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_less_equal() noexcept
|
||||
{
|
||||
return lane_mask{::cuda::ptx::get_sreg_lanemask_le()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lanes greater than the current lane set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_gt PTX special register.
|
||||
//!
|
||||
//! @return A lane mask with all lanes greater than the current lane set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits,
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_greater() noexcept
|
||||
{
|
||||
return lane_mask{::cuda::ptx::get_sreg_lanemask_gt()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lanes greater than or equal to the current lane set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with value of %%lanemask_ge PTX special register.
|
||||
//!
|
||||
//! @return A lane mask with all lanes greater than or equal to the current lane set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits,
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_greater_equal() noexcept
|
||||
{
|
||||
return lane_mask{::cuda::ptx::get_sreg_lanemask_ge()};
|
||||
}
|
||||
|
||||
//! @brief Returns a lane mask object with all lanes not equal to the current lane set.
|
||||
//!
|
||||
//! This function is equivalent to constructing a lane_mask object with a negated value of %%lanemask_eq PTX special
|
||||
//! register.
|
||||
//!
|
||||
//! @return A lane mask with all lanes not equal to the current lane set.
|
||||
//!
|
||||
//! @note This function may return a mask with 1s set even on inactive lane bits.
|
||||
[[nodiscard]] _CCCL_DEVICE_API static lane_mask all_not_equal() noexcept
|
||||
{
|
||||
return lane_mask{~::cuda::ptx::get_sreg_lanemask_eq()};
|
||||
}
|
||||
|
||||
//! @brief Bitwise AND operator for lane_mask.
|
||||
//!
|
||||
//! @param __lhs The left-hand side lane_mask.
|
||||
//! @param __rhs The right-hand side lane_mask.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the bitwise AND of the two lane_masks.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator&(lane_mask __lhs, lane_mask __rhs) noexcept
|
||||
{
|
||||
return lane_mask{__lhs.__value_ & __rhs.__value_};
|
||||
}
|
||||
|
||||
//! @brief Bitwise AND assignment operator for lane_mask.
|
||||
//!
|
||||
//! @param __v The lane_mask to AND with the current lane_mask.
|
||||
//!
|
||||
//! @return A reference to the current lane_mask after the AND operation.
|
||||
_CCCL_DEVICE_API constexpr lane_mask& operator&=(lane_mask __v) noexcept
|
||||
{
|
||||
return *this = *this & __v;
|
||||
}
|
||||
|
||||
//! @brief Bitwise OR operator for lane_mask.
|
||||
//!
|
||||
//! @param __lhs The left-hand side lane_mask.
|
||||
//! @param __rhs The right-hand side lane_mask.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the bitwise OR of the two lane_masks.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator|(lane_mask __lhs, lane_mask __rhs) noexcept
|
||||
{
|
||||
return lane_mask{__lhs.__value_ | __rhs.__value_};
|
||||
}
|
||||
|
||||
//! @brief Bitwise OR assignment operator for lane_mask.
|
||||
//!
|
||||
//! @param __v The lane_mask to OR with the current lane_mask.
|
||||
//!
|
||||
//! @return A reference to the current lane_mask after the OR operation.
|
||||
_CCCL_DEVICE_API constexpr lane_mask& operator|=(lane_mask __v) noexcept
|
||||
{
|
||||
return *this = *this | __v;
|
||||
}
|
||||
|
||||
//! @brief Bitwise XOR operator for lane_mask.
|
||||
//!
|
||||
//! @param __lhs The left-hand side lane_mask.
|
||||
//! @param __rhs The right-hand side lane_mask.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the bitwise XOR of the two lane_masks.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator^(lane_mask __lhs, lane_mask __rhs) noexcept
|
||||
{
|
||||
return lane_mask{__lhs.__value_ ^ __rhs.__value_};
|
||||
}
|
||||
|
||||
//! @brief Bitwise XOR assignment operator for lane_mask.
|
||||
//!
|
||||
//! @param __v The lane_mask to XOR with the current lane_mask.
|
||||
//!
|
||||
//! @return A reference to the current lane_mask after the XOR operation.
|
||||
_CCCL_DEVICE_API constexpr lane_mask& operator^=(lane_mask __v) noexcept
|
||||
{
|
||||
return *this = *this ^ __v;
|
||||
}
|
||||
|
||||
//! @brief Left shift operator for lane_mask.
|
||||
//!
|
||||
//! @param __mask The lane_mask to shift.
|
||||
//! @param __shift The number of bits to shift left.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the left-shifted lane_mask.
|
||||
//!
|
||||
//! @pre `__shift` must be in the range [0, 32).
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator<<(lane_mask __mask, int __shift) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
|
||||
return lane_mask{__mask.__value_ << __shift};
|
||||
}
|
||||
|
||||
//! @brief Left shift assignment operator for lane_mask.
|
||||
//!
|
||||
//! @param __shift The number of bits to shift left.
|
||||
//!
|
||||
//! @return A reference to the current lane_mask after the left shift operation.
|
||||
//!
|
||||
//! @pre `__shift` must be in the range [0, 32).
|
||||
_CCCL_DEVICE_API constexpr lane_mask& operator<<=(int __shift) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
|
||||
return *this = *this << __shift;
|
||||
}
|
||||
|
||||
//! @brief Right shift operator for lane_mask.
|
||||
//!
|
||||
//! @param __mask The lane_mask to shift.
|
||||
//! @param __shift The number of bits to shift right.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the right-shifted lane_mask.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator>>(lane_mask __mask, int __shift) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
|
||||
return lane_mask{__mask.__value_ >> __shift};
|
||||
}
|
||||
|
||||
//! @brief Right shift assignment operator for lane_mask.
|
||||
//!
|
||||
//! @param __shift The number of bits to shift right.
|
||||
//!
|
||||
//! @return A reference to the current lane_mask after the right shift operation.
|
||||
//!
|
||||
//! @pre `__shift` must be in the range [0, 32).
|
||||
_CCCL_DEVICE_API constexpr lane_mask& operator>>=(int __shift) noexcept
|
||||
{
|
||||
_CCCL_ASSERT(__shift >= 0 && __shift < 32, "shift must be in range [0, 32)");
|
||||
return *this = *this >> __shift;
|
||||
}
|
||||
|
||||
//! @brief Bitwise NOT operator for lane_mask.
|
||||
//!
|
||||
//! @param __mask The lane_mask to negate.
|
||||
//!
|
||||
//! @return A new lane_mask object representing the negated lane_mask.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr lane_mask operator~(lane_mask __mask) noexcept
|
||||
{
|
||||
return lane_mask{~__mask.__value_};
|
||||
}
|
||||
|
||||
//! @brief Equality operator for lane_mask.
|
||||
//!
|
||||
//! @param __lhs The left-hand side lane_mask.
|
||||
//! @param __rhs The right-hand side lane_mask.
|
||||
//!
|
||||
//! @return `true` if the two lane_masks are equal, `false` otherwise.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr bool operator==(lane_mask __lhs, lane_mask __rhs) noexcept
|
||||
{
|
||||
return __lhs.__value_ == __rhs.__value_;
|
||||
}
|
||||
|
||||
//! @brief Inequality operator for lane_mask.
|
||||
//!
|
||||
//! @param __lhs The left-hand side lane_mask.
|
||||
//! @param __rhs The right-hand side lane_mask.
|
||||
//!
|
||||
//! @return `true` if the two lane_masks are not equal, `false` otherwise.
|
||||
[[nodiscard]] _CCCL_DEVICE_API friend constexpr bool operator!=(lane_mask __lhs, lane_mask __rhs) noexcept
|
||||
{
|
||||
return !(__lhs == __rhs);
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
#endif // _CUDA___WARP_LANE_MASK_H
|
||||
@@ -0,0 +1,87 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___WARP_WARP_MATCH_H
|
||||
#define _CUDA___WARP_WARP_MATCH_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# include <cuda/__cmath/ceil_div.h>
|
||||
# include <cuda/__type_traits/is_bitwise_comparable.h>
|
||||
# include <cuda/__type_traits/is_trivially_copyable.h>
|
||||
# include <cuda/__warp/lane_mask.h>
|
||||
# include <cuda/std/__cstring/memcpy.h>
|
||||
# include <cuda/std/__memory/addressof.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
extern "C" _CCCL_DEVICE void __cuda__match_all_sync_is_not_supported_before_SM_70__();
|
||||
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API bool
|
||||
warp_match_all(const _Tp& __data, const lane_mask __lane_mask = lane_mask::all()) noexcept
|
||||
{
|
||||
static_assert(is_trivially_copyable_v<_Tp>, "data must be trivially copyable");
|
||||
_CCCL_ASSERT(__lane_mask != lane_mask::none(), "lane_mask must be non-zero");
|
||||
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, bool>)
|
||||
{
|
||||
const auto __mask = ::__ballot_sync(__lane_mask.value(), __data);
|
||||
return (__mask == __lane_mask.value() || __mask == 0);
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Tp), sizeof(::cuda::std::uint32_t));
|
||||
::cuda::std::uint32_t __array[__ratio]{};
|
||||
|
||||
# if defined(_CCCL_BUILTIN_CLEAR_PADDING)
|
||||
auto __data_copy = __data;
|
||||
_CCCL_BUILTIN_CLEAR_PADDING(&__data_copy);
|
||||
const auto __data_ptr = ::cuda::std::addressof(__data_copy);
|
||||
# else // ^^^ _CCCL_BUILTIN_CLEAR_PADDING ^^^ / vvv !_CCCL_BUILTIN_CLEAR_PADDING vvv
|
||||
static_assert(is_bitwise_comparable_v<_Tp>, "data must be bitwise comparable");
|
||||
const auto __data_ptr = ::cuda::std::addressof(__data);
|
||||
# endif // _CCCL_BUILTIN_CLEAR_PADDING
|
||||
::cuda::std::memcpy(__array, __data_ptr, sizeof(_Tp));
|
||||
|
||||
bool __ret = true;
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
int __pred = false;
|
||||
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_70,
|
||||
(::__match_all_sync(__lane_mask.value(), __array[i], &__pred);),
|
||||
(::cuda::device::__cuda__match_all_sync_is_not_supported_before_SM_70__();));
|
||||
__ret = __ret && __pred;
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
#endif // _CUDA___WARP_WARP_MATCH_H
|
||||
@@ -0,0 +1,97 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___WARP_WARP_MATCH_ANY_H
|
||||
#define _CUDA___WARP_WARP_MATCH_ANY_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# include <cuda/__cmath/ceil_div.h>
|
||||
# include <cuda/__type_traits/is_bitwise_comparable.h>
|
||||
# include <cuda/__type_traits/is_trivially_copyable.h>
|
||||
# include <cuda/__warp/lane_mask.h>
|
||||
# include <cuda/std/__cstring/memcpy.h>
|
||||
# include <cuda/std/__memory/addressof.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
extern "C" _CCCL_DEVICE void __cuda__match_any_sync_is_not_supported_before_SM_70__();
|
||||
|
||||
//! @brief Returns the mask of lanes with the same bitwise value as the calling lane.
|
||||
//!
|
||||
//! @param[in] __data The data to compare across lanes.
|
||||
//! @param[in] __lane_mask The mask of participating lanes.
|
||||
//!
|
||||
//! @return A lane mask containing lanes in `__lane_mask` whose `__data` matches the calling lane's data.
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API lane_mask
|
||||
warp_match_any(const _Tp& __data, const lane_mask __lane_mask = lane_mask::all()) noexcept
|
||||
{
|
||||
static_assert(is_trivially_copyable_v<_Tp>, "data must be trivially copyable");
|
||||
_CCCL_ASSERT(__lane_mask != lane_mask::none(), "lane_mask must be non-zero");
|
||||
|
||||
if constexpr (::cuda::std::is_same_v<_Tp, bool>)
|
||||
{
|
||||
auto __mask = ::__ballot_sync(__lane_mask.value(), __data);
|
||||
if (!__data)
|
||||
{
|
||||
__mask = (~__mask) & __lane_mask.value();
|
||||
}
|
||||
return lane_mask{__mask};
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Tp), sizeof(::cuda::std::uint32_t));
|
||||
::cuda::std::uint32_t __array[__ratio]{};
|
||||
|
||||
# if defined(_CCCL_BUILTIN_CLEAR_PADDING)
|
||||
auto __data_copy = __data;
|
||||
_CCCL_BUILTIN_CLEAR_PADDING(&__data_copy);
|
||||
const auto __data_ptr = ::cuda::std::addressof(__data_copy);
|
||||
# else // ^^^ _CCCL_BUILTIN_CLEAR_PADDING ^^^ / vvv !_CCCL_BUILTIN_CLEAR_PADDING vvv
|
||||
static_assert(is_bitwise_comparable_v<_Tp>, "data must be bitwise comparable");
|
||||
const auto __data_ptr = ::cuda::std::addressof(__data);
|
||||
# endif // _CCCL_BUILTIN_CLEAR_PADDING
|
||||
::cuda::std::memcpy(__array, __data_ptr, sizeof(_Tp));
|
||||
|
||||
lane_mask __ret = __lane_mask;
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
::cuda::std::uint32_t __match_any_result = 0;
|
||||
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_70,
|
||||
(__match_any_result = ::__match_any_sync(__lane_mask.value(), __array[i]);),
|
||||
(::cuda::device::__cuda__match_any_sync_is_not_supported_before_SM_70__();));
|
||||
__ret &= lane_mask{__match_any_result};
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
#endif // _CUDA___WARP_WARP_MATCH_ANY_H
|
||||
249
cccl_upstream/libcudacxx/include/cuda/__warp/warp_shuffle.h
Normal file
249
cccl_upstream/libcudacxx/include/cuda/__warp/warp_shuffle.h
Normal file
@@ -0,0 +1,249 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___WARP_WARP_SHUFFLE_H
|
||||
#define _CUDA___WARP_WARP_SHUFFLE_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
# if __cccl_ptx_isa >= 600
|
||||
|
||||
# include <cuda/__cmath/ceil_div.h>
|
||||
# include <cuda/__cmath/pow2.h>
|
||||
# include <cuda/__ptx/instructions/get_sreg.h>
|
||||
# include <cuda/__ptx/instructions/shfl_sync.h>
|
||||
# include <cuda/std/__memory/addressof.h>
|
||||
# include <cuda/std/__type_traits/enable_if.h>
|
||||
# include <cuda/std/__type_traits/integral_constant.h>
|
||||
# include <cuda/std/__type_traits/is_default_constructible.h>
|
||||
# include <cuda/std/__type_traits/is_pointer.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
template <typename _Tp>
|
||||
struct warp_shuffle_result
|
||||
{
|
||||
_Tp data;
|
||||
bool pred;
|
||||
|
||||
template <typename _Up = _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API operator ::cuda::std::enable_if_t<!::cuda::std::is_array_v<_Up>, _Up>() const
|
||||
{
|
||||
return data;
|
||||
}
|
||||
};
|
||||
|
||||
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_idx(
|
||||
const _Tp& __data, int __src_lane, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
|
||||
{
|
||||
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
|
||||
constexpr auto __warp_size = 32u;
|
||||
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
|
||||
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
|
||||
"non-void pointers are not allowed to prevent bug-prone code");
|
||||
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
|
||||
"_Width must be a power of 2 and less or equal to the warp size");
|
||||
|
||||
if constexpr (_Width == 1)
|
||||
{
|
||||
return warp_shuffle_result<_Up>{__data, true};
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
|
||||
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
|
||||
bool __pred;
|
||||
uint32_t __array[__ratio];
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
|
||||
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
__array[i] = ::cuda::ptx::shfl_sync_idx(__array[i], __pred, __src_lane, __clamp_segmask, __lane_mask);
|
||||
}
|
||||
warp_shuffle_result<_Up> __result;
|
||||
__result.pred = __pred;
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
|
||||
return __result;
|
||||
}
|
||||
}
|
||||
|
||||
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
|
||||
warp_shuffle_idx(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
|
||||
{
|
||||
return ::cuda::device::warp_shuffle_idx(__data, __src_lane, 0xFFFFFFFF, __width);
|
||||
}
|
||||
|
||||
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Tp> warp_shuffle_up(
|
||||
const _Tp& __data, int __delta, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
|
||||
{
|
||||
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
|
||||
constexpr auto __warp_size = 32u;
|
||||
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
|
||||
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
|
||||
"non-void pointers are not allowed to prevent bug-prone code");
|
||||
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
|
||||
"_Width must be a power of 2 and less or equal to the warp size");
|
||||
|
||||
if constexpr (_Width == 1)
|
||||
{
|
||||
_CCCL_ASSERT(__delta == 0, "delta must be 0 when Width == 1");
|
||||
return warp_shuffle_result<_Up>{__data, true};
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_ASSERT(__delta >= 0 && __delta < _Width, "delta must be in the range [0, _Width)");
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
|
||||
auto __clamp_segmask = (__warp_size - _Width) << 8;
|
||||
bool __pred;
|
||||
uint32_t __array[__ratio];
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
|
||||
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
__array[i] = ::cuda::ptx::shfl_sync_up(__array[i], __pred, __delta, __clamp_segmask, __lane_mask);
|
||||
}
|
||||
warp_shuffle_result<_Up> __result;
|
||||
__result.pred = __pred;
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
|
||||
return __result;
|
||||
}
|
||||
}
|
||||
|
||||
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
|
||||
warp_shuffle_up(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
|
||||
{
|
||||
return ::cuda::device::warp_shuffle_up(__data, __src_lane, 0xFFFFFFFF, __width);
|
||||
}
|
||||
|
||||
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_down(
|
||||
const _Tp& __data, int __delta, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
|
||||
{
|
||||
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
|
||||
constexpr auto __warp_size = 32u;
|
||||
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
|
||||
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
|
||||
"non-void pointers are not allowed to prevent bug-prone code");
|
||||
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
|
||||
"_Width must be a power of 2 and less or equal to the warp size");
|
||||
|
||||
if constexpr (_Width == 1)
|
||||
{
|
||||
_CCCL_ASSERT(__delta == 0, "__delta must be 0 when Width == 1");
|
||||
return warp_shuffle_result<_Up>{__data, true};
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_ASSERT(__delta >= 0 && __delta < _Width, "__delta must be in the range [0, _Width)");
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
|
||||
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
|
||||
bool __pred;
|
||||
uint32_t __array[__ratio];
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
|
||||
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
__array[i] = ::cuda::ptx::shfl_sync_down(__array[i], __pred, __delta, __clamp_segmask, __lane_mask);
|
||||
}
|
||||
warp_shuffle_result<_Up> __result;
|
||||
__result.pred = __pred;
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
|
||||
return __result;
|
||||
}
|
||||
}
|
||||
|
||||
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Tp>
|
||||
warp_shuffle_down(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
|
||||
{
|
||||
return ::cuda::device::warp_shuffle_down(__data, __src_lane, 0xFFFFFFFF, __width);
|
||||
}
|
||||
|
||||
template <int _Width = 32, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up> warp_shuffle_xor(
|
||||
const _Tp& __data, int __xor_mask, uint32_t __lane_mask = 0xFFFFFFFF, ::cuda::std::integral_constant<int, _Width> = {})
|
||||
{
|
||||
static_assert(::cuda::std::is_default_constructible_v<_Tp>, "_Tp must be default constructible");
|
||||
constexpr auto __warp_size = 32u;
|
||||
constexpr bool __is_void_ptr = ::cuda::std::is_same_v<_Up, void*> || ::cuda::std::is_same_v<_Up, const void*>;
|
||||
static_assert(!::cuda::std::is_pointer_v<_Up> || __is_void_ptr,
|
||||
"non-void pointers are not allowed to prevent bug-prone code");
|
||||
static_assert(::cuda::is_power_of_two(_Width) && _Width >= 1 && _Width <= __warp_size,
|
||||
"_Width must be a power of 2 and less or equal to the warp size");
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_70,
|
||||
([[maybe_unused]] int __pred1; _CCCL_ASSERT(::__match_all_sync(::__activemask(), __xor_mask, &__pred1),
|
||||
"all active lanes must have the same delta");))
|
||||
if constexpr (_Width == 1)
|
||||
{
|
||||
_CCCL_ASSERT(__xor_mask == 0, "delta must be 0 when Width == 1");
|
||||
return warp_shuffle_result<_Up>{__data, true};
|
||||
}
|
||||
else
|
||||
{
|
||||
_CCCL_ASSERT(__xor_mask >= 1 && __xor_mask < _Width, "delta must be in the range [1, _Width)");
|
||||
constexpr int __ratio = ::cuda::ceil_div(sizeof(_Up), sizeof(uint32_t));
|
||||
auto __clamp_segmask = (_Width - 1u) | ((__warp_size - _Width) << 8);
|
||||
bool __pred;
|
||||
uint32_t __array[__ratio];
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(__array), static_cast<const void*>(::cuda::std::addressof(__data)), sizeof(_Up));
|
||||
|
||||
_CCCL_PRAGMA_UNROLL_FULL()
|
||||
for (int i = 0; i < __ratio; ++i)
|
||||
{
|
||||
__array[i] = ::cuda::ptx::shfl_sync_bfly(__array[i], __pred, __xor_mask, __clamp_segmask, __lane_mask);
|
||||
}
|
||||
warp_shuffle_result<_Up> __result;
|
||||
__result.pred = __pred;
|
||||
::cuda::std::memcpy(
|
||||
static_cast<void*>(::cuda::std::addressof(__result.data)), static_cast<void*>(__array), sizeof(_Up));
|
||||
return __result;
|
||||
}
|
||||
}
|
||||
|
||||
template <int _Width, typename _Tp, typename _Up = ::cuda::std::remove_cv_t<_Tp>>
|
||||
[[nodiscard]] _CCCL_DEVICE_API warp_shuffle_result<_Up>
|
||||
warp_shuffle_xor(const _Tp& __data, int __src_lane, ::cuda::std::integral_constant<int, _Width> __width)
|
||||
{
|
||||
return ::cuda::device::warp_shuffle_xor(__data, __src_lane, 0xFFFFFFFF, __width);
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_DEVICE
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
# endif // __cccl_ptx_isa >= 600
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
#endif // _CUDA___WARP_WARP_SHUFFLE_H
|
||||
Reference in New Issue
Block a user