[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
@@ -0,0 +1,90 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H
|
||||
#define _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
# include <cuda/__hierarchy/hierarchy_levels.h>
|
||||
# include <cuda/std/tuple>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
/**
|
||||
* @brief Returns a tuple of dim3 compatible objects that can be used to launch
|
||||
* a kernel
|
||||
*
|
||||
* This function returns a tuple of hierarchy_query_result objects that contain
|
||||
* dimensions from the supplied hierarchy, that can be used to launch that
|
||||
* hierarchy. It is meant to allow for easy usage of hierarchy dimensions with
|
||||
* the <<<>>> launch syntax or cudaLaunchKernelEx in case of a cluster launch.
|
||||
* Contained hierarchy_query_result objects are results of extents() member
|
||||
* function on the hierarchy passed in. The returned tuple has three elements if
|
||||
* cluster_level is present in the hierarchy (extents(block, grid),
|
||||
* extents(cluster, block), extents(thread, block)). Otherwise it contains only
|
||||
* two elements, without the middle one related to the cluster.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* using namespace cuda;
|
||||
*
|
||||
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
|
||||
* block_dims<8, 8, 8>()); auto [grid_dimensions, cluster_dimensions,
|
||||
* block_dimensions] = get_launch_dimensions(hierarchy);
|
||||
* assert(grid_dimensions.x == 256);
|
||||
* assert(cluster_dimensions.x == 4);
|
||||
* assert(block_dimensions.x == 8);
|
||||
* assert(block_dimensions.y == 8);
|
||||
* assert(block_dimensions.z == 8);
|
||||
* @endcode
|
||||
* @par
|
||||
*
|
||||
* @param __hierarchy
|
||||
* Hierarchy that the launch dimensions are requested for
|
||||
*/
|
||||
template <class _BottomLevel, class... _LevelDescs>
|
||||
[[nodiscard]] _CCCL_HOST_API constexpr auto
|
||||
get_launch_dimensions(const hierarchy<_BottomLevel, _LevelDescs...>& __hierarchy)
|
||||
{
|
||||
if constexpr (hierarchy<_BottomLevel, _LevelDescs...>::has_level(cluster))
|
||||
{
|
||||
return ::cuda::std::make_tuple(
|
||||
::dim3{block.dims(grid, __hierarchy)},
|
||||
::dim3{block.dims(cluster, __hierarchy)},
|
||||
::dim3{gpu_thread.dims(block, __hierarchy)});
|
||||
}
|
||||
else
|
||||
{
|
||||
return ::cuda::std::make_tuple(::dim3{block.dims(grid, __hierarchy)}, ::dim3{gpu_thread.dims(block, __hierarchy)});
|
||||
}
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
#endif // _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H
|
||||
@@ -0,0 +1,543 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H
|
||||
#define _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/level_dimensions.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
# include <cuda/std/__type_traits/type_list.h>
|
||||
# include <cuda/std/__utility/integer_sequence.h>
|
||||
# include <cuda/std/tuple>
|
||||
|
||||
# include <nv/target>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
template <typename _Level>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __as_level(_Level __lvl) noexcept -> _Level
|
||||
{
|
||||
return __lvl;
|
||||
}
|
||||
|
||||
template <typename _LevelFn>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __as_level(_LevelFn* __fn) noexcept -> decltype(__fn())
|
||||
{
|
||||
return {};
|
||||
}
|
||||
} // namespace __detail
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
template <class... _Levels>
|
||||
struct __can_stack_checker
|
||||
{
|
||||
template <class... _LevelsShifted>
|
||||
static constexpr bool __can_stack = (__detail::__can_rhs_stack_on_lhs<_LevelsShifted, _Levels> && ...);
|
||||
};
|
||||
|
||||
template <class _LUnit, class _L1, class... _Levels>
|
||||
inline constexpr bool __can_stack =
|
||||
__can_stack_checker<__level_type_of<_L1>,
|
||||
__level_type_of<_Levels>...>::template __can_stack<__level_type_of<_Levels>..., _LUnit>;
|
||||
|
||||
template <::cuda::std::size_t... _Id>
|
||||
_CCCL_API constexpr auto __reverse_indices(::cuda::std::index_sequence<_Id...>) noexcept
|
||||
{
|
||||
return ::cuda::std::index_sequence<(sizeof...(_Id) - 1 - _Id)...>();
|
||||
}
|
||||
|
||||
template <class _LUnit, bool _Reversed = false>
|
||||
struct __make_hierarchy
|
||||
{
|
||||
template <class _Levels, ::cuda::std::size_t... _Ids>
|
||||
[[nodiscard]] _CCCL_NODEBUG_API static constexpr auto
|
||||
__apply_reverse(const _Levels& __ls, ::cuda::std::index_sequence<_Ids...>) noexcept
|
||||
{
|
||||
return __make_hierarchy<_LUnit, true>()(::cuda::std::get<_Ids>(__ls)...);
|
||||
}
|
||||
|
||||
template <class... _Levels2>
|
||||
[[nodiscard]] _CCCL_API constexpr auto operator()(const _Levels2&... __ls) const noexcept
|
||||
{
|
||||
using _UnitOrDefault = ::cuda::std::conditional_t<
|
||||
::cuda::std::is_same_v<void, _LUnit>,
|
||||
__default_unit_below<::cuda::std::__type_index_c<sizeof...(_Levels2) - 1, __level_type_of<_Levels2>...>>,
|
||||
_LUnit>;
|
||||
if constexpr (__can_stack<_UnitOrDefault, _Levels2...>)
|
||||
{
|
||||
return hierarchy(_UnitOrDefault{}, __ls...);
|
||||
}
|
||||
else if constexpr (!_Reversed)
|
||||
{
|
||||
return __apply_reverse(::cuda::std::tie(__ls...),
|
||||
__reverse_indices(::cuda::std::index_sequence_for<_Levels2...>()));
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(__can_stack<_UnitOrDefault, _Levels2...>,
|
||||
"Provided levels can't create a valid hierarchy when "
|
||||
"stacked in the provided order or reversed");
|
||||
_CCCL_UNREACHABLE();
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class _LUnit>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __get_levels_range_end() noexcept
|
||||
{
|
||||
return ::cuda::std::make_tuple();
|
||||
}
|
||||
|
||||
// Find LUnit in Levels... and discard the rest
|
||||
// maybe_unused needed for MSVC
|
||||
template <class _LUnit, class _LDims, class... _Levels>
|
||||
[[nodiscard]] _CCCL_API constexpr auto
|
||||
__get_levels_range_end(const _LDims& __lvl, [[maybe_unused]] const _Levels&... __levels) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_LUnit, __level_type_of<_LDims>>)
|
||||
{
|
||||
return ::cuda::std::make_tuple();
|
||||
}
|
||||
else
|
||||
{
|
||||
return ::cuda::std::tuple_cat(::cuda::std::tie(__lvl), __get_levels_range_end<_LUnit>(__levels...));
|
||||
}
|
||||
}
|
||||
|
||||
// Find the LTop in Levels... and discard the preceding ones
|
||||
template <class _LTop, class _LUnit, class _LTopDims, class... _Levels>
|
||||
[[nodiscard]] _CCCL_API constexpr auto
|
||||
__get_levels_range_start(const _LTopDims& __ltop, const _Levels&... __levels) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_LTop, __level_type_of<_LTopDims>>)
|
||||
{
|
||||
return __get_levels_range_end<_LUnit>(__ltop, __levels...);
|
||||
}
|
||||
else
|
||||
{
|
||||
return __get_levels_range_start<_LTop, _LUnit>(__levels...);
|
||||
}
|
||||
}
|
||||
|
||||
// Creates a new hierarchy from Levels... cutting out levels between LTop and
|
||||
// LUnit
|
||||
template <class _LTop, class _LUnit, class... _Levels>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __get_levels_range(const _Levels&... __levels) noexcept
|
||||
{
|
||||
return __get_levels_range_start<_LTop, _LUnit>(__levels...);
|
||||
}
|
||||
} // namespace __detail
|
||||
|
||||
// Artificial empty hierarchy to make it possible for the config type to be
|
||||
// empty, seems easier than checking everywhere in hierarchy APIs if its not
|
||||
// empty. Any usage of an empty hierarchy other than combine should lead to an
|
||||
// error anyway
|
||||
struct __empty_hierarchy
|
||||
{
|
||||
template <class _Other>
|
||||
[[nodiscard]] _CCCL_API _Other combine(const _Other& __other) const
|
||||
{
|
||||
return __other;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Type representing a hierarchy of CUDA threads
|
||||
*
|
||||
* This type combines a number of hierarchy_level_desc objects to represent
|
||||
* dimensions of a (possibly partial) hierarchy of CUDA threads. It supports
|
||||
* accessing individual levels or queries combining dimensions of multiple
|
||||
* levels. This type should not be created directly and make_hierarchy function
|
||||
* should be used instead. For every level, the unit for its dimensions is
|
||||
* implied by the next level in the hierarchy, except for the last type, for
|
||||
* which its the BottomUnit template argument. In case the BottomUnit type is
|
||||
* thread_level, the hierarchy is considered complete and there exist an alias
|
||||
* template for it named hierarchy, that only takes the Levels...
|
||||
* template argument.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* auto hierarchy = make_hierarchy(grid_dims(256), block_dims<8, 8, 8>());
|
||||
* assert(hierarchy.level(grid).dims.x == 256);
|
||||
* static_assert(hierarchy.count(thread, block) == 8 * 8 * 8);
|
||||
* @endcode
|
||||
* @par
|
||||
*
|
||||
* @tparam BottomUnit
|
||||
* Type indicating what is the unit of the last level in the hierarchy
|
||||
*
|
||||
* @tparam Levels
|
||||
* Template parameter pack with the types of levels in the hierarchy, must be
|
||||
* hierarchy_level_desc instances or types derived from it
|
||||
*/
|
||||
template <class _BottomUnit, class... _LevelDescs>
|
||||
class hierarchy
|
||||
{
|
||||
static_assert(__is_hierarchy_level_v<_BottomUnit>);
|
||||
static_assert(__detail::__can_stack<_BottomUnit, typename _LevelDescs::level_type...>);
|
||||
|
||||
template <class, class...>
|
||||
friend class hierarchy;
|
||||
|
||||
::cuda::std::tuple<_LevelDescs...> __descs_;
|
||||
|
||||
// This being static is a bit of a hack to make extents_type working without
|
||||
// incomplete class member access
|
||||
template <class _Unit, class _Level>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto
|
||||
__levels_range_static(const ::cuda::std::tuple<_LevelDescs...>& __levels) noexcept
|
||||
{
|
||||
static_assert(hierarchy::has_level<_Level>());
|
||||
static_assert(__has_bottom_unit_or_level_v<_Unit, hierarchy<_BottomUnit, _LevelDescs...>>);
|
||||
static_assert(__detail::__legal_unit_for_level<_Unit, _Level>);
|
||||
auto __fn = __detail::__get_levels_range<_Level, _Unit, _LevelDescs...>;
|
||||
return ::cuda::std::apply(__fn, __levels);
|
||||
}
|
||||
|
||||
// TODO is this useful enough to expose?
|
||||
template <class _Unit, class _Level>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __levels_range() const noexcept
|
||||
{
|
||||
return __levels_range_static<_Unit, _Level>(__descs_);
|
||||
}
|
||||
|
||||
template <class _Unit>
|
||||
struct __fragment_helper
|
||||
{
|
||||
template <class... _Selected>
|
||||
[[nodiscard]] _CCCL_API constexpr auto operator()(const _Selected&... __levels) const noexcept
|
||||
{
|
||||
return hierarchy<_Unit, _Selected...>(__levels...);
|
||||
}
|
||||
};
|
||||
|
||||
public:
|
||||
template <class _Level>
|
||||
static constexpr auto __level_idx =
|
||||
::cuda::std::__find_exactly_one_t<_Level, typename _LevelDescs::level_type...>::value;
|
||||
|
||||
using bottom_unit_type = _BottomUnit;
|
||||
using top_level_type = ::cuda::std::__type_index_c<0, typename _LevelDescs::level_type...>;
|
||||
|
||||
template <class _Level>
|
||||
using level_desc_type = ::cuda::std::__type_index_c<__level_idx<_Level>, _LevelDescs...>;
|
||||
|
||||
template <class _Level>
|
||||
[[nodiscard]] _CCCL_API static constexpr bool has_level(const _Level& = _Level{}) noexcept
|
||||
{
|
||||
return (::cuda::std::is_same_v<_Level, typename _LevelDescs::level_type> || ...);
|
||||
}
|
||||
|
||||
_CCCL_API constexpr hierarchy(const _LevelDescs&... __lds) noexcept
|
||||
: __descs_(__lds...)
|
||||
{}
|
||||
|
||||
_CCCL_TEMPLATE(class _BottomUnit2 = _BottomUnit)
|
||||
_CCCL_REQUIRES((!::cuda::std::is_same_v<void, _BottomUnit2>) )
|
||||
_CCCL_API constexpr hierarchy(const _BottomUnit2&, const _LevelDescs&... __lds) noexcept
|
||||
: __descs_(__lds...)
|
||||
{}
|
||||
|
||||
_CCCL_API constexpr hierarchy(const ::cuda::std::tuple<_LevelDescs...>& __lds) noexcept
|
||||
: __descs_(__lds)
|
||||
{}
|
||||
|
||||
_CCCL_TEMPLATE(class _BottomUnit2 = _BottomUnit)
|
||||
_CCCL_REQUIRES((!::cuda::std::is_same_v<void, _BottomUnit2>) )
|
||||
_CCCL_API constexpr hierarchy(const _BottomUnit2&, const ::cuda::std::tuple<_LevelDescs...>& __lds) noexcept
|
||||
: __descs_(__lds)
|
||||
{}
|
||||
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool operator==(const hierarchy& __lhs, const hierarchy& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__descs_ == __rhs.__descs_;
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool operator!=(const hierarchy& __lhs, const hierarchy& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__descs_ != __rhs.__descs_;
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Get a fragment of this hierarchy
|
||||
*
|
||||
* This member function can be used to get a fragment of the hierarchy its
|
||||
* called on. It returns a hierarchy that includes levels starting
|
||||
* with the level specified in Level and ending with a level before Unit.
|
||||
* Toegether with hierarchy_add_level function it can be used to create a new
|
||||
* hierarchy that is a modification of an existing hierarchy.
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
|
||||
* block_dims<8, 8, 8>()); auto fragment = hierarchy.fragment(block, grid);
|
||||
* auto new_hierarchy = hierarchy_add_level(fragment, block_dims<128>());
|
||||
* static_assert(new_hierarchy.count(thread, block) == 128);
|
||||
* @endcode
|
||||
* @par
|
||||
*
|
||||
* @tparam Unit
|
||||
* Type indicating what should be the unit of the resulting fragment
|
||||
*
|
||||
* @tparam Level
|
||||
* Type indicating what should be the top most level of the resulting
|
||||
* fragment
|
||||
*/
|
||||
template <typename _Unit, typename _Level>
|
||||
_CCCL_API constexpr auto fragment(const _Unit& = _Unit(), const _Level& = _Level()) const noexcept
|
||||
{
|
||||
auto __selected = __levels_range<_Unit, _Level>();
|
||||
// TODO fragment can't do constexpr queries because we use references here,
|
||||
// can we create copies of the levels in some cases and move to the
|
||||
// constructor?
|
||||
return ::cuda::std::apply(__fragment_helper<_Unit>(), __selected);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Returns level description associated with a specified hierarchy
|
||||
* level in this hierarchy.
|
||||
*
|
||||
* This function returns a copy of the object associated with the specified
|
||||
* level, that was passed into the hierarchy on its creation. Level need to be
|
||||
* levels present in this hierarchy.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* using namespace cuda;
|
||||
*
|
||||
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
|
||||
* block_dims<8, 8, 8>());
|
||||
* static_assert(decltype(hierarchy.level(cluster).dims)::static_extent(0) ==
|
||||
* 4);
|
||||
* @endcode
|
||||
* @par
|
||||
*
|
||||
* @tparam Level
|
||||
* Specifies the requested level
|
||||
*/
|
||||
template <typename _Level>
|
||||
[[nodiscard]] _CCCL_API constexpr const level_desc_type<_Level>& level(const _Level&) const noexcept
|
||||
{
|
||||
static_assert(hierarchy::has_level<_Level>());
|
||||
return ::cuda::std::get<__level_idx<_Level>>(__descs_);
|
||||
}
|
||||
|
||||
//! @brief Returns a new hierarchy with combined levels of this and the other
|
||||
//! supplied hierarchy
|
||||
//!
|
||||
//! This function combines this hierarchy with the supplied hierarchy, the
|
||||
//! resulting hierarchy holds levels present in both hierarchies. In case of
|
||||
//! overlap of levels this hierarchy is prioritized, so the result always
|
||||
//! holds all levels from this hierarchy and non-overlapping levels from the
|
||||
//! other hierarchy.
|
||||
//!
|
||||
//! @param __other The other hierarchy to be combined with this hierarchy
|
||||
//!
|
||||
//! @return Hierarchy holding the combined levels from both hierarchies
|
||||
template <class _OtherUnit, class... _OtherLevels>
|
||||
[[nodiscard]] _CCCL_API constexpr auto combine(const hierarchy<_OtherUnit, _OtherLevels...>& __other) const
|
||||
{
|
||||
using _BottomLevel = __level_type_of<::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, _LevelDescs...>>;
|
||||
using _OtherHierarchy = hierarchy<_OtherUnit, _OtherLevels...>;
|
||||
using _OtherTopLevel = typename _OtherHierarchy::top_level_type;
|
||||
using _OtherBottomLevel =
|
||||
__level_type_of<::cuda::std::__type_index_c<sizeof...(_OtherLevels) - 1, _OtherLevels...>>;
|
||||
if constexpr (__detail::__can_rhs_stack_on_lhs<_OtherTopLevel, _BottomLevel>)
|
||||
{
|
||||
// Easily stackable case, example this is (grid), other is (cluster,
|
||||
// block)
|
||||
return ::cuda::std::apply(__fragment_helper<_OtherUnit>(), ::cuda::std::tuple_cat(__descs_, __other.__descs_));
|
||||
}
|
||||
else if constexpr (_OtherHierarchy::template has_level<_BottomLevel>()
|
||||
&& (!_OtherHierarchy::template has_level<top_level_type>()
|
||||
|| ::cuda::std::is_same_v<top_level_type, _OtherTopLevel>) )
|
||||
{
|
||||
// Overlap with this on the top, e.g. this is (grid, cluster), other is
|
||||
// (cluster, block), can fully overlap Do we have some CCCL tuple utils
|
||||
// that can select all but the first?
|
||||
auto __to_add_with_one_too_many = __other.template __levels_range<_OtherUnit, _BottomLevel>();
|
||||
auto __to_add = ::cuda::std::apply(
|
||||
[](auto&&, auto&&... __rest) {
|
||||
return ::cuda::std::make_tuple(__rest...);
|
||||
},
|
||||
__to_add_with_one_too_many);
|
||||
return ::cuda::std::apply(__fragment_helper<_OtherUnit>(), ::cuda::std::tuple_cat(__descs_, __to_add));
|
||||
}
|
||||
else
|
||||
{
|
||||
if constexpr (__detail::__can_rhs_stack_on_lhs<top_level_type, _OtherBottomLevel>)
|
||||
{
|
||||
// Easily stackable case again, just reversed
|
||||
return ::cuda::std::apply(__fragment_helper<_BottomUnit>(), ::cuda::std::tuple_cat(__other.__descs_, __descs_));
|
||||
}
|
||||
else
|
||||
{
|
||||
// Overlap with this on the bottom, e.g. this is (cluster, block), other
|
||||
// is (grid, cluster), can fully overlap
|
||||
static_assert(hierarchy::has_level<_OtherBottomLevel>()
|
||||
&& (!_OtherHierarchy::template has_level<_BottomLevel>()
|
||||
|| ::cuda::std::is_same_v<_BottomLevel, _OtherBottomLevel>),
|
||||
"Can't combine the hierarchies");
|
||||
|
||||
auto __to_add = __other.template __levels_range<top_level_type, _OtherTopLevel>();
|
||||
return ::cuda::std::apply(__fragment_helper<_BottomUnit>(), ::cuda::std::tuple_cat(__to_add, __descs_));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
# ifndef _CCCL_DOXYGEN_INVOKED // Do not document
|
||||
[[nodiscard]] _CCCL_API constexpr hierarchy combine([[maybe_unused]] __empty_hierarchy __empty) const
|
||||
{
|
||||
return *this;
|
||||
}
|
||||
# endif // _CCCL_DOXYGEN_INVOKED
|
||||
|
||||
# if !_CCCL_COMPILER(NVRTC)
|
||||
template <class _NewLevel, class _Unit, class... _LevelDescs2>
|
||||
friend constexpr auto hierarchy_add_level(const hierarchy<_Unit, _LevelDescs2...>& hierarchy, _NewLevel __lnew);
|
||||
# endif // !_CCCL_COMPILER(NVRTC)
|
||||
};
|
||||
|
||||
_CCCL_TEMPLATE(class... _LevelDescs)
|
||||
_CCCL_REQUIRES(::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
|
||||
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _LevelDescs&...)
|
||||
-> hierarchy<__detail::__default_unit_below<
|
||||
::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, __level_type_of<_LevelDescs>...>>,
|
||||
_LevelDescs...>;
|
||||
|
||||
_CCCL_TEMPLATE(class _BottomUnit, class... _LevelDescs)
|
||||
_CCCL_REQUIRES(
|
||||
__is_hierarchy_level_v<_BottomUnit> _CCCL_AND ::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
|
||||
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _BottomUnit&, const _LevelDescs&...)
|
||||
-> hierarchy<_BottomUnit, _LevelDescs...>;
|
||||
|
||||
_CCCL_TEMPLATE(class... _LevelDescs)
|
||||
_CCCL_REQUIRES(::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
|
||||
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const ::cuda::std::tuple<_LevelDescs...>&)
|
||||
-> hierarchy<__detail::__default_unit_below<
|
||||
::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, __level_type_of<_LevelDescs>...>>,
|
||||
_LevelDescs...>;
|
||||
|
||||
_CCCL_TEMPLATE(class _BottomUnit, class... _LevelDescs)
|
||||
_CCCL_REQUIRES(
|
||||
__is_hierarchy_level_v<_BottomUnit> _CCCL_AND ::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
|
||||
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _BottomUnit&, const ::cuda::std::tuple<_LevelDescs...>&)
|
||||
-> hierarchy<_BottomUnit, _LevelDescs...>;
|
||||
|
||||
# if !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
// TODO consider having LUnit optional argument for template argument deduction
|
||||
/**
|
||||
* @brief Creates a hierarchy from passed in levels.
|
||||
*
|
||||
* This function takes any number of hierarchy_level_desc or derived objects
|
||||
* and creates a hierarchy out of them. Levels need to be in ascending
|
||||
* or descending order and the lowest level needs to be valid for thread_level
|
||||
* unit.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* using namespace cuda;
|
||||
*
|
||||
* auto hierarchy1 = make_hierarchy(grid_dims(256), cluster_dims<4>(),
|
||||
* block_dims<8, 8, 8>()); auto hierarchy2 = make_hierarchy(block_dims<8, 8,
|
||||
* 8>(), cluster_dims<4>(), grid_dims(256));
|
||||
* static_assert(cuda::std::is_same_v<decltype(hierarchy1),
|
||||
* decltype(hierarchy2)>);
|
||||
* @endcode
|
||||
* @par
|
||||
*/
|
||||
template <class _LUnit = void, class _L1, class... _LevelDescs>
|
||||
constexpr auto make_hierarchy(_L1 __l1, _LevelDescs... __ls) noexcept
|
||||
{
|
||||
return __detail::__make_hierarchy<_LUnit>()(__detail::__as_level(__l1), __detail::__as_level(__ls)...);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Add a level to a hierarchy
|
||||
*
|
||||
* This function returns a new hierarchy, that is a copy of the supplied
|
||||
* hierarchy with the supplied level added to it. This function will examine the
|
||||
* supplied level and add it either at the top or at the bottom of the
|
||||
* hierarchy, depending on what levels above and below it are valid for it.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* using namespace cuda;
|
||||
*
|
||||
* auto partial1 = make_hierarchy<block_level>(grid_dims(256),
|
||||
* cluster_dims<4>()); auto hierarchy1 = hierarchy_add_level(partial1,
|
||||
* block_dims<8, 8, 8>()); auto partial2 =
|
||||
* make_hierarchy<thread_level>(block_dims<8, 8, 8>(), cluster_dims<4>()); auto
|
||||
* hierarchy2 = hierarchy_add_level(partial2, grid_dims(256));
|
||||
* static_assert(cuda::std::is_same_v<decltype(hierarchy1),
|
||||
* decltype(hierarchy2)>);
|
||||
* @endcode
|
||||
* @par
|
||||
*/
|
||||
template <class _NewLevel, class _Unit, class... _LevelDescs>
|
||||
constexpr auto hierarchy_add_level(const hierarchy<_Unit, _LevelDescs...>& __hierarchy, _NewLevel __lnew)
|
||||
{
|
||||
auto __new_level = __detail::__as_level(__lnew);
|
||||
using __added_level = decltype(__new_level);
|
||||
using __top_level = __level_type_of<::cuda::std::__type_index_c<0, _LevelDescs...>>;
|
||||
using __bottom_level = __level_type_of<::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, _LevelDescs...>>;
|
||||
|
||||
if constexpr (__detail::__can_rhs_stack_on_lhs<__top_level, __level_type_of<__added_level>>)
|
||||
{
|
||||
return hierarchy<_Unit, __added_level, _LevelDescs...>(
|
||||
::cuda::std::tuple_cat(::cuda::std::make_tuple(__new_level), __hierarchy.__descs_));
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(__detail::__can_rhs_stack_on_lhs<__level_type_of<__added_level>, __bottom_level>,
|
||||
"Not supported order of levels in hierarchy");
|
||||
using __new_unit = __detail::__default_unit_below<__level_type_of<__added_level>>;
|
||||
return hierarchy<__new_unit, _LevelDescs..., __added_level>(
|
||||
::cuda::std::tuple_cat(__hierarchy.__descs_, ::cuda::std::make_tuple(__new_level)));
|
||||
}
|
||||
}
|
||||
# endif // !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H
|
||||
@@ -0,0 +1,285 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H
|
||||
#define _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/hierarchy_query_result.h>
|
||||
# include <cuda/__hierarchy/queries/count.h>
|
||||
# include <cuda/__hierarchy/queries/extents.h>
|
||||
# include <cuda/__hierarchy/queries/index.h>
|
||||
# include <cuda/__hierarchy/queries/rank.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__mdspan/extents.h>
|
||||
# include <cuda/std/__type_traits/is_integer.h>
|
||||
|
||||
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
|
||||
# include <cuda/experimental/__group/concepts.cuh>
|
||||
# include <cuda/experimental/__group/fwd.cuh>
|
||||
# include <cuda/experimental/__group/queries.cuh>
|
||||
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// Used to either pass-through the hierarchy argument or unpack it from launch configuration
|
||||
_CCCL_TEMPLATE(class _Type)
|
||||
_CCCL_REQUIRES(__is_or_has_hierarchy_member_v<_Type>)
|
||||
[[nodiscard]] _CCCL_API constexpr auto& __unpack_hierarchy_if_needed(const _Type& __instance) noexcept
|
||||
{
|
||||
if constexpr (__is_hierarchy_v<_Type>)
|
||||
{
|
||||
return __instance;
|
||||
}
|
||||
else
|
||||
{
|
||||
return __instance.hierarchy();
|
||||
}
|
||||
}
|
||||
|
||||
template <class _Level>
|
||||
struct hierarchy_level_base
|
||||
{
|
||||
using level_type = _Level;
|
||||
|
||||
template <class _InLevel>
|
||||
using __default_md_query_type = ::cuda::std::uint32_t;
|
||||
template <class _InLevel>
|
||||
using __default_1d_query_type = typename _InLevel::__product_type;
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto dims(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return _Level::template dims_as<__default_md_query_type<_InLevel>>(
|
||||
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto static_dims(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __static_dims_impl(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto extents(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return _Level::template extents_as<__default_md_query_type<_InLevel>>(
|
||||
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto static_count(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __static_count_impl(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto count(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return _Level::template count_as<__default_1d_query_type<_InLevel>>(
|
||||
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto index(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return _Level::template index_as<__default_md_query_type<_InLevel>>(
|
||||
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto rank(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return _Level::template rank_as<__default_1d_query_type<_InLevel>>(
|
||||
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
|
||||
__is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto dims_as(const _InLevel& __level, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __dims_as_impl<_Tp>(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
|
||||
__is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto extents_as(const _InLevel&, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __extents_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
|
||||
__is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto count_as(const _InLevel&, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __count_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
|
||||
__is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto index_as(const _InLevel&, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __index_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
|
||||
__is_or_has_hierarchy_member_v<_Hierarchy>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto rank_as(const _InLevel&, const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __rank_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
|
||||
}
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_TEMPLATE(class _Group)
|
||||
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static constexpr ::cuda::std::size_t static_count(const _Group&) noexcept
|
||||
{
|
||||
return ::cuda::experimental::__static_count_query_group<_Level, _Group>();
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Group)
|
||||
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static constexpr auto count(const _Group& __group) noexcept
|
||||
{
|
||||
return count_as<__default_1d_query_type<typename _Group::unit_type>>(__group);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Group)
|
||||
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static auto rank(const _Group& __group) noexcept
|
||||
{
|
||||
return rank_as<__default_1d_query_type<typename _Group::unit_type>>(__group);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _Group)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static constexpr _Tp count_as(const _Group& __group) noexcept
|
||||
{
|
||||
return ::cuda::experimental::__count_query_group<_Tp, _Level>(__group);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _Group)
|
||||
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static _Tp rank_as(const _Group& __group) noexcept
|
||||
{
|
||||
return ::cuda::experimental::__rank_query_group<_Tp, _Level>(__group);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Group)
|
||||
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr bool is_root_rank(const _Group& __group) noexcept
|
||||
{
|
||||
return _Level::rank(__group) == 0;
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Group)
|
||||
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
|
||||
[[nodiscard]] _CCCL_API static constexpr bool is_part_of(const _Group& __group) noexcept
|
||||
{
|
||||
// todo: static_assert that the _Level <= _Group::unit_type
|
||||
return ::cuda::experimental::__is_part_of_group<_Level>(__group);
|
||||
}
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
|
||||
|
||||
private:
|
||||
template <class>
|
||||
friend struct __native_hierarchy_level_base;
|
||||
|
||||
_CCCL_EXEC_CHECK_DISABLE
|
||||
template <class _Tp, class... _Args>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __dims_as_impl(const _Args&... __args) noexcept
|
||||
{
|
||||
auto __exts = _Level::template extents_as<_Tp>(__args...);
|
||||
using _Exts = decltype(__exts);
|
||||
|
||||
hierarchy_query_result<_Tp> __ret{1, 1, 1};
|
||||
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
|
||||
{
|
||||
__ret[__i] = __exts.extent(__i);
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
|
||||
template <class... _Args>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __static_dims_impl(const _Args&... __args) noexcept
|
||||
{
|
||||
using _Exts = decltype(_Level::extents(__args...));
|
||||
|
||||
hierarchy_query_result<::cuda::std::size_t> __ret{1, 1, 1};
|
||||
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
|
||||
{
|
||||
__ret[__i] = _Exts::static_extent(__i);
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
|
||||
template <class... _Args>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __static_count_impl(const _Args&... __args) noexcept
|
||||
{
|
||||
using _Exts = decltype(_Level::extents(__args...));
|
||||
|
||||
if constexpr (_Exts::rank_dynamic() == 0)
|
||||
{
|
||||
::cuda::std::size_t __ret{1};
|
||||
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
|
||||
{
|
||||
__ret *= _Exts::static_extent(__i);
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
else
|
||||
{
|
||||
return ::cuda::std::dynamic_extent;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H
|
||||
@@ -0,0 +1,123 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_HIERARCHY_LEVELS_H
|
||||
#define _CUDA___HIERARCHY_HIERARCHY_LEVELS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/native_hierarchy_level_base.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
# include <cuda/std/__type_traits/type_list.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <nv/target>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES thread_level : __native_hierarchy_level_base<thread_level>
|
||||
{
|
||||
using __product_type = ::cuda::std::uint32_t;
|
||||
using __allowed_above = __allowed_levels<block_level>;
|
||||
using __allowed_below = __allowed_levels<>;
|
||||
|
||||
using __next_native_level = block_level;
|
||||
};
|
||||
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES warp_level : __native_hierarchy_level_base<warp_level>
|
||||
{
|
||||
using __product_type = ::cuda::std::uint32_t;
|
||||
|
||||
using __next_native_level = block_level;
|
||||
};
|
||||
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES block_level : __native_hierarchy_level_base<block_level>
|
||||
{
|
||||
using __product_type = ::cuda::std::uint32_t;
|
||||
using __allowed_above = __allowed_levels<grid_level, cluster_level>;
|
||||
using __allowed_below = __allowed_levels<thread_level>;
|
||||
|
||||
using __next_native_level = cluster_level;
|
||||
};
|
||||
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES cluster_level : __native_hierarchy_level_base<cluster_level>
|
||||
{
|
||||
using __product_type = ::cuda::std::uint32_t;
|
||||
using __allowed_above = __allowed_levels<grid_level>;
|
||||
using __allowed_below = __allowed_levels<block_level>;
|
||||
|
||||
using __next_native_level = grid_level;
|
||||
};
|
||||
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES grid_level : __native_hierarchy_level_base<grid_level>
|
||||
{
|
||||
using __product_type = ::cuda::std::uint64_t;
|
||||
using __allowed_above = __allowed_levels<>;
|
||||
using __allowed_below = __allowed_levels<block_level, cluster_level>;
|
||||
};
|
||||
|
||||
_CCCL_GLOBAL_CONSTANT thread_level gpu_thread;
|
||||
_CCCL_GLOBAL_CONSTANT warp_level warp;
|
||||
_CCCL_GLOBAL_CONSTANT block_level block;
|
||||
_CCCL_GLOBAL_CONSTANT cluster_level cluster;
|
||||
_CCCL_GLOBAL_CONSTANT grid_level grid;
|
||||
|
||||
// Struct to represent levels allowed below or above a certain level,
|
||||
// used for hierarchy sorting, validation and for hierarchy traversal
|
||||
template <typename... _Levels>
|
||||
struct __allowed_levels
|
||||
{
|
||||
using __default_unit = ::cuda::std::__type_index_c<0, _Levels..., void>;
|
||||
};
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
template <typename LevelType>
|
||||
using __default_unit_below = typename LevelType::__allowed_below::__default_unit;
|
||||
|
||||
template <class _QueryLevel, class _AllowedLevels>
|
||||
inline constexpr bool __is_level_allowed = false;
|
||||
|
||||
template <class _QueryLevel, class... _Levels>
|
||||
inline constexpr bool __is_level_allowed<_QueryLevel, __allowed_levels<_Levels...>> =
|
||||
(::cuda::std::is_same_v<_QueryLevel, _Levels> || ...);
|
||||
|
||||
template <class _L1, class _L2>
|
||||
inline constexpr bool __can_rhs_stack_on_lhs =
|
||||
__is_level_allowed<_L1, typename _L2::__allowed_below> || __is_level_allowed<_L2, typename _L1::__allowed_above>;
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
inline constexpr bool __legal_unit_for_level =
|
||||
__can_rhs_stack_on_lhs<_Unit, _Level> || __legal_unit_for_level<_Unit, __default_unit_below<_Level>>;
|
||||
|
||||
template <class _Unit>
|
||||
inline constexpr bool __legal_unit_for_level<_Unit, void> = false;
|
||||
} // namespace __detail
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_HIERARCHY_LEVELS_H
|
||||
@@ -0,0 +1,156 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H
|
||||
#define _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
template <class _Tp>
|
||||
struct hierarchy_query_result
|
||||
{
|
||||
using value_type = _Tp;
|
||||
|
||||
_Tp x;
|
||||
_Tp y;
|
||||
_Tp z;
|
||||
|
||||
[[nodiscard]] _CCCL_API constexpr _Tp& operator[](::cuda::std::size_t __i) noexcept
|
||||
{
|
||||
if (__i == 0)
|
||||
{
|
||||
return x;
|
||||
}
|
||||
else if (__i == 1)
|
||||
{
|
||||
return y;
|
||||
}
|
||||
else
|
||||
{
|
||||
return z;
|
||||
}
|
||||
}
|
||||
[[nodiscard]] _CCCL_API constexpr const _Tp& operator[](::cuda::std::size_t __i) const noexcept
|
||||
{
|
||||
if (__i == 0)
|
||||
{
|
||||
return x;
|
||||
}
|
||||
else if (__i == 1)
|
||||
{
|
||||
return y;
|
||||
}
|
||||
else
|
||||
{
|
||||
return z;
|
||||
}
|
||||
}
|
||||
|
||||
// Hide SFINAE conversion operators from Doxygen. The _CCCL_TEMPLATE/_CCCL_REQUIRES
|
||||
// macros expand to enable_if_t expressions that Breathe renders as invalid C++ template
|
||||
// parameter lists, causing Sphinx parse errors on the generated struct page.
|
||||
# ifndef _CCCL_DOXYGEN_INVOKED
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, signed char>)
|
||||
_CCCL_API constexpr operator char3() const noexcept
|
||||
{
|
||||
return {static_cast<signed char>(x), static_cast<signed char>(y), static_cast<signed char>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, short>)
|
||||
_CCCL_API constexpr operator short3() const noexcept
|
||||
{
|
||||
return {static_cast<short>(x), static_cast<short>(y), static_cast<short>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, int>)
|
||||
_CCCL_API constexpr operator int3() const noexcept
|
||||
{
|
||||
return {static_cast<int>(x), static_cast<int>(y), static_cast<int>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, long>)
|
||||
_CCCL_API constexpr operator long3() const noexcept
|
||||
{
|
||||
return {static_cast<long>(x), static_cast<long>(y), static_cast<long>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, long long>)
|
||||
_CCCL_API constexpr operator longlong3() const noexcept
|
||||
{
|
||||
return {static_cast<long long>(x), static_cast<long long>(y), static_cast<long long>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned char>)
|
||||
_CCCL_API constexpr operator uchar3() const noexcept
|
||||
{
|
||||
return {static_cast<unsigned char>(x), static_cast<unsigned char>(y), static_cast<unsigned char>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned short>)
|
||||
_CCCL_API constexpr operator ushort3() const noexcept
|
||||
{
|
||||
return {static_cast<unsigned short>(x), static_cast<unsigned short>(y), static_cast<unsigned short>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned>)
|
||||
_CCCL_API constexpr operator uint3() const noexcept
|
||||
{
|
||||
return {static_cast<unsigned>(x), static_cast<unsigned>(y), static_cast<unsigned>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned long>)
|
||||
_CCCL_API constexpr operator ulong3() const noexcept
|
||||
{
|
||||
return {static_cast<unsigned long>(x), static_cast<unsigned long>(y), static_cast<unsigned long>(z)};
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp2 = _Tp)
|
||||
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned long long>)
|
||||
_CCCL_API constexpr operator ulonglong3() const noexcept
|
||||
{
|
||||
return {static_cast<unsigned long long>(x), static_cast<unsigned long long>(y), static_cast<unsigned long long>(z)};
|
||||
}
|
||||
# endif // !_CCCL_DOXYGEN_INVOKED
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H
|
||||
@@ -0,0 +1,240 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H
|
||||
#define _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/hierarchy_levels.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__mdspan/extents.h>
|
||||
# include <cuda/std/__type_traits/is_integer.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
namespace __detail
|
||||
{
|
||||
/* Keeping it around in case issues like
|
||||
https://github.com/NVIDIA/cccl/issues/522 template <typename T, size_t...
|
||||
Extents> struct extents_corrected : public ::cuda::std::extents<T, Extents...> {
|
||||
using ::cuda::std::extents<T, Extents...>::extents;
|
||||
|
||||
template <typename ::cuda::std::extents<T, Extents...>::rank_type Id>
|
||||
_CCCL_API constexpr auto extent_corrected() const {
|
||||
if constexpr (::cuda::std::extents<T, Extents...>::static_extent(Id) !=
|
||||
::cuda::std::dynamic_extent) { return this->static_extent(Id);
|
||||
}
|
||||
else {
|
||||
return this->extent(Id);
|
||||
}
|
||||
}
|
||||
};
|
||||
*/
|
||||
|
||||
template <class _Dims>
|
||||
struct __dimensions_handler
|
||||
{
|
||||
static constexpr bool __is_type_supported = ::cuda::std::__cccl_is_integer_v<_Dims>;
|
||||
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __translate(const _Dims& __dims) noexcept
|
||||
{
|
||||
return ::cuda::std::extents<dimensions_index_type, ::cuda::std::dynamic_extent, 1, 1>(
|
||||
static_cast<unsigned>(__dims));
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __dimensions_handler<::dim3>
|
||||
{
|
||||
static constexpr bool __is_type_supported = true;
|
||||
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __translate(const ::dim3& __dims) noexcept
|
||||
{
|
||||
return ::cuda::std::dims<3, dimensions_index_type>(__dims.x, __dims.y, __dims.z);
|
||||
}
|
||||
};
|
||||
|
||||
template <class _Dims, _Dims _Val>
|
||||
struct __dimensions_handler<::cuda::std::integral_constant<_Dims, _Val>>
|
||||
{
|
||||
static constexpr bool __is_type_supported = ::cuda::std::__cccl_is_integer_v<_Dims>;
|
||||
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __translate(const _Dims& __dims) noexcept
|
||||
{
|
||||
return ::cuda::std::extents<dimensions_index_type, static_cast<::cuda::std::size_t>(__dims), 1, 1>();
|
||||
}
|
||||
};
|
||||
} // namespace __detail
|
||||
|
||||
/**
|
||||
* @brief Type representing dimensions of a level in a thread hierarchy.
|
||||
*
|
||||
* This type combines a level type like grid_level or block_level with
|
||||
* a cuda::std::extents object to describe dimensions of a level in a thread
|
||||
* hierarchy. This type is not intended to be created explicitly and *_dims
|
||||
* functions creating them should be used instead. They will translate the input
|
||||
* arguments to a correct cuda::std::extents to be stored inside
|
||||
* hierarchy_level_desc.
|
||||
* While this type can be used to access the stored dimensions,
|
||||
* the main usage is to pass a number of hierarchy_level_desc objects
|
||||
* to make_hierarchy function in order to create a hierarchy.
|
||||
* This type does not store what the unit is for the stored dimensions,
|
||||
* it is instead implied by the level below it in a hierarchy object.
|
||||
* In case there is a need to store more information about a specific level,
|
||||
* for example some library-specific information, this type can be derived
|
||||
* from and the resulting type can be used to build the hierarchy.
|
||||
*
|
||||
* @par Snippet
|
||||
* @code
|
||||
* #include <cudax/hierarchy_dimensions.cuh>
|
||||
*
|
||||
* auto hierarchy = make_hierarchy(grid_dims(256), block_dims<8, 8, 8>());
|
||||
* assert(hierarchy.level(grid).dims.x == 256);
|
||||
* @endcode
|
||||
* @par
|
||||
*
|
||||
* @tparam Level
|
||||
* Type indicating which hierarchy level this is
|
||||
*
|
||||
* @tparam Dimensions
|
||||
* Type holding the dimensions of this level
|
||||
*/
|
||||
template <class _Level, class _Exts>
|
||||
class _CCCL_DECLSPEC_EMPTY_BASES hierarchy_level_desc : __hierarchy_level_desc_base
|
||||
{
|
||||
static_assert(__is_hierarchy_level_v<_Level>);
|
||||
static_assert(::cuda::std::__is_cuda_std_extents_v<_Exts>);
|
||||
|
||||
// Needs alignas to work around an issue with tuple
|
||||
alignas(16) _Exts __exts_; // Unit for dimensions is implicit
|
||||
|
||||
public:
|
||||
using level_type = _Level;
|
||||
using extents_type = _Exts;
|
||||
|
||||
_CCCL_HIDE_FROM_ABI constexpr hierarchy_level_desc() noexcept = default;
|
||||
|
||||
_CCCL_API constexpr hierarchy_level_desc(const _Exts& __exts) noexcept
|
||||
: __exts_(__exts)
|
||||
{}
|
||||
|
||||
[[nodiscard]] _CCCL_API constexpr _Exts extents() const noexcept
|
||||
{
|
||||
return __exts_;
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool
|
||||
operator==(const hierarchy_level_desc& __lhs, const hierarchy_level_desc& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__exts_ == __rhs.__exts_;
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool
|
||||
operator!=(const hierarchy_level_desc& __lhs, const hierarchy_level_desc& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__exts_ != __rhs.__exts_;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing grid_level
|
||||
*
|
||||
* This function creates a statically sized level from up to three template
|
||||
* arguments.
|
||||
*/
|
||||
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
|
||||
[[nodiscard]] _CCCL_API constexpr auto grid_dims() noexcept
|
||||
{
|
||||
return hierarchy_level_desc<grid_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing grid_level
|
||||
*
|
||||
* This function creates the level from an integral or dim3 argument.
|
||||
*/
|
||||
template <class _Dims>
|
||||
[[nodiscard]] _CCCL_API constexpr auto grid_dims(_Dims __dims) noexcept
|
||||
{
|
||||
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
|
||||
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
|
||||
return hierarchy_level_desc<grid_level, decltype(__translated_dims)>(__translated_dims);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing cluster_level
|
||||
*
|
||||
* This function creates a statically sized level from up to three template
|
||||
* arguments.
|
||||
*/
|
||||
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
|
||||
[[nodiscard]] _CCCL_API constexpr auto cluster_dims() noexcept
|
||||
{
|
||||
return hierarchy_level_desc<cluster_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing cluster_level
|
||||
*
|
||||
* This function creates the level from an integral or dim3 argument.
|
||||
*/
|
||||
template <class _Dims>
|
||||
[[nodiscard]] _CCCL_API constexpr auto cluster_dims(_Dims __dims) noexcept
|
||||
{
|
||||
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
|
||||
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
|
||||
return hierarchy_level_desc<cluster_level, decltype(__translated_dims)>(__translated_dims);
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing block_level
|
||||
*
|
||||
* This function creates a statically sized level from up to three template
|
||||
* arguments.
|
||||
*/
|
||||
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
|
||||
[[nodiscard]] _CCCL_API constexpr auto block_dims() noexcept
|
||||
{
|
||||
return hierarchy_level_desc<block_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
|
||||
}
|
||||
|
||||
/**
|
||||
* @brief Creates an instance of hierarchy_level_desc describing block_level
|
||||
*
|
||||
* This function creates the level from an integral or dim3 argument.
|
||||
*/
|
||||
template <class _Dims>
|
||||
[[nodiscard]] _CCCL_API constexpr auto block_dims(_Dims __dims) noexcept
|
||||
{
|
||||
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
|
||||
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
|
||||
return hierarchy_level_desc<block_level, decltype(__translated_dims)>(__translated_dims);
|
||||
}
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H
|
||||
@@ -0,0 +1,175 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H
|
||||
#define _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/hierarchy_level_base.h>
|
||||
# include <cuda/__hierarchy/hierarchy_query_result.h>
|
||||
# include <cuda/__hierarchy/queries/count.h>
|
||||
# include <cuda/__hierarchy/queries/extents.h>
|
||||
# include <cuda/__hierarchy/queries/index.h>
|
||||
# include <cuda/__hierarchy/queries/rank.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__mdspan/extents.h>
|
||||
# include <cuda/std/__type_traits/is_integer.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
|
||||
// to warn about applying [[nodiscard]] to a function that returns void.
|
||||
_CCCL_DIAG_PUSH
|
||||
# if _CCCL_CUDA_COMPILER(NVCC)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
|
||||
# endif // _CCCL_CUDA_COMPILER(NVCC)
|
||||
|
||||
template <class _Level>
|
||||
struct _CCCL_DECLSPEC_EMPTY_BASES __native_hierarchy_level_base : hierarchy_level_base<_Level>
|
||||
{
|
||||
using __base_type = hierarchy_level_base<_Level>;
|
||||
using __base_type::count;
|
||||
using __base_type::count_as;
|
||||
using __base_type::dims;
|
||||
using __base_type::dims_as;
|
||||
using __base_type::extents;
|
||||
using __base_type::extents_as;
|
||||
using __base_type::static_count;
|
||||
using __base_type::static_dims;
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
using __base_type::index;
|
||||
using __base_type::index_as;
|
||||
using __base_type::rank;
|
||||
using __base_type::rank_as;
|
||||
|
||||
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
|
||||
using __base_type::is_part_of;
|
||||
using __base_type::is_root_rank;
|
||||
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto dims(const _InLevel& __level) noexcept
|
||||
{
|
||||
return _Level::template dims_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto static_dims(const _InLevel& __level) noexcept
|
||||
{
|
||||
return __base_type::__static_dims_impl(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto extents(const _InLevel& __level) noexcept
|
||||
{
|
||||
return _Level::template extents_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto static_count(const _InLevel& __level) noexcept
|
||||
{
|
||||
return __base_type::__static_count_impl(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto count(const _InLevel& __level) noexcept
|
||||
{
|
||||
return _Level::template count_as<typename __base_type::template __default_1d_query_type<_InLevel>>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto index(const _InLevel& __level) noexcept
|
||||
{
|
||||
return _Level::template index_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto rank(const _InLevel& __level) noexcept
|
||||
{
|
||||
return _Level::template rank_as<typename __base_type::template __default_1d_query_type<_InLevel>>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto dims_as(const _InLevel& __level) noexcept
|
||||
{
|
||||
return __base_type::template __dims_as_impl<_Tp>(__level);
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto extents_as(const _InLevel&) noexcept
|
||||
{
|
||||
return __extents_query_native<_Level, _InLevel>::template __call<_Tp>();
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto count_as(const _InLevel&) noexcept
|
||||
{
|
||||
return __count_query_native<_Level, _InLevel>::template __call<_Tp>();
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto index_as(const _InLevel&) noexcept
|
||||
{
|
||||
return __index_query_native<_Level, _InLevel>::template __call<_Tp>();
|
||||
}
|
||||
|
||||
_CCCL_TEMPLATE(class _Tp, class _InLevel)
|
||||
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp rank_as(const _InLevel&) noexcept
|
||||
{
|
||||
return __rank_query_native<_Level, _InLevel>::template __call<_Tp>();
|
||||
}
|
||||
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
};
|
||||
|
||||
_CCCL_DIAG_POP
|
||||
|
||||
template <>
|
||||
struct __native_hierarchy_level_base<grid_level> : hierarchy_level_base<grid_level>
|
||||
{};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H
|
||||
@@ -0,0 +1,113 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_QUERIES_COUNT_H
|
||||
#define _CUDA___HIERARCHY_QUERIES_COUNT_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// native hierarchy queries
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
|
||||
// to warn about applying [[nodiscard]] to a function that returns void.
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
|
||||
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
|
||||
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __count_query_native
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
const auto __exts = __extents_query_native<_Unit, _Level>::template __call<_Tp>();
|
||||
|
||||
_Tp __ret = 1;
|
||||
for (::cuda::std::size_t __i = 0; __i < __exts.rank(); ++__i)
|
||||
{
|
||||
__ret *= __exts.extent(__i);
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __count_query_native<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
unsigned __count = 1;
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__count = ::__clusterSizeInBlocks();))
|
||||
return static_cast<_Tp>(__count);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __count_query_native<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
return static_cast<_Tp>(static_cast<_Tp>(gridDim.x) * gridDim.y * gridDim.z);
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_DIAG_POP
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// hierarchy queries
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __count_query
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr _Tp __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
const auto __exts = __extents_query<_Unit, _Level>::template __call<_Tp>(__hier);
|
||||
|
||||
_Tp __ret = 1;
|
||||
for (::cuda::std::size_t __i = 0; __i < __exts.rank(); ++__i)
|
||||
{
|
||||
__ret *= __exts.extent(__i);
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_QUERIES_COUNT_H
|
||||
@@ -0,0 +1,359 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_QUERIES_EXTENTS_H
|
||||
#define _CUDA___HIERARCHY_QUERIES_EXTENTS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__cmath/ceil_div.h>
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__algorithm/max.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__mdspan/extents.h>
|
||||
# include <cuda/std/__utility/integer_sequence.h>
|
||||
# include <cuda/std/array>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// helpers
|
||||
|
||||
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::size_t
|
||||
__hierarchy_static_extents_mul_helper(::cuda::std::size_t __lhs, ::cuda::std::size_t __rhs) noexcept
|
||||
{
|
||||
if (__lhs == ::cuda::std::dynamic_extent || __rhs == ::cuda::std::dynamic_extent)
|
||||
{
|
||||
return ::cuda::std::dynamic_extent;
|
||||
}
|
||||
else
|
||||
{
|
||||
return __lhs * __rhs;
|
||||
}
|
||||
}
|
||||
|
||||
template <class _ResultIndex, class _LhsExts, class _RhsExts, ::cuda::std::size_t... _Is>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __hierarchy_static_extents_mul(::cuda::std::index_sequence<_Is...>) noexcept
|
||||
{
|
||||
return ::cuda::std::extents<
|
||||
_ResultIndex,
|
||||
::cuda::__hierarchy_static_extents_mul_helper((_Is < _LhsExts::rank()) ? _LhsExts::static_extent(_Is) : 1,
|
||||
(_Is < _RhsExts::rank()) ? _RhsExts::static_extent(_Is) : 1)...>{};
|
||||
}
|
||||
|
||||
//! @brief Multiplies 2 extents in column major order together, returning a new extents type. If the ranks don't match,
|
||||
//! the extent with lower rank is padded with 1s on the right to match the rank of the other.
|
||||
//!
|
||||
//! @param __lhs The left hand side extents to multiply.
|
||||
//! @param __rhs The right hand side extents to multiply.
|
||||
//!
|
||||
//! @return The result of multiplying the extents together.
|
||||
template <class _Index, ::cuda::std::size_t... _LhsExts, ::cuda::std::size_t... _RhsExts>
|
||||
[[nodiscard]] _CCCL_API constexpr auto
|
||||
__hierarchy_extents_mul(const ::cuda::std::extents<_Index, _LhsExts...>& __lhs,
|
||||
const ::cuda::std::extents<_Index, _RhsExts...>& __rhs) noexcept
|
||||
{
|
||||
using _Lhs = ::cuda::std::extents<_Index, _LhsExts...>;
|
||||
using _Rhs = ::cuda::std::extents<_Index, _RhsExts...>;
|
||||
|
||||
constexpr auto __rank = ::cuda::std::max(_Lhs::rank(), _Rhs::rank());
|
||||
using _Ret =
|
||||
decltype(::cuda::__hierarchy_static_extents_mul<_Index, _Lhs, _Rhs>(::cuda::std::make_index_sequence<__rank>{}));
|
||||
|
||||
::cuda::std::array<_Index, __rank> __ret{};
|
||||
for (::cuda::std::size_t __i = 0; __i < __rank; ++__i)
|
||||
{
|
||||
if (_Ret::static_extent(__i) == ::cuda::std::dynamic_extent)
|
||||
{
|
||||
__ret[__i] = static_cast<_Index>((__i < _Lhs::rank()) ? __lhs.extent(__i) : 1)
|
||||
* static_cast<_Index>((__i < _Rhs::rank()) ? __rhs.extent(__i) : 1);
|
||||
}
|
||||
else
|
||||
{
|
||||
__ret[__i] = static_cast<_Index>(_Ret::static_extent(__i));
|
||||
}
|
||||
}
|
||||
return _Ret{__ret};
|
||||
}
|
||||
|
||||
template <class _Index, class _OrgIndex, ::cuda::std::size_t... _StaticExts>
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::std::extents<_Index, _StaticExts...>
|
||||
__hierarchy_extents_cast(::cuda::std::extents<_OrgIndex, _StaticExts...> __org_exts) noexcept
|
||||
{
|
||||
using _OrgExts = ::cuda::std::extents<_OrgIndex, _StaticExts...>;
|
||||
::cuda::std::array<_Index, _OrgExts::rank()> __ret{};
|
||||
for (::cuda::std::size_t __i = 0; __i < _OrgExts::rank(); ++__i)
|
||||
{
|
||||
if (_OrgExts::static_extent(__i) == ::cuda::std::dynamic_extent)
|
||||
{
|
||||
__ret[__i] = static_cast<_Index>(__org_exts.extent(__i));
|
||||
}
|
||||
else
|
||||
{
|
||||
__ret[__i] = static_cast<_Index>(_OrgExts::static_extent(__i));
|
||||
}
|
||||
}
|
||||
return ::cuda::std::extents<_Index, _StaticExts...>{__ret};
|
||||
}
|
||||
|
||||
template <class _Tp, class _Unit, class _Level, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API constexpr auto __extents_query_generic(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
static_assert(__has_bottom_unit_or_level_v<_Unit, _Hierarchy> || __is_native_hierarchy_level_v<_Unit>,
|
||||
"_Hierarchy doesn't contain _Unit");
|
||||
static_assert(_Hierarchy::has_level(_Level{}) || __is_native_hierarchy_level_v<_Level>,
|
||||
"_Hierarchy doesn't contain _Level");
|
||||
|
||||
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
|
||||
using _CurrExts = decltype(::cuda::__hierarchy_extents_cast<_Tp>(__hier.level(_NextLevel{}).extents()));
|
||||
|
||||
// Remove dependency on runtime storage. This makes the queries work for hierarchy levels with all static extents
|
||||
// in constant evaluated context.
|
||||
_CurrExts __curr_exts{};
|
||||
if constexpr (_CurrExts::rank_dynamic() > 0)
|
||||
{
|
||||
__curr_exts = ::cuda::__hierarchy_extents_cast<_Tp>(__hier.level(_NextLevel{}).extents());
|
||||
}
|
||||
|
||||
if constexpr (!::cuda::std::is_same_v<_NextLevel, _Level>)
|
||||
{
|
||||
const auto __next_exts = __extents_query<_NextLevel, _Level>::template __call<_Tp>(__hier);
|
||||
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
|
||||
}
|
||||
else
|
||||
{
|
||||
return __curr_exts;
|
||||
}
|
||||
}
|
||||
|
||||
// native hierarchy queries
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
|
||||
// to warn about applying [[nodiscard]] to a function that returns void.
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
|
||||
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
|
||||
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __extents_query_native
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static auto __call() noexcept
|
||||
{
|
||||
static_assert(__is_natively_reachable_hierarchy_level_v<_Unit, _Level>, "_Level must be reachable from _Unit");
|
||||
|
||||
using _NextLevel = typename _Unit::__next_native_level;
|
||||
const auto __next_exts = __extents_query_native<_NextLevel, _Level>::template __call<_Tp>();
|
||||
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::extents<_Tp, 32> __call() noexcept
|
||||
{
|
||||
return {};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<thread_level, block_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
|
||||
{
|
||||
return ::cuda::std::dims<3, _Tp>{
|
||||
static_cast<_Tp>(blockDim.x), static_cast<_Tp>(blockDim.y), static_cast<_Tp>(blockDim.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<warp_level, block_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<1, _Tp> __call() noexcept
|
||||
{
|
||||
const auto __thread_count = blockDim.x * blockDim.y * blockDim.z;
|
||||
return ::cuda::std::dims<1, _Tp>{static_cast<_Tp>(::cuda::ceil_div(__thread_count, 32))};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
|
||||
{
|
||||
::dim3 __dims{1u, 1u, 1u};
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__dims = ::__clusterDim();))
|
||||
return ::cuda::std::dims<3, _Tp>{static_cast<_Tp>(__dims.x), static_cast<_Tp>(__dims.y), static_cast<_Tp>(__dims.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
|
||||
{
|
||||
return ::cuda::std::dims<3, _Tp>{
|
||||
static_cast<_Tp>(gridDim.x), static_cast<_Tp>(gridDim.y), static_cast<_Tp>(gridDim.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query_native<cluster_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
|
||||
{
|
||||
::dim3 __dims{gridDim};
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__dims = ::__clusterGridDimInClusters();))
|
||||
return ::cuda::std::dims<3, _Tp>{static_cast<_Tp>(__dims.x), static_cast<_Tp>(__dims.y), static_cast<_Tp>(__dims.z)};
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_DIAG_POP
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// hierarchy queries
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __extents_query
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return ::cuda::__extents_query_generic<_Tp, _Unit, _Level>(__hier);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr ::cuda::std::extents<_Tp, 32> __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
static_assert(__has_bottom_unit_or_level_v<thread_level, _Hierarchy>, "_Hierarchy doesn't contain thread_level");
|
||||
static_assert(_Hierarchy::template has_level<block_level>(), "_Hierarchy doesn't contain block_level");
|
||||
return {};
|
||||
}
|
||||
};
|
||||
|
||||
template <class _Level>
|
||||
struct __extents_query<warp_level, _Level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
auto __block_exts = __extents_query<thread_level, block_level>::template __call<_Tp>(__hier);
|
||||
using _BlockExts = decltype(__block_exts);
|
||||
|
||||
if constexpr (_BlockExts::rank_dynamic() == 0)
|
||||
{
|
||||
constexpr auto __static_thread_count =
|
||||
_BlockExts::static_extent(0) * _BlockExts::static_extent(1) * _BlockExts::static_extent(2);
|
||||
static_assert(__static_thread_count >= 32, "_Hierarchy doesn't contain enough threads to fill a single warp");
|
||||
|
||||
constexpr auto __static_warp_count = ::cuda::ceil_div(__static_thread_count, 32);
|
||||
::cuda::std::extents<_Tp, __static_warp_count> __curr_exts{};
|
||||
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
|
||||
{
|
||||
return __curr_exts;
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto __next_exts = __extents_query<block_level, _Level>::template __call<_Tp>(__hier);
|
||||
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
|
||||
}
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto __thread_count = __block_exts.extent(0) * __block_exts.extent(1) * __block_exts.extent(2);
|
||||
_CCCL_ASSERT(__thread_count >= 32, "_Hierarchy doesn't contain enough threads to fill a single warp");
|
||||
|
||||
const auto __warp_count = static_cast<_Tp>(::cuda::ceil_div(__thread_count, 32));
|
||||
::cuda::std::dims<1, _Tp> __curr_exts{__warp_count};
|
||||
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
|
||||
{
|
||||
return __curr_exts;
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto __next_exts = __extents_query<block_level, _Level>::template __call<_Tp>(__hier);
|
||||
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
|
||||
}
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
if constexpr (_Hierarchy::template has_level<cluster_level>())
|
||||
{
|
||||
return ::cuda::__extents_query_generic<_Tp, block_level, cluster_level>(__hier);
|
||||
}
|
||||
else
|
||||
{
|
||||
static_assert(__has_bottom_unit_or_level_v<block_level, _Hierarchy>, "_Hierarchy doesn't contain block_level");
|
||||
static_assert(_Hierarchy::template has_level<grid_level>(), "_Hierarchy doesn't contain grid_level");
|
||||
return ::cuda::std::extents<_Tp, 1, 1, 1>{};
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __extents_query<cluster_level, grid_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
if constexpr (_Hierarchy::template has_level<cluster_level>())
|
||||
{
|
||||
return ::cuda::__extents_query_generic<_Tp, cluster_level, grid_level>(__hier);
|
||||
}
|
||||
else
|
||||
{
|
||||
return __extents_query<block_level, grid_level>::template __call<_Tp>(__hier);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_QUERIES_EXTENTS_H
|
||||
@@ -0,0 +1,268 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_QUERIES_INDEX_H
|
||||
#define _CUDA___HIERARCHY_QUERIES_INDEX_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__cmath/ceil_div.h>
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/hierarchy_query_result.h>
|
||||
# include <cuda/__hierarchy/queries/extents.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
# include <cuda/std/__mdspan/extents.h>
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
# include <cuda/__ptx/instructions/get_sreg.h>
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
|
||||
// to warn about applying [[nodiscard]] to a function that returns void.
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
|
||||
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
|
||||
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
|
||||
// native hierarchy queries
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __index_query_native
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
static_assert(__is_natively_reachable_hierarchy_level_v<_Unit, _Level>, "_Level must be reachable from _Unit");
|
||||
|
||||
using _NextLevel = typename _Unit::__next_native_level;
|
||||
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
const auto __next_idx = __index_query_native<_NextLevel, _Level>::template __call<_Tp>();
|
||||
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
|
||||
hierarchy_query_result<_Tp> __ret{};
|
||||
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
|
||||
{
|
||||
__ret[__i] = __curr_idx[__i] + ((__i < __curr_exts.rank()) ? __curr_exts.extent(__i) : 1) * __next_idx[__i];
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
// todo(dabayer): Is it worth using cuda::ptx::get_sreg_laneid() here? Doesn't it prevent some other optimizations
|
||||
// due to using inline ptx?
|
||||
return {static_cast<_Tp>(::cuda::ptx::get_sreg_laneid()), 0, 0};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<thread_level, block_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
return {static_cast<_Tp>(threadIdx.x), static_cast<_Tp>(threadIdx.y), static_cast<_Tp>(threadIdx.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<warp_level, block_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
const auto __thread_rank = (threadIdx.z * blockDim.y + threadIdx.y) * blockDim.x + threadIdx.x;
|
||||
return {static_cast<_Tp>(__thread_rank / 32), 0, 0};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
::dim3 __idx{0u, 0u, 0u};
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__idx = ::__clusterRelativeBlockIdx();))
|
||||
return {static_cast<_Tp>(__idx.x), static_cast<_Tp>(__idx.y), static_cast<_Tp>(__idx.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
return {static_cast<_Tp>(blockIdx.x), static_cast<_Tp>(blockIdx.y), static_cast<_Tp>(blockIdx.z)};
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query_native<cluster_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
|
||||
{
|
||||
::dim3 __idx{blockIdx};
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__idx = ::__clusterIdx();))
|
||||
return {static_cast<_Tp>(__idx.x), static_cast<_Tp>(__idx.y), static_cast<_Tp>(__idx.z)};
|
||||
}
|
||||
};
|
||||
|
||||
// hierarchy queries
|
||||
|
||||
template <class _Tp, class _Unit, class _NextLevel, class _Level, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API hierarchy_query_result<_Tp> __index_query_generic(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
if constexpr (::cuda::std::is_same_v<_Level, _NextLevel>)
|
||||
{
|
||||
using _CurrExts = decltype(__extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier));
|
||||
auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
|
||||
{
|
||||
if (__i >= _CurrExts::rank() || _CurrExts::static_extent(__i) == 1)
|
||||
{
|
||||
__curr_idx[__i] = 0;
|
||||
}
|
||||
}
|
||||
return __curr_idx;
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto __curr_exts = __extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
|
||||
const auto __next_idx = __index_query<_NextLevel, _Level>::template __call<_Tp>(__hier);
|
||||
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
|
||||
hierarchy_query_result<_Tp> __ret{};
|
||||
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
|
||||
{
|
||||
__ret[__i] = __curr_idx[__i] + ((__i < __curr_exts.rank()) ? __curr_exts.extent(__i) : 1) * __next_idx[__i];
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
}
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __index_query
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
static_assert(__has_bottom_unit_or_level_v<_Unit, _Hierarchy> || __is_native_hierarchy_level_v<_Unit>,
|
||||
"_Hierarchy doesn't contain _Unit");
|
||||
static_assert(_Hierarchy::template has_level<_Level>() || __is_native_hierarchy_level_v<_Level>,
|
||||
"_Hierarchy doesn't contain _Level");
|
||||
|
||||
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
|
||||
return ::cuda::__index_query_generic<_Tp, _Unit, _NextLevel, _Level>(__hier);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __index_query_native<thread_level, warp_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <class _Level>
|
||||
struct __index_query<warp_level, _Level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
const auto __block_exts = __extents_query<thread_level, block_level>::template __call<unsigned>(__hier);
|
||||
const auto __thread_idx = __index_query<thread_level, block_level>::template __call<unsigned>(__hier);
|
||||
const auto __thread_rank =
|
||||
(__thread_idx.z * __block_exts.extent(1) + __thread_idx.y) * __block_exts.extent(0) + __thread_idx.x;
|
||||
const auto __warp_rank = __thread_rank / 32;
|
||||
|
||||
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
|
||||
{
|
||||
return {static_cast<_Tp>(__warp_rank), 0, 0};
|
||||
}
|
||||
else
|
||||
{
|
||||
const auto __thread_count = __block_exts.extent(0) * __block_exts.extent(1) * __block_exts.extent(2);
|
||||
const auto __warp_count = ::cuda::ceil_div(__thread_count, 32);
|
||||
const auto __next_idx = __index_query<block_level, _Level>::template __call<_Tp>(__hier);
|
||||
return {static_cast<_Tp>(__next_idx.x * __warp_count + __warp_rank), __next_idx.y, __next_idx.z};
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
return __index_query_native<block_level, cluster_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
return __index_query_native<block_level, grid_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __index_query<cluster_level, grid_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return __index_query_native<cluster_level, grid_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_DIAG_POP
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_QUERIES_INDEX_H
|
||||
215
cccl_upstream/libcudacxx/include/cuda/__hierarchy/queries/rank.h
Normal file
215
cccl_upstream/libcudacxx/include/cuda/__hierarchy/queries/rank.h
Normal file
@@ -0,0 +1,215 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_QUERIES_RANK_H
|
||||
#define _CUDA___HIERARCHY_QUERIES_RANK_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/__hierarchy/hierarchy_query_result.h>
|
||||
# include <cuda/__hierarchy/queries/count.h>
|
||||
# include <cuda/__hierarchy/queries/extents.h>
|
||||
# include <cuda/__hierarchy/queries/index.h>
|
||||
# include <cuda/__hierarchy/traits.h>
|
||||
# include <cuda/std/__cstddef/types.h>
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
# include <cuda/__ptx/instructions/get_sreg.h>
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
# if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
|
||||
// to warn about applying [[nodiscard]] to a function that returns void.
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
|
||||
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
|
||||
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
|
||||
|
||||
// native hierarchy queries
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __rank_query_native
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
using _NextLevel = typename _Unit::__next_native_level;
|
||||
|
||||
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
|
||||
_Tp __ret = 0;
|
||||
if constexpr (!::cuda::std::is_same_v<_Level, _NextLevel>)
|
||||
{
|
||||
__ret = __rank_query_native<_NextLevel, _Level>::template __call<_Tp>()
|
||||
* __count_query_native<_Unit, _NextLevel>::template __call<_Tp>();
|
||||
}
|
||||
|
||||
for (::cuda::std::size_t __i = __curr_exts.rank(); __i > 0; --__i)
|
||||
{
|
||||
_Tp __inc = __curr_idx[__i - 1];
|
||||
for (::cuda::std::size_t __j = __i - 1; __j > 0; --__j)
|
||||
{
|
||||
__inc *= __curr_exts.extent(__j - 1);
|
||||
}
|
||||
__ret += __inc;
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query_native<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
return static_cast<_Tp>(::cuda::ptx::get_sreg_laneid());
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query_native<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
unsigned __rank = 0;
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_90, (__rank = ::__clusterRelativeBlockRank();))
|
||||
return static_cast<_Tp>(__rank);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query_native<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
|
||||
{
|
||||
return static_cast<_Tp>((static_cast<_Tp>(blockIdx.z) * gridDim.y + blockIdx.y) * gridDim.x + blockIdx.x);
|
||||
}
|
||||
};
|
||||
|
||||
// hierarchy queries
|
||||
|
||||
template <class _Tp, class _Unit, class _NextLevel, class _Level, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API _Tp __rank_query_generic(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
const auto __curr_exts = __extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
|
||||
const auto __curr_idx = __index_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
|
||||
|
||||
_Tp __ret = 0;
|
||||
if constexpr (!::cuda::std::is_same_v<_Level, _NextLevel>)
|
||||
{
|
||||
__ret = __rank_query<_NextLevel, _Level>::template __call<_Tp>(__hier)
|
||||
* __count_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
|
||||
}
|
||||
|
||||
for (::cuda::std::size_t __i = __curr_exts.rank(); __i > 0; --__i)
|
||||
{
|
||||
_Tp __inc = __curr_idx[__i - 1];
|
||||
for (::cuda::std::size_t __j = __i - 1; __j > 0; --__j)
|
||||
{
|
||||
__inc *= __curr_exts.extent(__j - 1);
|
||||
}
|
||||
__ret += __inc;
|
||||
}
|
||||
return __ret;
|
||||
}
|
||||
|
||||
template <class _Unit, class _Level>
|
||||
struct __rank_query
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
|
||||
return ::cuda::__rank_query_generic<_Tp, _Unit, _NextLevel, _Level>(__hier);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query<thread_level, warp_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
return __rank_query_native<thread_level, warp_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <class _Level>
|
||||
struct __rank_query<warp_level, _Level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return ::cuda::__rank_query_generic<_Tp, warp_level, block_level, _Level>(__hier);
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query<block_level, cluster_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
return __rank_query_native<block_level, cluster_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query<block_level, grid_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
|
||||
{
|
||||
return __rank_query_native<block_level, grid_level>::template __call<_Tp>();
|
||||
}
|
||||
};
|
||||
|
||||
template <>
|
||||
struct __rank_query<cluster_level, grid_level>
|
||||
{
|
||||
template <class _Tp, class _Hierarchy>
|
||||
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
|
||||
{
|
||||
return ::cuda::__rank_query_generic<_Tp, cluster_level, grid_level, grid_level>(__hier);
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_DIAG_POP
|
||||
# endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_QUERIES_RANK_H
|
||||
114
cccl_upstream/libcudacxx/include/cuda/__hierarchy/traits.h
Normal file
114
cccl_upstream/libcudacxx/include/cuda/__hierarchy/traits.h
Normal file
@@ -0,0 +1,114 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___HIERARCHY_TRAITS_H
|
||||
#define _CUDA___HIERARCHY_TRAITS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/hierarchy.h>
|
||||
# include <cuda/std/__tuple_dir/get.h>
|
||||
# include <cuda/std/__type_traits/is_same.h>
|
||||
# include <cuda/std/__type_traits/remove_cvref.h>
|
||||
# include <cuda/std/__type_traits/type_list.h>
|
||||
# include <cuda/std/__type_traits/void_t.h>
|
||||
# include <cuda/std/cstdint>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// __is_natively_reachable_hierarchy_level_v
|
||||
|
||||
template <class _FromLevel, class _CurrLevel, class _ToLevel, class = void>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v = false;
|
||||
template <class _FromLevel, class _CurrLevel, class _ToLevel>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<
|
||||
_FromLevel,
|
||||
_CurrLevel,
|
||||
_ToLevel,
|
||||
::cuda::std::void_t<typename _CurrLevel::__next_native_level>> =
|
||||
__is_natively_reachable_hierarchy_level_helper_v<_FromLevel, typename _CurrLevel::__next_native_level, _ToLevel>;
|
||||
template <class _Level, class _ToLevel>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<_Level, _Level, _ToLevel> = false;
|
||||
template <class _FromLevel, class _Level>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<_FromLevel, _Level, _Level> = true;
|
||||
|
||||
template <class _FromLevel, class _ToLevel, class = void>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_v = false;
|
||||
template <class _FromLevel, class _ToLevel>
|
||||
inline constexpr bool __is_natively_reachable_hierarchy_level_v<
|
||||
_FromLevel,
|
||||
_ToLevel,
|
||||
::cuda::std::void_t<typename _FromLevel::__next_native_level>> =
|
||||
__is_native_hierarchy_level_v<_ToLevel>
|
||||
&& __is_natively_reachable_hierarchy_level_helper_v<_FromLevel, typename _FromLevel::__next_native_level, _ToLevel>;
|
||||
|
||||
// __level_type_of
|
||||
|
||||
template <class _LevelDesc>
|
||||
using __level_type_of = typename _LevelDesc::level_type;
|
||||
|
||||
// __has_bottom_unit_or_level_v
|
||||
|
||||
template <class _QueryLevel, class _Hierarchy>
|
||||
inline constexpr bool __has_bottom_unit_or_level_v =
|
||||
::cuda::std::is_same_v<_QueryLevel, typename _Hierarchy::bottom_unit_type>
|
||||
|| _Hierarchy::template has_level<_QueryLevel>();
|
||||
|
||||
// __next_hierarchy_level
|
||||
|
||||
template <class _Level, class _Hierarchy>
|
||||
struct __next_hierarchy_level;
|
||||
|
||||
template <class _Level, class _BottomUnit, class... _LevelDescs>
|
||||
struct __next_hierarchy_level<_Level, hierarchy<_BottomUnit, _LevelDescs...>>
|
||||
{
|
||||
static constexpr ::cuda::std::size_t __level_idx =
|
||||
hierarchy<_BottomUnit, _LevelDescs...>::template __level_idx<_Level>;
|
||||
using __type = ::cuda::std::__type_index_c<__level_idx - 1, typename _LevelDescs::level_type...>;
|
||||
};
|
||||
|
||||
template <class _Level, class... _LevelDescs>
|
||||
struct __next_hierarchy_level<_Level, hierarchy<_Level, _LevelDescs...>>
|
||||
{
|
||||
using __type = ::cuda::std::__type_index_c<(sizeof...(_LevelDescs) - 1), typename _LevelDescs::level_type...>;
|
||||
};
|
||||
|
||||
template <class _Level, class _Hierarchy>
|
||||
using __next_hierarchy_level_t = typename __next_hierarchy_level<_Level, _Hierarchy>::__type;
|
||||
|
||||
template <class _Type>
|
||||
_CCCL_CONCEPT_FRAGMENT(__has_hierarchy_member_,
|
||||
requires(const _Type& __instance)(requires(
|
||||
::cuda::__is_hierarchy_v<::cuda::std::remove_cvref_t<decltype(__instance.hierarchy())>>)));
|
||||
template <class _Type>
|
||||
_CCCL_CONCEPT __has_hierarchy_member = _CCCL_FRAGMENT(__has_hierarchy_member_, _Type);
|
||||
|
||||
template <class _Type>
|
||||
inline constexpr bool __is_or_has_hierarchy_member_v = __has_hierarchy_member<_Type> || __is_hierarchy_v<_Type>;
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___HIERARCHY_TRAITS_H
|
||||
Reference in New Issue
Block a user