[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,90 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H
#define _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
# include <cuda/__hierarchy/hierarchy_levels.h>
# include <cuda/std/tuple>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
/**
* @brief Returns a tuple of dim3 compatible objects that can be used to launch
* a kernel
*
* This function returns a tuple of hierarchy_query_result objects that contain
* dimensions from the supplied hierarchy, that can be used to launch that
* hierarchy. It is meant to allow for easy usage of hierarchy dimensions with
* the <<<>>> launch syntax or cudaLaunchKernelEx in case of a cluster launch.
* Contained hierarchy_query_result objects are results of extents() member
* function on the hierarchy passed in. The returned tuple has three elements if
* cluster_level is present in the hierarchy (extents(block, grid),
* extents(cluster, block), extents(thread, block)). Otherwise it contains only
* two elements, without the middle one related to the cluster.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* using namespace cuda;
*
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
* block_dims<8, 8, 8>()); auto [grid_dimensions, cluster_dimensions,
* block_dimensions] = get_launch_dimensions(hierarchy);
* assert(grid_dimensions.x == 256);
* assert(cluster_dimensions.x == 4);
* assert(block_dimensions.x == 8);
* assert(block_dimensions.y == 8);
* assert(block_dimensions.z == 8);
* @endcode
* @par
*
* @param __hierarchy
* Hierarchy that the launch dimensions are requested for
*/
template <class _BottomLevel, class... _LevelDescs>
[[nodiscard]] _CCCL_HOST_API constexpr auto
get_launch_dimensions(const hierarchy<_BottomLevel, _LevelDescs...>& __hierarchy)
{
if constexpr (hierarchy<_BottomLevel, _LevelDescs...>::has_level(cluster))
{
return ::cuda::std::make_tuple(
::dim3{block.dims(grid, __hierarchy)},
::dim3{block.dims(cluster, __hierarchy)},
::dim3{gpu_thread.dims(block, __hierarchy)});
}
else
{
return ::cuda::std::make_tuple(::dim3{block.dims(grid, __hierarchy)}, ::dim3{gpu_thread.dims(block, __hierarchy)});
}
}
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
#endif // _CUDA___HIERARCHY_GET_LAUNCH_DIMENSIONS_H

View File

@@ -0,0 +1,543 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H
#define _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/level_dimensions.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/__type_traits/type_list.h>
# include <cuda/std/__utility/integer_sequence.h>
# include <cuda/std/tuple>
# include <nv/target>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
namespace __detail
{
template <typename _Level>
[[nodiscard]] _CCCL_API constexpr auto __as_level(_Level __lvl) noexcept -> _Level
{
return __lvl;
}
template <typename _LevelFn>
[[nodiscard]] _CCCL_API constexpr auto __as_level(_LevelFn* __fn) noexcept -> decltype(__fn())
{
return {};
}
} // namespace __detail
namespace __detail
{
template <class... _Levels>
struct __can_stack_checker
{
template <class... _LevelsShifted>
static constexpr bool __can_stack = (__detail::__can_rhs_stack_on_lhs<_LevelsShifted, _Levels> && ...);
};
template <class _LUnit, class _L1, class... _Levels>
inline constexpr bool __can_stack =
__can_stack_checker<__level_type_of<_L1>,
__level_type_of<_Levels>...>::template __can_stack<__level_type_of<_Levels>..., _LUnit>;
template <::cuda::std::size_t... _Id>
_CCCL_API constexpr auto __reverse_indices(::cuda::std::index_sequence<_Id...>) noexcept
{
return ::cuda::std::index_sequence<(sizeof...(_Id) - 1 - _Id)...>();
}
template <class _LUnit, bool _Reversed = false>
struct __make_hierarchy
{
template <class _Levels, ::cuda::std::size_t... _Ids>
[[nodiscard]] _CCCL_NODEBUG_API static constexpr auto
__apply_reverse(const _Levels& __ls, ::cuda::std::index_sequence<_Ids...>) noexcept
{
return __make_hierarchy<_LUnit, true>()(::cuda::std::get<_Ids>(__ls)...);
}
template <class... _Levels2>
[[nodiscard]] _CCCL_API constexpr auto operator()(const _Levels2&... __ls) const noexcept
{
using _UnitOrDefault = ::cuda::std::conditional_t<
::cuda::std::is_same_v<void, _LUnit>,
__default_unit_below<::cuda::std::__type_index_c<sizeof...(_Levels2) - 1, __level_type_of<_Levels2>...>>,
_LUnit>;
if constexpr (__can_stack<_UnitOrDefault, _Levels2...>)
{
return hierarchy(_UnitOrDefault{}, __ls...);
}
else if constexpr (!_Reversed)
{
return __apply_reverse(::cuda::std::tie(__ls...),
__reverse_indices(::cuda::std::index_sequence_for<_Levels2...>()));
}
else
{
static_assert(__can_stack<_UnitOrDefault, _Levels2...>,
"Provided levels can't create a valid hierarchy when "
"stacked in the provided order or reversed");
_CCCL_UNREACHABLE();
}
}
};
template <class _LUnit>
[[nodiscard]] _CCCL_API constexpr auto __get_levels_range_end() noexcept
{
return ::cuda::std::make_tuple();
}
// Find LUnit in Levels... and discard the rest
// maybe_unused needed for MSVC
template <class _LUnit, class _LDims, class... _Levels>
[[nodiscard]] _CCCL_API constexpr auto
__get_levels_range_end(const _LDims& __lvl, [[maybe_unused]] const _Levels&... __levels) noexcept
{
if constexpr (::cuda::std::is_same_v<_LUnit, __level_type_of<_LDims>>)
{
return ::cuda::std::make_tuple();
}
else
{
return ::cuda::std::tuple_cat(::cuda::std::tie(__lvl), __get_levels_range_end<_LUnit>(__levels...));
}
}
// Find the LTop in Levels... and discard the preceding ones
template <class _LTop, class _LUnit, class _LTopDims, class... _Levels>
[[nodiscard]] _CCCL_API constexpr auto
__get_levels_range_start(const _LTopDims& __ltop, const _Levels&... __levels) noexcept
{
if constexpr (::cuda::std::is_same_v<_LTop, __level_type_of<_LTopDims>>)
{
return __get_levels_range_end<_LUnit>(__ltop, __levels...);
}
else
{
return __get_levels_range_start<_LTop, _LUnit>(__levels...);
}
}
// Creates a new hierarchy from Levels... cutting out levels between LTop and
// LUnit
template <class _LTop, class _LUnit, class... _Levels>
[[nodiscard]] _CCCL_API constexpr auto __get_levels_range(const _Levels&... __levels) noexcept
{
return __get_levels_range_start<_LTop, _LUnit>(__levels...);
}
} // namespace __detail
// Artificial empty hierarchy to make it possible for the config type to be
// empty, seems easier than checking everywhere in hierarchy APIs if its not
// empty. Any usage of an empty hierarchy other than combine should lead to an
// error anyway
struct __empty_hierarchy
{
template <class _Other>
[[nodiscard]] _CCCL_API _Other combine(const _Other& __other) const
{
return __other;
}
};
/**
* @brief Type representing a hierarchy of CUDA threads
*
* This type combines a number of hierarchy_level_desc objects to represent
* dimensions of a (possibly partial) hierarchy of CUDA threads. It supports
* accessing individual levels or queries combining dimensions of multiple
* levels. This type should not be created directly and make_hierarchy function
* should be used instead. For every level, the unit for its dimensions is
* implied by the next level in the hierarchy, except for the last type, for
* which its the BottomUnit template argument. In case the BottomUnit type is
* thread_level, the hierarchy is considered complete and there exist an alias
* template for it named hierarchy, that only takes the Levels...
* template argument.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* auto hierarchy = make_hierarchy(grid_dims(256), block_dims<8, 8, 8>());
* assert(hierarchy.level(grid).dims.x == 256);
* static_assert(hierarchy.count(thread, block) == 8 * 8 * 8);
* @endcode
* @par
*
* @tparam BottomUnit
* Type indicating what is the unit of the last level in the hierarchy
*
* @tparam Levels
* Template parameter pack with the types of levels in the hierarchy, must be
* hierarchy_level_desc instances or types derived from it
*/
template <class _BottomUnit, class... _LevelDescs>
class hierarchy
{
static_assert(__is_hierarchy_level_v<_BottomUnit>);
static_assert(__detail::__can_stack<_BottomUnit, typename _LevelDescs::level_type...>);
template <class, class...>
friend class hierarchy;
::cuda::std::tuple<_LevelDescs...> __descs_;
// This being static is a bit of a hack to make extents_type working without
// incomplete class member access
template <class _Unit, class _Level>
[[nodiscard]] _CCCL_API static constexpr auto
__levels_range_static(const ::cuda::std::tuple<_LevelDescs...>& __levels) noexcept
{
static_assert(hierarchy::has_level<_Level>());
static_assert(__has_bottom_unit_or_level_v<_Unit, hierarchy<_BottomUnit, _LevelDescs...>>);
static_assert(__detail::__legal_unit_for_level<_Unit, _Level>);
auto __fn = __detail::__get_levels_range<_Level, _Unit, _LevelDescs...>;
return ::cuda::std::apply(__fn, __levels);
}
// TODO is this useful enough to expose?
template <class _Unit, class _Level>
[[nodiscard]] _CCCL_API constexpr auto __levels_range() const noexcept
{
return __levels_range_static<_Unit, _Level>(__descs_);
}
template <class _Unit>
struct __fragment_helper
{
template <class... _Selected>
[[nodiscard]] _CCCL_API constexpr auto operator()(const _Selected&... __levels) const noexcept
{
return hierarchy<_Unit, _Selected...>(__levels...);
}
};
public:
template <class _Level>
static constexpr auto __level_idx =
::cuda::std::__find_exactly_one_t<_Level, typename _LevelDescs::level_type...>::value;
using bottom_unit_type = _BottomUnit;
using top_level_type = ::cuda::std::__type_index_c<0, typename _LevelDescs::level_type...>;
template <class _Level>
using level_desc_type = ::cuda::std::__type_index_c<__level_idx<_Level>, _LevelDescs...>;
template <class _Level>
[[nodiscard]] _CCCL_API static constexpr bool has_level(const _Level& = _Level{}) noexcept
{
return (::cuda::std::is_same_v<_Level, typename _LevelDescs::level_type> || ...);
}
_CCCL_API constexpr hierarchy(const _LevelDescs&... __lds) noexcept
: __descs_(__lds...)
{}
_CCCL_TEMPLATE(class _BottomUnit2 = _BottomUnit)
_CCCL_REQUIRES((!::cuda::std::is_same_v<void, _BottomUnit2>) )
_CCCL_API constexpr hierarchy(const _BottomUnit2&, const _LevelDescs&... __lds) noexcept
: __descs_(__lds...)
{}
_CCCL_API constexpr hierarchy(const ::cuda::std::tuple<_LevelDescs...>& __lds) noexcept
: __descs_(__lds)
{}
_CCCL_TEMPLATE(class _BottomUnit2 = _BottomUnit)
_CCCL_REQUIRES((!::cuda::std::is_same_v<void, _BottomUnit2>) )
_CCCL_API constexpr hierarchy(const _BottomUnit2&, const ::cuda::std::tuple<_LevelDescs...>& __lds) noexcept
: __descs_(__lds)
{}
[[nodiscard]] _CCCL_API friend constexpr bool operator==(const hierarchy& __lhs, const hierarchy& __rhs) noexcept
{
return __lhs.__descs_ == __rhs.__descs_;
}
[[nodiscard]] _CCCL_API friend constexpr bool operator!=(const hierarchy& __lhs, const hierarchy& __rhs) noexcept
{
return __lhs.__descs_ != __rhs.__descs_;
}
/**
* @brief Get a fragment of this hierarchy
*
* This member function can be used to get a fragment of the hierarchy its
* called on. It returns a hierarchy that includes levels starting
* with the level specified in Level and ending with a level before Unit.
* Toegether with hierarchy_add_level function it can be used to create a new
* hierarchy that is a modification of an existing hierarchy.
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
* block_dims<8, 8, 8>()); auto fragment = hierarchy.fragment(block, grid);
* auto new_hierarchy = hierarchy_add_level(fragment, block_dims<128>());
* static_assert(new_hierarchy.count(thread, block) == 128);
* @endcode
* @par
*
* @tparam Unit
* Type indicating what should be the unit of the resulting fragment
*
* @tparam Level
* Type indicating what should be the top most level of the resulting
* fragment
*/
template <typename _Unit, typename _Level>
_CCCL_API constexpr auto fragment(const _Unit& = _Unit(), const _Level& = _Level()) const noexcept
{
auto __selected = __levels_range<_Unit, _Level>();
// TODO fragment can't do constexpr queries because we use references here,
// can we create copies of the levels in some cases and move to the
// constructor?
return ::cuda::std::apply(__fragment_helper<_Unit>(), __selected);
}
/**
* @brief Returns level description associated with a specified hierarchy
* level in this hierarchy.
*
* This function returns a copy of the object associated with the specified
* level, that was passed into the hierarchy on its creation. Level need to be
* levels present in this hierarchy.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* using namespace cuda;
*
* auto hierarchy = make_hierarchy(grid_dims(256), cluster_dims<4>(),
* block_dims<8, 8, 8>());
* static_assert(decltype(hierarchy.level(cluster).dims)::static_extent(0) ==
* 4);
* @endcode
* @par
*
* @tparam Level
* Specifies the requested level
*/
template <typename _Level>
[[nodiscard]] _CCCL_API constexpr const level_desc_type<_Level>& level(const _Level&) const noexcept
{
static_assert(hierarchy::has_level<_Level>());
return ::cuda::std::get<__level_idx<_Level>>(__descs_);
}
//! @brief Returns a new hierarchy with combined levels of this and the other
//! supplied hierarchy
//!
//! This function combines this hierarchy with the supplied hierarchy, the
//! resulting hierarchy holds levels present in both hierarchies. In case of
//! overlap of levels this hierarchy is prioritized, so the result always
//! holds all levels from this hierarchy and non-overlapping levels from the
//! other hierarchy.
//!
//! @param __other The other hierarchy to be combined with this hierarchy
//!
//! @return Hierarchy holding the combined levels from both hierarchies
template <class _OtherUnit, class... _OtherLevels>
[[nodiscard]] _CCCL_API constexpr auto combine(const hierarchy<_OtherUnit, _OtherLevels...>& __other) const
{
using _BottomLevel = __level_type_of<::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, _LevelDescs...>>;
using _OtherHierarchy = hierarchy<_OtherUnit, _OtherLevels...>;
using _OtherTopLevel = typename _OtherHierarchy::top_level_type;
using _OtherBottomLevel =
__level_type_of<::cuda::std::__type_index_c<sizeof...(_OtherLevels) - 1, _OtherLevels...>>;
if constexpr (__detail::__can_rhs_stack_on_lhs<_OtherTopLevel, _BottomLevel>)
{
// Easily stackable case, example this is (grid), other is (cluster,
// block)
return ::cuda::std::apply(__fragment_helper<_OtherUnit>(), ::cuda::std::tuple_cat(__descs_, __other.__descs_));
}
else if constexpr (_OtherHierarchy::template has_level<_BottomLevel>()
&& (!_OtherHierarchy::template has_level<top_level_type>()
|| ::cuda::std::is_same_v<top_level_type, _OtherTopLevel>) )
{
// Overlap with this on the top, e.g. this is (grid, cluster), other is
// (cluster, block), can fully overlap Do we have some CCCL tuple utils
// that can select all but the first?
auto __to_add_with_one_too_many = __other.template __levels_range<_OtherUnit, _BottomLevel>();
auto __to_add = ::cuda::std::apply(
[](auto&&, auto&&... __rest) {
return ::cuda::std::make_tuple(__rest...);
},
__to_add_with_one_too_many);
return ::cuda::std::apply(__fragment_helper<_OtherUnit>(), ::cuda::std::tuple_cat(__descs_, __to_add));
}
else
{
if constexpr (__detail::__can_rhs_stack_on_lhs<top_level_type, _OtherBottomLevel>)
{
// Easily stackable case again, just reversed
return ::cuda::std::apply(__fragment_helper<_BottomUnit>(), ::cuda::std::tuple_cat(__other.__descs_, __descs_));
}
else
{
// Overlap with this on the bottom, e.g. this is (cluster, block), other
// is (grid, cluster), can fully overlap
static_assert(hierarchy::has_level<_OtherBottomLevel>()
&& (!_OtherHierarchy::template has_level<_BottomLevel>()
|| ::cuda::std::is_same_v<_BottomLevel, _OtherBottomLevel>),
"Can't combine the hierarchies");
auto __to_add = __other.template __levels_range<top_level_type, _OtherTopLevel>();
return ::cuda::std::apply(__fragment_helper<_BottomUnit>(), ::cuda::std::tuple_cat(__to_add, __descs_));
}
}
}
# ifndef _CCCL_DOXYGEN_INVOKED // Do not document
[[nodiscard]] _CCCL_API constexpr hierarchy combine([[maybe_unused]] __empty_hierarchy __empty) const
{
return *this;
}
# endif // _CCCL_DOXYGEN_INVOKED
# if !_CCCL_COMPILER(NVRTC)
template <class _NewLevel, class _Unit, class... _LevelDescs2>
friend constexpr auto hierarchy_add_level(const hierarchy<_Unit, _LevelDescs2...>& hierarchy, _NewLevel __lnew);
# endif // !_CCCL_COMPILER(NVRTC)
};
_CCCL_TEMPLATE(class... _LevelDescs)
_CCCL_REQUIRES(::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _LevelDescs&...)
-> hierarchy<__detail::__default_unit_below<
::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, __level_type_of<_LevelDescs>...>>,
_LevelDescs...>;
_CCCL_TEMPLATE(class _BottomUnit, class... _LevelDescs)
_CCCL_REQUIRES(
__is_hierarchy_level_v<_BottomUnit> _CCCL_AND ::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _BottomUnit&, const _LevelDescs&...)
-> hierarchy<_BottomUnit, _LevelDescs...>;
_CCCL_TEMPLATE(class... _LevelDescs)
_CCCL_REQUIRES(::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const ::cuda::std::tuple<_LevelDescs...>&)
-> hierarchy<__detail::__default_unit_below<
::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, __level_type_of<_LevelDescs>...>>,
_LevelDescs...>;
_CCCL_TEMPLATE(class _BottomUnit, class... _LevelDescs)
_CCCL_REQUIRES(
__is_hierarchy_level_v<_BottomUnit> _CCCL_AND ::cuda::std::__fold_and_v<__is_hierarchy_level_desc_v<_LevelDescs>...>)
_CCCL_DEDUCTION_GUIDE_ATTRIBUTES hierarchy(const _BottomUnit&, const ::cuda::std::tuple<_LevelDescs...>&)
-> hierarchy<_BottomUnit, _LevelDescs...>;
# if !_CCCL_COMPILER(NVRTC)
// TODO consider having LUnit optional argument for template argument deduction
/**
* @brief Creates a hierarchy from passed in levels.
*
* This function takes any number of hierarchy_level_desc or derived objects
* and creates a hierarchy out of them. Levels need to be in ascending
* or descending order and the lowest level needs to be valid for thread_level
* unit.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* using namespace cuda;
*
* auto hierarchy1 = make_hierarchy(grid_dims(256), cluster_dims<4>(),
* block_dims<8, 8, 8>()); auto hierarchy2 = make_hierarchy(block_dims<8, 8,
* 8>(), cluster_dims<4>(), grid_dims(256));
* static_assert(cuda::std::is_same_v<decltype(hierarchy1),
* decltype(hierarchy2)>);
* @endcode
* @par
*/
template <class _LUnit = void, class _L1, class... _LevelDescs>
constexpr auto make_hierarchy(_L1 __l1, _LevelDescs... __ls) noexcept
{
return __detail::__make_hierarchy<_LUnit>()(__detail::__as_level(__l1), __detail::__as_level(__ls)...);
}
/**
* @brief Add a level to a hierarchy
*
* This function returns a new hierarchy, that is a copy of the supplied
* hierarchy with the supplied level added to it. This function will examine the
* supplied level and add it either at the top or at the bottom of the
* hierarchy, depending on what levels above and below it are valid for it.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* using namespace cuda;
*
* auto partial1 = make_hierarchy<block_level>(grid_dims(256),
* cluster_dims<4>()); auto hierarchy1 = hierarchy_add_level(partial1,
* block_dims<8, 8, 8>()); auto partial2 =
* make_hierarchy<thread_level>(block_dims<8, 8, 8>(), cluster_dims<4>()); auto
* hierarchy2 = hierarchy_add_level(partial2, grid_dims(256));
* static_assert(cuda::std::is_same_v<decltype(hierarchy1),
* decltype(hierarchy2)>);
* @endcode
* @par
*/
template <class _NewLevel, class _Unit, class... _LevelDescs>
constexpr auto hierarchy_add_level(const hierarchy<_Unit, _LevelDescs...>& __hierarchy, _NewLevel __lnew)
{
auto __new_level = __detail::__as_level(__lnew);
using __added_level = decltype(__new_level);
using __top_level = __level_type_of<::cuda::std::__type_index_c<0, _LevelDescs...>>;
using __bottom_level = __level_type_of<::cuda::std::__type_index_c<sizeof...(_LevelDescs) - 1, _LevelDescs...>>;
if constexpr (__detail::__can_rhs_stack_on_lhs<__top_level, __level_type_of<__added_level>>)
{
return hierarchy<_Unit, __added_level, _LevelDescs...>(
::cuda::std::tuple_cat(::cuda::std::make_tuple(__new_level), __hierarchy.__descs_));
}
else
{
static_assert(__detail::__can_rhs_stack_on_lhs<__level_type_of<__added_level>, __bottom_level>,
"Not supported order of levels in hierarchy");
using __new_unit = __detail::__default_unit_below<__level_type_of<__added_level>>;
return hierarchy<__new_unit, _LevelDescs..., __added_level>(
::cuda::std::tuple_cat(__hierarchy.__descs_, ::cuda::std::make_tuple(__new_level)));
}
}
# endif // !_CCCL_COMPILER(NVRTC)
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_HIERARCHY_DIMENSIONS_H

View File

@@ -0,0 +1,285 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H
#define _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/hierarchy_query_result.h>
# include <cuda/__hierarchy/queries/count.h>
# include <cuda/__hierarchy/queries/extents.h>
# include <cuda/__hierarchy/queries/index.h>
# include <cuda/__hierarchy/queries/rank.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__mdspan/extents.h>
# include <cuda/std/__type_traits/is_integer.h>
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
# include <cuda/experimental/__group/concepts.cuh>
# include <cuda/experimental/__group/fwd.cuh>
# include <cuda/experimental/__group/queries.cuh>
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// Used to either pass-through the hierarchy argument or unpack it from launch configuration
_CCCL_TEMPLATE(class _Type)
_CCCL_REQUIRES(__is_or_has_hierarchy_member_v<_Type>)
[[nodiscard]] _CCCL_API constexpr auto& __unpack_hierarchy_if_needed(const _Type& __instance) noexcept
{
if constexpr (__is_hierarchy_v<_Type>)
{
return __instance;
}
else
{
return __instance.hierarchy();
}
}
template <class _Level>
struct hierarchy_level_base
{
using level_type = _Level;
template <class _InLevel>
using __default_md_query_type = ::cuda::std::uint32_t;
template <class _InLevel>
using __default_1d_query_type = typename _InLevel::__product_type;
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto dims(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return _Level::template dims_as<__default_md_query_type<_InLevel>>(
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto static_dims(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return __static_dims_impl(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto extents(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return _Level::template extents_as<__default_md_query_type<_InLevel>>(
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto static_count(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return __static_count_impl(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto count(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return _Level::template count_as<__default_1d_query_type<_InLevel>>(
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
# if _CCCL_CUDA_COMPILATION()
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto index(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return _Level::template index_as<__default_md_query_type<_InLevel>>(
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(__is_hierarchy_level_v<_InLevel> _CCCL_AND __is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto rank(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return _Level::template rank_as<__default_1d_query_type<_InLevel>>(
__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
# endif // _CCCL_CUDA_COMPILATION()
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
__is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto dims_as(const _InLevel& __level, const _Hierarchy& __hier) noexcept
{
return __dims_as_impl<_Tp>(__level, ::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
__is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto extents_as(const _InLevel&, const _Hierarchy& __hier) noexcept
{
return __extents_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
__is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_API static constexpr auto count_as(const _InLevel&, const _Hierarchy& __hier) noexcept
{
return __count_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
}
# if _CCCL_CUDA_COMPILATION()
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
__is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto index_as(const _InLevel&, const _Hierarchy& __hier) noexcept
{
return __index_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
}
_CCCL_TEMPLATE(class _Tp, class _InLevel, class _Hierarchy)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND __is_hierarchy_level_v<_InLevel> _CCCL_AND
__is_or_has_hierarchy_member_v<_Hierarchy>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto rank_as(const _InLevel&, const _Hierarchy& __hier) noexcept
{
return __rank_query<_Level, _InLevel>::template __call<_Tp>(::cuda::__unpack_hierarchy_if_needed(__hier));
}
# endif // _CCCL_CUDA_COMPILATION()
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
# if _CCCL_CUDA_COMPILATION()
_CCCL_TEMPLATE(class _Group)
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static constexpr ::cuda::std::size_t static_count(const _Group&) noexcept
{
return ::cuda::experimental::__static_count_query_group<_Level, _Group>();
}
_CCCL_TEMPLATE(class _Group)
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static constexpr auto count(const _Group& __group) noexcept
{
return count_as<__default_1d_query_type<typename _Group::unit_type>>(__group);
}
_CCCL_TEMPLATE(class _Group)
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static auto rank(const _Group& __group) noexcept
{
return rank_as<__default_1d_query_type<typename _Group::unit_type>>(__group);
}
_CCCL_TEMPLATE(class _Tp, class _Group)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static constexpr _Tp count_as(const _Group& __group) noexcept
{
return ::cuda::experimental::__count_query_group<_Tp, _Level>(__group);
}
_CCCL_TEMPLATE(class _Tp, class _Group)
_CCCL_REQUIRES(::cuda::std::__cccl_is_integer_v<_Tp> _CCCL_AND ::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static _Tp rank_as(const _Group& __group) noexcept
{
return ::cuda::experimental::__rank_query_group<_Tp, _Level>(__group);
}
_CCCL_TEMPLATE(class _Group)
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr bool is_root_rank(const _Group& __group) noexcept
{
return _Level::rank(__group) == 0;
}
_CCCL_TEMPLATE(class _Group)
_CCCL_REQUIRES(::cuda::experimental::is_group<_Group>)
[[nodiscard]] _CCCL_API static constexpr bool is_part_of(const _Group& __group) noexcept
{
// todo: static_assert that the _Level <= _Group::unit_type
return ::cuda::experimental::__is_part_of_group<_Level>(__group);
}
# endif // _CCCL_CUDA_COMPILATION()
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
private:
template <class>
friend struct __native_hierarchy_level_base;
_CCCL_EXEC_CHECK_DISABLE
template <class _Tp, class... _Args>
[[nodiscard]] _CCCL_API static constexpr auto __dims_as_impl(const _Args&... __args) noexcept
{
auto __exts = _Level::template extents_as<_Tp>(__args...);
using _Exts = decltype(__exts);
hierarchy_query_result<_Tp> __ret{1, 1, 1};
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
{
__ret[__i] = __exts.extent(__i);
}
return __ret;
}
template <class... _Args>
[[nodiscard]] _CCCL_API static constexpr auto __static_dims_impl(const _Args&... __args) noexcept
{
using _Exts = decltype(_Level::extents(__args...));
hierarchy_query_result<::cuda::std::size_t> __ret{1, 1, 1};
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
{
__ret[__i] = _Exts::static_extent(__i);
}
return __ret;
}
template <class... _Args>
[[nodiscard]] _CCCL_API static constexpr auto __static_count_impl(const _Args&... __args) noexcept
{
using _Exts = decltype(_Level::extents(__args...));
if constexpr (_Exts::rank_dynamic() == 0)
{
::cuda::std::size_t __ret{1};
for (::cuda::std::size_t __i = 0; __i < _Exts::rank(); ++__i)
{
__ret *= _Exts::static_extent(__i);
}
return __ret;
}
else
{
return ::cuda::std::dynamic_extent;
}
}
};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_HIERARCHY_LEVEL_BASE_H

View File

@@ -0,0 +1,123 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_HIERARCHY_LEVELS_H
#define _CUDA___HIERARCHY_HIERARCHY_LEVELS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/native_hierarchy_level_base.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/__type_traits/type_list.h>
# include <cuda/std/cstdint>
# include <nv/target>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
struct _CCCL_DECLSPEC_EMPTY_BASES thread_level : __native_hierarchy_level_base<thread_level>
{
using __product_type = ::cuda::std::uint32_t;
using __allowed_above = __allowed_levels<block_level>;
using __allowed_below = __allowed_levels<>;
using __next_native_level = block_level;
};
struct _CCCL_DECLSPEC_EMPTY_BASES warp_level : __native_hierarchy_level_base<warp_level>
{
using __product_type = ::cuda::std::uint32_t;
using __next_native_level = block_level;
};
struct _CCCL_DECLSPEC_EMPTY_BASES block_level : __native_hierarchy_level_base<block_level>
{
using __product_type = ::cuda::std::uint32_t;
using __allowed_above = __allowed_levels<grid_level, cluster_level>;
using __allowed_below = __allowed_levels<thread_level>;
using __next_native_level = cluster_level;
};
struct _CCCL_DECLSPEC_EMPTY_BASES cluster_level : __native_hierarchy_level_base<cluster_level>
{
using __product_type = ::cuda::std::uint32_t;
using __allowed_above = __allowed_levels<grid_level>;
using __allowed_below = __allowed_levels<block_level>;
using __next_native_level = grid_level;
};
struct _CCCL_DECLSPEC_EMPTY_BASES grid_level : __native_hierarchy_level_base<grid_level>
{
using __product_type = ::cuda::std::uint64_t;
using __allowed_above = __allowed_levels<>;
using __allowed_below = __allowed_levels<block_level, cluster_level>;
};
_CCCL_GLOBAL_CONSTANT thread_level gpu_thread;
_CCCL_GLOBAL_CONSTANT warp_level warp;
_CCCL_GLOBAL_CONSTANT block_level block;
_CCCL_GLOBAL_CONSTANT cluster_level cluster;
_CCCL_GLOBAL_CONSTANT grid_level grid;
// Struct to represent levels allowed below or above a certain level,
// used for hierarchy sorting, validation and for hierarchy traversal
template <typename... _Levels>
struct __allowed_levels
{
using __default_unit = ::cuda::std::__type_index_c<0, _Levels..., void>;
};
namespace __detail
{
template <typename LevelType>
using __default_unit_below = typename LevelType::__allowed_below::__default_unit;
template <class _QueryLevel, class _AllowedLevels>
inline constexpr bool __is_level_allowed = false;
template <class _QueryLevel, class... _Levels>
inline constexpr bool __is_level_allowed<_QueryLevel, __allowed_levels<_Levels...>> =
(::cuda::std::is_same_v<_QueryLevel, _Levels> || ...);
template <class _L1, class _L2>
inline constexpr bool __can_rhs_stack_on_lhs =
__is_level_allowed<_L1, typename _L2::__allowed_below> || __is_level_allowed<_L2, typename _L1::__allowed_above>;
template <class _Unit, class _Level>
inline constexpr bool __legal_unit_for_level =
__can_rhs_stack_on_lhs<_Unit, _Level> || __legal_unit_for_level<_Unit, __default_unit_below<_Level>>;
template <class _Unit>
inline constexpr bool __legal_unit_for_level<_Unit, void> = false;
} // namespace __detail
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_HIERARCHY_LEVELS_H

View File

@@ -0,0 +1,156 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H
#define _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <class _Tp>
struct hierarchy_query_result
{
using value_type = _Tp;
_Tp x;
_Tp y;
_Tp z;
[[nodiscard]] _CCCL_API constexpr _Tp& operator[](::cuda::std::size_t __i) noexcept
{
if (__i == 0)
{
return x;
}
else if (__i == 1)
{
return y;
}
else
{
return z;
}
}
[[nodiscard]] _CCCL_API constexpr const _Tp& operator[](::cuda::std::size_t __i) const noexcept
{
if (__i == 0)
{
return x;
}
else if (__i == 1)
{
return y;
}
else
{
return z;
}
}
// Hide SFINAE conversion operators from Doxygen. The _CCCL_TEMPLATE/_CCCL_REQUIRES
// macros expand to enable_if_t expressions that Breathe renders as invalid C++ template
// parameter lists, causing Sphinx parse errors on the generated struct page.
# ifndef _CCCL_DOXYGEN_INVOKED
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, signed char>)
_CCCL_API constexpr operator char3() const noexcept
{
return {static_cast<signed char>(x), static_cast<signed char>(y), static_cast<signed char>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, short>)
_CCCL_API constexpr operator short3() const noexcept
{
return {static_cast<short>(x), static_cast<short>(y), static_cast<short>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, int>)
_CCCL_API constexpr operator int3() const noexcept
{
return {static_cast<int>(x), static_cast<int>(y), static_cast<int>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, long>)
_CCCL_API constexpr operator long3() const noexcept
{
return {static_cast<long>(x), static_cast<long>(y), static_cast<long>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, long long>)
_CCCL_API constexpr operator longlong3() const noexcept
{
return {static_cast<long long>(x), static_cast<long long>(y), static_cast<long long>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned char>)
_CCCL_API constexpr operator uchar3() const noexcept
{
return {static_cast<unsigned char>(x), static_cast<unsigned char>(y), static_cast<unsigned char>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned short>)
_CCCL_API constexpr operator ushort3() const noexcept
{
return {static_cast<unsigned short>(x), static_cast<unsigned short>(y), static_cast<unsigned short>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned>)
_CCCL_API constexpr operator uint3() const noexcept
{
return {static_cast<unsigned>(x), static_cast<unsigned>(y), static_cast<unsigned>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned long>)
_CCCL_API constexpr operator ulong3() const noexcept
{
return {static_cast<unsigned long>(x), static_cast<unsigned long>(y), static_cast<unsigned long>(z)};
}
_CCCL_TEMPLATE(class _Tp2 = _Tp)
_CCCL_REQUIRES(::cuda::std::is_same_v<_Tp2, unsigned long long>)
_CCCL_API constexpr operator ulonglong3() const noexcept
{
return {static_cast<unsigned long long>(x), static_cast<unsigned long long>(y), static_cast<unsigned long long>(z)};
}
# endif // !_CCCL_DOXYGEN_INVOKED
};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_HIERARCHY_QUERY_RESULT_H

View File

@@ -0,0 +1,240 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H
#define _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/hierarchy_levels.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__mdspan/extents.h>
# include <cuda/std/__type_traits/is_integer.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
namespace __detail
{
/* Keeping it around in case issues like
https://github.com/NVIDIA/cccl/issues/522 template <typename T, size_t...
Extents> struct extents_corrected : public ::cuda::std::extents<T, Extents...> {
using ::cuda::std::extents<T, Extents...>::extents;
template <typename ::cuda::std::extents<T, Extents...>::rank_type Id>
_CCCL_API constexpr auto extent_corrected() const {
if constexpr (::cuda::std::extents<T, Extents...>::static_extent(Id) !=
::cuda::std::dynamic_extent) { return this->static_extent(Id);
}
else {
return this->extent(Id);
}
}
};
*/
template <class _Dims>
struct __dimensions_handler
{
static constexpr bool __is_type_supported = ::cuda::std::__cccl_is_integer_v<_Dims>;
[[nodiscard]] _CCCL_API static constexpr auto __translate(const _Dims& __dims) noexcept
{
return ::cuda::std::extents<dimensions_index_type, ::cuda::std::dynamic_extent, 1, 1>(
static_cast<unsigned>(__dims));
}
};
template <>
struct __dimensions_handler<::dim3>
{
static constexpr bool __is_type_supported = true;
[[nodiscard]] _CCCL_API static constexpr auto __translate(const ::dim3& __dims) noexcept
{
return ::cuda::std::dims<3, dimensions_index_type>(__dims.x, __dims.y, __dims.z);
}
};
template <class _Dims, _Dims _Val>
struct __dimensions_handler<::cuda::std::integral_constant<_Dims, _Val>>
{
static constexpr bool __is_type_supported = ::cuda::std::__cccl_is_integer_v<_Dims>;
[[nodiscard]] _CCCL_API static constexpr auto __translate(const _Dims& __dims) noexcept
{
return ::cuda::std::extents<dimensions_index_type, static_cast<::cuda::std::size_t>(__dims), 1, 1>();
}
};
} // namespace __detail
/**
* @brief Type representing dimensions of a level in a thread hierarchy.
*
* This type combines a level type like grid_level or block_level with
* a cuda::std::extents object to describe dimensions of a level in a thread
* hierarchy. This type is not intended to be created explicitly and *_dims
* functions creating them should be used instead. They will translate the input
* arguments to a correct cuda::std::extents to be stored inside
* hierarchy_level_desc.
* While this type can be used to access the stored dimensions,
* the main usage is to pass a number of hierarchy_level_desc objects
* to make_hierarchy function in order to create a hierarchy.
* This type does not store what the unit is for the stored dimensions,
* it is instead implied by the level below it in a hierarchy object.
* In case there is a need to store more information about a specific level,
* for example some library-specific information, this type can be derived
* from and the resulting type can be used to build the hierarchy.
*
* @par Snippet
* @code
* #include <cudax/hierarchy_dimensions.cuh>
*
* auto hierarchy = make_hierarchy(grid_dims(256), block_dims<8, 8, 8>());
* assert(hierarchy.level(grid).dims.x == 256);
* @endcode
* @par
*
* @tparam Level
* Type indicating which hierarchy level this is
*
* @tparam Dimensions
* Type holding the dimensions of this level
*/
template <class _Level, class _Exts>
class _CCCL_DECLSPEC_EMPTY_BASES hierarchy_level_desc : __hierarchy_level_desc_base
{
static_assert(__is_hierarchy_level_v<_Level>);
static_assert(::cuda::std::__is_cuda_std_extents_v<_Exts>);
// Needs alignas to work around an issue with tuple
alignas(16) _Exts __exts_; // Unit for dimensions is implicit
public:
using level_type = _Level;
using extents_type = _Exts;
_CCCL_HIDE_FROM_ABI constexpr hierarchy_level_desc() noexcept = default;
_CCCL_API constexpr hierarchy_level_desc(const _Exts& __exts) noexcept
: __exts_(__exts)
{}
[[nodiscard]] _CCCL_API constexpr _Exts extents() const noexcept
{
return __exts_;
}
[[nodiscard]] _CCCL_API friend constexpr bool
operator==(const hierarchy_level_desc& __lhs, const hierarchy_level_desc& __rhs) noexcept
{
return __lhs.__exts_ == __rhs.__exts_;
}
[[nodiscard]] _CCCL_API friend constexpr bool
operator!=(const hierarchy_level_desc& __lhs, const hierarchy_level_desc& __rhs) noexcept
{
return __lhs.__exts_ != __rhs.__exts_;
}
};
/**
* @brief Creates an instance of hierarchy_level_desc describing grid_level
*
* This function creates a statically sized level from up to three template
* arguments.
*/
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
[[nodiscard]] _CCCL_API constexpr auto grid_dims() noexcept
{
return hierarchy_level_desc<grid_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
}
/**
* @brief Creates an instance of hierarchy_level_desc describing grid_level
*
* This function creates the level from an integral or dim3 argument.
*/
template <class _Dims>
[[nodiscard]] _CCCL_API constexpr auto grid_dims(_Dims __dims) noexcept
{
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
return hierarchy_level_desc<grid_level, decltype(__translated_dims)>(__translated_dims);
}
/**
* @brief Creates an instance of hierarchy_level_desc describing cluster_level
*
* This function creates a statically sized level from up to three template
* arguments.
*/
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
[[nodiscard]] _CCCL_API constexpr auto cluster_dims() noexcept
{
return hierarchy_level_desc<cluster_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
}
/**
* @brief Creates an instance of hierarchy_level_desc describing cluster_level
*
* This function creates the level from an integral or dim3 argument.
*/
template <class _Dims>
[[nodiscard]] _CCCL_API constexpr auto cluster_dims(_Dims __dims) noexcept
{
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
return hierarchy_level_desc<cluster_level, decltype(__translated_dims)>(__translated_dims);
}
/**
* @brief Creates an instance of hierarchy_level_desc describing block_level
*
* This function creates a statically sized level from up to three template
* arguments.
*/
template <::cuda::std::size_t _XDim, ::cuda::std::size_t _YDim = 1, ::cuda::std::size_t _ZDim = 1>
[[nodiscard]] _CCCL_API constexpr auto block_dims() noexcept
{
return hierarchy_level_desc<block_level, ::cuda::std::extents<dimensions_index_type, _XDim, _YDim, _ZDim>>();
}
/**
* @brief Creates an instance of hierarchy_level_desc describing block_level
*
* This function creates the level from an integral or dim3 argument.
*/
template <class _Dims>
[[nodiscard]] _CCCL_API constexpr auto block_dims(_Dims __dims) noexcept
{
static_assert(__detail::__dimensions_handler<_Dims>::__is_type_supported);
auto __translated_dims = __detail::__dimensions_handler<_Dims>::__translate(__dims);
return hierarchy_level_desc<block_level, decltype(__translated_dims)>(__translated_dims);
}
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_LEVEL_DIMENSIONS_H

View File

@@ -0,0 +1,175 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H
#define _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/hierarchy_level_base.h>
# include <cuda/__hierarchy/hierarchy_query_result.h>
# include <cuda/__hierarchy/queries/count.h>
# include <cuda/__hierarchy/queries/extents.h>
# include <cuda/__hierarchy/queries/index.h>
# include <cuda/__hierarchy/queries/rank.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__mdspan/extents.h>
# include <cuda/std/__type_traits/is_integer.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
// to warn about applying [[nodiscard]] to a function that returns void.
_CCCL_DIAG_PUSH
# if _CCCL_CUDA_COMPILER(NVCC)
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
# endif // _CCCL_CUDA_COMPILER(NVCC)
template <class _Level>
struct _CCCL_DECLSPEC_EMPTY_BASES __native_hierarchy_level_base : hierarchy_level_base<_Level>
{
using __base_type = hierarchy_level_base<_Level>;
using __base_type::count;
using __base_type::count_as;
using __base_type::dims;
using __base_type::dims_as;
using __base_type::extents;
using __base_type::extents_as;
using __base_type::static_count;
using __base_type::static_dims;
# if _CCCL_CUDA_COMPILATION()
using __base_type::index;
using __base_type::index_as;
using __base_type::rank;
using __base_type::rank_as;
# if defined(_CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX)
using __base_type::is_part_of;
using __base_type::is_root_rank;
# endif // _CUDAX_ENABLE_GROUP_FEATURES_IN_LIBCUDACXX
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto dims(const _InLevel& __level) noexcept
{
return _Level::template dims_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto static_dims(const _InLevel& __level) noexcept
{
return __base_type::__static_dims_impl(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto extents(const _InLevel& __level) noexcept
{
return _Level::template extents_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static constexpr auto static_count(const _InLevel& __level) noexcept
{
return __base_type::__static_count_impl(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto count(const _InLevel& __level) noexcept
{
return _Level::template count_as<typename __base_type::template __default_1d_query_type<_InLevel>>(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto index(const _InLevel& __level) noexcept
{
return _Level::template index_as<typename __base_type::template __default_md_query_type<_InLevel>>(__level);
}
_CCCL_TEMPLATE(class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto rank(const _InLevel& __level) noexcept
{
return _Level::template rank_as<typename __base_type::template __default_1d_query_type<_InLevel>>(__level);
}
_CCCL_TEMPLATE(class _Tp, class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto dims_as(const _InLevel& __level) noexcept
{
return __base_type::template __dims_as_impl<_Tp>(__level);
}
_CCCL_TEMPLATE(class _Tp, class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto extents_as(const _InLevel&) noexcept
{
return __extents_query_native<_Level, _InLevel>::template __call<_Tp>();
}
_CCCL_TEMPLATE(class _Tp, class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto count_as(const _InLevel&) noexcept
{
return __count_query_native<_Level, _InLevel>::template __call<_Tp>();
}
_CCCL_TEMPLATE(class _Tp, class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static auto index_as(const _InLevel&) noexcept
{
return __index_query_native<_Level, _InLevel>::template __call<_Tp>();
}
_CCCL_TEMPLATE(class _Tp, class _InLevel)
_CCCL_REQUIRES(__is_native_hierarchy_level_v<_InLevel>)
[[nodiscard]] _CCCL_DEVICE_API static _Tp rank_as(const _InLevel&) noexcept
{
return __rank_query_native<_Level, _InLevel>::template __call<_Tp>();
}
# endif // _CCCL_CUDA_COMPILATION()
};
_CCCL_DIAG_POP
template <>
struct __native_hierarchy_level_base<grid_level> : hierarchy_level_base<grid_level>
{};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_NATIVE_HIERARCHY_LEVEL_BASE_H

View File

@@ -0,0 +1,113 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_QUERIES_COUNT_H
#define _CUDA___HIERARCHY_QUERIES_COUNT_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// native hierarchy queries
# if _CCCL_CUDA_COMPILATION()
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
// to warn about applying [[nodiscard]] to a function that returns void.
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
template <class _Unit, class _Level>
struct __count_query_native
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
const auto __exts = __extents_query_native<_Unit, _Level>::template __call<_Tp>();
_Tp __ret = 1;
for (::cuda::std::size_t __i = 0; __i < __exts.rank(); ++__i)
{
__ret *= __exts.extent(__i);
}
return __ret;
}
};
template <>
struct __count_query_native<block_level, cluster_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
unsigned __count = 1;
NV_IF_TARGET(NV_PROVIDES_SM_90, (__count = ::__clusterSizeInBlocks();))
return static_cast<_Tp>(__count);
}
};
template <>
struct __count_query_native<block_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
return static_cast<_Tp>(static_cast<_Tp>(gridDim.x) * gridDim.y * gridDim.z);
}
};
_CCCL_DIAG_POP
# endif // _CCCL_CUDA_COMPILATION()
// hierarchy queries
template <class _Unit, class _Level>
struct __count_query
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr _Tp __call(const _Hierarchy& __hier) noexcept
{
const auto __exts = __extents_query<_Unit, _Level>::template __call<_Tp>(__hier);
_Tp __ret = 1;
for (::cuda::std::size_t __i = 0; __i < __exts.rank(); ++__i)
{
__ret *= __exts.extent(__i);
}
return __ret;
}
};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_QUERIES_COUNT_H

View File

@@ -0,0 +1,359 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_QUERIES_EXTENTS_H
#define _CUDA___HIERARCHY_QUERIES_EXTENTS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__cmath/ceil_div.h>
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__algorithm/max.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__mdspan/extents.h>
# include <cuda/std/__utility/integer_sequence.h>
# include <cuda/std/array>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// helpers
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL ::cuda::std::size_t
__hierarchy_static_extents_mul_helper(::cuda::std::size_t __lhs, ::cuda::std::size_t __rhs) noexcept
{
if (__lhs == ::cuda::std::dynamic_extent || __rhs == ::cuda::std::dynamic_extent)
{
return ::cuda::std::dynamic_extent;
}
else
{
return __lhs * __rhs;
}
}
template <class _ResultIndex, class _LhsExts, class _RhsExts, ::cuda::std::size_t... _Is>
[[nodiscard]] _CCCL_API constexpr auto __hierarchy_static_extents_mul(::cuda::std::index_sequence<_Is...>) noexcept
{
return ::cuda::std::extents<
_ResultIndex,
::cuda::__hierarchy_static_extents_mul_helper((_Is < _LhsExts::rank()) ? _LhsExts::static_extent(_Is) : 1,
(_Is < _RhsExts::rank()) ? _RhsExts::static_extent(_Is) : 1)...>{};
}
//! @brief Multiplies 2 extents in column major order together, returning a new extents type. If the ranks don't match,
//! the extent with lower rank is padded with 1s on the right to match the rank of the other.
//!
//! @param __lhs The left hand side extents to multiply.
//! @param __rhs The right hand side extents to multiply.
//!
//! @return The result of multiplying the extents together.
template <class _Index, ::cuda::std::size_t... _LhsExts, ::cuda::std::size_t... _RhsExts>
[[nodiscard]] _CCCL_API constexpr auto
__hierarchy_extents_mul(const ::cuda::std::extents<_Index, _LhsExts...>& __lhs,
const ::cuda::std::extents<_Index, _RhsExts...>& __rhs) noexcept
{
using _Lhs = ::cuda::std::extents<_Index, _LhsExts...>;
using _Rhs = ::cuda::std::extents<_Index, _RhsExts...>;
constexpr auto __rank = ::cuda::std::max(_Lhs::rank(), _Rhs::rank());
using _Ret =
decltype(::cuda::__hierarchy_static_extents_mul<_Index, _Lhs, _Rhs>(::cuda::std::make_index_sequence<__rank>{}));
::cuda::std::array<_Index, __rank> __ret{};
for (::cuda::std::size_t __i = 0; __i < __rank; ++__i)
{
if (_Ret::static_extent(__i) == ::cuda::std::dynamic_extent)
{
__ret[__i] = static_cast<_Index>((__i < _Lhs::rank()) ? __lhs.extent(__i) : 1)
* static_cast<_Index>((__i < _Rhs::rank()) ? __rhs.extent(__i) : 1);
}
else
{
__ret[__i] = static_cast<_Index>(_Ret::static_extent(__i));
}
}
return _Ret{__ret};
}
template <class _Index, class _OrgIndex, ::cuda::std::size_t... _StaticExts>
[[nodiscard]] _CCCL_API constexpr ::cuda::std::extents<_Index, _StaticExts...>
__hierarchy_extents_cast(::cuda::std::extents<_OrgIndex, _StaticExts...> __org_exts) noexcept
{
using _OrgExts = ::cuda::std::extents<_OrgIndex, _StaticExts...>;
::cuda::std::array<_Index, _OrgExts::rank()> __ret{};
for (::cuda::std::size_t __i = 0; __i < _OrgExts::rank(); ++__i)
{
if (_OrgExts::static_extent(__i) == ::cuda::std::dynamic_extent)
{
__ret[__i] = static_cast<_Index>(__org_exts.extent(__i));
}
else
{
__ret[__i] = static_cast<_Index>(_OrgExts::static_extent(__i));
}
}
return ::cuda::std::extents<_Index, _StaticExts...>{__ret};
}
template <class _Tp, class _Unit, class _Level, class _Hierarchy>
[[nodiscard]] _CCCL_API constexpr auto __extents_query_generic(const _Hierarchy& __hier) noexcept
{
static_assert(__has_bottom_unit_or_level_v<_Unit, _Hierarchy> || __is_native_hierarchy_level_v<_Unit>,
"_Hierarchy doesn't contain _Unit");
static_assert(_Hierarchy::has_level(_Level{}) || __is_native_hierarchy_level_v<_Level>,
"_Hierarchy doesn't contain _Level");
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
using _CurrExts = decltype(::cuda::__hierarchy_extents_cast<_Tp>(__hier.level(_NextLevel{}).extents()));
// Remove dependency on runtime storage. This makes the queries work for hierarchy levels with all static extents
// in constant evaluated context.
_CurrExts __curr_exts{};
if constexpr (_CurrExts::rank_dynamic() > 0)
{
__curr_exts = ::cuda::__hierarchy_extents_cast<_Tp>(__hier.level(_NextLevel{}).extents());
}
if constexpr (!::cuda::std::is_same_v<_NextLevel, _Level>)
{
const auto __next_exts = __extents_query<_NextLevel, _Level>::template __call<_Tp>(__hier);
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
}
else
{
return __curr_exts;
}
}
// native hierarchy queries
# if _CCCL_CUDA_COMPILATION()
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
// to warn about applying [[nodiscard]] to a function that returns void.
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
template <class _Unit, class _Level>
struct __extents_query_native
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static auto __call() noexcept
{
static_assert(__is_natively_reachable_hierarchy_level_v<_Unit, _Level>, "_Level must be reachable from _Unit");
using _NextLevel = typename _Unit::__next_native_level;
const auto __next_exts = __extents_query_native<_NextLevel, _Level>::template __call<_Tp>();
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
}
};
template <>
struct __extents_query_native<thread_level, warp_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::extents<_Tp, 32> __call() noexcept
{
return {};
}
};
template <>
struct __extents_query_native<thread_level, block_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
{
return ::cuda::std::dims<3, _Tp>{
static_cast<_Tp>(blockDim.x), static_cast<_Tp>(blockDim.y), static_cast<_Tp>(blockDim.z)};
}
};
template <>
struct __extents_query_native<warp_level, block_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<1, _Tp> __call() noexcept
{
const auto __thread_count = blockDim.x * blockDim.y * blockDim.z;
return ::cuda::std::dims<1, _Tp>{static_cast<_Tp>(::cuda::ceil_div(__thread_count, 32))};
}
};
template <>
struct __extents_query_native<block_level, cluster_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
{
::dim3 __dims{1u, 1u, 1u};
NV_IF_TARGET(NV_PROVIDES_SM_90, (__dims = ::__clusterDim();))
return ::cuda::std::dims<3, _Tp>{static_cast<_Tp>(__dims.x), static_cast<_Tp>(__dims.y), static_cast<_Tp>(__dims.z)};
}
};
template <>
struct __extents_query_native<block_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
{
return ::cuda::std::dims<3, _Tp>{
static_cast<_Tp>(gridDim.x), static_cast<_Tp>(gridDim.y), static_cast<_Tp>(gridDim.z)};
}
};
template <>
struct __extents_query_native<cluster_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static ::cuda::std::dims<3, _Tp> __call() noexcept
{
::dim3 __dims{gridDim};
NV_IF_TARGET(NV_PROVIDES_SM_90, (__dims = ::__clusterGridDimInClusters();))
return ::cuda::std::dims<3, _Tp>{static_cast<_Tp>(__dims.x), static_cast<_Tp>(__dims.y), static_cast<_Tp>(__dims.z)};
}
};
_CCCL_DIAG_POP
# endif // _CCCL_CUDA_COMPILATION()
// hierarchy queries
template <class _Unit, class _Level>
struct __extents_query
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
{
return ::cuda::__extents_query_generic<_Tp, _Unit, _Level>(__hier);
}
};
template <>
struct __extents_query<thread_level, warp_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr ::cuda::std::extents<_Tp, 32> __call(const _Hierarchy&) noexcept
{
static_assert(__has_bottom_unit_or_level_v<thread_level, _Hierarchy>, "_Hierarchy doesn't contain thread_level");
static_assert(_Hierarchy::template has_level<block_level>(), "_Hierarchy doesn't contain block_level");
return {};
}
};
template <class _Level>
struct __extents_query<warp_level, _Level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
{
auto __block_exts = __extents_query<thread_level, block_level>::template __call<_Tp>(__hier);
using _BlockExts = decltype(__block_exts);
if constexpr (_BlockExts::rank_dynamic() == 0)
{
constexpr auto __static_thread_count =
_BlockExts::static_extent(0) * _BlockExts::static_extent(1) * _BlockExts::static_extent(2);
static_assert(__static_thread_count >= 32, "_Hierarchy doesn't contain enough threads to fill a single warp");
constexpr auto __static_warp_count = ::cuda::ceil_div(__static_thread_count, 32);
::cuda::std::extents<_Tp, __static_warp_count> __curr_exts{};
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
{
return __curr_exts;
}
else
{
const auto __next_exts = __extents_query<block_level, _Level>::template __call<_Tp>(__hier);
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
}
}
else
{
const auto __thread_count = __block_exts.extent(0) * __block_exts.extent(1) * __block_exts.extent(2);
_CCCL_ASSERT(__thread_count >= 32, "_Hierarchy doesn't contain enough threads to fill a single warp");
const auto __warp_count = static_cast<_Tp>(::cuda::ceil_div(__thread_count, 32));
::cuda::std::dims<1, _Tp> __curr_exts{__warp_count};
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
{
return __curr_exts;
}
else
{
const auto __next_exts = __extents_query<block_level, _Level>::template __call<_Tp>(__hier);
return ::cuda::__hierarchy_extents_mul(__curr_exts, __next_exts);
}
}
}
};
template <>
struct __extents_query<block_level, cluster_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
{
if constexpr (_Hierarchy::template has_level<cluster_level>())
{
return ::cuda::__extents_query_generic<_Tp, block_level, cluster_level>(__hier);
}
else
{
static_assert(__has_bottom_unit_or_level_v<block_level, _Hierarchy>, "_Hierarchy doesn't contain block_level");
static_assert(_Hierarchy::template has_level<grid_level>(), "_Hierarchy doesn't contain grid_level");
return ::cuda::std::extents<_Tp, 1, 1, 1>{};
}
}
};
template <>
struct __extents_query<cluster_level, grid_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_API static constexpr auto __call(const _Hierarchy& __hier) noexcept
{
if constexpr (_Hierarchy::template has_level<cluster_level>())
{
return ::cuda::__extents_query_generic<_Tp, cluster_level, grid_level>(__hier);
}
else
{
return __extents_query<block_level, grid_level>::template __call<_Tp>(__hier);
}
}
};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_QUERIES_EXTENTS_H

View File

@@ -0,0 +1,268 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_QUERIES_INDEX_H
#define _CUDA___HIERARCHY_QUERIES_INDEX_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__cmath/ceil_div.h>
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/hierarchy_query_result.h>
# include <cuda/__hierarchy/queries/extents.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__cstddef/types.h>
# include <cuda/std/__mdspan/extents.h>
# if _CCCL_CUDA_COMPILATION()
# include <cuda/__ptx/instructions/get_sreg.h>
# endif // _CCCL_CUDA_COMPILATION()
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
# if _CCCL_CUDA_COMPILATION()
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
// to warn about applying [[nodiscard]] to a function that returns void.
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
// native hierarchy queries
template <class _Unit, class _Level>
struct __index_query_native
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
static_assert(__is_natively_reachable_hierarchy_level_v<_Unit, _Level>, "_Level must be reachable from _Unit");
using _NextLevel = typename _Unit::__next_native_level;
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
const auto __next_idx = __index_query_native<_NextLevel, _Level>::template __call<_Tp>();
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
hierarchy_query_result<_Tp> __ret{};
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
{
__ret[__i] = __curr_idx[__i] + ((__i < __curr_exts.rank()) ? __curr_exts.extent(__i) : 1) * __next_idx[__i];
}
return __ret;
}
};
template <>
struct __index_query_native<thread_level, warp_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
// todo(dabayer): Is it worth using cuda::ptx::get_sreg_laneid() here? Doesn't it prevent some other optimizations
// due to using inline ptx?
return {static_cast<_Tp>(::cuda::ptx::get_sreg_laneid()), 0, 0};
}
};
template <>
struct __index_query_native<thread_level, block_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
return {static_cast<_Tp>(threadIdx.x), static_cast<_Tp>(threadIdx.y), static_cast<_Tp>(threadIdx.z)};
}
};
template <>
struct __index_query_native<warp_level, block_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
const auto __thread_rank = (threadIdx.z * blockDim.y + threadIdx.y) * blockDim.x + threadIdx.x;
return {static_cast<_Tp>(__thread_rank / 32), 0, 0};
}
};
template <>
struct __index_query_native<block_level, cluster_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
::dim3 __idx{0u, 0u, 0u};
NV_IF_TARGET(NV_PROVIDES_SM_90, (__idx = ::__clusterRelativeBlockIdx();))
return {static_cast<_Tp>(__idx.x), static_cast<_Tp>(__idx.y), static_cast<_Tp>(__idx.z)};
}
};
template <>
struct __index_query_native<block_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
return {static_cast<_Tp>(blockIdx.x), static_cast<_Tp>(blockIdx.y), static_cast<_Tp>(blockIdx.z)};
}
};
template <>
struct __index_query_native<cluster_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call() noexcept
{
::dim3 __idx{blockIdx};
NV_IF_TARGET(NV_PROVIDES_SM_90, (__idx = ::__clusterIdx();))
return {static_cast<_Tp>(__idx.x), static_cast<_Tp>(__idx.y), static_cast<_Tp>(__idx.z)};
}
};
// hierarchy queries
template <class _Tp, class _Unit, class _NextLevel, class _Level, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API hierarchy_query_result<_Tp> __index_query_generic(const _Hierarchy& __hier) noexcept
{
if constexpr (::cuda::std::is_same_v<_Level, _NextLevel>)
{
using _CurrExts = decltype(__extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier));
auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
{
if (__i >= _CurrExts::rank() || _CurrExts::static_extent(__i) == 1)
{
__curr_idx[__i] = 0;
}
}
return __curr_idx;
}
else
{
const auto __curr_exts = __extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
const auto __next_idx = __index_query<_NextLevel, _Level>::template __call<_Tp>(__hier);
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
hierarchy_query_result<_Tp> __ret{};
for (::cuda::std::size_t __i = 0; __i < 3; ++__i)
{
__ret[__i] = __curr_idx[__i] + ((__i < __curr_exts.rank()) ? __curr_exts.extent(__i) : 1) * __next_idx[__i];
}
return __ret;
}
}
template <class _Unit, class _Level>
struct __index_query
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
{
static_assert(__has_bottom_unit_or_level_v<_Unit, _Hierarchy> || __is_native_hierarchy_level_v<_Unit>,
"_Hierarchy doesn't contain _Unit");
static_assert(_Hierarchy::template has_level<_Level>() || __is_native_hierarchy_level_v<_Level>,
"_Hierarchy doesn't contain _Level");
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
return ::cuda::__index_query_generic<_Tp, _Unit, _NextLevel, _Level>(__hier);
}
};
template <>
struct __index_query<thread_level, warp_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
{
return __index_query_native<thread_level, warp_level>::template __call<_Tp>();
}
};
template <class _Level>
struct __index_query<warp_level, _Level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
{
const auto __block_exts = __extents_query<thread_level, block_level>::template __call<unsigned>(__hier);
const auto __thread_idx = __index_query<thread_level, block_level>::template __call<unsigned>(__hier);
const auto __thread_rank =
(__thread_idx.z * __block_exts.extent(1) + __thread_idx.y) * __block_exts.extent(0) + __thread_idx.x;
const auto __warp_rank = __thread_rank / 32;
if constexpr (::cuda::std::is_same_v<_Level, block_level>)
{
return {static_cast<_Tp>(__warp_rank), 0, 0};
}
else
{
const auto __thread_count = __block_exts.extent(0) * __block_exts.extent(1) * __block_exts.extent(2);
const auto __warp_count = ::cuda::ceil_div(__thread_count, 32);
const auto __next_idx = __index_query<block_level, _Level>::template __call<_Tp>(__hier);
return {static_cast<_Tp>(__next_idx.x * __warp_count + __warp_rank), __next_idx.y, __next_idx.z};
}
}
};
template <>
struct __index_query<block_level, cluster_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy&) noexcept
{
return __index_query_native<block_level, cluster_level>::template __call<_Tp>();
}
};
template <>
struct __index_query<block_level, grid_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy&) noexcept
{
return __index_query_native<block_level, grid_level>::template __call<_Tp>();
}
};
template <>
struct __index_query<cluster_level, grid_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static hierarchy_query_result<_Tp> __call(const _Hierarchy& __hier) noexcept
{
return __index_query_native<cluster_level, grid_level>::template __call<_Tp>();
}
};
_CCCL_DIAG_POP
# endif // _CCCL_CUDA_COMPILATION()
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_QUERIES_INDEX_H

View File

@@ -0,0 +1,215 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_QUERIES_RANK_H
#define _CUDA___HIERARCHY_QUERIES_RANK_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/__hierarchy/hierarchy_query_result.h>
# include <cuda/__hierarchy/queries/count.h>
# include <cuda/__hierarchy/queries/extents.h>
# include <cuda/__hierarchy/queries/index.h>
# include <cuda/__hierarchy/traits.h>
# include <cuda/std/__cstddef/types.h>
# if _CCCL_CUDA_COMPILATION()
# include <cuda/__ptx/instructions/get_sreg.h>
# endif // _CCCL_CUDA_COMPILATION()
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
# if _CCCL_CUDA_COMPILATION()
// cudafe++ makes the queries (that are device only) return void when compiling for host, which causes host compilers
// to warn about applying [[nodiscard]] to a function that returns void.
_CCCL_DIAG_PUSH
_CCCL_DIAG_SUPPRESS_NVHPC(nodiscard_doesnt_apply)
# if _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
_CCCL_DIAG_SUPPRESS_GCC("-Wattributes")
_CCCL_DIAG_SUPPRESS_CLANG("-Wignored-attributes")
# endif // _CCCL_CUDA_COMPILER(NVCC, <, 13, 0)
// native hierarchy queries
template <class _Unit, class _Level>
struct __rank_query_native
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
using _NextLevel = typename _Unit::__next_native_level;
const auto __curr_exts = __extents_query_native<_Unit, _NextLevel>::template __call<_Tp>();
const auto __curr_idx = __index_query_native<_Unit, _NextLevel>::template __call<_Tp>();
_Tp __ret = 0;
if constexpr (!::cuda::std::is_same_v<_Level, _NextLevel>)
{
__ret = __rank_query_native<_NextLevel, _Level>::template __call<_Tp>()
* __count_query_native<_Unit, _NextLevel>::template __call<_Tp>();
}
for (::cuda::std::size_t __i = __curr_exts.rank(); __i > 0; --__i)
{
_Tp __inc = __curr_idx[__i - 1];
for (::cuda::std::size_t __j = __i - 1; __j > 0; --__j)
{
__inc *= __curr_exts.extent(__j - 1);
}
__ret += __inc;
}
return __ret;
}
};
template <>
struct __rank_query_native<thread_level, warp_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
return static_cast<_Tp>(::cuda::ptx::get_sreg_laneid());
}
};
template <>
struct __rank_query_native<block_level, cluster_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
unsigned __rank = 0;
NV_IF_TARGET(NV_PROVIDES_SM_90, (__rank = ::__clusterRelativeBlockRank();))
return static_cast<_Tp>(__rank);
}
};
template <>
struct __rank_query_native<block_level, grid_level>
{
template <class _Tp>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call() noexcept
{
return static_cast<_Tp>((static_cast<_Tp>(blockIdx.z) * gridDim.y + blockIdx.y) * gridDim.x + blockIdx.x);
}
};
// hierarchy queries
template <class _Tp, class _Unit, class _NextLevel, class _Level, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API _Tp __rank_query_generic(const _Hierarchy& __hier) noexcept
{
const auto __curr_exts = __extents_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
const auto __curr_idx = __index_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
_Tp __ret = 0;
if constexpr (!::cuda::std::is_same_v<_Level, _NextLevel>)
{
__ret = __rank_query<_NextLevel, _Level>::template __call<_Tp>(__hier)
* __count_query<_Unit, _NextLevel>::template __call<_Tp>(__hier);
}
for (::cuda::std::size_t __i = __curr_exts.rank(); __i > 0; --__i)
{
_Tp __inc = __curr_idx[__i - 1];
for (::cuda::std::size_t __j = __i - 1; __j > 0; --__j)
{
__inc *= __curr_exts.extent(__j - 1);
}
__ret += __inc;
}
return __ret;
}
template <class _Unit, class _Level>
struct __rank_query
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
{
using _NextLevel = __next_hierarchy_level_t<_Unit, _Hierarchy>;
return ::cuda::__rank_query_generic<_Tp, _Unit, _NextLevel, _Level>(__hier);
}
};
template <>
struct __rank_query<thread_level, warp_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
{
return __rank_query_native<thread_level, warp_level>::template __call<_Tp>();
}
};
template <class _Level>
struct __rank_query<warp_level, _Level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
{
return ::cuda::__rank_query_generic<_Tp, warp_level, block_level, _Level>(__hier);
}
};
template <>
struct __rank_query<block_level, cluster_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
{
return __rank_query_native<block_level, cluster_level>::template __call<_Tp>();
}
};
template <>
struct __rank_query<block_level, grid_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy&) noexcept
{
return __rank_query_native<block_level, grid_level>::template __call<_Tp>();
}
};
template <>
struct __rank_query<cluster_level, grid_level>
{
template <class _Tp, class _Hierarchy>
[[nodiscard]] _CCCL_DEVICE_API static _Tp __call(const _Hierarchy& __hier) noexcept
{
return ::cuda::__rank_query_generic<_Tp, cluster_level, grid_level, grid_level>(__hier);
}
};
_CCCL_DIAG_POP
# endif // _CCCL_CUDA_COMPILATION()
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_QUERIES_RANK_H

View File

@@ -0,0 +1,114 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___HIERARCHY_TRAITS_H
#define _CUDA___HIERARCHY_TRAITS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/hierarchy.h>
# include <cuda/std/__tuple_dir/get.h>
# include <cuda/std/__type_traits/is_same.h>
# include <cuda/std/__type_traits/remove_cvref.h>
# include <cuda/std/__type_traits/type_list.h>
# include <cuda/std/__type_traits/void_t.h>
# include <cuda/std/cstdint>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// __is_natively_reachable_hierarchy_level_v
template <class _FromLevel, class _CurrLevel, class _ToLevel, class = void>
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v = false;
template <class _FromLevel, class _CurrLevel, class _ToLevel>
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<
_FromLevel,
_CurrLevel,
_ToLevel,
::cuda::std::void_t<typename _CurrLevel::__next_native_level>> =
__is_natively_reachable_hierarchy_level_helper_v<_FromLevel, typename _CurrLevel::__next_native_level, _ToLevel>;
template <class _Level, class _ToLevel>
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<_Level, _Level, _ToLevel> = false;
template <class _FromLevel, class _Level>
inline constexpr bool __is_natively_reachable_hierarchy_level_helper_v<_FromLevel, _Level, _Level> = true;
template <class _FromLevel, class _ToLevel, class = void>
inline constexpr bool __is_natively_reachable_hierarchy_level_v = false;
template <class _FromLevel, class _ToLevel>
inline constexpr bool __is_natively_reachable_hierarchy_level_v<
_FromLevel,
_ToLevel,
::cuda::std::void_t<typename _FromLevel::__next_native_level>> =
__is_native_hierarchy_level_v<_ToLevel>
&& __is_natively_reachable_hierarchy_level_helper_v<_FromLevel, typename _FromLevel::__next_native_level, _ToLevel>;
// __level_type_of
template <class _LevelDesc>
using __level_type_of = typename _LevelDesc::level_type;
// __has_bottom_unit_or_level_v
template <class _QueryLevel, class _Hierarchy>
inline constexpr bool __has_bottom_unit_or_level_v =
::cuda::std::is_same_v<_QueryLevel, typename _Hierarchy::bottom_unit_type>
|| _Hierarchy::template has_level<_QueryLevel>();
// __next_hierarchy_level
template <class _Level, class _Hierarchy>
struct __next_hierarchy_level;
template <class _Level, class _BottomUnit, class... _LevelDescs>
struct __next_hierarchy_level<_Level, hierarchy<_BottomUnit, _LevelDescs...>>
{
static constexpr ::cuda::std::size_t __level_idx =
hierarchy<_BottomUnit, _LevelDescs...>::template __level_idx<_Level>;
using __type = ::cuda::std::__type_index_c<__level_idx - 1, typename _LevelDescs::level_type...>;
};
template <class _Level, class... _LevelDescs>
struct __next_hierarchy_level<_Level, hierarchy<_Level, _LevelDescs...>>
{
using __type = ::cuda::std::__type_index_c<(sizeof...(_LevelDescs) - 1), typename _LevelDescs::level_type...>;
};
template <class _Level, class _Hierarchy>
using __next_hierarchy_level_t = typename __next_hierarchy_level<_Level, _Hierarchy>::__type;
template <class _Type>
_CCCL_CONCEPT_FRAGMENT(__has_hierarchy_member_,
requires(const _Type& __instance)(requires(
::cuda::__is_hierarchy_v<::cuda::std::remove_cvref_t<decltype(__instance.hierarchy())>>)));
template <class _Type>
_CCCL_CONCEPT __has_hierarchy_member = _CCCL_FRAGMENT(__has_hierarchy_member_, _Type);
template <class _Type>
inline constexpr bool __is_or_has_hierarchy_member_v = __has_hierarchy_member<_Type> || __is_hierarchy_v<_Type>;
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___HIERARCHY_TRAITS_H