//===----------------------------------------------------------------------===// // // Part of CUDA Experimental in CUDA C++ Core Libraries, // under the Apache License v2.0 with LLVM Exceptions. // See https://llvm.org/LICENSE.txt for license information. // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception // SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES. // //===----------------------------------------------------------------------===// #include #include #include #include #include #include #include #include #include #include #include "group_testing.cuh" namespace { template __device__ void test_group_as(Config config) { using NsSeq = cuda::std::integer_sequence; constexpr unsigned ns[]{static_cast(Ns)...}; constexpr cuda::std::size_t ngroups = sizeof...(Ns); cuda::std::size_t group_starts[ngroups]; cuda::std::exclusive_scan(cuda::std::begin(ns), cuda::std::end(ns), group_starts, cuda::std::size_t{}); // Test static Ns. { using Mapping = cudax::group_as, true>; // Test default constructor. { static_assert(cuda::std::is_trivially_default_constructible_v); static_assert(cuda::std::is_empty_v); Mapping mapping; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test the mapping is constructible from the Ns sequence. { static_assert(cuda::std::is_nothrow_constructible_v); cudax::group_as mapping{NsSeq{}}; static_assert(cuda::std::is_same_v); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test the mapping is not constructible from Ns sequence and non_exhaustive_t. static_assert(!cuda::std::is_constructible_v); // Test static_group_count(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_group_count())); static_assert(Mapping::static_group_count() == ngroups); // Test static_unit_count(). { static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_unit_count(cuda::std::size_t{}))); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(Mapping::static_unit_count(i) == ns[i]); } } // Test is_always_exhaustive(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::is_always_exhaustive())); static_assert(Mapping::is_always_exhaustive()); // Test unit_count(). { static_assert( cuda::std::is_same_v().unit_count(cuda::std::size_t{}))>); static_assert(noexcept(cuda::std::declval().unit_count(cuda::std::size_t{}))); const Mapping mapping; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test map(...). { const cudax::this_warp parent_group{config}; const ThreadsInWarpMappingResult prev_mapping_result; static_assert(cudax::__group_mapping_result().map( cuda::gpu_thread, parent_group, prev_mapping_result))>); static_assert( noexcept(cuda::std::declval().map(cuda::gpu_thread, parent_group, prev_mapping_result))); const Mapping mapping; auto result = mapping.map(cuda::gpu_thread, parent_group, prev_mapping_result); using Result = decltype(result); const auto rank_in_warp = cuda::gpu_thread.rank(parent_group); unsigned group_rank_ref = ngroups - 1; unsigned rank_ref = rank_in_warp - group_starts[group_rank_ref]; for (unsigned i = 1; i < ngroups; ++i) { if (rank_in_warp < group_starts[i]) { group_rank_ref = i - 1; rank_ref = rank_in_warp - group_starts[i - 1]; break; } } static_assert(Result::static_group_count() == ngroups); CHECK(result.group_count() == static_cast(ngroups)); CHECK(result.group_rank() == group_rank_ref); static_assert(Result::static_unit_count() == cuda::std::dynamic_extent); CHECK(result.unit_count() == ns[group_rank_ref]); CHECK(result.unit_rank() == rank_ref); const auto lane_mask_ref = (ns[group_rank_ref] < 32) ? ((1u << ns[group_rank_ref]) - 1) << group_starts[group_rank_ref] : ~0; CHECK(result.lane_mask() == cuda::device::lane_mask{lane_mask_ref}); CHECK(result.is_valid()); static_assert(Result::is_always_exhaustive()); static_assert(Result::is_always_contiguous()); } } // Test dynamic Ns. { using Mapping = cudax::group_as, true>; // Test default constructor. static_assert(!cuda::std::is_default_constructible_v); // Test the mapping is constructible from the Ns array. { static_assert(cuda::std::is_nothrow_constructible_v); cudax::group_as mapping{ns}; static_assert(cuda::std::is_same_v); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test the mapping is not constructible from Ns array and non_exhaustive_t. static_assert(!cuda::std::is_constructible_v); // Test static_group_count(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_group_count())); static_assert(Mapping::static_group_count() == ngroups); // Test static_unit_count(). { static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_unit_count(cuda::std::size_t{}))); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(Mapping::static_unit_count(i) == cuda::std::dynamic_extent); } } // Test is_always_exhaustive(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::is_always_exhaustive())); static_assert(Mapping::is_always_exhaustive()); // Test unit_count(). { static_assert( cuda::std::is_same_v().unit_count(cuda::std::size_t{}))>); static_assert(noexcept(cuda::std::declval().unit_count(cuda::std::size_t{}))); const Mapping mapping{ns}; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test map(...). { const cudax::this_warp parent_group{config}; const ThreadsInWarpMappingResult prev_mapping_result; static_assert(cudax::__group_mapping_result().map( cuda::gpu_thread, parent_group, prev_mapping_result))>); static_assert( noexcept(cuda::std::declval().map(cuda::gpu_thread, parent_group, prev_mapping_result))); const Mapping mapping{ns}; auto result = mapping.map(cuda::gpu_thread, parent_group, prev_mapping_result); using Result = decltype(result); const auto rank_in_warp = cuda::gpu_thread.rank_as(parent_group); unsigned group_rank_ref = ngroups - 1; unsigned rank_ref = rank_in_warp - group_starts[group_rank_ref]; for (unsigned i = 1; i < ngroups; ++i) { if (rank_in_warp < group_starts[i]) { group_rank_ref = i - 1; rank_ref = rank_in_warp - group_starts[i - 1]; break; } } static_assert(Result::static_group_count() == ngroups); CHECK(result.group_count() == static_cast(ngroups)); CHECK(result.group_rank() == group_rank_ref); static_assert(Result::static_unit_count() == cuda::std::dynamic_extent); CHECK(result.unit_count() == ns[group_rank_ref]); CHECK(result.unit_rank() == rank_ref); const auto lane_mask_ref = (ns[group_rank_ref] < 32) ? ((1u << ns[group_rank_ref]) - 1) << group_starts[group_rank_ref] : ~0; CHECK(result.lane_mask() == cuda::device::lane_mask{lane_mask_ref}); CHECK(result.is_valid()); static_assert(Result::is_always_exhaustive()); static_assert(Result::is_always_contiguous()); } } } template __device__ void test_group_as_non_exhaustive(Config config) { using NsSeq = cuda::std::integer_sequence; constexpr unsigned ns[]{static_cast(Ns)...}; constexpr cuda::std::size_t ngroups = sizeof...(Ns); cuda::std::size_t group_starts[ngroups]; cuda::std::exclusive_scan(cuda::std::begin(ns), cuda::std::end(ns), group_starts, cuda::std::size_t{}); const auto ns_sum = cuda::std::accumulate(cuda::std::begin(ns), cuda::std::end(ns), 0u); // Test static Ns. { using Mapping = cudax::group_as, false>; // Test default constructor. { static_assert(cuda::std::is_trivially_default_constructible_v); static_assert(cuda::std::is_empty_v); Mapping mapping; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test the mapping is not constructible from the Ns sequence. static_assert(!cuda::std::is_constructible_v); // Test the mapping is constructible from Ns sequence and non_exhaustive_t. { static_assert(cuda::std::is_nothrow_constructible_v); cudax::group_as mapping{NsSeq{}, cudax::non_exhaustive}; static_assert(cuda::std::is_same_v); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test static_group_count(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_group_count())); static_assert(Mapping::static_group_count() == ngroups); // Test static_unit_count(). { static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_unit_count(cuda::std::size_t{}))); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(Mapping::static_unit_count(i) == ns[i]); } } // Test is_always_exhaustive(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::is_always_exhaustive())); static_assert(!Mapping::is_always_exhaustive()); // Test unit_count(). { static_assert( cuda::std::is_same_v().unit_count(cuda::std::size_t{}))>); static_assert(noexcept(cuda::std::declval().unit_count(cuda::std::size_t{}))); const Mapping mapping; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test map(...). { const cudax::this_warp parent_group{config}; const ThreadsInWarpMappingResult prev_mapping_result; static_assert(cudax::__group_mapping_result().map( cuda::gpu_thread, parent_group, prev_mapping_result))>); static_assert( noexcept(cuda::std::declval().map(cuda::gpu_thread, parent_group, prev_mapping_result))); const Mapping mapping; auto result = mapping.map(cuda::gpu_thread, parent_group, prev_mapping_result); using Result = decltype(result); const auto rank_in_warp = cuda::gpu_thread.rank(parent_group); const auto is_valid_ref = (rank_in_warp < ns_sum); static_assert(Result::static_group_count() == ngroups); static_assert(Result::static_unit_count() == cuda::std::dynamic_extent); static_assert(!Result::is_always_exhaustive()); static_assert(Result::is_always_contiguous()); CHECK(result.group_count() == static_cast(ngroups)); CHECK(result.is_valid() == is_valid_ref); if (is_valid_ref) { unsigned group_rank_ref = ngroups - 1; unsigned rank_ref = rank_in_warp - group_starts[group_rank_ref]; for (unsigned i = 1; i < ngroups; ++i) { if (rank_in_warp < group_starts[i]) { group_rank_ref = i - 1; rank_ref = rank_in_warp - group_starts[i - 1]; break; } } CHECK(result.group_rank() == group_rank_ref); CHECK(result.unit_count() == ns[group_rank_ref]); CHECK(result.unit_rank() == rank_ref); const auto lane_mask_ref = (ns[group_rank_ref] < 32) ? ((1u << ns[group_rank_ref]) - 1) << group_starts[group_rank_ref] : ~0; CHECK(result.lane_mask() == cuda::device::lane_mask{lane_mask_ref}); } } } // Test dynamic Ns. { using Mapping = cudax::group_as, false>; // Test default constructor. static_assert(!cuda::std::is_default_constructible_v); // Test the mapping is not constructible from the Ns array. static_assert(!cuda::std::is_constructible_v); // Test the mapping is constructible from Ns array and non_exhaustive_t. { static_assert(cuda::std::is_nothrow_constructible_v); cudax::group_as mapping{ns, cudax::non_exhaustive}; static_assert(cuda::std::is_same_v); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test static_group_count(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_group_count())); static_assert(Mapping::static_group_count() == ngroups); // Test static_unit_count(). { static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::static_unit_count(cuda::std::size_t{}))); for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(Mapping::static_unit_count(i) == cuda::std::dynamic_extent); } } // Test is_always_exhaustive(). static_assert(cuda::std::is_same_v); static_assert(noexcept(Mapping::is_always_exhaustive())); static_assert(!Mapping::is_always_exhaustive()); // Test unit_count(). { static_assert( cuda::std::is_same_v().unit_count(cuda::std::size_t{}))>); static_assert(noexcept(cuda::std::declval().unit_count(cuda::std::size_t{}))); const Mapping mapping{ns, cudax::non_exhaustive}; for (cuda::std::size_t i = 0; i < ngroups; ++i) { CHECK(mapping.unit_count(i) == ns[i]); } } // Test map(...). { const cudax::this_warp parent_group{config}; const ThreadsInWarpMappingResult prev_mapping_result; static_assert(cudax::__group_mapping_result().map( cuda::gpu_thread, parent_group, prev_mapping_result))>); static_assert( noexcept(cuda::std::declval().map(cuda::gpu_thread, parent_group, prev_mapping_result))); const Mapping mapping{ns, cudax::non_exhaustive}; auto result = mapping.map(cuda::gpu_thread, parent_group, prev_mapping_result); using Result = decltype(result); const auto rank_in_warp = cuda::gpu_thread.rank(parent_group); const auto is_valid_ref = (rank_in_warp < ns_sum); static_assert(Result::static_group_count() == ngroups); static_assert(Result::static_unit_count() == cuda::std::dynamic_extent); static_assert(!Result::is_always_exhaustive()); static_assert(Result::is_always_contiguous()); CHECK(result.group_count() == static_cast(ngroups)); CHECK(result.is_valid() == is_valid_ref); if (is_valid_ref) { unsigned group_rank_ref = ngroups - 1; unsigned rank_ref = rank_in_warp - group_starts[group_rank_ref]; for (unsigned i = 1; i < ngroups; ++i) { if (rank_in_warp < group_starts[i]) { group_rank_ref = i - 1; rank_ref = rank_in_warp - group_starts[i - 1]; break; } } CHECK(result.group_rank() == group_rank_ref); CHECK(result.unit_count() == ns[group_rank_ref]); CHECK(result.unit_rank() == rank_ref); const auto lane_mask_ref = (ns[group_rank_ref] < 32) ? ((1u << ns[group_rank_ref]) - 1) << group_starts[group_rank_ref] : ~0; CHECK(result.lane_mask() == cuda::device::lane_mask{lane_mask_ref}); } } } } struct TestKernel { template __device__ void operator()(const Config& config) { test_group_as<1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1>( config); test_group_as<2, 4, 8, 16, 2>(config); test_group_as<3, 5, 1, 1, 22>(config); test_group_as<31, 1>(config); test_group_as<32>(config); test_group_as_non_exhaustive<1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1, 1>( config); test_group_as_non_exhaustive<2, 4, 8, 16, 2>(config); test_group_as_non_exhaustive<3, 5, 1, 1, 22>(config); test_group_as_non_exhaustive<31, 1>(config); test_group_as_non_exhaustive<32>(config); test_group_as_non_exhaustive<31>(config); test_group_as_non_exhaustive<4, 6, 8>(config); test_group_as_non_exhaustive<2, 2, 3, 1, 14>(config); } }; } // namespace C2H_TEST("Group-as mapping", "[group]") { const auto device = cuda::devices[0]; const cuda::stream stream{device}; { const auto config = cuda::make_config(cuda::grid_dims<1>(), cuda::block_dims<8, 4>()); cuda::launch(stream, config, TestKernel{}); } { const auto config = cuda::make_config(cuda::grid_dims<1>(), cuda::block_dims(dim3{8, 4})); cuda::launch(stream, config, TestKernel{}); } stream.sync(); }