[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
84
cccl_upstream/cub/test/catch2_test_warp_mask.cu
Normal file
84
cccl_upstream/cub/test/catch2_test_warp_mask.cu
Normal file
@@ -0,0 +1,84 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <cub/util_arch.cuh>
|
||||
#include <cub/util_ptx.cuh>
|
||||
|
||||
#include <cuda/__cmath/pow2.h>
|
||||
|
||||
#include <c2h/catch2_test_helper.h>
|
||||
|
||||
template <int logical_warp_threads>
|
||||
struct total_warps_t
|
||||
{
|
||||
private:
|
||||
static constexpr unsigned int total_warps =
|
||||
(::cuda::is_power_of_two(logical_warp_threads)) ? cub::detail::warp_threads / logical_warp_threads : 1;
|
||||
|
||||
public:
|
||||
static constexpr unsigned int value()
|
||||
{
|
||||
return total_warps;
|
||||
}
|
||||
};
|
||||
|
||||
bool is_lane_involved(unsigned int member_mask, unsigned int lane)
|
||||
{
|
||||
return member_mask & (1 << lane);
|
||||
}
|
||||
|
||||
using logical_warp_threads = c2h::iota<1, 32>;
|
||||
using power_of_two_warp_threads = c2h::enum_type_list<int, 1, 2, 4, 8, 16, 32>;
|
||||
|
||||
C2H_TEST("Warp mask ignores lanes before current logical warp", "[mask][warp]", power_of_two_warp_threads)
|
||||
{
|
||||
constexpr int logical_warp_thread = c2h::get<0, TestType>::value;
|
||||
constexpr unsigned int total_warps = total_warps_t<logical_warp_thread>::value();
|
||||
|
||||
for (unsigned int warp_id = 0; warp_id < total_warps; warp_id++)
|
||||
{
|
||||
const unsigned int warp_mask = cub::WarpMask<logical_warp_thread>(warp_id);
|
||||
const unsigned int warp_begin = logical_warp_thread * warp_id;
|
||||
|
||||
for (unsigned int prev_warp_lane = 0; prev_warp_lane < warp_begin; prev_warp_lane++)
|
||||
{
|
||||
REQUIRE_FALSE(is_lane_involved(warp_mask, prev_warp_lane));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
C2H_TEST("Warp mask involves lanes of current logical warp", "[mask][warp]", logical_warp_threads)
|
||||
{
|
||||
constexpr int logical_warp_thread = c2h::get<0, TestType>::value;
|
||||
constexpr unsigned int total_warps = total_warps_t<logical_warp_thread>::value();
|
||||
|
||||
for (unsigned int warp_id = 0; warp_id < total_warps; warp_id++)
|
||||
{
|
||||
const unsigned int warp_mask = cub::WarpMask<logical_warp_thread>(warp_id);
|
||||
const unsigned int warp_begin = logical_warp_thread * warp_id;
|
||||
const unsigned int warp_end = warp_begin + logical_warp_thread;
|
||||
|
||||
for (unsigned int warp_lane = warp_begin; warp_lane < warp_end; warp_lane++)
|
||||
{
|
||||
REQUIRE(is_lane_involved(warp_mask, warp_lane));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
C2H_TEST("Warp mask ignores lanes after current logical warp", "[mask][warp]", logical_warp_threads)
|
||||
{
|
||||
constexpr int logical_warp_thread = c2h::get<0, TestType>::value;
|
||||
constexpr unsigned int total_warps = total_warps_t<logical_warp_thread>::value();
|
||||
|
||||
for (unsigned int warp_id = 0; warp_id < total_warps; warp_id++)
|
||||
{
|
||||
const unsigned int warp_mask = cub::WarpMask<logical_warp_thread>(warp_id);
|
||||
const unsigned int warp_begin = logical_warp_thread * warp_id;
|
||||
const unsigned int warp_end = warp_begin + logical_warp_thread;
|
||||
|
||||
for (unsigned int post_warp_lane = warp_end; post_warp_lane < cub::detail::warp_threads; post_warp_lane++)
|
||||
{
|
||||
REQUIRE_FALSE(is_lane_involved(warp_mask, post_warp_lane));
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user