CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
138 lines
3.5 KiB
C++
138 lines
3.5 KiB
C++
//===----------------------------------------------------------------------===//
|
|
//
|
|
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
|
// See https://llvm.org/LICENSE.txt for license information.
|
|
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
|
//
|
|
//===----------------------------------------------------------------------===//
|
|
|
|
// XFAIL: enable-tile
|
|
// error: asm statement is unsupported in tile code
|
|
|
|
// UNSUPPORTED: windows && pre-sm-70
|
|
|
|
#include <cuda/atomic>
|
|
#include <cuda/std/cassert>
|
|
|
|
#include "test_macros.h"
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T store(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.store(in + 1, cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
T old = T(7);
|
|
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
T old = T(7);
|
|
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T exchange(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
T out = x.exchange(T(1), cuda::memory_order_relaxed);
|
|
return out + x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_add(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_add(T(1), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_sub(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_sub(T(1), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_and(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_and(T(1), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_or(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_or(T(1), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_xor(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_xor(T(1), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_min(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_min(T(7), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC T fetch_max(T in)
|
|
{
|
|
cuda::atomic<T> x(in);
|
|
x.fetch_max(T(7), cuda::memory_order_relaxed);
|
|
return x.load(cuda::memory_order_relaxed);
|
|
}
|
|
|
|
template <typename T>
|
|
TEST_DEVICE_FUNC inline void tests()
|
|
{
|
|
const T tid = threadIdx.x;
|
|
assert(tid + T(1) == store(tid));
|
|
assert(T(1) + tid == exchange(tid));
|
|
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
|
|
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
|
|
assert((tid + T(1)) == fetch_add(tid));
|
|
assert((tid & T(1)) == fetch_and(tid));
|
|
assert((tid | T(1)) == fetch_or(tid));
|
|
assert((tid ^ T(1)) == fetch_xor(tid));
|
|
assert(min(tid, T(7)) == fetch_min(tid));
|
|
assert(max(tid, T(7)) == fetch_max(tid));
|
|
assert(T(tid - T(1)) == fetch_sub(tid));
|
|
}
|
|
|
|
int main(int arg, char** argv)
|
|
{
|
|
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
|
|
NV_IF_ELSE_TARGET(
|
|
NV_IS_HOST,
|
|
(cuda_thread_count = 64;),
|
|
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
|
|
tests<int32_t>();
|
|
tests<int64_t>();))
|
|
#endif
|
|
return 0;
|
|
}
|