Files
project_6/cccl_upstream/libcudacxx/test/libcudacxx/cuda/atomics/atomic.local.pass.cpp
EngineX CI 56fd68e7dd [INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
2026-07-30 09:35:51 +00:00

138 lines
3.5 KiB
C++

//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: windows && pre-sm-70
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
template <typename T>
TEST_DEVICE_FUNC T store(T in)
{
cuda::atomic<T> x(in);
x.store(in + 1, cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T exchange(T in)
{
cuda::atomic<T> x(in);
T out = x.exchange(T(1), cuda::memory_order_relaxed);
return out + x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_add(T in)
{
cuda::atomic<T> x(in);
x.fetch_add(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_sub(T in)
{
cuda::atomic<T> x(in);
x.fetch_sub(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_and(T in)
{
cuda::atomic<T> x(in);
x.fetch_and(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_or(T in)
{
cuda::atomic<T> x(in);
x.fetch_or(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_xor(T in)
{
cuda::atomic<T> x(in);
x.fetch_xor(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_min(T in)
{
cuda::atomic<T> x(in);
x.fetch_min(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_max(T in)
{
cuda::atomic<T> x(in);
x.fetch_max(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC inline void tests()
{
const T tid = threadIdx.x;
assert(tid + T(1) == store(tid));
assert(T(1) + tid == exchange(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
assert((tid + T(1)) == fetch_add(tid));
assert((tid & T(1)) == fetch_and(tid));
assert((tid | T(1)) == fetch_or(tid));
assert((tid ^ T(1)) == fetch_xor(tid));
assert(min(tid, T(7)) == fetch_min(tid));
assert(max(tid, T(7)) == fetch_max(tid));
assert(T(tid - T(1)) == fetch_sub(tid));
}
int main(int arg, char** argv)
{
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
NV_IF_ELSE_TARGET(
NV_IS_HOST,
(cuda_thread_count = 64;),
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
tests<int32_t>();
tests<int64_t>();))
#endif
return 0;
}