[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,70 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
template <typename T>
TEST_FUNC __noinline__ void test_global_implicit_property(T ap, cudaAccessProperty cp)
{
// Test implicit conversions
cudaAccessProperty v = ap;
assert(cp == v);
// Test default, copy constructor, and copy-assignent
cuda::access_property o(ap);
cuda::access_property d;
d = ap;
// Test explicit conversion to i64
uint64_t x = (uint64_t) o;
uint64_t y = (uint64_t) d;
assert(x == y);
}
TEST_FUNC __noinline__ void test_global()
{
cuda::access_property o(cuda::access_property::global{});
uint64_t x = (uint64_t) o;
unused(x);
}
TEST_FUNC __noinline__ void test_shared()
{
(void) cuda::access_property::shared{};
}
static_assert(sizeof(cuda::access_property::shared) == 1);
static_assert(sizeof(cuda::access_property::global) == 1);
static_assert(sizeof(cuda::access_property::persisting) == 1);
static_assert(sizeof(cuda::access_property::normal) == 1);
static_assert(sizeof(cuda::access_property::streaming) == 1);
static_assert(sizeof(cuda::access_property) == 8);
static_assert(alignof(cuda::access_property::shared) == 1);
static_assert(alignof(cuda::access_property::global) == 1);
static_assert(alignof(cuda::access_property::persisting) == 1);
static_assert(alignof(cuda::access_property::normal) == 1);
static_assert(alignof(cuda::access_property::streaming) == 1);
static_assert(alignof(cuda::access_property) == 8);
int main(int argc, char** argv)
{
test_global_implicit_property(cuda::access_property::normal{}, cudaAccessProperty::cudaAccessPropertyNormal);
test_global_implicit_property(cuda::access_property::streaming{}, cudaAccessProperty::cudaAccessPropertyStreaming);
test_global_implicit_property(cuda::access_property::persisting{}, cudaAccessProperty::cudaAccessPropertyPersisting);
test_global();
test_shared();
return 0;
}

View File

@@ -0,0 +1,54 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
// UNSUPPORTED: clang-14, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
#include <cuda/annotated_ptr>
#include "test_macros.h"
TEST_FUNC constexpr bool test_constexpr()
{
using namespace cuda;
access_property a{}; // default constructor
access_property b{a}; // copy constructor
access_property c{cuda::std::move(a)}; // move constructor
// user-declared ctor
access_property d1{access_property::global{}};
access_property d2{access_property::normal{}};
access_property d3{access_property::streaming{}};
access_property d4{access_property::persisting{}};
auto p1 = static_cast<cudaAccessProperty>(access_property::normal{});
auto p2 = static_cast<cudaAccessProperty>(access_property::streaming{});
auto p3 = static_cast<cudaAccessProperty>(access_property::persisting{});
// fraction ctor
access_property e1{access_property::normal{}, 1.0f};
access_property e2{access_property::streaming{}, 1.0f};
access_property e3{access_property::persisting{}, 1.0f};
access_property e4{access_property::normal{}, 1.0f, access_property::streaming{}};
access_property e5{access_property::persisting{}, 1.0f, access_property::streaming{}};
b = a; // copy assignment
b = cuda::std::move(a); // move assignment
auto value = static_cast<uint64_t>(a);
unused(p1, p2, p3, b, c, d1, d2, d3, d4, e1, e2, e3, e4, e5, value);
return true;
}
int main(int, char**)
{
static_assert(test_constexpr());
return 0;
}

View File

@@ -0,0 +1,26 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
TEST_FUNC __noinline__ void test_access_property_fail()
{
cuda::access_property o = cuda::access_property::normal{};
// Test implicit conversion fails
std::uint64_t x;
x = o;
unused(o);
}
int main(int argc, char** argv)
{
test_access_property_fail();
return 0;
}

View File

@@ -0,0 +1,148 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// UNSUPPORTED: pre-sm-80
#include <cuda/annotated_ptr>
#include <cuda/cmath>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename Prop>
TEST_DEVICE_FUNC constexpr cuda::__l2_evict_t to_enum()
{
if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::normal>)
{
return cuda::__l2_evict_t::_L2_Evict_Normal_Demote;
}
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::streaming>)
{
return cuda::__l2_evict_t::_L2_Evict_First;
}
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::persisting>)
{
return cuda::__l2_evict_t::_L2_Evict_Last;
}
else // if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::global>)
{
return cuda::__l2_evict_t::_L2_Evict_Unchanged;
}
}
//----------------------------------------------------------------------------------------------------------------------
// test range
template <typename Primary, typename Secondary = void, int I = 1>
TEST_DEVICE_FUNC void test_fraction_constexpr()
{
if constexpr (I > 16)
{
return;
}
else
{
constexpr auto fraction = static_cast<float>(I) * (1.0f / 16.0f);
auto policy = cuda::__createpolicy_fraction(to_enum<Primary>(), to_enum<Secondary>(), fraction);
if constexpr (cuda::std::is_void_v<Secondary>)
{
constexpr cuda::access_property property{Primary{}, fraction};
assert(static_cast<uint64_t>(property) == policy);
}
else
{
constexpr cuda::access_property property{Primary{}, fraction, Secondary{}};
assert(static_cast<uint64_t>(property) == policy);
}
test_fraction_constexpr<Primary, Secondary, I + 1>();
}
}
__global__ void test_fraction()
{
test_fraction_constexpr<cuda::access_property::normal>();
test_fraction_constexpr<cuda::access_property::streaming>();
test_fraction_constexpr<cuda::access_property::persisting>();
test_fraction_constexpr<cuda::access_property::normal, cuda::access_property::streaming>();
test_fraction_constexpr<cuda::access_property::persisting, cuda::access_property::streaming>();
}
//----------------------------------------------------------------------------------------------------------------------
// test range
template <typename Primary, typename Secondary>
__global__ void test_range_kernel(void* ptr, uint64_t property, uint32_t primary_size, uint32_t total_size)
{
auto policy = __createpolicy_range(to_enum<Primary>(), to_enum<Secondary>(), ptr, primary_size, total_size);
if (static_cast<uint64_t>(property) != policy)
{
printf(" primary_size = %u, total_size = %u\n", primary_size, total_size);
printf(" primary = %u, secondary = %u\n", (int) to_enum<Primary>(), (int) to_enum<Secondary>());
printf(" 0x%llX vs 0x%llX\n", static_cast<unsigned long long>(policy), static_cast<unsigned long long>(property));
}
assert(static_cast<uint64_t>(property) == policy);
}
template <typename Primary, typename Secondary = void>
void test_range_launch(void* ptr, uint32_t primary_size, uint32_t total_size)
{
cuda::access_property property;
if constexpr (cuda::std::is_void_v<Secondary>)
{
property = cuda::access_property{ptr, primary_size, total_size, Primary{}};
}
else
{
property = cuda::access_property{ptr, primary_size, total_size, Primary{}, Secondary{}};
}
test_range_kernel<Primary, Secondary><<<1, 1>>>(ptr, static_cast<uint64_t>(property), primary_size, total_size);
}
void test_range()
{
int* ptr = nullptr;
ptr++;
for (uint32_t total_size = 1, i = 0; i <= 31; i++, total_size <<= 1)
{
for (uint32_t primary_size = 1, j = 0; j <= i; j++, primary_size <<= 1)
{
test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
test_range_launch<cuda::access_property::streaming, cuda::access_property::streaming>(
ptr, primary_size, total_size);
test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(
ptr, primary_size, total_size);
}
}
// PTX createpolicy_range and access_property behaviors don't match (for now)
// uint32_t primary_size = 0xFFFFFFFF;
// uint32_t total_size = 0xFFFFFFFF;
// test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
// test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(ptr, primary_size,
// total_size);
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, (test_range();))
NV_IF_TARGET(NV_IS_HOST, (test_fraction<<<1, 1>>>();))
return 0;
}

View File

@@ -0,0 +1,164 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::global>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::persisting>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::normal>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::streaming>) == sizeof(uintptr_t),
"annotated_ptr<T, global> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property>) == 2 * sizeof(uintptr_t),
"annotated_ptr<T,access_property> must be 2 * pointer size");
// NOTE: we could make these smaller in the future (e.g. 32-bit) but that would be an ABI breaking change:
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::shared>) == sizeof(uintptr_t),
"annotated_ptr<T, shared> must be pointer size");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::global>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::persisting>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::normal>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::streaming>) == alignof(int*),
"annotated_ptr must align with int*");
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property>) == alignof(int*),
"annotated_ptr must align with int*");
// NOTE: we could lower the alignment in the future but that would be an ABI breaking change:
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::shared>) == alignof(int*),
"annotated_ptr must align with int*");
#define N 128
struct S
{
int x;
TEST_FUNC S& operator=(int o)
{
this->x = o;
return *this;
}
};
template <typename In, typename T>
TEST_FUNC __noinline__ void test_read_access(In i, T* r)
{
assert(i);
assert(i - i == 0);
assert((bool) i);
const In o = i;
// assert(i->x == 0); // FAILS with shmem
// assert(o->x == 0); // FAILS with shmem
for (int n = 0; n < N; ++n)
{
assert(i[n].x == n);
assert(&i[n] == &i[n]);
assert(&i[n] == &r[n]);
assert(o[n].x == n);
assert(&o[n] == &o[n]);
assert(&o[n] == &r[n]);
}
}
template <typename In>
TEST_FUNC __noinline__ void test_write_access(In i)
{
assert(i);
assert((bool) i);
const In o = i;
for (int n = 0; n < N; ++n)
{
i[n].x = 2 * n;
assert(i[n].x == 2 * n);
assert(i[n].x == 2 * n);
i[n].x = n;
o[n].x = 2 * n;
assert(o[n].x == 2 * n);
assert(o[n].x == 2 * n);
o[n].x = n;
}
}
TEST_FUNC __noinline__ void all_tests()
{
S* arr = global_alloc<S, N>();
test_read_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::normal>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::streaming>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::persisting>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::global>(arr), arr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property>(arr), arr);
test_write_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr));
test_write_access(cuda::annotated_ptr<S, cuda::access_property>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr));
NV_IF_TARGET(
NV_IS_DEVICE,
(S* sarr = shared_alloc<S, N>(); // Allocating shared memory is only supported on device
test_read_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr), sarr);
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::shared>(sarr), sarr);
test_write_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr));
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr));))
}
int main(int argc, char** argv)
{
all_tests();
return 0;
}

View File

@@ -0,0 +1,157 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
TEST_DEVICE_FUNC void annotated_ptr_timing_dev(int* in, int* out)
{
cuda::access_property ap(cuda::access_property::persisting{});
// Retrieve global id
int i = blockIdx.x * blockDim.x + threadIdx.x;
cuda::annotated_ptr<int, cuda::access_property> in_ann{in, ap};
cuda::annotated_ptr<int, cuda::access_property> out_ann{out, ap};
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
DPRINTF("&out[i]:%p = &in_ann[i]:%p for i = %d\n", &out_ann[i], &in_ann[i], i);
out_ann[i] = in_ann[i];
};
__global__ void annotated_ptr_timing(int* in, int* out)
{
annotated_ptr_timing_dev(in, out);
}
TEST_DEVICE_FUNC void ptr_timing_dev(int* in, int* out)
{
// Retrieve global id
int i = blockIdx.x * blockDim.x + threadIdx.x;
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
out[i] = in[i];
};
__global__ void ptr_timing(int* in, int* out)
{
ptr_timing_dev(in, out);
};
TEST_FUNC __noinline__ void bench()
{
#ifndef __CUDA_ARCH__
static const size_t ARR_SZ = 1 << 22;
static const size_t THREAD_CNT = 128;
static const size_t BLOCK_CNT = ARR_SZ / THREAD_CNT;
const dim3 threads(THREAD_CNT, 1, 1), blocks(BLOCK_CNT, 1, 1);
cudaEvent_t start, stop;
#else
static const size_t ARR_SZ = 1 << 10;
#endif
int* arr0 = nullptr;
int* arr1 = nullptr;
float annotated_time = 0.f, pointer_time = 0.f;
#ifdef __CUDA_ARCH__
arr0 = (int*) malloc(ARR_SZ * sizeof(int));
arr1 = (int*) malloc(ARR_SZ * sizeof(int));
#else
assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
assert_rt(cudaDeviceSynchronize());
#endif
#ifdef __CUDA_ARCH__
ptr_timing_dev(arr0, arr1);
#else
ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaDeviceSynchronize());
#endif
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
#ifdef __CUDA_ARCH__
ptr_timing_dev(arr0, arr1);
#else
assert_rt(cudaDeviceSynchronize());
assert_rt(cudaEventCreate(&start));
assert_rt(cudaEventCreate(&stop));
assert_rt(cudaEventRecord(start));
ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaEventRecord(stop));
assert_rt(cudaEventSynchronize(stop));
assert_rt(cudaEventElapsedTime(&pointer_time, start, stop));
assert_rt(cudaEventDestroy(start));
assert_rt(cudaEventDestroy(stop));
assert_rt(cudaDeviceSynchronize());
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
#endif
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
(annotated_ptr_timing_dev(arr0, arr1);),
(assert_rt(cudaDeviceSynchronize()); annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaDeviceSynchronize());))
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
NV_IF_ELSE_TARGET(
NV_IS_DEVICE,
(annotated_ptr_timing_dev(arr0, arr1);),
(assert_rt(cudaDeviceSynchronize()); assert_rt(cudaEventCreate(&start)); assert_rt(cudaEventCreate(&stop));
assert_rt(cudaEventRecord(start));
annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
assert_rt(cudaEventRecord(stop));
assert_rt(cudaEventSynchronize(stop));
assert_rt(cudaEventElapsedTime(&annotated_time, start, stop));
assert_rt(cudaEventDestroy(start));
assert_rt(cudaEventDestroy(stop));
assert_rt(cudaDeviceSynchronize());
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}))
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
printf("array(ms):%f, arrotated_ptr(ms):%f\n", pointer_time, annotated_time);
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_IS_DEVICE, (bench();))
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: nvrtc
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
// UNSUPPORTED: clang-14, gcc-12, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
// UNSUPPORTED: msvc && nvcc-12.0
#include <cuda/annotated_ptr>
#include "test_macros.h"
TEST_FUNC constexpr bool test_public_methods()
{
using namespace cuda;
using annotated_ptr = cuda::annotated_ptr<const int, access_property::persisting>;
using annotated_smem_ptr [[maybe_unused]] = cuda::annotated_ptr<const int, access_property::shared>;
annotated_ptr a{}; // default constructor
annotated_ptr b{a}; // copy constructor
annotated_ptr c{cuda::std::move(a)}; // move constructor
NV_IF_TARGET(NV_IS_DEVICE, (annotated_smem_ptr d{nullptr};)) // pointer constructor
b = a; // copy assignment
b = cuda::std::move(a); // move assignment
auto diff = a - b;
auto pred = static_cast<bool>(a);
auto prop = a.__property();
unused(c);
unused(diff);
unused(pred);
unused(prop);
return true;
}
TEST_FUNC constexpr bool test_interleave_values()
{
using namespace cuda;
constexpr auto normal = __l2_interleave(__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto streaming = __l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto persisting = __l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
constexpr auto normal_demote =
__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
static_assert(normal == __l2_interleave_normal);
static_assert(streaming == __l2_interleave_streaming);
static_assert(persisting == __l2_interleave_persisting);
static_assert(normal_demote == __l2_interleave_normal_demote);
return true;
}
int main(int, char**)
{
static_assert(test_interleave_values());
static_assert(test_public_methods());
return 0;
}

View File

@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
template <typename T, typename P>
TEST_FUNC __noinline__ void test_ctor(T* ptr)
{
// default ctor, cpy and cpy assignment
cuda::annotated_ptr<T, P> def;
{
cuda::annotated_ptr<T, P> temp;
temp = def;
unused(temp);
}
cuda::annotated_ptr<T, P> other(def);
unused(other);
// from ptr
cuda::annotated_ptr<T, P> a(ptr);
assert(a);
// cpy ctor & assign to cv
cuda::annotated_ptr<const T, P> c(def);
cuda::annotated_ptr<volatile T, P> d(def);
cuda::annotated_ptr<const volatile T, P> e(def);
c = def;
d = def;
e = def;
// from c|v to c|v|cv
cuda::annotated_ptr<const T, P> f(c);
cuda::annotated_ptr<volatile T, P> g(d);
cuda::annotated_ptr<const volatile T, P> h(e);
f = c;
g = d;
h = e;
unused(f, g, h);
// to cv
cuda::annotated_ptr<const volatile T, P> i(c);
cuda::annotated_ptr<const volatile T, P> j(d);
i = c;
j = d;
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_global_ctor()
{
T* rp = nullptr;
rp++;
test_ctor<T, P>(rp);
// from ptr + prop
P p;
cuda::annotated_ptr<T, cuda::access_property> a(rp, p);
cuda::annotated_ptr<const T, cuda::access_property> b(rp, p);
cuda::annotated_ptr<volatile T, cuda::access_property> c(rp, p);
cuda::annotated_ptr<const volatile T, cuda::access_property> d(rp, p);
}
TEST_FUNC __noinline__ void test_global_ctors()
{
test_global_ctor<int, cuda::access_property::normal>();
test_global_ctor<int, cuda::access_property::streaming>();
test_global_ctor<int, cuda::access_property::persisting>();
test_global_ctor<int, cuda::access_property::global>();
test_global_ctor<int, cuda::access_property>();
NV_IF_TARGET(NV_IS_DEVICE, (__shared__ int smem_value; test_ctor<int, cuda::access_property::shared>(&smem_value);))
}
int main(int argc, char** argv)
{
test_global_ctors();
return 0;
}

View File

@@ -0,0 +1,49 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
template <typename T, typename P>
TEST_FUNC __noinline__ void test_ctor()
{
// default ctor, cpy and cpy assignment
cuda::annotated_ptr<T, P> def;
def = def;
cuda::annotated_ptr<T, P> other(def);
// from ptr
T* rp = nullptr;
cuda::annotated_ptr<T, P> a(rp);
assert(!a);
// cpy ctor & assign to cv
cuda::annotated_ptr<const T, P> c(def);
cuda::annotated_ptr<volatile T, P> d(def);
cuda::annotated_ptr<const volatile T, P> e(def);
c = e; // FAIL
d = d; // FAIL
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_global_ctor()
{
test_ctor<T, P>();
}
TEST_FUNC __noinline__ void test_global_ctors()
{
test_global_ctor<int, cuda::access_property::normal>();
}
int main(int argc, char** argv)
{
test_global_ctors();
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
int main(int argc, char** argv)
{
cuda::access_property ap(cuda::access_property::persisting{});
int* array0 = new int[9];
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
array_anno_ptr = shared_ptr; // fail to compile, as expected
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "utils.h"
int main(int argc, char** argv)
{
cuda::access_property ap(cuda::access_property::persisting{});
int* array0 = new int[9];
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
array_anno_ptr = shared_ptr; // fail to compile, as expected
return 0;
}

View File

@@ -0,0 +1,26 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// NVRTC does not do host side testing
// UNSUPPORTED: nvrtc
#include "utils.h"
TEST_FUNC static void fails_from_host()
{
int a;
__nv_associate_access_property(&a, uint64_t{0});
}
int main(int argc, char** argv)
{
// calling from host needs to fail and kill the app
fails_from_host();
return 0;
}

View File

@@ -0,0 +1,40 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "test_macros.h"
#include "utils.h"
template <typename T, typename U>
TEST_DEVICE_FUNC __noinline__ void shared_mem_test_dev()
{
T* smem = shared_alloc<T, 128>();
smem[10] = 42;
cuda::annotated_ptr<U, cuda::access_property::shared> p{smem + 10};
assert(*p == 42);
}
TEST_DEVICE_FUNC __noinline__ void test_all()
{
shared_mem_test_dev<int, int>();
shared_mem_test_dev<int, const int>();
shared_mem_test_dev<int, volatile int>();
shared_mem_test_dev<int, const volatile int>();
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_IS_DEVICE, (test_all();))
return 0;
}

View File

@@ -0,0 +1,60 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
constexpr size_t array_size = 128;
template <typename T, typename P>
TEST_FUNC __noinline__ void test(P ap)
{
T* arr = global_alloc<T, array_size>();
cuda::apply_access_property(arr, array_size * sizeof(T), ap);
for (size_t i = 0; i < array_size; ++i)
{
assert(static_cast<size_t>(arr[i]) == i);
}
dealloc<T>(arr);
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_aligned(P ap)
{
T* arr = global_alloc<T, array_size>();
cuda::apply_access_property(arr, cuda::aligned_size_t<sizeof(T)>(array_size * sizeof(T)), ap);
for (size_t i = 0; i < array_size; ++i)
{
assert(static_cast<size_t>(arr[i]) == i);
}
dealloc<T>(arr);
}
TEST_FUNC __noinline__ void test_all()
{
test<int>(cuda::access_property::normal{});
test<int>(cuda::access_property::persisting{});
test_aligned<int>(cuda::access_property::normal{});
test_aligned<int>(cuda::access_property::persisting{});
}
int main(int argc, char** argv)
{
test_all();
return 0;
}

View File

@@ -0,0 +1,59 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include "utils.h"
#define ARR_SZ 128
template <typename T, typename P>
TEST_FUNC __noinline__ void test(P ap)
{
T* arr = global_alloc<T, ARR_SZ>();
arr = cuda::associate_access_property(arr, ap);
for (int i = 0; i < ARR_SZ; ++i)
{
assert(arr[i] == i);
}
dealloc<T>(arr);
}
template <typename T, typename P>
TEST_FUNC __noinline__ void test_shared(P ap)
{
T* arr = shared_alloc<T, ARR_SZ>();
arr = cuda::associate_access_property(arr, ap);
for (int i = 0; i < ARR_SZ; ++i)
{
assert(arr[i] == i);
}
}
TEST_FUNC __noinline__ void test_all()
{
test<int>(cuda::access_property::normal{});
test<int>(cuda::access_property::persisting{});
test<int>(cuda::access_property::streaming{});
test<int>(cuda::access_property::global{});
test<int>(cuda::access_property{});
NV_IF_TARGET(NV_IS_DEVICE, (test_shared<int>(cuda::access_property::shared{});))
}
int main(int argc, char** argv)
{
test_all();
return 0;
}

View File

@@ -0,0 +1,121 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: pre-sm-70
#include <cooperative_groups.h>
#include "utils.h"
// TODO: global-shared
// TODO: read const
TEST_FUNC __noinline__ void test_memcpy_async()
{
size_t ARR_SZ = 1 << 10;
int* arr0 = nullptr;
int* arr1 = nullptr;
cuda::access_property ap(cuda::access_property::persisting{});
cuda::barrier<cuda::thread_scope_system> bar0, bar1, bar2, bar3;
init(&bar0, 1);
init(&bar1, 1);
init(&bar2, 1);
init(&bar3, 1);
NV_IF_ELSE_TARGET(
NV_IS_DEVICE,
(arr0 = (int*) malloc(ARR_SZ * sizeof(int)); arr1 = (int*) malloc(ARR_SZ * sizeof(int));),
(assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
assert_rt(cudaDeviceSynchronize());))
cuda::annotated_ptr<int, cuda::access_property> ann0{arr0, ap};
cuda::annotated_ptr<int, cuda::access_property> ann1{arr1, ap};
// cuda::annotated_ptr<const int, cuda::access_property> cann0{arr0, ap};
for (size_t i = 0; i < ARR_SZ; ++i)
{
arr0[i] = static_cast<int>(i);
arr1[i] = 0;
}
cuda::memcpy_async(ann1, ann0, ARR_SZ * sizeof(int), bar0);
// cuda::memcpy_async(ann1, cann0, ARR_SZ * sizeof(int), bar0);
bar0.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
cuda::memcpy_async(arr1, ann0, ARR_SZ * sizeof(int), bar1);
// cuda::memcpy_async(arr1, cann0, ARR_SZ * sizeof(int), bar1);
bar1.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i)
{
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == static_cast<int>(i));
}
arr1[i] = 0;
}
NV_IF_TARGET(
NV_IS_DEVICE,
(
auto group = cooperative_groups::this_thread_block();
cuda::memcpy_async(group, ann1, ann0, ARR_SZ * sizeof(int), bar2);
// cuda::memcpy_async(group, ann1, cann0, ARR_SZ * sizeof(int), bar2);
bar2.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == (int) i);
}
arr1[i] = 0;
}
cuda::memcpy_async(group, arr1, ann0, ARR_SZ * sizeof(int), bar3);
// cuda::memcpy_async(group, arr1, cann0, ARR_SZ * sizeof(int), bar3);
bar3.arrive_and_wait();
for (size_t i = 0; i < ARR_SZ; ++i) {
if (arr1[i] != (int) i)
{
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
assert(arr1[i] == (int) i);
}
arr1[i] = 0;
}))
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
}
int main(int argc, char** argv)
{
test_memcpy_async();
return 0;
}

View File

@@ -0,0 +1,80 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "test_macros.h"
TEST_DIAG_SUPPRESS_MSVC(4505)
#include <cuda/annotated_ptr>
#include <cuda/std/cassert>
#include <nv/target>
#if defined(DEBUG)
# define DPRINTF(...) \
{ \
printf(__VA_ARGS__); \
}
#else
# define DPRINTF(...) \
do \
{ \
} while (false)
#endif
TEST_FUNC void assert_rt_wrap(cudaError_t code, const char* file, int line)
{
if (code != cudaSuccess)
{
#if !TEST_COMPILER(NVRTC)
NV_IF_ELSE_TARGET(NV_IS_HOST,
(printf("assert: %s %s %d\n", cudaGetErrorString(code), file, line);),
(printf("assert: error=%d %s %d\n", code, file, line);))
#endif // !TEST_COMPILER(NVRTC)
assert(code == cudaSuccess);
}
}
#define assert_rt(ret) \
{ \
assert_rt_wrap((ret), __FILE__, __LINE__); \
}
template <typename T, int N>
TEST_FUNC __noinline__ T* global_alloc()
{
T* arr = nullptr;
NV_IF_ELSE_TARGET(
NV_IS_DEVICE, (arr = (T*) malloc(N * sizeof(T));), (assert_rt(cudaMallocManaged((void**) &arr, N * sizeof(T)));))
for (int i = 0; i < N; ++i)
{
arr[i] = i;
}
return arr;
}
template <typename T, int N>
TEST_DEVICE_FUNC __noinline__ T* shared_alloc()
{
__shared__ T data[N];
for (int i = 0; i < N; ++i)
{
data[i] = i;
}
return data;
}
template <typename T>
TEST_FUNC __noinline__ void dealloc(T* arr)
{
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr);), assert_rt(cudaFree(arr));)
}

View File

@@ -0,0 +1,198 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct minimal_comparable_value
{
int value;
};
TEST_FUNC constexpr bool operator<(minimal_comparable_value lhs, minimal_comparable_value rhs)
{
return lhs.value < rhs.value;
}
TEST_FUNC constexpr bool operator==(minimal_comparable_value lhs, minimal_comparable_value rhs)
{
return lhs.value == rhs.value;
}
namespace cuda::std
{
template <>
class numeric_limits<minimal_comparable_value>
{
public:
static constexpr bool is_specialized = true;
TEST_FUNC static constexpr minimal_comparable_value lowest() noexcept
{
return {0};
}
TEST_FUNC static constexpr minimal_comparable_value max() noexcept
{
return {100};
}
};
} // namespace cuda::std
TEST_FUNC constexpr bool test()
{
// --- static_bounds ---
// Basic static bounds
{
constexpr auto b = cuda::args::static_bounds<1, 4096>{};
static_assert(b.lower() == 1);
static_assert(b.upper() == 4096);
}
// Exact static bounds
{
constexpr auto b = cuda::args::static_bounds<42, 42>{};
static_assert(b.lower() == 42);
static_assert(b.upper() == 42);
}
// Long type deduced from NTTPs
{
static_assert(cuda::std::is_same_v<decltype(cuda::args::static_bounds<0L, 1000L>::lower()), long>);
}
#if TEST_HAS_CLASS_NTTP
// Static bounds preserve their original NTTP types
{
constexpr auto b = cuda::args::bounds<1.0f, 8.0f>();
static_assert(b.lower() == 1.0f);
static_assert(b.upper() == 8);
static_assert(cuda::std::is_same_v<decltype(b.lower()), float>);
static_assert(cuda::std::is_same_v<decltype(b.upper()), float>);
}
#endif // TEST_HAS_CLASS_NTTP
// --- runtime_bounds ---
// Basic runtime bounds
{
auto b = cuda::args::runtime_bounds{10, 100};
assert(b.lower() == 10);
assert(b.upper() == 100);
static_assert(cuda::std::is_same_v<decltype(b.lower()), int>);
}
// Default runtime bounds span the element type's numeric_limits range
{
constexpr cuda::args::runtime_bounds<int> b{};
static_assert(b.lower() == cuda::std::numeric_limits<int>::lowest());
static_assert(b.upper() == (cuda::std::numeric_limits<int>::max)());
}
// --- argument_bounds factory functions ---
// Static via factory
{
constexpr auto b = cuda::args::bounds<1, 8>();
static_assert(b.lower() == 1);
static_assert(b.upper() == 8);
static_assert(cuda::args::__is_static_bounds_cv_v<decltype(b)>);
static_assert(!cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
}
// Runtime via factory
{
auto b = cuda::args::bounds(10, 100);
assert(b.lower() == 10);
assert(b.upper() == 100);
static_assert(!cuda::args::__is_static_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
}
// Runtime bounds only require operator< and operator==.
{
constexpr auto b = cuda::args::bounds(minimal_comparable_value{10}, minimal_comparable_value{20});
static_assert(b.lower() == minimal_comparable_value{10});
static_assert(b.upper() == minimal_comparable_value{20});
}
// Static and runtime bounds intersection
{
static_assert(cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<1, 100>>(
cuda::args::runtime_bounds<int>{50, 200}));
static_assert(!cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<100, 200>>(
cuda::args::runtime_bounds<int>{0, 50}));
}
// Runtime bounds validation with no static bounds only requires operator< and operator==.
{
minimal_comparable_value values[] = {{10}, {20}};
[[maybe_unused]] auto arg = cuda::args::deferred_sequence{
cuda::std::span<minimal_comparable_value>{values, 2},
cuda::args::bounds(minimal_comparable_value{5}, minimal_comparable_value{50})};
}
// Unsigned no-bounds arguments must not instantiate a pointless `value < 0` comparison.
{
unsigned int value = 0;
[[maybe_unused]] auto arg = cuda::args::deferred{&value};
}
#if TEST_HAS_CLASS_NTTP
// Static/runtime bounds intersection only requires operator< and operator==.
{
using static_bounds_t = cuda::args::static_bounds<minimal_comparable_value{10}, minimal_comparable_value{50}>;
constexpr auto runtime_bounds = cuda::args::bounds(minimal_comparable_value{20}, minimal_comparable_value{40});
static_assert(cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(runtime_bounds));
static_assert(!cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(
cuda::args::bounds(minimal_comparable_value{60}, minimal_comparable_value{70})));
cuda::args::__validate_static_element_bounds<minimal_comparable_value, static_bounds_t>(
minimal_comparable_value{30});
cuda::args::__validate_runtime_element_bounds(minimal_comparable_value{30}, runtime_bounds);
minimal_comparable_value values[] = {{20}, {30}};
[[maybe_unused]] auto arg = cuda::args::__immediate_sequence{
cuda::std::span<minimal_comparable_value>{values, 2}, static_bounds_t{}, runtime_bounds};
}
#endif // TEST_HAS_CLASS_NTTP
// Non-bounds type
{
static_assert(!cuda::args::__is_bounds_v<int>);
}
// Bounds types accepted by argument wrapper template parameters
{
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::no_bounds>);
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::static_bounds<1, 8>>);
static_assert(!cuda::args::__valid_static_bounds_v<int, cuda::args::runtime_bounds<int>>);
static_assert(!cuda::args::__valid_static_bounds_v<int, int>);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,185 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/iterator>
#include <cuda/std/array>
#include <cuda/std/complex>
#include <cuda/std/expected>
#include <cuda/std/limits>
#include <cuda/std/mdspan>
#include <cuda/std/optional>
#include <cuda/std/span>
#include <cuda/std/tuple>
#include <cuda/std/type_traits>
#include <cuda/std/utility>
#include "test_iterators.h"
#include "test_macros.h"
enum class color
{
red,
green,
blue
};
template <class _Tp>
struct element_type_like
{
using element_type = _Tp;
};
template <class _Tp>
struct range_like
{
using iterator = _Tp*;
};
template <class _Tp>
struct value_type_like
{
using value_type = _Tp;
};
struct non_sequence_value
{};
TEST_FUNC void test()
{
// --- __is_sequence_v ---
// builtin and class type are not sequences
static_assert(!cuda::args::__is_sequence_v<int>);
static_assert(!cuda::args::__is_sequence_v<color>);
static_assert(!cuda::args::__is_sequence_v<non_sequence_value>);
static_assert(!cuda::args::__is_sequence_v<range_like<int>>);
static_assert(!cuda::args::__is_sequence_v<element_type_like<int>>);
static_assert(!cuda::args::__is_sequence_v<value_type_like<int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::complex<float>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::pair<float, int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::tuple<float, int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::optional<int>>);
static_assert(!cuda::args::__is_sequence_v<cuda::std::expected<int, int>>);
// iterators and pointers can be sequences if they are at least random access
static_assert(cuda::args::__is_sequence_v<int*>);
static_assert(cuda::args::__is_sequence_v<const int*>);
static_assert(cuda::args::__is_sequence_v<cuda::counting_iterator<int>>);
static_assert(!cuda::args::__is_sequence_v<bidirectional_iterator<int*>>);
// ranges and arrays are sequences
static_assert(cuda::args::__is_sequence_v<int[]>);
static_assert(cuda::args::__is_sequence_v<const int[]>);
static_assert(cuda::args::__is_sequence_v<int[42]>);
static_assert(cuda::args::__is_sequence_v<const int[42]>);
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int, 1>>);
static_assert(cuda::args::__is_sequence_v<const cuda::std::span<int, 1>&>);
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int>>);
static_assert(cuda::args::__is_sequence_v<cuda::std::array<int, 3>>);
// --- __element_type_of_t ---
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<const cuda::std::span<int, 1>&>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<int*>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::counting_iterator<int>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::array<int, 3>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<range_like<int>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<element_type_like<int>>, int>);
static_assert(
cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::mdspan<const int, cuda::std::extents<int, 1>>>, int>);
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<value_type_like<int>>, int>);
// --- argument_traits: is_deferred ---
static_assert(!cuda::args::__traits<int>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::immediate<int>>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_deferred);
static_assert(!cuda::args::__traits<cuda::args::constant<42>>::is_deferred);
#if TEST_HAS_CLASS_NTTP
static_assert(!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_deferred);
#endif // TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>::is_deferred);
static_assert(cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_deferred);
// --- argument_traits: is_single_value ---
static_assert(cuda::args::__traits<int>::is_single_value);
static_assert(cuda::args::__traits<int*>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int*>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<cuda::counting_iterator<int>>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::deferred<int*>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_single_value);
// --- argument_traits: value_type ---
static_assert(cuda::std::is_same_v<cuda::args::__traits<int>::value_type, int>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type, int>);
static_assert(
cuda::std::is_same_v<cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::value_type,
cuda::std::span<int>>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<42>>::value_type, int>);
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<10, float>>::value_type, float>);
#if TEST_HAS_CLASS_NTTP
static_assert(cuda::std::is_same_v<
cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::value_type,
cuda::std::array<int, 3>>);
#endif // TEST_HAS_CLASS_NTTP
// --- argument_traits: lowest / highest ---
static_assert(cuda::args::__traits<int>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<int>::highest == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__traits<const int>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<int&>::highest == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__traits<float>::lowest == cuda::std::numeric_limits<float>::lowest());
static_assert(cuda::args::__traits<float>::highest == (cuda::std::numeric_limits<float>::max)());
static_assert(cuda::args::__traits<const cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>>::lowest == 1);
static_assert(cuda::args::__traits<cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>&>::highest == 8);
static_assert(
cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>, cuda::args::static_bounds<1, 8>>>::highest
== 8);
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::lowest == 10.0f);
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::highest == 10.0f);
#if TEST_HAS_CLASS_NTTP
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::lowest == 1);
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::highest == 3);
#endif // TEST_HAS_CLASS_NTTP
// --- Free function bounds on plain values ---
static_assert(cuda::args::__lowest_(42) == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__highest_(42) == (cuda::std::numeric_limits<int>::max)());
static_assert(cuda::args::__lowest_(1.0f) == cuda::std::numeric_limits<float>::lowest());
static_assert(cuda::args::__highest_(1.0f) == (cuda::std::numeric_limits<float>::max)());
// --- Scalar and sequence wrappers expose distinct single-value traits ---
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,174 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/iterator>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
TEST_FUNC constexpr bool test()
{
// Deferred single value via span<T, 1>
{
int val = 42;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
assert(cuda::args::__unwrap(def)[0] == 42);
assert(cuda::args::__access::__arg(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<decltype(def)>::highest == (cuda::std::numeric_limits<int>::max)());
}
// Deferred single value with static bounds
{
int val = 42;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 1000>()};
assert(cuda::args::__unwrap(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 1000);
}
// Deferred single value via pointer
{
int val = 42;
using def_t = cuda::args::deferred<int*, cuda::args::static_bounds<0, 100>>;
static_assert(cuda::args::__traits<def_t>::lowest == 0);
static_assert(cuda::args::__traits<def_t>::highest == 100);
// Also verify construction works
auto def = cuda::args::deferred{&val, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def) == &val);
}
// Deferred single value via fancy iterator
{
auto it = cuda::counting_iterator<int>{42};
auto def = cuda::args::deferred{it, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def)[0] == 42);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
static_assert(cuda::args::__traits<decltype(def)>::is_single_value);
}
// Deferred single value with both bounds, runtime bounds first
{
int val = 42;
auto def =
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 256>()};
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 256);
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
cuda::args::__access::__runtime_bounds(def) = cuda::args::bounds(5, 90);
assert(cuda::args::__highest_(def) == 90);
}
// Deferred sequence via fancy iterator
{
auto it = cuda::counting_iterator<int>{10};
auto def = cuda::args::deferred_sequence{it, cuda::args::bounds<0, 100>()};
assert(cuda::args::__unwrap(def)[0] == 10);
assert(cuda::args::__unwrap(def)[2] == 12);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
static_assert(!cuda::args::__traits<decltype(def)>::is_single_value);
}
// Deferred sequence with both bounds
{
int arr[4] = {10, 20, 30, 40};
auto def = cuda::args::deferred_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 4096>(), cuda::args::bounds(5, 100)};
assert(cuda::args::__access::__arg(def).size() == 4);
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
}
// Deferred sequence with both bounds, runtime bounds first
{
int arr[4] = {10, 20, 30, 40};
auto def = cuda::args::deferred_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 4096>()};
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(def)>::highest == 4096);
assert(cuda::args::__lowest_(def) == 5);
assert(cuda::args::__highest_(def) == 100);
}
// Traits: deferred is single value
{
using traits = cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>;
static_assert(traits::is_deferred);
static_assert(traits::is_single_value);
}
// Traits: deferred with pointer is also single value
{
using traits = cuda::args::__traits<cuda::args::deferred<int*>>;
static_assert(traits::is_deferred);
static_assert(traits::is_single_value);
}
// Traits: deferred_sequence is not single value
{
using traits = cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>;
static_assert(traits::is_deferred);
static_assert(!traits::is_single_value);
}
// Unwrap: deferred
{
int val = 99;
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
auto& v = cuda::args::__unwrap(def);
assert(v[0] == 99);
}
// Unwrap: deferred_sequence
{
int arr[3] = {10, 20, 30};
auto def = cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}};
const auto& v = cuda::args::__unwrap(def);
assert(v.size() == 3);
assert(v[1] == 20);
}
// Unwrap: rvalue deferred returns by value
{
int val = 99;
auto v = cuda::args::__unwrap(cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}});
assert(v[0] == 99);
}
// Unwrap: rvalue deferred_sequence returns by value
{
int arr[3] = {10, 20, 30};
auto v = cuda::args::__unwrap(cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}});
assert(v.size() == 3);
assert(v[2] == 30);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,18 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
[[maybe_unused]] cuda::args::deferred_sequence<int> invalid_arg{0};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using traits = cuda::args::__traits<cuda::args::deferred_sequence<int>>;
[[maybe_unused]] constexpr bool invalid_traits = traits::is_deferred;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,181 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/cassert>
#include <cuda/std/limits>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct non_sequence_value
{
int payload;
};
TEST_FUNC constexpr bool test()
{
// Uniform scalar via CTAD
{
auto da = cuda::args::immediate{5};
assert(cuda::args::__unwrap(da) == 5);
assert(cuda::args::__access::__arg(da) == 5);
static_assert(cuda::args::__traits<decltype(da)>::lowest == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__traits<decltype(da)>::highest == (cuda::std::numeric_limits<int>::max)());
assert(cuda::args::__lowest_(da) == 5);
assert(cuda::args::__highest_(da) == 5);
cuda::args::__access::__arg(da) = 6;
assert(cuda::args::__unwrap(da) == 6);
}
// Uniform scalar with static bounds
{
auto da = cuda::args::immediate{5, cuda::args::bounds<1, 8>()};
assert(cuda::args::__unwrap(da) == 5);
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 8);
assert(cuda::args::__lowest_(da) == 5);
assert(cuda::args::__highest_(da) == 5);
}
// Non-sequence values are accepted without scalar-only restrictions
{
auto da = cuda::args::immediate{non_sequence_value{7}};
assert(cuda::args::__unwrap(da).payload == 7);
}
// Pointer-like types can still represent a single value when explicitly wrapped that way
{
int value = 11;
auto da = cuda::args::immediate{&value};
static_assert(cuda::args::__traits<decltype(da)>::is_single_value);
assert(*cuda::args::__unwrap(da) == 11);
}
// Per-segment span with runtime bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds(1L, 100L)};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__access::__arg(da).size() == 4);
assert(cuda::args::__access::__runtime_bounds(da).lower() == 1);
assert(cuda::args::__access::__runtime_bounds(da).upper() == 100);
assert(cuda::args::__lowest_(da) == 1);
assert(cuda::args::__highest_(da) == 100);
cuda::args::__access::__runtime_bounds(da) = cuda::args::bounds(1, 90);
assert(cuda::args::__highest_(da) == 90);
}
// Per-segment span with both bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 256>(), cuda::args::bounds(10, 200)};
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
assert(cuda::args::__lowest_(da) == 10);
assert(cuda::args::__highest_(da) == 200);
}
// Per-segment span with both bounds, runtime bounds first
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{
cuda::std::span<int>{arr, 4}, cuda::args::bounds(10, 200), cuda::args::bounds<1, 256>()};
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
assert(cuda::args::__lowest_(da) == 10);
assert(cuda::args::__highest_(da) == 200);
}
// Per-segment via span
{
int arr[4] = {1, 2, 3, 4};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__unwrap(da)[0] == 1);
assert(cuda::args::__unwrap(da)[3] == 4);
}
// Per-segment with static bounds
{
int arr[4] = {10, 20, 30, 40};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 100>()};
assert(cuda::args::__unwrap(da).size() == 4);
assert(cuda::args::__unwrap(da)[2] == 30);
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
static_assert(cuda::args::__traits<decltype(da)>::highest == 100);
}
// Traits
{
using traits = cuda::args::__traits<cuda::args::immediate<int>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, int>);
}
// Sequence traits
{
using traits = cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>;
static_assert(!traits::is_deferred);
static_assert(!traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::span<int>>);
}
// __is_sequence_v on unwrapped types
{
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type>);
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
}
// Unwrap: scalar
{
auto da = cuda::args::immediate{7};
auto& v = cuda::args::__unwrap(da);
assert(v == 7);
v = 8;
assert(cuda::args::__unwrap(da) == 8);
}
// Unwrap: span
{
int arr[3] = {10, 20, 30};
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}};
const auto& v = cuda::args::__unwrap(da);
assert(v.size() == 3);
assert(v[1] == 20);
}
// Unwrap: rvalue scalar returns by value
{
const auto& v = cuda::args::__unwrap(cuda::args::immediate{7});
assert(v == 7);
}
// Unwrap: rvalue span returns by value
{
int arr[3] = {10, 20, 30};
auto v = cuda::args::__unwrap(cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}});
assert(v.size() == 3);
assert(v[2] == 30);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,23 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
// A type without a cuda::std::numeric_limits specialization has no meaningful implicit bounds. Default-constructing
// runtime_bounds for such a type must be rejected at compile time instead of silently producing a degenerate range.
struct unspecialized_type
{};
[[maybe_unused]] cuda::args::runtime_bounds<unspecialized_type> invalid_bounds{};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,197 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
#include <cuda/std/array>
#include <cuda/std/limits>
#include <cuda/std/type_traits>
#include "test_macros.h"
struct non_sequence_value
{
int payload;
};
enum class dependent_direction
{
min,
max
};
template <dependent_direction Value>
struct dependent_direction_tag
{
static constexpr auto value = Value;
};
template <class Tag>
TEST_FUNC void test_dependent_constant_type()
{
constexpr auto direction = Tag::value;
using constant_t = cuda::args::constant<direction>;
// Regression: NVCC bug generated a host stub using a cv/ref-qualified constant type while device registration used
// the unqualified type, causing cudaErrorInvalidDeviceFunction when launching the kernel.
static_assert(cuda::std::is_same_v<typename constant_t::value_type, dependent_direction>);
static_assert(cuda::std::is_same_v<constant_t, cuda::args::constant<Tag::value, dependent_direction>>);
}
TEST_FUNC void test()
{
// Basic value
{
constexpr auto sa = cuda::args::constant<42>{};
static_assert(cuda::args::__unwrap(sa) == 42);
static_assert(cuda::std::is_same_v<decltype(sa)::value_type, int>);
}
// Different types
{
constexpr auto sa_long = cuda::args::constant<100L>{};
static_assert(cuda::args::__unwrap(sa_long) == 100L);
static_assert(cuda::std::is_same_v<decltype(sa_long)::value_type, long>);
constexpr auto sa_float = cuda::args::constant<10, float>{};
static_assert(cuda::args::__unwrap(sa_float) == 10.0f);
static_assert(cuda::std::is_same_v<decltype(sa_float)::value_type, float>);
static_assert(cuda::std::is_same_v<decltype(cuda::args::__unwrap(sa_float)), float>);
}
// Negative value
{
constexpr auto sa_neg = cuda::args::constant<-1>{};
static_assert(cuda::args::__unwrap(sa_neg) == -1);
}
// Dependent value
{
test_dependent_constant_type<dependent_direction_tag<dependent_direction::max>>();
}
#if TEST_HAS_CLASS_NTTP
// Non-sequence values are accepted without scalar-only restrictions
{
constexpr auto sa = cuda::args::constant<non_sequence_value{7}>{};
static_assert(cuda::args::__unwrap(sa).payload == 7);
}
#endif // TEST_HAS_CLASS_NTTP
#if TEST_HAS_CLASS_NTTP
// Array sequence
{
constexpr auto sa_arr = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
static_assert(cuda::args::__unwrap(sa_arr)[0] == 128);
static_assert(cuda::args::__unwrap(sa_arr)[1] == 256);
static_assert(cuda::args::__unwrap(sa_arr)[2] == 512);
static_assert(cuda::std::is_same_v<decltype(sa_arr)::value_type, cuda::std::array<int, 3>>);
}
#endif // TEST_HAS_CLASS_NTTP
// Bounds: scalar
{
constexpr auto sa = cuda::args::constant<42>{};
static_assert(cuda::args::__lowest_(sa) == 42);
static_assert(cuda::args::__highest_(sa) == 42);
}
#if TEST_HAS_CLASS_NTTP
// Bounds: array sequence computes lowest/highest of elements
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
static_assert(cuda::args::__lowest_(sa) == 128);
static_assert(cuda::args::__highest_(sa) == 512);
}
#endif // TEST_HAS_CLASS_NTTP
#if TEST_HAS_CLASS_NTTP
// Bounds: empty array sequence has unconstrained element bounds
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 0>{}>{};
static_assert(cuda::args::__lowest_(sa) == cuda::std::numeric_limits<int>::lowest());
static_assert(cuda::args::__highest_(sa) == (cuda::std::numeric_limits<int>::max)());
}
#endif // TEST_HAS_CLASS_NTTP
// Traits
{
using traits = cuda::args::__traits<cuda::args::constant<42>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_constant);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, int>);
static_assert(traits::lowest == 42);
static_assert(traits::highest == 42);
}
// Traits: explicit constant value type
{
using traits = cuda::args::__traits<cuda::args::constant<10, float>>;
static_assert(!traits::is_deferred);
static_assert(traits::is_constant);
static_assert(traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, float>);
static_assert(cuda::std::is_same_v<traits::element_type, float>);
static_assert(traits::lowest == 10.0f);
static_assert(traits::highest == 10.0f);
}
#if TEST_HAS_CLASS_NTTP
// Sequence traits
{
using traits = cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>;
static_assert(traits::is_constant);
static_assert(!traits::is_deferred);
static_assert(!traits::is_single_value);
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::array<int, 3>>);
static_assert(cuda::std::is_same_v<traits::element_type, int>);
}
#endif // TEST_HAS_CLASS_NTTP
// Single value: scalar is single, sequence is not
{
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::constant<42>>::value_type>);
#if TEST_HAS_CLASS_NTTP
static_assert(
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
#endif // TEST_HAS_CLASS_NTTP
}
// Unwrap: scalar
{
constexpr auto sa = cuda::args::constant<42>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val == 42);
}
// Unwrap: scalar with explicit value type
{
constexpr auto sa = cuda::args::constant<10, float>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val == 10.0f);
static_assert(cuda::std::is_same_v<decltype(val), const float>);
}
#if TEST_HAS_CLASS_NTTP
// Unwrap: sequence
{
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{10, 20, 30}>{};
constexpr auto val = cuda::args::__unwrap(sa);
static_assert(val[0] == 10);
static_assert(val[2] == 30);
}
#endif // TEST_HAS_CLASS_NTTP
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using arg_t = cuda::args::immediate<int, cuda::args::runtime_bounds<int>>;
[[maybe_unused]] arg_t invalid_arg{0};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,20 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
using arg_t = cuda::args::immediate<unsigned char, cuda::args::static_bounds<0, 1000>>;
[[maybe_unused]] constexpr auto invalid_highest = cuda::args::__traits<arg_t>::highest;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,18 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
[[maybe_unused]] constexpr auto invalid_bounds = cuda::args::static_bounds<0, 1L>{};
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,24 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/argument>
// Reading the implicit bounds of __traits for an element type without a cuda::std::numeric_limits specialization must
// fail to compile rather than silently yielding a value-initialized (and therefore meaningless) bound. This exercises
// the __traits_impl primary-template path, which is the bound surface read by generic consumers.
struct unspecialized_type
{};
[[maybe_unused]] constexpr auto invalid_lowest = cuda::args::__traits<unspecialized_type>::lowest;
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,214 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// Integration test: demonstrates how an algorithm consumes argument wrappers
// to make compile-time and runtime resource decisions.
// All argument types (plain values, constants, immediate values, deferred values) work uniformly
// through the free functions.
#include <cuda/argument>
#include <cuda/std/algorithm>
#include <cuda/std/array>
#include <cuda/std/cassert>
#include <cuda/std/span>
#include "test_macros.h"
constexpr int shared_memory_capacity = 256;
constexpr int default_max_segment_size = 1024;
enum class algorithm_variant
{
shared_memory,
global_memory
};
// Static scaling: choose algorithm variant at compile time.
template <class _SegSizeArg>
TEST_FUNC constexpr algorithm_variant select_variant(_SegSizeArg)
{
if constexpr (cuda::args::__traits<_SegSizeArg>::highest <= shared_memory_capacity)
{
return algorithm_variant::shared_memory;
}
else
{
return algorithm_variant::global_memory;
}
}
// Dynamic scaling: compute buffer size at runtime, clamped to default.
template <class _SegSizeArg>
TEST_FUNC constexpr int compute_buffer_size(_SegSizeArg __seg_size, int __num_segments)
{
auto __highest = cuda::std::min(default_max_segment_size, static_cast<int>(cuda::args::__highest_(__seg_size)));
return __highest * __num_segments;
}
// Process: use the actual unwrapped value.
template <class _SegSizeArg>
TEST_FUNC constexpr int process_segments(_SegSizeArg __seg_size)
{
const auto& __val = cuda::args::__unwrap(__seg_size);
if constexpr (cuda::args::__traits<_SegSizeArg>::is_single_value)
{
return static_cast<int>(__val);
}
else
{
int __total = 0;
for (size_t __i = 0; __i < __val.size(); ++__i)
{
__total += static_cast<int>(__val[__i]);
}
return __total;
}
}
TEST_FUNC constexpr bool test()
{
// Plain scalar: no bounds, global memory, buffer clamped to default
{
static_assert(select_variant(100) == algorithm_variant::global_memory);
assert(compute_buffer_size(100, 4) == default_max_segment_size * 4);
assert(process_segments(100) == 100);
}
#if 0 // FIXME(miscco): This should not work
// Plain span: per-segment, no bounds, global memory
{
int sizes[3] = {64, 128, 96};
auto seg = cuda::std::span<int>{sizes, 3};
assert(select_variant(seg) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg, 3) == default_max_segment_size * 3);
assert(process_segments(seg) == 64 + 128 + 96);
}
#endif
// constant: scalar, fits in shared memory, buffer = value
{
constexpr auto seg_size = cuda::args::constant<128>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 128 * 4);
assert(process_segments(seg_size) == 128);
}
#if TEST_HAS_CLASS_NTTP
// __constant_sequence: array sequence, highest fits in shared memory
{
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 256}>{};
static_assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_sizes, 3) == 256 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 256);
}
// __constant_sequence: array sequence, highest exceeds shared memory, buffer clamped
{
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 512}>{};
static_assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_sizes, 3) == 512 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 512);
}
#endif // TEST_HAS_CLASS_NTTP
// immediate: tight static bounds, shared memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 256>()};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// immediate: wide static bounds, global memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 4096>()};
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// immediate: no bounds, global memory, buffer = value
{
constexpr auto seg_size = cuda::args::immediate{100};
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
assert(process_segments(seg_size) == 100);
}
// __immediate_sequence: per-segment span with runtime bounds only
{
int sizes[3] = {64, 128, 96};
auto seg_sizes = cuda::args::__immediate_sequence{cuda::std::span<int>{sizes, 3}, cuda::args::bounds(1, 200)};
assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 96);
}
// __immediate_sequence: per-segment span with both bounds
{
int sizes[3] = {64, 128, 96};
auto seg_sizes = cuda::args::__immediate_sequence{
cuda::std::span<int>{sizes, 3}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
static_assert(cuda::args::__traits<decltype(seg_sizes)>::highest <= shared_memory_capacity);
assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
assert(process_segments(seg_sizes) == 64 + 128 + 96);
}
// deferred: uniform, bounds for decisions only
{
int val = 100;
auto seg_size =
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
static_assert(cuda::args::__traits<decltype(seg_size)>::highest <= shared_memory_capacity);
assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(compute_buffer_size(seg_size, 4) == 200 * 4);
}
// --- Floating point cases ---
// Plain float: no bounds
{
static_assert(select_variant(1.0f) == algorithm_variant::global_memory);
assert(process_segments(1.0f) == 1);
}
// constant float using an integer NTTP and explicit value type
{
constexpr auto seg_size = cuda::args::constant<128, float>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 128);
}
#if TEST_HAS_CLASS_NTTP
// constant float (float NTTPs require C++20)
{
constexpr auto seg_size = cuda::args::constant<128.0f>{};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 128);
}
// immediate float with static bounds
{
constexpr auto seg_size = cuda::args::immediate{100.0f, cuda::args::bounds<1.0f, 256.0f>()};
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
assert(process_segments(seg_size) == 100);
}
#endif // TEST_HAS_CLASS_NTTP
return true;
}
int main(int, char**)
{
test();
return 0;
}

View File

@@ -0,0 +1,115 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// UNSUPPORTED: nvcc-11, nvcc-12
// <cuda/atomic>
// TODO: Add support for new half
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Fetch min
{
using A = cuda::atomic<T, ThreadScope>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(T(-5)) == T(-1));
printf("%i == %i\n", (int) t.load(), (int) T(-5));
NV_IF_TARGET(NV_IS_HOST, (fflush(stdout);))
assert(t.load() == T(-5));
}
{
using A = cuda::atomic<T, ThreadScope>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(T(-5)) == T(-1));
assert(t.load() == T(-5));
}
// Test not lesser
{
using A = cuda::atomic<T, ThreadScope>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T, ThreadScope>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
// Fetch max
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST,
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();),
NV_PROVIDES_SM_70,
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestFn<__half, shared_memory_selector, cuda::thread_scope::thread_scope_thread>()();
TestFn<__half, global_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
return 0;
}

View File

@@ -0,0 +1,140 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T,
template <typename, typename> class Selector,
cuda::thread_scope ThreadScope,
bool Signed = cuda::std::is_signed<T>::value>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Test greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(1);
assert(t.fetch_max(2) == T(1));
assert(t.load() == T(2));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_max(2) == T(3));
assert(t.load() == T(3));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn<T, Selector, ThreadScope, true>
{
TEST_FUNC void operator()() const
{
// Call unsigned tests
TestFn<T, Selector, ThreadScope, false>()();
// Test greater, but with signed math
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-5);
assert(t.fetch_max(-1) == T(-5));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-5);
assert(t.fetch_max(-1) == T(-5));
assert(t.load() == T(-1));
}
// Test not greater
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_max(-5) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_max(-5) == T(-1));
assert(t.load() == T(-1));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFnDispatch
{
TEST_FUNC void operator()() const
{
TestFn<T, Selector, ThreadScope>()();
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
NV_PROVIDES_SM_70,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,140 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "atomic_helpers.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
template <class T,
template <typename, typename> class Selector,
cuda::thread_scope ThreadScope,
bool Signed = cuda::std::is_signed<T>::value>
struct TestFn
{
TEST_FUNC void operator()() const
{
// Test lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(5);
assert(t.fetch_min(4) == T(5));
assert(t.load() == T(4));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(5);
assert(t.fetch_min(4) == T(5));
assert(t.load() == T(4));
}
// Test not lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(3);
assert(t.fetch_min(4) == T(3));
assert(t.load() == T(3));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(3);
assert(t.fetch_min(4) == T(3));
assert(t.load() == T(3));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFn<T, Selector, ThreadScope, true>
{
TEST_FUNC void operator()() const
{
// Call unsigned tests
TestFn<T, Selector, ThreadScope, false>()();
// Test lesser, but with signed math
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(-5) == T(-1));
assert(t.load() == T(-5));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(-5) == T(-1));
assert(t.load() == T(-5));
}
// Test not lesser
{
using A = cuda::atomic<T>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
{
using A = cuda::atomic<T>;
Selector<volatile A, constructor_initializer> sel;
volatile A& t = *sel.construct();
t = T(-1);
assert(t.fetch_min(4) == T(-1));
assert(t.load() == T(-1));
}
}
};
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
struct TestFnDispatch
{
TEST_FUNC void operator()() const
{
TestFn<T, Selector, ThreadScope>()();
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
NV_PROVIDES_SM_70,
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
NV_IF_TARGET(NV_IS_DEVICE,
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,102 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
#ifndef ATOMIC_HELPERS_H
#define ATOMIC_HELPERS_H
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
struct UserAtomicType
{
int i;
TEST_FUNC explicit UserAtomicType(int d = 0) noexcept
: i(d)
{}
TEST_FUNC friend bool operator==(const UserAtomicType& x, const UserAtomicType& y)
{
return x.i == y.i;
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachIntegralType
{
TEST_FUNC void operator()() const
{
TestFunctor<char, Selector, Scope>()();
TestFunctor<signed char, Selector, Scope>()();
TestFunctor<unsigned char, Selector, Scope>()();
TestFunctor<short, Selector, Scope>()();
TestFunctor<unsigned short, Selector, Scope>()();
TestFunctor<int, Selector, Scope>()();
TestFunctor<unsigned int, Selector, Scope>()();
TestFunctor<long, Selector, Scope>()();
TestFunctor<unsigned long, Selector, Scope>()();
TestFunctor<long long, Selector, Scope>()();
TestFunctor<unsigned long long, Selector, Scope>()();
TestFunctor<wchar_t, Selector, Scope>();
TestFunctor<char16_t, Selector, Scope>()();
TestFunctor<char32_t, Selector, Scope>()();
TestFunctor<int8_t, Selector, Scope>()();
TestFunctor<uint8_t, Selector, Scope>()();
TestFunctor<int16_t, Selector, Scope>()();
TestFunctor<uint16_t, Selector, Scope>()();
TestFunctor<int32_t, Selector, Scope>()();
TestFunctor<uint32_t, Selector, Scope>()();
TestFunctor<int64_t, Selector, Scope>()();
TestFunctor<uint64_t, Selector, Scope>()();
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachFloatingPointType
{
TEST_FUNC void operator()() const
{
TestFunctor<float, Selector, Scope>()();
TestFunctor<double, Selector, Scope>()();
}
};
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
template <typename, typename> class Selector,
cuda::thread_scope Scope
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
= cuda::thread_scope_system
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
>
struct TestEachAtomicType
{
TEST_FUNC void operator()() const
{
TestEachIntegralType<TestFunctor, Selector, Scope>()();
TestEachFloatingPointType<TestFunctor, Selector, Scope>()();
TestFunctor<UserAtomicType, Selector, Scope>()();
TestFunctor<int*, Selector, Scope>()();
TestFunctor<const int*, Selector, Scope>()();
}
};
#endif // ATOMIC_HELPER_H

View File

@@ -0,0 +1,13 @@
// -*- C++ -*-
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,137 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: windows && pre-sm-70
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
template <typename T>
TEST_DEVICE_FUNC T store(T in)
{
cuda::atomic<T> x(in);
x.store(in + 1, cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
{
cuda::atomic<T> x(in);
T old = T(7);
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T exchange(T in)
{
cuda::atomic<T> x(in);
T out = x.exchange(T(1), cuda::memory_order_relaxed);
return out + x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_add(T in)
{
cuda::atomic<T> x(in);
x.fetch_add(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_sub(T in)
{
cuda::atomic<T> x(in);
x.fetch_sub(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_and(T in)
{
cuda::atomic<T> x(in);
x.fetch_and(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_or(T in)
{
cuda::atomic<T> x(in);
x.fetch_or(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_xor(T in)
{
cuda::atomic<T> x(in);
x.fetch_xor(T(1), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_min(T in)
{
cuda::atomic<T> x(in);
x.fetch_min(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC T fetch_max(T in)
{
cuda::atomic<T> x(in);
x.fetch_max(T(7), cuda::memory_order_relaxed);
return x.load(cuda::memory_order_relaxed);
}
template <typename T>
TEST_DEVICE_FUNC inline void tests()
{
const T tid = threadIdx.x;
assert(tid + T(1) == store(tid));
assert(T(1) + tid == exchange(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
assert((tid + T(1)) == fetch_add(tid));
assert((tid & T(1)) == fetch_and(tid));
assert((tid | T(1)) == fetch_or(tid));
assert((tid ^ T(1)) == fetch_xor(tid));
assert(min(tid, T(7)) == fetch_min(tid));
assert(max(tid, T(7)) == fetch_max(tid));
assert(T(tid - T(1)) == fetch_sub(tid));
}
int main(int arg, char** argv)
{
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
NV_IF_ELSE_TARGET(
NV_IS_HOST,
(cuda_thread_count = 64;),
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
tests<int32_t>();
tests<int64_t>();))
#endif
return 0;
}

View File

@@ -0,0 +1,119 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// UNSUPPORTED: nvrtc
// <cuda/atomic>
#define _LIBCUDACXX_FORCE_PTX_AUTOMATIC_STORAGE_PATH 1 // Force using the PTX is_local atomics path
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "test_macros.h"
/*
Test goals:
Pre-load registers with values that will be used to trigger the wrong codepath in local device atomics.
This test is architecture and driver dependent. It is not possible to reproduce this when compiled to SASS on 12.0, but
will repro on 12.8.
Compiled to SASS is an important point, compiling to PTX will show the failure to initialize the local test flag for
isspacep.local to 0, but that might be compiled out by the JIT compiler in the driver
*/
__global__ void __launch_bounds__(1024) device_test(char* gmem)
{
constexpr int threads = 1024;
__shared__ int hidx;
__shared__ int histogram[threads];
cuda::atomic<int, cuda::thread_scope_thread> xatom(0);
constexpr int passes = 16;
constexpr int ops = 32;
constexpr int expected = passes * ops;
if (threadIdx.x == 0)
{
hidx = 0;
memset(histogram, sizeof(histogram), 0);
}
__syncthreads();
for (xatom = 0; xatom.load() < passes; xatom++)
{
using A = cuda::atomic_ref<int, cuda::std::thread_scope_block>;
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 0]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 1]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 2]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 3]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 4]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 5]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 6]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 7]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 8]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 9]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 10]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 11]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 12]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 13]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 14]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 15]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 16]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 17]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 18]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 19]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 20]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 21]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 22]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 23]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 24]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 25]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 26]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 27]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 28]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 29]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 30]);
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 31]);
}
__syncthreads();
if (histogram[threadIdx.x] != expected)
{
printf("[%i] = %i\r\n", threadIdx.x, histogram[threadIdx.x]);
}
assert(histogram[threadIdx.x] == expected);
}
void launch_kernel()
{
cudaError_t err;
char* inptr = nullptr;
CUDA_CALL(err, cudaGetLastError());
CUDA_CALL(err, cudaMalloc(&inptr, 1024));
CUDA_CALL(err, cudaMemset(inptr, 1, 1024));
device_test<<<1, 1024>>>(inptr);
CUDA_CALL(err, cudaGetLastError());
CUDA_CALL(err, cudaDeviceSynchronize());
}
int main(int arg, char** argv)
{
NV_IF_TARGET(NV_IS_HOST, (launch_kernel();))
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: windows
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
// Check that atomics on host may be constructed
template <class T>
TEST_FUNC void do_test()
{
T v(0);
cuda::atomic_ref<T> a(v);
}
int main(int, char**)
{
do_test<__int128_t>();
do_test<__uint128_t>();
return 0;
}

View File

@@ -0,0 +1,33 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: nvrtc
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include "test_macros.h"
// Check that host atomics fail to build
template <class T>
TEST_FUNC void do_test()
{
T v(0);
cuda::atomic_ref<T> a(v);
a.store(1);
assert(a++ == 1);
assert(a.load() == 2);
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, (do_test<__int128_t>(); do_test<__uint128_t>();))
return 0;
}

View File

@@ -0,0 +1,67 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: pre-sm-70
// UNSUPPORTED: windows
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "cuda_space_selector.h"
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr T combine_literal(uint64_t lower, uint64_t upper)
{
return T(lower) | (T(upper) << 64);
}
template <template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
TEST_FUNC void test()
{
{
using T = __int128_t;
using A = cuda::atomic_ref<T, ThreadScope>;
Selector<T, constructor_initializer> sel;
T& t = *sel.construct();
t = T(0);
A atom(t);
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
atom.store(test_v, cuda::std::memory_order_release);
assert(atom.load() == test_v);
}
{
using T = __uint128_t;
using A = cuda::atomic_ref<T, ThreadScope>;
Selector<T, constructor_initializer> sel;
T& t = *sel.construct();
t = T(0);
A atom(t);
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
atom.store(test_v);
assert(atom.load() == test_v);
}
}
int main(int, char**)
{
#if __cccl_ptx_isa >= 840
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70,
(test<local_memory_selector, cuda::thread_scope_thread>(); test<shared_memory_selector, cuda::thread_scope_block>();
test<global_memory_selector, cuda::thread_scope_block>();
test<global_memory_selector, cuda::thread_scope_device>();))
#endif
return 0;
}

View File

@@ -0,0 +1,99 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
#include <cuda/atomic>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "test_macros.h"
/*
Test goals:
Interleaved 8b/16b access to a 32b window while there is thread contention.
for 8b:
Launch 1024 threads, fetch_add(1) each window, value at end of kernel should be 0xFF..FF. This checks for corruption
caused by interleaved access to different parts of the window.
for 16b:
Launch 1024 threads, fetch_add(1), checking for 0x01FF01FF.
*/
template <class T, int Inc>
TEST_FUNC void fetch_add_into_window(T* window, uint16_t* atomHistory)
{
using Atom = cuda::atomic_ref<T, cuda::thread_scope_block>;
Atom a(*window);
*atomHistory = a.fetch_add(Inc);
}
template <class T>
TEST_DEVICE_FUNC void device_do_test(uint32_t expected)
{
constexpr uint32_t threadCount = 1024;
constexpr uint32_t histogramResultCount = 256 * sizeof(T);
constexpr uint32_t histogramEntriesPerThread = 4 / sizeof(T);
__shared__ uint16_t atomHistory[threadCount];
__shared__ uint8_t atomHistogram[histogramResultCount];
__shared__ uint32_t atomicStorage;
cuda::atomic_ref<uint32_t, cuda::thread_scope_block> bucket(atomicStorage);
constexpr uint32_t offsetMask = ((4 / sizeof(T)) - 1);
// Access offset is interleaved meaning threads 4, 5, 6, 7 access window 0, 1, 2, 3 and so on.
const uint32_t threadOffset = threadIdx.x & offsetMask;
if (threadIdx.x == 0)
{
memset(atomHistogram, 0, histogramResultCount);
bucket.store(0);
}
__syncthreads();
T* window = reinterpret_cast<T*>(&atomicStorage) + threadOffset;
fetch_add_into_window<T, 1>(window, atomHistory + threadIdx.x);
__syncthreads();
if (threadIdx.x == 0)
{
// For each thread, add its atomic result into the corresponding bucket
for (uint32_t i = 0; i < threadCount; i++)
{
atomHistogram[atomHistory[i]]++;
}
// Check that each bucket has exactly (4 / sizeof(T)) entries
// This checks that atomic fetch operations were sequential. i.e. 4xfetch_add(1) returns [0, 1, 2, 3]
for (uint32_t i = 0; i < histogramResultCount; i++)
{
assert(atomHistogram[i] == histogramEntriesPerThread);
}
printf("expected: 0x%X\r\n", expected);
printf("result: 0x%X\r\n", bucket.load());
assert(bucket.load() == expected);
}
}
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST,
(cuda_thread_count = 1024;),
NV_IS_DEVICE,
(device_do_test<uint8_t>(0); device_do_test<uint16_t>(0x02000200);));
return 0;
}

View File

@@ -0,0 +1,77 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
//
//===----------------------------------------------------------------------===//
//
// XFAIL: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
// UNSUPPORTED: windows && pre-sm-70
// <cuda/atomic>
// cuda::atomic<key>
// Original test issue:
// https://github.com/NVIDIA/libcudacxx/issues/160
#include <cuda/atomic>
#include "cuda_space_selector.h"
#include "test_macros.h"
template <template <typename, typename> class Selector>
struct TestFn
{
TEST_FUNC void operator()() const
{
{
struct key
{
int32_t a;
int32_t b;
};
using A = cuda::std::atomic<key>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
cuda::std::atomic_init(&t, key{1, 2});
auto r = t.load();
auto d = key{5, 5};
t.store(r);
(void) t.exchange(r);
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
}
{
struct alignas(8) key
{
int32_t a;
int32_t b;
};
using A = cuda::std::atomic<key>;
Selector<A, constructor_initializer> sel;
A& t = *sel.construct();
cuda::std::atomic_init(&t, key{1, 2});
auto r = t.load();
auto d = key{5, 5};
t.store(r);
(void) t.exchange(r);
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
}
}
};
int main(int, char**)
{
NV_DISPATCH_TARGET(NV_IS_HOST, TestFn<local_memory_selector>()();
, NV_PROVIDES_SM_70, TestFn<local_memory_selector>()();)
NV_IF_TARGET(NV_IS_DEVICE, (TestFn<shared_memory_selector>()(); TestFn<global_memory_selector>()();))
return 0;
}

View File

@@ -0,0 +1,87 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef TEST_ARRIVE_TX_H_
#define TEST_ARRIVE_TX_H_
#include <cuda/barrier>
#include <cuda/memory>
#include <cuda/std/utility>
#include "concurrent_agents.h"
#include "cuda_space_selector.h"
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
template <typename Barrier>
inline TEST_DEVICE_FUNC void mbarrier_complete_tx(Barrier& b, int transaction_count)
{
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90,
(
if (cuda::device::is_address_from(cuda::device::barrier_native_handle(b), cuda::device::address_space::shared)) {
asm volatile(
"mbarrier.complete_tx.relaxed.cta.shared::cta.b64 [%0], %1;"
:
: "r"((unsigned int) __cvta_generic_to_shared(cuda::device::barrier_native_handle(b))), "r"(transaction_count)
: "memory");
} else { __trap(); }),
NV_ANY_TARGET,
(
// On architectures pre-SM90 (and on host), we drop the transaction count
// update. The barriers do not keep track of transaction counts.
__trap();));
}
template <bool split_arrive_and_expect>
TEST_DEVICE_FUNC void thread(cuda::barrier<cuda::thread_scope_block>& b, int arrives_per_thread)
{
constexpr int tx_count = 1;
typename cuda::barrier<cuda::thread_scope_block>::arrival_token tok;
if _CCCL_CONSTEXPR_CXX20 (split_arrive_and_expect)
{
cuda::device::barrier_expect_tx(b, tx_count);
tok = b.arrive(arrives_per_thread);
}
else
{
tok = cuda::device::barrier_arrive_tx(b, arrives_per_thread, tx_count);
}
// Manually increase the transaction count of the barrier.
mbarrier_complete_tx(b, tx_count);
b.wait(cuda::std::move(tok));
}
template <bool split_arrive_and_expect>
TEST_DEVICE_FUNC void test()
{
NV_DISPATCH_TARGET(
NV_IS_DEVICE,
(
// Run all threads, each arriving with arrival count 1
using barrier_t = cuda::barrier<cuda::thread_scope_block>;
shared_memory_selector<barrier_t, constructor_initializer> sel_1;
barrier_t* bar_1 = sel_1.construct(blockDim.x);
__syncthreads();
thread<split_arrive_and_expect>(*bar_1, 1);
// Run all threads, each arriving with arrival count 2
shared_memory_selector<barrier_t, constructor_initializer> sel_2;
barrier_t* bar_2 = sel_2.construct(2 * blockDim.x);
__syncthreads();
thread<split_arrive_and_expect>(*bar_2, 2);));
}
#endif // TEST_ARRIVE_TX_H_

View File

@@ -0,0 +1,56 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: clang && !nvcc
// UNSUPPORTED: no_execute
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cooperative_groups.h>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// When PR #416 is merged, uncomment this line:
// cuda_cluster_size = 2;
),
NV_IS_DEVICE,
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
if (threadIdx.x == 0) { init(&bar, blockDim.x); } namespace cg = cooperative_groups;
auto cluster = cg::this_cluster();
cluster.sync();
// This test currently fails at this point because support for
// clusters has not yet been added.
cuda::barrier<cuda::thread_scope_block> * remote_bar;
remote_bar = cluster.map_shared_rank(&bar, cluster.block_rank() ^ 1);
// When PR #416 is merged, this should fail here because the barrier
// is in device memory.
auto token = cuda::device::barrier_arrive_tx(*remote_bar, 1, 0);));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 256;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: no_execute
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
TEST_DEVICE_FUNC uint64_t bar_storage;
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
// Should fail because the barrier is in device memory.
[[maybe_unused]] auto token = cuda::device::barrier_arrive_tx(*bar_ptr, 1, 0);));
return 0;
}

View File

@@ -0,0 +1,28 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#ifndef __cccl_lib_local_barrier_arrive_tx
static_assert(false, "should define __cccl_lib_local_barrier_arrive_tx");
#endif // __cccl_lib_local_barrier_arrive_tx
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-70
// <cuda/barrier>
#include <cuda/barrier>
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
if (threadIdx.x == 0) { init(&bar, blockDim.x); } __syncthreads();
// barrier_arrive_tx should fail on SM70 and SM80, because it is hidden.
auto token = cuda::device::barrier_arrive_tx(bar, 1, 0);
#ifdef __cccl_lib_local_barrier_arrive_tx
static_assert(false, "Fail manually for SM90 and up.");
#endif // __cccl_lib_local_barrier_arrive_tx
));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 2;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 32;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,107 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/utility> // cuda::std::move
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
static constexpr int buf_len = 1024;
alignas(128) TEST_GLOBAL_VARIABLE int gmem_buffer[buf_len];
TEST_DEVICE_FUNC void test()
{
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
gmem_buffer[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
__threadfence();
__syncthreads();
// TEST: Add i to buffer[i]
alignas(16) __shared__ int smem_buffer[buf_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
cde::cp_async_bulk_global_to_shared(smem_buffer, gmem_buffer, sizeof(smem_buffer), bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Update in shared memory
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
smem_buffer[i] += i;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cde::cp_async_bulk_shared_to_global(gmem_buffer, smem_buffer, sizeof(smem_buffer));
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
__threadfence();
__syncthreads();
// TEAR-DOWN: check that global memory is correct
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
assert(gmem_buffer[i] == 2 * i);
}
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512;));
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test();));
return 0;
}

View File

@@ -0,0 +1,28 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#ifndef __cccl_lib_experimental_ctk12_cp_async_exposure
static_assert(false, "should define __cccl_lib_experimental_ctk12_cp_async_exposure");
#endif
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,95 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
// Kernels below are intended to be compiled, but not run. This is to check if
// all generated PTX is valid.
__global__ void test_bulk_tensor(CUtensorMap* map)
{
__shared__ int smem;
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
cde::cp_async_bulk_tensor_1d_global_to_shared(&smem, map, 0, bar);
cde::cp_async_bulk_tensor_2d_global_to_shared(&smem, map, 0, 0, bar);
cde::cp_async_bulk_tensor_3d_global_to_shared(&smem, map, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_4d_global_to_shared(&smem, map, 0, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_5d_global_to_shared(&smem, map, 0, 0, 0, 0, 0, bar);
cde::cp_async_bulk_tensor_1d_shared_to_global(map, 0, &smem);
cde::cp_async_bulk_tensor_2d_shared_to_global(map, 0, 0, &smem);
cde::cp_async_bulk_tensor_3d_shared_to_global(map, 0, 0, 0, &smem);
cde::cp_async_bulk_tensor_4d_shared_to_global(map, 0, 0, 0, 0, &smem);
cde::cp_async_bulk_tensor_5d_shared_to_global(map, 0, 0, 0, 0, 0, &smem);
}
__global__ void test_bulk(void* gmem)
{
__shared__ int smem;
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
cde::cp_async_bulk_global_to_shared(&smem, gmem, 1024, bar);
cde::cp_async_bulk_shared_to_global(gmem, &smem, 1024);
}
__global__ void test_fences_async_group(void*)
{
cde::fence_proxy_async_shared_cta();
cde::cp_async_bulk_commit_group();
// Wait for up to 8 groups
cde::cp_async_bulk_wait_group_read<0>();
cde::cp_async_bulk_wait_group_read<1>();
cde::cp_async_bulk_wait_group_read<2>();
cde::cp_async_bulk_wait_group_read<3>();
cde::cp_async_bulk_wait_group_read<4>();
cde::cp_async_bulk_wait_group_read<5>();
cde::cp_async_bulk_wait_group_read<6>();
cde::cp_async_bulk_wait_group_read<7>();
cde::cp_async_bulk_wait_group_read<8>();
}
int main(int, char**)
{
return 0;
}

View File

@@ -0,0 +1,221 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: clang && !nvcc
// UNSUPPORTED: nvrtc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/utility> // cuda::std::move
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// NVRTC does not support cuda.h (due to import of stdlib.h)
#if !TEST_COMPILER(NVRTC)
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
#endif // !TEST_COMPILER(NVRTC)
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
constexpr size_t GMEM_WIDTH = 1024; // Width of tensor (in # elements)
constexpr size_t GMEM_HEIGHT = 1024; // Height of tensor (in # elements)
constexpr size_t gmem_len = GMEM_WIDTH * GMEM_HEIGHT;
constexpr int SMEM_WIDTH = 32; // Width of shared memory buffer (in # elements)
constexpr int SMEM_HEIGHT = 8; // Height of shared memory buffer (in # elements)
static constexpr int buf_len = SMEM_HEIGHT * SMEM_WIDTH;
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
// We need a type with a size. On NVRTC, cuda.h cannot be imported, so we don't
// have access to the definition of CUTensorMap (only to the declaration of CUtensorMap inside
// cuda/barrier). So we use this type instead and reinterpret_cast in the
// kernel.
struct fake_cutensormap
{
alignas(64) uint64_t opaque[16];
};
__constant__ fake_cutensormap global_fake_tensor_map;
TEST_DEVICE_FUNC void test(int base_i, int base_j)
{
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < static_cast<int>(gmem_len); i += blockDim.x)
{
gmem_tensor[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
__threadfence();
__syncthreads();
// TEST: Add i to buffer[i]
alignas(128) __shared__ int smem_buffer[buf_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
// Fastest moving coordinate first.
cde::cp_async_bulk_tensor_2d_global_to_shared(smem_buffer, global_tensor_map, base_j, base_i, bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Check smem
for (int i = 0; i < SMEM_HEIGHT; ++i)
{
for (int j = 0; j < SMEM_HEIGHT; ++j)
{
const int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
const int smem_lin_idx = i * SMEM_WIDTH + j;
assert(smem_buffer[smem_lin_idx] == gmem_lin_idx);
}
}
__syncthreads();
// Update smem
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
{
smem_buffer[i] = 2 * smem_buffer[i] + 1;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cde::cp_async_bulk_tensor_2d_shared_to_global(global_tensor_map, base_j, base_i, smem_buffer);
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
__threadfence();
__syncthreads();
// TEAR-DOWN: check that global memory is correct
for (int i = 0; i < SMEM_HEIGHT; ++i)
{
for (int j = 0; j < SMEM_HEIGHT; ++j)
{
int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
}
}
__syncthreads();
}
#if !TEST_COMPILER(NVRTC)
# if _CCCL_CTK_BELOW(12, 5)
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
}
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code =
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
}
# endif // _CCCL_CTK_AT_LEAST(12, 5)
#endif // !TEST_COMPILER(NVRTC)
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512;
int* tensor_ptr = nullptr;
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor);
assert(code == cudaSuccess && "getsymboladdress failed.");
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
CUtensorMap local_tensor_map{};
// rank is the number of dimensions of the array.
constexpr uint32_t rank = 2;
uint64_t size[rank] = {GMEM_WIDTH, GMEM_HEIGHT};
// The stride is the number of bytes to traverse from the first element of one row to the next.
// It must be a multiple of 16.
uint64_t stride[rank - 1] = {GMEM_WIDTH * sizeof(int)};
// The box_size is the size of the shared memory buffer that is used as the
// destination of a TMA transfer.
uint32_t box_size[rank] = {SMEM_WIDTH, SMEM_HEIGHT};
// The distance between elements in units of sizeof(element). A stride of 2
// can be used to load only the real component of a complex-valued tensor, for instance.
uint32_t elem_stride[rank] = {1, 1};
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
// Create the tensor descriptor.
CUresult res = cuTensorMapEncodeTiled(
&local_tensor_map, // CUtensorMap *tensorMap,
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
rank, // cuuint32_t tensorRank,
tensor_ptr, // void *globalAddress,
size, // const cuuint64_t *globalDim,
stride, // const cuuint64_t *globalStrides,
box_size, // const cuuint32_t *boxDim,
elem_stride, // const cuuint32_t *elementStrides,
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
assert(code == cudaSuccess && "memcpytosymbol failed.");));
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test(0, 0); test(4, 0); test(4, 4);));
return 0;
}

View File

@@ -0,0 +1,64 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS{256};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS_DEV{256};
constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS{32};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS_DEV{32};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> TEST_SMEM_COORDS[] = {{0}, {4}, {8}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,69 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS{8, 11};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS_DEV{8, 11};
constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS{4, 2};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS_DEV{4, 2};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> TEST_SMEM_COORDS[] = {
{0, 0},
{4, 1},
{4, 5},
{0, 5},
};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,64 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS{8, 11, 13};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS_DEV{8, 11, 13};
constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS{4, 2, 4};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS_DEV{4, 2, 4};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> TEST_SMEM_COORDS[] = {{0, 0, 0}, {4, 1, 3}, {4, 5, 1}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS{8, 11, 13, 3};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS_DEV{8, 11, 13, 3};
constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS{4, 2, 4, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS_DEV{4, 2, 4, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> TEST_SMEM_COORDS[] = {
{0, 0, 0, 0}, {4, 1, 3, 0}, {4, 8, 7, 2}, {4, 5, 1, 1}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
// UNSUPPORTED: nvrtc
// XFAIL: clang && !nvcc
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include <cuda/barrier>
#include <cuda/std/array>
#include "cp_async_bulk_tensor_generic.h"
#include "test_macros.h"
// Define the size of contiguous tensor in global and shared memory.
//
// Note that the first dimension is the one with stride 1. This one must be a
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
// offset.
//
// We have a separate variable for host and device because a constexpr
// cuda::std::array cannot be shared between host and device as some of its
// member functions take a const reference, which is unsupported by nvcc.
constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS{8, 11, 13, 3, 3};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS_DEV{8, 11, 13, 3, 3};
constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS{4, 2, 4, 1, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS_DEV{4, 2, 4, 1, 1};
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> TEST_SMEM_COORDS[] = {
{0, 0, 0, 0, 0}, {4, 1, 3, 0, 1}, {4, 5, 1, 1, 2}};
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're launching
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
NV_IS_DEVICE,
(for (auto smem_coord : TEST_SMEM_COORDS) {
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
}));
return 0;
}

View File

@@ -0,0 +1,349 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// <cuda/barrier>
#ifndef TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
#define TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
#include <cuda/barrier>
#include <cuda/ptx>
#include <cuda/std/array>
#include <cuda/std/utility> // cuda::std::move
namespace ptx = cuda::ptx;
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
// NVRTC does not support cuda.h (due to import of stdlib.h)
#if !TEST_COMPILER(NVRTC)
# include <cstdio>
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
#endif // ! TEST_COMPILER(NVRTC)
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
using barrier = cuda::barrier<cuda::thread_scope_block>;
namespace cde = cuda::device::experimental;
/*
* This header supports the 1d, 2d, ..., 5d test of the TMA PTX wrappers.
*
* The functions below help convert Nd coordinates into something useful.
*
*/
// Compute the total number of elements in a tensor
template <class T, size_t num_dims>
constexpr TEST_FUNC int tensor_len(cuda::std::array<T, num_dims> dims)
{
T len = 1;
for (T d : dims)
{
len *= d;
}
return static_cast<int>(len);
}
// Function to convert:
// a linear index into a shared memory tensor
// into
// a linear index into a global memory tensor.
template <size_t num_dims>
inline TEST_DEVICE_FUNC int smem_lin_idx_to_gmem_lin_idx(
int smem_lin_idx,
cuda::std::array<uint32_t, num_dims> smem_coord,
cuda::std::array<uint32_t, num_dims> smem_dims,
cuda::std::array<uint64_t, num_dims> gmem_dims)
{
assert(smem_coord.size() == smem_dims.size());
assert(smem_coord.size() == gmem_dims.size());
int gmem_lin_idx = 0;
int gmem_stride = 1;
for (int i = 0; i < (int) smem_coord.size(); ++i)
{
int smem_i_idx = smem_lin_idx % smem_dims.begin()[i];
gmem_lin_idx += (smem_coord.begin()[i] + smem_i_idx) * gmem_stride;
smem_lin_idx /= smem_dims.begin()[i];
gmem_stride *= gmem_dims.begin()[i];
}
return gmem_lin_idx;
}
template <size_t num_dims>
TEST_DEVICE_FUNC inline void cp_tensor_global_to_shared(
CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem, barrier& bar)
{
switch (indices.size())
{
case 1:
cde::cp_async_bulk_tensor_1d_global_to_shared(smem, tensor_map, indices[0], bar);
break;
case 2:
cde::cp_async_bulk_tensor_2d_global_to_shared(smem, tensor_map, indices[0], indices[1], bar);
break;
case 3:
cde::cp_async_bulk_tensor_3d_global_to_shared(smem, tensor_map, indices[0], indices[1], indices[2], bar);
break;
case 4:
cde::cp_async_bulk_tensor_4d_global_to_shared(
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], bar);
break;
case 5:
cde::cp_async_bulk_tensor_5d_global_to_shared(
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], bar);
break;
default:
assert(false && "Wrong number of dimensions.");
}
}
template <size_t num_dims>
TEST_DEVICE_FUNC inline void
cp_tensor_shared_to_global(CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem)
{
switch (indices.size())
{
case 1:
cde::cp_async_bulk_tensor_1d_shared_to_global(tensor_map, indices[0], smem);
break;
case 2:
cde::cp_async_bulk_tensor_2d_shared_to_global(tensor_map, indices[0], indices[1], smem);
break;
case 3:
cde::cp_async_bulk_tensor_3d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], smem);
break;
case 4:
cde::cp_async_bulk_tensor_4d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], indices[3], smem);
break;
case 5:
cde::cp_async_bulk_tensor_5d_shared_to_global(
tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], smem);
break;
default:
assert(false && "Wrong number of dimensions.");
}
}
// To define a tensor map in constant memory, we need a type with a size. On
// NVRTC, cuda.h cannot be imported, so we don't have access to the definition
// of CUTensorMap (only to the declaration of CUtensorMap inside cuda/barrier).
// So we use this type instead and reinterpret_cast in the kernel.
struct fake_cutensormap
{
alignas(64) uint64_t opaque[16];
};
__constant__ fake_cutensormap global_fake_tensor_map;
/*
* This test has as primary purpose to make sure that the indices in the mapping
* from C++ to PTX didn't get mixed up.
*
* How does it test this?
*
* 1. It fills a global memory tensor with linear coordinates 0, 1, ...
* 2. It loads a tile into shared memory at some coordinate (x, y, ... )
* 3. It checks that the coordinates that were received in shared memory match the expected.
* 4. It modifies the coordinates (c = 2 * c + 1)
* 5. It writes the tile back to global memory
* 6. It checks that all the values in global are properly modified.
*/
template <size_t smem_len, size_t num_dims>
TEST_DEVICE_FUNC void
test(cuda::std::array<uint32_t, num_dims> smem_coord,
cuda::std::array<uint32_t, num_dims> smem_dims,
cuda::std::array<uint64_t, num_dims> gmem_dims,
int* gmem_tensor,
int gmem_len)
{
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
// SETUP: fill global memory buffer
for (int i = threadIdx.x; i < gmem_len; i += blockDim.x)
{
gmem_tensor[i] = i;
}
// Ensure that writes to global memory are visible to others, including
// those in the async proxy.
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
// fence.proxy.async.global should suffice, but I am keeping the threadfence
// out of an abundance of caution.
__threadfence();
ptx::fence_proxy_async(ptx::space_global);
__syncthreads();
// TEST: Add i to buffer[i]
alignas(128) __shared__ int smem_buffer[smem_len];
#if _CCCL_CUDA_COMPILER(CLANG)
__shared__ char barrier_data[sizeof(barrier)];
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
__shared__ barrier bar;
#endif // !_CCCL_CUDA_COMPILER(CLANG)
if (threadIdx.x == 0)
{
init(&bar, blockDim.x);
}
__syncthreads();
// Load data:
uint64_t token;
if (threadIdx.x == 0)
{
// Fastest moving coordinate first.
cp_tensor_global_to_shared(global_tensor_map, smem_coord, smem_buffer, bar);
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
}
else
{
token = bar.arrive();
}
bar.wait(cuda::std::move(token));
// Check smem
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
assert(smem_buffer[i] == gmem_lin_idx);
}
__syncthreads();
// Update smem
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
smem_buffer[i] = 2 * smem_buffer[i] + 1;
}
cde::fence_proxy_async_shared_cta();
__syncthreads();
// Write back to global memory:
if (threadIdx.x == 0)
{
cp_tensor_shared_to_global(global_tensor_map, smem_coord, smem_buffer);
cde::cp_async_bulk_commit_group();
cde::cp_async_bulk_wait_group_read<0>();
}
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
// fence.proxy.async.global should suffice, but I am keeping the threadfence
// out of an abundance of caution.
__threadfence();
ptx::fence_proxy_async(ptx::space_global);
__syncthreads();
// // TEAR-DOWN: check that global memory is correct
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
{
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
}
__syncthreads();
}
#if !TEST_COMPILER(NVRTC)
# if _CCCL_CTK_BELOW(12, 5)
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
}
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
{
void* driver_ptr = nullptr;
cudaDriverEntryPointQueryResult driver_status;
auto code =
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
assert(code == cudaSuccess && "Could not get driver API");
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
}
# endif // _CCCL_CTK_AT_LEAST(12, 5)
#endif // !TEST_COMPILER(NVRTC)
#if !TEST_COMPILER(NVRTC)
template <typename T, size_t num_dims>
CUtensorMap map_encode(T* tensor_ptr,
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
const cuda::std::array<uint32_t, num_dims>& smem_dims)
{
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
CUtensorMap tensor_map{};
// The stride is the number of bytes to traverse from the first element of one row to the next.
// It must be a multiple of 16.
// cuTensorMapEncodeTiled requires that the stride array is a valid pointer, so we add one superfluous element
// This is necessary for num_dims == 1
cuda::std::array<uint64_t, num_dims> stride;
uint64_t base_stride = sizeof(T);
for (size_t i = 0; i < stride.size() - 1; ++i)
{
base_stride *= gmem_dims[i];
stride[i] = base_stride;
}
// The distance between elements in units of sizeof(element). A stride of 2
// can be used to load only the real component of a complex-valued tensor, for instance.
cuda::std::array<uint32_t, num_dims> elem_stride; // = {1, .., 1};
for (size_t i = 0; i < elem_stride.size(); ++i)
{
elem_stride[i] = 1;
}
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
// Create the tensor descriptor.
CUresult res = cuTensorMapEncodeTiled(
&tensor_map, // CUtensorMap *tensorMap,
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
num_dims, // cuuint32_t tensorRank,
tensor_ptr, // void *globalAddress,
gmem_dims.data(), // const cuuint64_t *globalDim,
stride.data(), // const cuuint64_t *globalStrides,
smem_dims.data(), // const cuuint32_t *boxDim,
elem_stride.data(), // const cuuint32_t *elementStrides,
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
return tensor_map;
}
template <typename T, size_t num_dims>
void init_tensor_map(const T& gmem_tensor_symbol,
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
const cuda::std::array<uint32_t, num_dims>& smem_dims)
{
// Get pointer to gmem_tensor to create tensor map.
int* tensor_ptr = nullptr;
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor_symbol);
assert(code == cudaSuccess && "Could not get symbol address.");
// Create tensor map
CUtensorMap local_tensor_map = map_encode(tensor_ptr, gmem_dims, smem_dims);
// Copy it to device
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
assert(code == cudaSuccess && "Could not copy symbol to device.");
}
#endif // ! TEST_COMPILER(NVRTC)
#endif // TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 256;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,42 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// UNSUPPORTED: no_execute
// <cuda/barrier>
#include <cuda/barrier>
#include "test_macros.h"
// Suppress warning about barrier in shared memory
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
[[maybe_unused]] TEST_GLOBAL_VARIABLE uint64_t bar_storage;
int main(int, char**)
{
NV_IF_TARGET(
NV_IS_DEVICE,
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
// Should fail because the barrier is in device memory.
cuda::device::barrier_expect_tx(*bar_ptr, 1);));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 2;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,34 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
//
// UNSUPPORTED: libcpp-has-no-threads
// UNSUPPORTED: pre-sm-90
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
// <cuda/barrier>
#include "arrive_tx.h"
int main(int, char**)
{
NV_DISPATCH_TARGET(
NV_IS_HOST,
(
// Required by concurrent_agents_launch to know how many we're
// launching. This can only be an int, because the nvrtc tests use grep
// to figure out how many threads to launch.
cuda_thread_count = 32;),
NV_IS_DEVICE,
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
return 0;
}

View File

@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: pre-sm-70
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include <cuda/barrier>
#include "cuda_space_selector.h"
template <cuda::thread_scope Sco, template <typename, typename> class BarrierSelector>
TEST_FUNC void test()
{
cuda::barrier<Sco> b(3);
init(&b, 2);
auto token = b.arrive();
b.arrive_and_wait();
b.wait(std::move(token));
}
template <cuda::thread_scope Sco>
TEST_FUNC void test_select_barrier()
{
test<Sco, local_memory_selector>();
NV_IF_TARGET(NV_IS_DEVICE, (test<Sco, shared_memory_selector>(); test<Sco, global_memory_selector>();))
}
int main(int argc, char** argv)
{
test_select_barrier<cuda::thread_scope_system>();
test_select_barrier<cuda::thread_scope_device>();
test_select_barrier<cuda::thread_scope_block>();
test_select_barrier<cuda::thread_scope_thread>();
return 0;
}

View File

@@ -0,0 +1,41 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: pre-sm-80
// UNSUPPORTED: enable-tile
// error: asm statement is unsupported in tile code
#include <cuda/barrier>
#include "cuda_space_selector.h"
#include "test_macros.h"
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
TEST_NV_DIAG_SUPPRESS(set_but_not_used)
TEST_DEVICE_FUNC void test()
{
__shared__ cuda::barrier<cuda::thread_scope_block>* b;
shared_memory_selector<cuda::barrier<cuda::thread_scope_block>, constructor_initializer> sel;
b = sel.construct(2);
[[maybe_unused]] uint64_t token;
asm volatile("mbarrier.arrive.b64 %0, [%1];" : "=l"(token) : "l"(cuda::device::barrier_native_handle(*b)) : "memory");
b->arrive_and_wait();
}
int main(int argc, char** argv)
{
NV_IF_TARGET(NV_PROVIDES_SM_80, test();)
return 0;
}

View File

@@ -0,0 +1,62 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr bool test()
{
using nl = cuda::std::numeric_limits<T>;
constexpr T all_ones = static_cast<T>(~T{0});
constexpr T half_low = all_ones >> (nl::digits / 2u);
constexpr T half_high = static_cast<T>(all_ones << (nl::digits / 2u));
static_assert(cuda::bit_reverse(all_ones) == all_ones);
static_assert(cuda::bit_reverse(T{0}) == T{0});
static_assert(cuda::bit_reverse(half_low) == half_high);
static_assert(cuda::bit_reverse(T{0b11001001}) == (T{0b10010011} << (nl::digits - 8u)));
static_assert(cuda::bit_reverse(T{T{0b10010011} << (nl::digits - 8u)}) == T{0b11001001});
unused(all_ones);
unused(half_low);
unused(half_high);
return true;
}
TEST_FUNC constexpr bool test()
{
test<unsigned char>();
test<unsigned short>();
test<unsigned>();
test<unsigned long>();
test<unsigned long long>();
test<uint8_t>();
test<uint16_t>();
test<uint32_t>();
test<uint64_t>();
test<size_t>();
test<uintmax_t>();
test<uintptr_t>();
#if _CCCL_HAS_INT128()
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int, char**)
{
assert(test());
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,32 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
int main(int, char**)
{
using T = uint32_t;
static_assert(cuda::bitfield_insert(T{0}, T{0}, -1, 1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, -1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, 33));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 32, 1));
static_assert(cuda::bitfield_insert(T{0}, T{0}, 20, 20));
static_assert(cuda::bitfield_extract(T{0}, -1, 1));
static_assert(cuda::bitfield_extract(T{0}, 0, -1));
static_assert(cuda::bitfield_extract(T{0}, 0, 33));
static_assert(cuda::bitfield_extract(T{0}, 32, 1));
static_assert(cuda::bitfield_extract(T{0}, 20, 20));
return 0;
}

View File

@@ -0,0 +1,84 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr bool test()
{
using nl = cuda::std::numeric_limits<T>;
constexpr T all_ones = static_cast<T>(~T{0});
unused(all_ones);
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 1) == 1);
assert(cuda::bitfield_insert(T{0}, all_ones, 1, 1) == 0b10);
assert(cuda::bitfield_insert(T{0b10}, all_ones, 0, 1) == 0b11);
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 0) == all_ones);
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 1) == all_ones);
assert(cuda::bitfield_insert(all_ones, all_ones, 2, 1) == all_ones);
assert(cuda::bitfield_insert(all_ones, T{0b1000}, 1, 2) == (all_ones & static_cast<T>(~T{0b110})));
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 2) == 0b11);
assert(cuda::bitfield_insert(T{0}, all_ones, 3, 2) == 0b11000);
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 3, 2) == 0b10111000);
assert(cuda::bitfield_insert(T{0b10100000}, T{0b11}, 3, 2) == 0b10111000);
assert(cuda::bitfield_insert(T{0}, all_ones, nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 0, nl::digits) == all_ones);
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, nl::digits, 0) == T{0b10100000});
assert(cuda::bitfield_extract(T{0}, 3, 4) == 0);
assert(cuda::bitfield_extract(T{0b1011}, 0, 1) == 1);
assert(cuda::bitfield_extract(T{0b1011}, 1, 1) == 1);
assert(cuda::bitfield_extract(T{0b1011}, 2, 2) == 0b10);
assert(cuda::bitfield_extract(all_ones, 0, 0) == 0);
assert(cuda::bitfield_extract(all_ones, 0, 4) == 0b1111);
assert(cuda::bitfield_extract(all_ones, 2, 4) == 0b1111);
assert(cuda::bitfield_extract(T{0b1010010}, 0, 2) == 0b10);
assert(cuda::bitfield_extract(T{0b10101100}, 3, 2) == 1);
assert(cuda::bitfield_extract(T{0b10100000}, 3, 3) == 0b100);
assert(cuda::bitfield_extract(T{all_ones}, nl::digits - 1, 1) == 1);
assert(cuda::bitfield_extract(T{0b10100000}, 0, nl::digits) == T{0b10100000});
assert(cuda::bitfield_extract(T{0b10100000}, nl::digits, 0) == 0);
return true;
}
TEST_FUNC constexpr bool test()
{
test<unsigned char>();
test<unsigned short>();
test<unsigned>();
test<unsigned long>();
test<unsigned long long>();
test<uint8_t>();
test<uint16_t>();
test<uint32_t>();
test<uint64_t>();
test<size_t>();
test<uintmax_t>();
test<uintptr_t>();
#if _CCCL_HAS_INT128()
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int, char**)
{
assert(test());
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/bit>
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/type_traits>
#include "test_macros.h"
template <typename T>
TEST_FUNC constexpr bool test()
{
using nl = cuda::std::numeric_limits<T>;
constexpr T all_ones = static_cast<T>(~T{0});
unused(all_ones);
assert(cuda::bitmask<T>(0, 0) == 0);
assert(cuda::bitmask<T>(0, 1) == 1);
assert(cuda::bitmask<T>(1, 0) == 0);
assert(cuda::bitmask<T>(1, 1) == 0b10);
assert(cuda::bitmask<T>(0, 2) == 0b11);
assert(cuda::bitmask<T>(2, 2) == 0b1100);
assert(cuda::bitmask<T>(0, 2) == 0b11);
assert(cuda::bitmask<T>(3, 2) == 0b11000);
assert(cuda::bitmask<T>(nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
assert(cuda::bitmask<T>(0, nl::digits) == all_ones);
assert(cuda::bitmask<T>(nl::digits, 0) == 0);
return true;
}
TEST_FUNC constexpr bool test()
{
test<unsigned char>();
test<unsigned short>();
test<unsigned>();
test<unsigned long>();
test<unsigned long long>();
test<uint8_t>();
test<uint16_t>();
test<uint32_t>();
test<uint64_t>();
test<size_t>();
test<uintmax_t>();
test<uintptr_t>();
#if _CCCL_HAS_INT128()
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int, char**)
{
assert(test());
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,11 @@
#include <cuda/std/type_traits>
#include <cstdio>
#include <c2h/catch2_test_helper.h>
C2H_TEST("libcudacxx can be used", "")
{
printf("CCCL version: %d.%d.%d\n", CCCL_MAJOR_VERSION, CCCL_MINOR_VERSION, CCCL_PATCH_VERSION);
REQUIRE(cuda::std::true_type::value);
}

View File

@@ -0,0 +1,89 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH
#define TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH
#include <cuda/algorithm>
#include <cuda/buffer>
#include <cuda/devices>
#include <cuda/memory_resource>
#include <cuda/std/mdspan>
#include <cuda/stream>
#include <testing.cuh>
#include <utility.cuh>
inline constexpr uint8_t fill_byte = 1;
inline constexpr uint32_t buffer_size = 42;
using legacy_pinned_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_pinned_memory_resource>;
using legacy_managed_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_managed_memory_resource>;
template <typename T>
auto make_pinned_memory_buffer(cuda::stream_ref stream, std::size_t size, cuda::device_ref device = cuda::devices[0])
{
legacy_pinned_resource resource{cuda::mr::legacy_pinned_memory_resource{device}};
return cuda::make_buffer<T>(stream, resource, size, cuda::no_init);
}
template <typename T>
auto make_managed_memory_buffer(cuda::stream_ref stream, std::size_t size, cuda::device_ref device = cuda::devices[0])
{
legacy_managed_resource resource{cuda::mr::legacy_managed_memory_resource{cudaMemAttachGlobal, device}};
return cuda::make_buffer<T>(stream, resource, size, cuda::no_init);
}
inline int get_expected_value(uint8_t pattern_byte)
{
int result;
memset(&result, pattern_byte, sizeof(int));
return result;
}
template <typename Result>
void check_result_and_erase(cuda::stream_ref stream, Result&& result, uint8_t pattern_byte = fill_byte)
{
int expected = get_expected_value(pattern_byte);
stream.sync();
for (int& i : result)
{
CCCLRT_REQUIRE(i == expected);
i = 0;
}
}
template <typename Layout = cuda::std::layout_right, typename Extents>
auto make_buffer_for_mdspan(cuda::stream_ref stream, Extents extents, char value = 0)
{
auto mapping = typename Layout::template mapping<decltype(extents)>{extents};
auto buffer = make_pinned_memory_buffer<int>(stream, mapping.required_span_size());
stream.sync();
memset(buffer.data(), value, buffer.size() * sizeof(int));
return buffer;
}
inline auto create_fake_strided_mdspan()
{
cuda::std::dextents<size_t, 3> dynamic_extents{1, 2, 3};
cuda::std::array<size_t, 3> strides{12, 4, 1};
#if _CCCL_CUDACC_BELOW(12, 6)
auto map = cuda::std::layout_stride::mapping{dynamic_extents, strides};
#else
cuda::std::layout_stride::mapping map{dynamic_extents, strides};
#endif
return cuda::std::mdspan<int, decltype(dynamic_extents), cuda::std::layout_stride>(nullptr, map);
};
#endif // TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH

View File

@@ -0,0 +1,303 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/devices>
#include <cuda/memory_pool>
#include "common.cuh"
#include "cuda/__algorithm/copy.h"
C2H_CCCLRT_TEST("1d Copy", "[algorithm]")
{
cuda::stream _stream{cuda::device_ref{0}};
SECTION("Device resource")
{
std::vector<int> host_vector(buffer_size);
{
auto buffer = cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
cuda::fill_bytes(_stream, buffer, fill_byte);
cuda::copy_bytes(_stream, buffer, host_vector);
check_result_and_erase(_stream, host_vector);
cuda::copy_bytes(_stream, buffer, host_vector);
check_result_and_erase(_stream, host_vector);
}
{
auto not_yet_const_buffer =
cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
cuda::fill_bytes(_stream, not_yet_const_buffer, fill_byte);
const auto& const_buffer = not_yet_const_buffer;
cuda::copy_bytes(_stream, const_buffer, host_vector);
check_result_and_erase(_stream, host_vector);
cuda::copy_bytes(_stream, const_buffer, cuda::std::span(host_vector));
check_result_and_erase(_stream, host_vector);
cuda::copy_configuration config;
config.src_location_hint = cuda::device_ref{0};
#if _CCCL_CTK_AT_LEAST(13, 0)
config.src_access_order = cuda::source_access_order::stream;
#else
config.src_access_order = cuda::source_access_order::any;
#endif
cuda::copy_bytes(_stream, const_buffer, host_vector, config);
check_result_and_erase(_stream, host_vector);
cuda::copy_bytes(_stream, const_buffer.first(0), host_vector);
}
}
SECTION("Host and managed resource")
{
{
auto host_buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
auto device_buffer = make_managed_memory_buffer<int>(_stream, buffer_size);
cuda::fill_bytes(_stream, host_buffer, fill_byte);
cuda::copy_bytes(_stream, host_buffer, device_buffer);
check_result_and_erase(_stream, device_buffer);
cuda::copy_bytes(_stream, host_buffer, device_buffer);
check_result_and_erase(_stream, device_buffer);
}
{
auto not_yet_const_host_buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
auto device_buffer = make_managed_memory_buffer<int>(_stream, buffer_size);
cuda::fill_bytes(_stream, not_yet_const_host_buffer, fill_byte);
const auto& const_host_buffer = not_yet_const_host_buffer;
cuda::copy_bytes(_stream, const_host_buffer, device_buffer);
check_result_and_erase(_stream, device_buffer);
cuda::copy_bytes(_stream, const_host_buffer, device_buffer);
check_result_and_erase(_stream, device_buffer);
}
}
SECTION("Asymmetric size")
{
auto host_buffer = make_pinned_memory_buffer<int>(_stream, 1);
cuda::fill_bytes(_stream, host_buffer, fill_byte);
::std::vector<int> vec(buffer_size, 0xbeef);
cuda::copy_bytes(_stream, host_buffer, vec);
_stream.sync();
CCCLRT_REQUIRE(vec[0] == get_expected_value(fill_byte));
CCCLRT_REQUIRE(vec[1] == 0xbeef);
}
}
C2H_CCCLRT_TEST("copy_bytes uses the stream device when current device differs", "[algorithm][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
#if _CCCL_CTK_AT_LEAST(13, 0)
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
if (!explicit_device.attribute(cuda::device_attributes::memory_pools_supported))
{
return;
}
cuda::stream stream{explicit_device};
int expected = get_expected_value(fill_byte);
int result{};
{
auto host_src = make_pinned_memory_buffer<int>(stream, 1, explicit_device);
auto host_dst = make_pinned_memory_buffer<int>(stream, 1, explicit_device);
auto src = cuda::make_device_buffer<int>(stream, explicit_device, 1, cuda::no_init);
auto dst = cuda::make_device_buffer<int>(stream, explicit_device, 1, cuda::no_init);
host_src.get_unsynchronized(0) = expected;
host_dst.get_unsynchronized(0) = 0;
{
cuda::__ensure_current_context guard(explicit_device);
cuda::copy_bytes(stream, host_src, src);
}
{
cuda::__ensure_current_context guard(current_device);
cuda::copy_bytes(stream, src, dst);
}
{
cuda::__ensure_current_context guard(explicit_device);
cuda::copy_bytes(stream, dst, host_dst);
}
stream.sync();
result = host_dst.get_unsynchronized(0);
}
stream.sync();
CCCLRT_REQUIRE(result == expected);
#endif // _CCCL_CTK_AT_LEAST(13, 0)
}
C2H_CCCLRT_TEST("copy_bytes can copy between peer device buffers", "[algorithm][multi_gpu]")
{
// Cross-device copy coverage requires at least two GPUs.
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref source_device{0};
auto peers = source_device.peers();
// This test exercises direct peer memory access; non-peer topologies have no legal device-to-device path to cover.
if (peers.empty())
{
return;
}
cuda::device_ref destination_device = peers.front();
// Device buffers are allocated from stream-ordered memory pools.
if (!source_device.attribute(cuda::device_attributes::memory_pools_supported)
|| !destination_device.attribute(cuda::device_attributes::memory_pools_supported))
{
return;
}
cuda::stream source_stream{source_device};
cuda::stream destination_stream{destination_device};
cuda::device_memory_pool source_pool{source_device};
cuda::device_memory_pool destination_pool{destination_device};
source_pool.enable_access_from(destination_device);
CCCLRT_REQUIRE(source_pool.is_accessible_from(destination_device));
auto source_resource = source_pool.as_ref();
auto destination_resource = destination_pool.as_ref();
int expected = get_expected_value(fill_byte);
int result{};
{
auto host_src = make_pinned_memory_buffer<int>(source_stream, 1, source_device);
auto host_dst = make_pinned_memory_buffer<int>(destination_stream, 1, destination_device);
auto src = cuda::make_buffer<int>(source_stream, source_resource, 1, cuda::no_init);
auto dst = cuda::make_buffer<int>(destination_stream, destination_resource, 1, cuda::no_init);
host_src.get_unsynchronized(0) = expected;
host_dst.get_unsynchronized(0) = 0;
{
cuda::__ensure_current_context guard(source_device);
cuda::copy_bytes(source_stream, host_src, src);
}
source_stream.sync();
cuda::copy_configuration config;
config.src_location_hint = source_device;
config.dst_location_hint = destination_device;
{
cuda::__ensure_current_context guard(destination_device);
cuda::copy_bytes(destination_stream, src, dst, config);
cuda::copy_bytes(destination_stream, dst, host_dst);
}
destination_stream.sync();
result = host_dst.get_unsynchronized(0);
}
source_stream.sync();
destination_stream.sync();
CCCLRT_REQUIRE(result == expected);
}
template <typename SrcLayout = cuda::std::layout_right,
typename DstLayout = SrcLayout,
typename SrcExtents,
typename DstExtents>
void test_mdspan_copy_bytes(
cuda::stream_ref stream, SrcExtents src_extents = SrcExtents(), DstExtents dst_extents = DstExtents())
{
auto src_buffer = make_buffer_for_mdspan<SrcLayout>(stream, src_extents, 1);
auto tmp_buffer = make_buffer_for_mdspan<SrcLayout>(stream, src_extents, 0);
auto dst_buffer = make_buffer_for_mdspan<DstLayout>(stream, dst_extents, 0);
cuda::std::mdspan<int, SrcExtents, SrcLayout> src(src_buffer.data(), src_extents);
cuda::std::mdspan<int, SrcExtents, SrcLayout> tmp(tmp_buffer.data(), src_extents);
cuda::std::mdspan<int, DstExtents, DstLayout> dst(dst_buffer.data(), dst_extents);
for (int i = 0; i < static_cast<int>(src.extent(1)); i++)
{
src(0, i) = i;
}
cuda::copy_bytes(stream, std::move(src), tmp);
cuda::copy_configuration config;
#if _CCCL_CTK_AT_LEAST(13, 0)
config.src_access_order = cuda::source_access_order::stream;
#else
config.src_access_order = cuda::source_access_order::any;
#endif
cuda::copy_bytes(stream, tmp, dst, config);
stream.sync();
for (int i = 0; i < static_cast<int>(dst.extent(1)); i++)
{
CCCLRT_REQUIRE(dst(0, i) == i);
}
}
C2H_CCCLRT_TEST("Mdspan copy", "[algorithm]")
{
cuda::stream stream{cuda::device_ref{0}};
SECTION("Different extents")
{
auto static_extents = cuda::std::extents<size_t, 3, 4>();
test_mdspan_copy_bytes(stream, static_extents, static_extents);
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, static_extents, static_extents);
auto dynamic_extents = cuda::std::dextents<size_t, 2>(3, 4);
test_mdspan_copy_bytes(stream, dynamic_extents, dynamic_extents);
test_mdspan_copy_bytes(stream, static_extents, dynamic_extents);
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, static_extents, dynamic_extents);
auto mixed_extents = cuda::std::extents<int, cuda::std::dynamic_extent, 4>(3);
test_mdspan_copy_bytes(stream, dynamic_extents, mixed_extents);
test_mdspan_copy_bytes(stream, mixed_extents, static_extents);
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, mixed_extents, static_extents);
}
}
C2H_CCCLRT_TEST("Non exhaustive mdspan copy_bytes", "[algorithm]")
{
cuda::stream stream{cuda::device_ref{0}};
{
auto fake_strided_mdspan = create_fake_strided_mdspan();
try
{
cuda::copy_bytes(stream, fake_strided_mdspan, fake_strided_mdspan);
}
catch (const ::std::invalid_argument& e)
{
CHECK(e.what() == ::std::string("copy_bytes supports only exhaustive mdspans"));
}
}
}

View File

@@ -0,0 +1,76 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include "common.cuh"
C2H_CCCLRT_TEST("Fill", "[algorithm]")
{
cuda::stream _stream{cuda::device_ref{0}};
SECTION("Host memory")
{
auto buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
cuda::fill_bytes(_stream, buffer, fill_byte);
check_result_and_erase(_stream, buffer);
}
SECTION("Device memory")
{
auto buffer = cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
cuda::fill_bytes(_stream, buffer, fill_byte);
auto host_vector = make_pinned_memory_buffer<int>(_stream, buffer_size);
cuda::copy_bytes(_stream, buffer, host_vector);
check_result_and_erase(_stream, host_vector);
auto span = buffer.first(0);
cuda::fill_bytes(_stream, span, fill_byte);
}
}
C2H_CCCLRT_TEST("Mdspan Fill", "[algorithm]")
{
cuda::stream stream{cuda::device_ref{0}};
{
cuda::std::dextents<size_t, 3> dynamic_extents{1, 2, 3};
auto buffer = make_buffer_for_mdspan(stream, dynamic_extents, 0);
cuda::std::mdspan<int, decltype(dynamic_extents)> dynamic_mdspan(buffer.data(), dynamic_extents);
cuda::fill_bytes(stream, dynamic_mdspan, fill_byte);
check_result_and_erase(stream, buffer);
}
{
cuda::std::extents<size_t, 2, cuda::std::dynamic_extent, 4> mixed_extents{1};
auto buffer = make_buffer_for_mdspan(stream, mixed_extents, 0);
cuda::std::mdspan<int, decltype(mixed_extents)> mixed_mdspan(buffer.data(), mixed_extents);
cuda::fill_bytes(stream, cuda::std::move(mixed_mdspan), fill_byte);
check_result_and_erase(stream, buffer);
}
}
C2H_CCCLRT_TEST("Non exhaustive mdspan fill_bytes", "[data_manipulation]")
{
cuda::stream stream{cuda::device_ref{0}};
{
auto fake_strided_mdspan = create_fake_strided_mdspan();
try
{
cuda::fill_bytes(stream, fake_strided_mdspan, fill_byte);
}
catch (const ::std::invalid_argument& e)
{
CHECK(e.what() == ::std::string("fill_bytes supports only exhaustive mdspans"));
}
}
}

View File

@@ -0,0 +1,103 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __COMMON_HOST_DEVICE_H__
#define __COMMON_HOST_DEVICE_H__
#include <cuda/hierarchy>
#include "testing.cuh"
template <typename Dims, typename Lambda>
void __global__ lambda_launcher(const Dims dims, const Lambda lambda)
{
lambda(dims);
}
template <typename Comparator, unsigned int FilterArch>
bool arch_filter(const cudaDeviceProp& props)
{
int act_arch = props.major * 10 + props.minor;
if (Comparator()(act_arch, FilterArch))
{
return true;
}
return false;
}
static bool skip_host_exec(bool (* /* filter */)(const cudaDeviceProp&))
{
return false;
}
static bool skip_device_exec(bool (*filter)(const cudaDeviceProp&))
{
cudaDeviceProp props;
CUDART(cudaGetDeviceProperties(&props, 0));
return filter(props);
}
template <typename Dims, typename Lambda, typename... Filters>
void test_host_dev(const Dims& dims, const Lambda& lambda, const Filters&... filters)
{
SECTION("Host execution")
{
if ((... && !skip_host_exec(filters)))
{
// host testing
lambda(dims);
}
}
SECTION("Device execution")
{
// Asymmetrical but cleaner
if ((... || skip_device_exec(filters)))
{
return;
}
cudaLaunchConfig_t config = {};
config.gridDim = {0};
cudaLaunchAttribute attrs[1];
config.attrs = &attrs[0];
config.blockDim = dim3{cuda::gpu_thread.dims(cuda::block, dims)};
config.gridDim = dim3{cuda::block.dims(cuda::grid, dims)};
if constexpr (Dims::has_level(cuda::cluster))
{
dim3 cluster_dims{cuda::block.dims(cuda::cluster, dims)};
config.attrs[config.numAttrs].id = cudaLaunchAttributeClusterDimension;
config.attrs[config.numAttrs].val.clusterDim = {cluster_dims.x, cluster_dims.y, cluster_dims.z};
config.numAttrs = 1;
}
else
{
config.numAttrs = 0;
}
// device testing
CUDART(cudaLaunchKernelEx(&config, lambda_launcher<Dims, Lambda>, dims, lambda));
CUDART(cudaDeviceSynchronize());
}
}
template <typename Fn, typename Tuple>
void apply_each(const Fn& fn, const Tuple& tuple)
{
cuda::std::apply(
[&](const auto&... elems) {
(fn(elems), ...);
},
tuple);
}
#endif // __COMMON_HOST_DEVICE_H__

View File

@@ -0,0 +1,120 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__
#define __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__
#include <cuda/std/detail/__config>
#include <cuda/__driver/driver_api.h>
#include <nv/target>
#include <exception> // IWYU pragma: keep
#include <sstream>
#include "test_macros.h"
#include "utility.cuh"
#include <c2h/catch2_test_helper.h>
#define CUDART(call) REQUIRE((call) == cudaSuccess)
#define CCCLRT_REQUIRE(condition) REQUIRE(condition)
#define CCCLRT_CHECK(condition) CHECK(condition)
#define CCCLRT_FAIL(message) FAIL(message)
#define CCCLRT_CHECK_FALSE(condition) CHECK_FALSE(condition)
// Explicit device side require macros for clang-cuda
#define CCCLRT_REQUIRE_DEVICE(condition) REQUIRE_DEVICE(condition)
#define CCCLRT_CHECK_DEVICE(condition) CHECK(condition)
#define CCCLRT_FAIL_DEVICE(message) FAIL(message)
#define CCCLRT_CHECK_FALSE_DEVICE(condition) CHECK_FALSE(condition)
TEST_FUNC constexpr bool operator==(const dim3& lhs, const dim3& rhs) noexcept
{
return (lhs.x == rhs.x) && (lhs.y == rhs.y) && (lhs.z == rhs.z);
}
namespace Catch
{
template <>
struct StringMaker<dim3>
{
static std::string convert(dim3 const& dims)
{
std::ostringstream oss;
oss << "(" << dims.x << ", " << dims.y << ", " << dims.z << ")";
return oss.str();
}
};
} // namespace Catch
namespace
{
namespace test
{
inline int count_driver_stack()
{
if (::cuda::__driver::__ctxGetCurrent() != nullptr)
{
auto ctx = ::cuda::__driver::__ctxPop();
auto result = 1 + count_driver_stack();
::cuda::__driver::__ctxPush(ctx);
return result;
}
else
{
return 0;
}
}
inline void empty_driver_stack()
{
while (::cuda::__driver::__ctxGetCurrent() != nullptr)
{
::cuda::__driver::__ctxPop();
}
}
inline int cuda_driver_version()
{
return ::cuda::__driver::__getVersion();
}
// Needs to be a template because we use template catch2 macro
template <typename Dummy = void>
struct ccclrt_test_fixture
{
ccclrt_test_fixture()
{
empty_driver_stack();
}
~ccclrt_test_fixture()
{
CCCLRT_CHECK(count_driver_stack() == 0);
}
};
} // namespace test
} // namespace
// Test macro that should be used in all cccl-rt tests
// It first empties the driver stack in case some other test has left it non-empty
// and then runs the test. At the end it checks if it remained empty, which ensures
// we don't accidentally initialize device 0 through CUDART usage and makes sure
// our APIs work with empty driver stack.
#define C2H_CCCLRT_TEST(NAME, TAGS, ...) C2H_TEST_WITH_FIXTURE(::test::ccclrt_test_fixture, NAME, TAGS, __VA_ARGS__)
#define C2H_CCCLRT_TEST_LIST(NAME, TAGS, ...) \
C2H_TEST_LIST_WITH_FIXTURE(::test::ccclrt_test_fixture, NAME, TAGS, __VA_ARGS__)
#endif // __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__

View File

@@ -0,0 +1,192 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef __COMMON_UTILITY_H__
#define __COMMON_UTILITY_H__
#include <cuda_runtime_api.h>
// cuda_runtime_api needs to come first
#include <cuda/__runtime/api_wrapper.h>
#include <cuda/__runtime/ensure_current_context.h>
#include <cuda/__stream/stream_ref.h>
#include <cuda/atomic>
#include <cuda/std/utility>
#include <new> // IWYU pragma: keep (needed for placement new)
#include "test_macros.h"
TEST_DEVICE_FUNC inline void ccclrt_require_impl(
bool condition, const char* condition_text, const char* filename, unsigned int linenum, const char* funcname)
{
if (!condition)
{
// TODO do warp aggregate prints for easier readability?
printf("%s:%u: %s: block: [%d,%d,%d], thread: [%d,%d,%d] Condition `%s` failed.\n",
filename,
linenum,
funcname,
blockIdx.x,
blockIdx.y,
blockIdx.z,
threadIdx.x,
threadIdx.y,
threadIdx.z,
condition_text);
__trap();
}
}
namespace
{
namespace test
{
template <typename T1, typename T2>
T1& assign(T1& t1, T2&& t2)
{
t1 = ::cuda::std::forward<T2>(t2);
return t1;
}
struct _malloc_pinned
{
private:
void* pv = nullptr;
public:
explicit _malloc_pinned(std::size_t size)
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
_CCCL_TRY_CUDA_API(::cudaMallocHost, "failed to allocate pinned memory", &pv, size);
}
~_malloc_pinned()
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
[[maybe_unused]] auto status = ::cudaFreeHost(pv);
}
template <class T>
T* get_as() const noexcept
{
return static_cast<T*>(pv);
}
};
template <class T>
struct pinned
{
private:
_malloc_pinned _mem;
public:
explicit pinned(T t)
: _mem(sizeof(T))
{
::new (_mem.get_as<void>()) T(std::move(t));
}
~pinned()
{
get()->~T();
}
T* get() noexcept
{
return _mem.get_as<T>();
}
const T* get() const noexcept
{
return _mem.get_as<T>();
}
T& operator*() noexcept
{
return *get();
}
const T& operator*() const noexcept
{
return *get();
}
};
template <int N>
struct assign_n
{
TEST_DEVICE_FUNC constexpr void operator()(int* pi) const noexcept
{
*pi = N;
}
};
template <int N>
struct verify_n
{
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
{
// TODO: fix clang CUDA require macro
// CCCLRT_REQUIRE(*pi == N);
ccclrt_require_impl(*pi == N, "*pi == N", __FILE__, __LINE__, __PRETTY_FUNCTION__);
}
};
using assign_42 = assign_n<42>;
using verify_42 = verify_n<42>;
struct atomic_add_one
{
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
{
cuda::atomic_ref atomic_pi(*pi);
atomic_pi.fetch_add(1);
}
};
struct atomic_sub_one
{
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
{
cuda::atomic_ref atomic_pi(*pi);
atomic_pi.fetch_sub(1);
}
};
struct spin_until_80
{
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
{
cuda::atomic_ref atomic_pi(*pi);
while (atomic_pi.load() != 80)
;
}
};
struct empty_kernel
{
TEST_DEVICE_FUNC void operator()() const noexcept {}
};
template <class Fn, class... Args>
static __global__ void kernel_launcher(Fn fn, Args... args)
{
fn(args...);
}
template <class Fn, class... Args>
void launch_kernel_single_thread(cuda::stream_ref stream, Fn fn, Args... args)
{
cuda::__ensure_current_context guard(stream);
kernel_launcher<<<1, 1, 0, stream.get()>>>(fn, args...);
assert(cudaGetLastError() == cudaSuccess);
}
} // namespace test
} // namespace
#endif // __COMMON_UTILITY_H__

View File

@@ -0,0 +1,65 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#include <cuda/std/array>
#include <cuda/std/cassert>
#include <cuda/std/cstddef>
#include "test_macros.h"
TEST_FUNC constexpr bool test()
{
// 1. Test signature.
static_assert(cuda::std::__is_cuda_std_array_v<decltype(cuda::__all_arch_ids())>);
static_assert(noexcept(cuda::__all_arch_ids()));
// 2. Test that all values are present.
const auto all_arch_ids = cuda::__all_arch_ids();
cuda::std::size_t i = 0;
assert(all_arch_ids[i++] == cuda::arch_id::sm_50);
assert(all_arch_ids[i++] == cuda::arch_id::sm_52);
assert(all_arch_ids[i++] == cuda::arch_id::sm_53);
assert(all_arch_ids[i++] == cuda::arch_id::sm_60);
assert(all_arch_ids[i++] == cuda::arch_id::sm_61);
assert(all_arch_ids[i++] == cuda::arch_id::sm_62);
assert(all_arch_ids[i++] == cuda::arch_id::sm_70);
assert(all_arch_ids[i++] == cuda::arch_id::sm_75);
assert(all_arch_ids[i++] == cuda::arch_id::sm_80);
assert(all_arch_ids[i++] == cuda::arch_id::sm_86);
assert(all_arch_ids[i++] == cuda::arch_id::sm_87);
assert(all_arch_ids[i++] == cuda::arch_id::sm_88);
assert(all_arch_ids[i++] == cuda::arch_id::sm_89);
assert(all_arch_ids[i++] == cuda::arch_id::sm_90);
assert(all_arch_ids[i++] == cuda::arch_id::sm_100);
assert(all_arch_ids[i++] == cuda::arch_id::sm_103);
assert(all_arch_ids[i++] == cuda::arch_id::sm_110);
assert(all_arch_ids[i++] == cuda::arch_id::sm_120);
assert(all_arch_ids[i++] == cuda::arch_id::sm_121);
assert(all_arch_ids[i++] == cuda::arch_id::sm_90a);
assert(all_arch_ids[i++] == cuda::arch_id::sm_100a);
assert(all_arch_ids[i++] == cuda::arch_id::sm_103a);
assert(all_arch_ids[i++] == cuda::arch_id::sm_110a);
assert(all_arch_ids[i++] == cuda::arch_id::sm_120a);
assert(all_arch_ids[i++] == cuda::arch_id::sm_121a);
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,151 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include <cuda/std/utility>
#include "test_macros.h"
TEST_DEVICE_FUNC void test_current()
{
// 1. Test cuda::device::current_arch_id() signature.
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::device::current_arch_id())>);
static_assert(noexcept(cuda::device::current_arch_id()));
// 2. Test cuda::device::current_arch_id() in constexpr context. Unsupported with nvc++ -cuda.
#if !_CCCL_CUDA_COMPILER(NVHPC)
if constexpr (cuda::device::current_arch_id() == cuda::arch_id{})
{
// cuda::arch_id{} is an invalid architecture, so this statement should be unrachable.
assert(false);
}
#endif // !_CCCL_CUDA_COMPILER(NVHPC)
// 3. Test cuda::device::current_arch_id() against the NV_IF_TARGET macros
[[maybe_unused]] cuda::arch_id arch = cuda::device::current_arch_id();
NV_DISPATCH_TARGET(
NV_HAS_FEATURE_SM_90a,
(assert(arch == cuda::arch_id::sm_90a); return;),
NV_HAS_FEATURE_SM_100a,
(assert(arch == cuda::arch_id::sm_100a); return;),
NV_HAS_FEATURE_SM_103a,
(assert(arch == cuda::arch_id::sm_103a); return;),
NV_HAS_FEATURE_SM_110a,
(assert(arch == cuda::arch_id::sm_110a); return;),
NV_HAS_FEATURE_SM_120a,
(assert(arch == cuda::arch_id::sm_120a); return;),
NV_HAS_FEATURE_SM_121a,
(assert(arch == cuda::arch_id::sm_121a); return;))
NV_DISPATCH_TARGET(
NV_IS_EXACTLY_SM_60,
(assert(arch == cuda::arch_id::sm_60); return;),
NV_IS_EXACTLY_SM_61,
(assert(arch == cuda::arch_id::sm_61); return;),
NV_IS_EXACTLY_SM_62,
(assert(arch == cuda::arch_id::sm_62); return;),
NV_IS_EXACTLY_SM_70,
(assert(arch == cuda::arch_id::sm_70); return;),
NV_IS_EXACTLY_SM_75,
(assert(arch == cuda::arch_id::sm_75); return;),
NV_IS_EXACTLY_SM_80,
(assert(arch == cuda::arch_id::sm_80); return;),
NV_IS_EXACTLY_SM_86,
(assert(arch == cuda::arch_id::sm_86); return;),
NV_IS_EXACTLY_SM_87,
(assert(arch == cuda::arch_id::sm_87); return;),
NV_IS_EXACTLY_SM_88,
(assert(arch == cuda::arch_id::sm_88); return;),
NV_IS_EXACTLY_SM_89,
(assert(arch == cuda::arch_id::sm_89); return;))
NV_DISPATCH_TARGET(
NV_IS_EXACTLY_SM_90,
(assert(arch == cuda::arch_id::sm_90); return;),
NV_IS_EXACTLY_SM_100,
(assert(arch == cuda::arch_id::sm_100); return;),
NV_IS_EXACTLY_SM_103,
(assert(arch == cuda::arch_id::sm_103); return;),
NV_IS_EXACTLY_SM_110,
(assert(arch == cuda::arch_id::sm_110); return;),
NV_IS_EXACTLY_SM_120,
(assert(arch == cuda::arch_id::sm_120); return;),
NV_IS_EXACTLY_SM_121,
(assert(arch == cuda::arch_id::sm_121); return;),
NV_ANY_TARGET,
(assert(false);) // fail for unknown architecture
)
}
TEST_FUNC constexpr bool test()
{
// 1. Test cuda::arch_id enum values.
static_assert(cuda::std::is_scoped_enum_v<cuda::arch_id>);
static_assert(cuda::std::is_same_v<cuda::std::underlying_type_t<cuda::arch_id>, int>);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_60) == 60);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_61) == 61);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_62) == 62);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_70) == 70);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_75) == 75);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_80) == 80);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_86) == 86);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_87) == 87);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_88) == 88);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_89) == 89);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_90) == 90);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_100) == 100);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_103) == 103);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_110) == 110);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_120) == 120);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_121) == 121);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_90a) == 90 * 100000);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_100a) == 100 * 100000);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_103a) == 103 * 100000);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_110a) == 110 * 100000);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_120a) == 120 * 100000);
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_121a) == 121 * 100000);
// 2. Test cuda::to_arch_id(cuda::compute_capability).
{
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::to_arch_id(cuda::compute_capability{}))>);
static_assert(noexcept(cuda::to_arch_id(cuda::compute_capability{})));
cuda::arch_id id_lowest = cuda::to_arch_id(cuda::compute_capability{60});
assert(id_lowest == cuda::arch_id::sm_60);
cuda::arch_id id_highest = cuda::to_arch_id(cuda::compute_capability{120});
assert(id_highest == cuda::arch_id::sm_120);
}
// 3. Test cuda::to_arch_specific_id(cuda::compute_capability).
{
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::to_arch_specific_id(cuda::compute_capability{}))>);
static_assert(noexcept(cuda::to_arch_specific_id(cuda::compute_capability{})));
cuda::arch_id id_lowest = cuda::to_arch_specific_id(cuda::compute_capability{90});
assert(id_lowest == cuda::arch_id::sm_90a);
cuda::arch_id id_highest = cuda::to_arch_specific_id(cuda::compute_capability{120});
assert(id_highest == cuda::arch_id::sm_120a);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
NV_IF_TARGET(NV_IS_DEVICE, (test_current();))
return 0;
}

View File

@@ -0,0 +1,63 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#if __cpp_lib_format >= 201907L
# include <format>
#endif // __cpp_lib_format >= 201907L
#include "literal.h"
#if __cpp_lib_format >= 201907L
template <class C>
void test()
{
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_60) == TEST_STRLIT(C, "sm_60"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_61) == TEST_STRLIT(C, "sm_61"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_62) == TEST_STRLIT(C, "sm_62"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_70) == TEST_STRLIT(C, "sm_70"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_75) == TEST_STRLIT(C, "sm_75"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_80) == TEST_STRLIT(C, "sm_80"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_86) == TEST_STRLIT(C, "sm_86"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_87) == TEST_STRLIT(C, "sm_87"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_88) == TEST_STRLIT(C, "sm_88"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_89) == TEST_STRLIT(C, "sm_89"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_90) == TEST_STRLIT(C, "sm_90"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_100) == TEST_STRLIT(C, "sm_100"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_103) == TEST_STRLIT(C, "sm_103"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_110) == TEST_STRLIT(C, "sm_110"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_120) == TEST_STRLIT(C, "sm_120"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_121) == TEST_STRLIT(C, "sm_121"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_90a) == TEST_STRLIT(C, "sm_90a"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_100a) == TEST_STRLIT(C, "sm_100a"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_103a) == TEST_STRLIT(C, "sm_103a"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_110a) == TEST_STRLIT(C, "sm_110a"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_120a) == TEST_STRLIT(C, "sm_120a"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_121a) == TEST_STRLIT(C, "sm_121a"));
}
void test()
{
test<char>();
test<wchar_t>();
}
#endif // __cpp_lib_format >= 201907L
int main(int, char**)
{
#if __cpp_lib_format >= 201907L
NV_IF_TARGET(NV_IS_HOST, (test();))
#endif // __cpp_lib_format >= 201907L
return 0;
}

View File

@@ -0,0 +1,182 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/devices>
#include <cuda/std/cassert>
#include <testing.cuh>
#include "test_macros.h"
template <class T>
TEST_DEVICE_FUNC T foo(const T& x)
{
return x;
}
template <cuda::arch_id Arch>
__global__ void arch_specific_kernel_mock_do_not_launch()
{
assert(Arch == cuda::device::current_arch_id());
// I will try to pack something like this into an API
if constexpr (cuda::arch_traits<Arch>().compute_capability != cuda::device::current_arch_traits().compute_capability)
{
return;
}
[[maybe_unused]] __shared__ int array[cuda::arch_traits<Arch>().max_shared_memory_per_block / sizeof(int)];
// constexpr is useless and I can't use intrinsics here :(
if (cuda::device::current_arch_traits().cluster_supported)
{
[[maybe_unused]] int dummy;
asm volatile("mov.u32 %0, %%cluster_ctarank;" : "=r"(dummy));
}
if (cuda::device::current_arch_traits().redux_intrinisic)
{
[[maybe_unused]] int dummy1 = 0, dummy2 = 0;
asm volatile("redux.sync.add.s32 %0, %1, 0xffffffff;" : "=r"(dummy1) : "r"(dummy2));
}
if (cuda::device::current_arch_traits().cp_async_supported)
{
asm volatile("cp.async.commit_group;");
}
// Confirm trait value is defined device code and usable as a reference
foo(cuda::arch_traits<Arch>().compute_capability);
foo(cuda::device::current_arch_traits().compute_capability);
}
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_70>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_75>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_80>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_86>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_87>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_88>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_89>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_90>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_100>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_103>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_110>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_120>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_121>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_90a>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_100a>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_103a>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_110a>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_120a>();
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_121a>();
template <int ComputeCapability>
void constexpr compare_static_and_dynamic()
{
constexpr cuda::compute_capability cc{ComputeCapability};
constexpr cuda::arch_traits_t static_traits = cuda::arch_traits<cuda::to_arch_id(cc)>();
constexpr cuda::arch_traits_t dynamic_traits = cuda::arch_traits_for(cuda::to_arch_id(cc));
static_assert(static_traits.arch_id == dynamic_traits.arch_id);
static_assert(static_traits.max_threads_per_block == dynamic_traits.max_threads_per_block);
static_assert(static_traits.max_block_dim_x == dynamic_traits.max_block_dim_x);
static_assert(static_traits.max_block_dim_y == dynamic_traits.max_block_dim_y);
static_assert(static_traits.max_block_dim_z == dynamic_traits.max_block_dim_z);
static_assert(static_traits.max_grid_dim_x == dynamic_traits.max_grid_dim_x);
static_assert(static_traits.max_grid_dim_y == dynamic_traits.max_grid_dim_y);
static_assert(static_traits.max_grid_dim_z == dynamic_traits.max_grid_dim_z);
static_assert(static_traits.warp_size == dynamic_traits.warp_size);
static_assert(static_traits.total_constant_memory == dynamic_traits.total_constant_memory);
static_assert(static_traits.max_resident_grids == dynamic_traits.max_resident_grids);
static_assert(static_traits.max_shared_memory_per_block == dynamic_traits.max_shared_memory_per_block);
static_assert(static_traits.gpu_overlap == dynamic_traits.gpu_overlap);
static_assert(static_traits.can_map_host_memory == dynamic_traits.can_map_host_memory);
static_assert(static_traits.concurrent_kernels == dynamic_traits.concurrent_kernels);
static_assert(static_traits.stream_priorities_supported == dynamic_traits.stream_priorities_supported);
static_assert(static_traits.global_l1_cache_supported == dynamic_traits.global_l1_cache_supported);
static_assert(static_traits.local_l1_cache_supported == dynamic_traits.local_l1_cache_supported);
static_assert(static_traits.max_registers_per_block == dynamic_traits.max_registers_per_block);
static_assert(static_traits.max_registers_per_multiprocessor == dynamic_traits.max_registers_per_multiprocessor);
static_assert(static_traits.compute_capability == dynamic_traits.compute_capability);
static_assert(static_traits.compute_capability_major == dynamic_traits.compute_capability_major);
static_assert(static_traits.compute_capability_minor == dynamic_traits.compute_capability_minor);
static_assert(static_traits.compute_capability == dynamic_traits.compute_capability);
static_assert(
static_traits.max_shared_memory_per_multiprocessor == dynamic_traits.max_shared_memory_per_multiprocessor);
static_assert(static_traits.max_blocks_per_multiprocessor == dynamic_traits.max_blocks_per_multiprocessor);
static_assert(static_traits.max_warps_per_multiprocessor == dynamic_traits.max_warps_per_multiprocessor);
static_assert(static_traits.max_threads_per_multiprocessor == dynamic_traits.max_threads_per_multiprocessor);
static_assert(static_traits.reserved_shared_memory_per_block == dynamic_traits.reserved_shared_memory_per_block);
static_assert(static_traits.max_shared_memory_per_block_optin == dynamic_traits.max_shared_memory_per_block_optin);
static_assert(static_traits.cluster_supported == dynamic_traits.cluster_supported);
static_assert(static_traits.redux_intrinisic == dynamic_traits.redux_intrinisic);
static_assert(static_traits.elect_intrinsic == dynamic_traits.elect_intrinsic);
static_assert(static_traits.cp_async_supported == dynamic_traits.cp_async_supported);
static_assert(static_traits.tma_supported == dynamic_traits.tma_supported);
}
C2H_CCCLRT_TEST("Traits", "[device]")
{
compare_static_and_dynamic<70>();
compare_static_and_dynamic<75>();
compare_static_and_dynamic<80>();
compare_static_and_dynamic<86>();
compare_static_and_dynamic<89>();
compare_static_and_dynamic<90>();
compare_static_and_dynamic<100>();
compare_static_and_dynamic<103>();
compare_static_and_dynamic<110>();
compare_static_and_dynamic<120>();
// Compare arch traits with attributes
for (const cuda::device_ref& dev : cuda::devices)
{
const auto cc = dev.attribute(cuda::device_attributes::compute_capability);
const auto traits = cuda::arch_traits_for(cc);
CCCLRT_REQUIRE(traits.max_threads_per_block == dev.attribute(cuda::device_attributes::max_threads_per_block));
CCCLRT_REQUIRE(traits.max_block_dim_x == dev.attribute(cuda::device_attributes::max_block_dim_x));
CCCLRT_REQUIRE(traits.max_block_dim_y == dev.attribute(cuda::device_attributes::max_block_dim_y));
CCCLRT_REQUIRE(traits.max_block_dim_z == dev.attribute(cuda::device_attributes::max_block_dim_z));
CCCLRT_REQUIRE(traits.max_grid_dim_x == dev.attribute(cuda::device_attributes::max_grid_dim_x));
CCCLRT_REQUIRE(traits.max_grid_dim_y == dev.attribute(cuda::device_attributes::max_grid_dim_y));
CCCLRT_REQUIRE(traits.max_grid_dim_z == dev.attribute(cuda::device_attributes::max_grid_dim_z));
CCCLRT_REQUIRE(traits.warp_size == dev.attribute(cuda::device_attributes::warp_size));
CCCLRT_REQUIRE(traits.total_constant_memory == dev.attribute(cuda::device_attributes::total_constant_memory));
CCCLRT_REQUIRE(
traits.max_shared_memory_per_block == dev.attribute(cuda::device_attributes::max_shared_memory_per_block));
CCCLRT_REQUIRE(traits.gpu_overlap == dev.attribute(cuda::device_attributes::gpu_overlap));
CCCLRT_REQUIRE(traits.can_map_host_memory == dev.attribute(cuda::device_attributes::can_map_host_memory));
CCCLRT_REQUIRE(traits.concurrent_kernels == dev.attribute(cuda::device_attributes::concurrent_kernels));
CCCLRT_REQUIRE(
traits.stream_priorities_supported == dev.attribute(cuda::device_attributes::stream_priorities_supported));
CCCLRT_REQUIRE(
traits.global_l1_cache_supported == dev.attribute(cuda::device_attributes::global_l1_cache_supported));
CCCLRT_REQUIRE(traits.local_l1_cache_supported == dev.attribute(cuda::device_attributes::local_l1_cache_supported));
CCCLRT_REQUIRE(traits.max_registers_per_block == dev.attribute(cuda::device_attributes::max_registers_per_block));
CCCLRT_REQUIRE(traits.max_registers_per_multiprocessor
== dev.attribute(cuda::device_attributes::max_registers_per_multiprocessor));
CCCLRT_REQUIRE(traits.compute_capability_major == dev.attribute(cuda::device_attributes::compute_capability_major));
CCCLRT_REQUIRE(traits.compute_capability_minor == dev.attribute(cuda::device_attributes::compute_capability_minor));
CCCLRT_REQUIRE(traits.compute_capability == dev.attribute(cuda::device_attributes::compute_capability));
CCCLRT_REQUIRE(traits.max_shared_memory_per_multiprocessor
== dev.attribute(cuda::device_attributes::max_shared_memory_per_multiprocessor));
CCCLRT_REQUIRE(
traits.max_blocks_per_multiprocessor == dev.attribute(cuda::device_attributes::max_blocks_per_multiprocessor));
CCCLRT_REQUIRE(
traits.max_threads_per_multiprocessor == dev.attribute(cuda::device_attributes::max_threads_per_multiprocessor));
CCCLRT_REQUIRE(traits.reserved_shared_memory_per_block
== dev.attribute(cuda::device_attributes::reserved_shared_memory_per_block));
CCCLRT_REQUIRE(traits.max_shared_memory_per_block_optin
== dev.attribute(cuda::device_attributes::max_shared_memory_per_block_optin));
}
}

View File

@@ -0,0 +1,256 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
#include <cuda/devices>
#include <cuda/std/cassert>
#include <cuda/std/type_traits>
#include "test_macros.h"
TEST_DEVICE_FUNC void test_current()
{
// 1. Test cuda::device::current_compute_capability() signature.
static_assert(cuda::std::is_same_v<cuda::compute_capability, decltype(cuda::device::current_compute_capability())>);
static_assert(noexcept(cuda::device::current_compute_capability()));
// 1. Test cuda::device::current_compute_capability() in constexpr context. Unsupported with nvc++ -cuda.
#if !_CCCL_CUDA_COMPILER(NVHPC)
if constexpr (cuda::device::current_compute_capability() == cuda::compute_capability{})
{
// cuda::current_compute_capability{} is an invalid compute capability, so this statement should be unrachable.
assert(false);
}
#endif // !_CCCL_CUDA_COMPILER(NVHPC)
// 2. Test cuda::device::current_compute_capability() against the NV_IF_TARGET macros
[[maybe_unused]] cuda::compute_capability cc = cuda::device::current_compute_capability();
NV_DISPATCH_TARGET(
NV_IS_EXACTLY_SM_60,
(assert(cc == cuda::compute_capability{60}); return;),
NV_IS_EXACTLY_SM_61,
(assert(cc == cuda::compute_capability{61}); return;),
NV_IS_EXACTLY_SM_62,
(assert(cc == cuda::compute_capability{62}); return;),
NV_IS_EXACTLY_SM_70,
(assert(cc == cuda::compute_capability{70}); return;),
NV_IS_EXACTLY_SM_75,
(assert(cc == cuda::compute_capability{75}); return;),
NV_IS_EXACTLY_SM_80,
(assert(cc == cuda::compute_capability{80}); return;),
NV_IS_EXACTLY_SM_86,
(assert(cc == cuda::compute_capability{86}); return;),
NV_IS_EXACTLY_SM_87,
(assert(cc == cuda::compute_capability{87}); return;),
NV_IS_EXACTLY_SM_88,
(assert(cc == cuda::compute_capability{88}); return;),
NV_IS_EXACTLY_SM_89,
(assert(cc == cuda::compute_capability{89}); return;))
NV_DISPATCH_TARGET(
NV_IS_EXACTLY_SM_90,
(assert(cc == cuda::compute_capability{90}); return;),
NV_IS_EXACTLY_SM_100,
(assert(cc == cuda::compute_capability{100}); return;),
NV_IS_EXACTLY_SM_103,
(assert(cc == cuda::compute_capability{103}); return;),
NV_IS_EXACTLY_SM_110,
(assert(cc == cuda::compute_capability{110}); return;),
NV_IS_EXACTLY_SM_120,
(assert(cc == cuda::compute_capability{120}); return;),
NV_IS_EXACTLY_SM_121,
(assert(cc == cuda::compute_capability{121}); return;),
NV_ANY_TARGET,
(assert(false);) // fail for unknown compute capability
)
}
TEST_FUNC constexpr bool test()
{
// 1. Test default constructor.
{
static_assert(cuda::std::is_nothrow_default_constructible_v<cuda::compute_capability>);
cuda::compute_capability cc;
assert(cc.get() == 0);
}
// 2. Test constructor from compute capability in format 10 * major + minor.
{
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, int>);
static_assert(!cuda::std::is_convertible_v<int, cuda::compute_capability>);
cuda::compute_capability cc{148};
assert(cc.get() == 148);
}
// 3. Test constructor from major and minor.
{
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, int, int>);
cuda::compute_capability cc{8, 9};
assert(cc.get() == 89);
}
// 4. Test constructor from cuda::arch_id.
{
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, cuda::arch_id>);
static_assert(!cuda::std::is_convertible_v<cuda::arch_id, cuda::compute_capability>);
cuda::compute_capability cc1{cuda::arch_id::sm_100};
assert(cc1.get() == 100);
cuda::compute_capability cc2{cuda::arch_id::sm_100a};
assert(cc2.get() == 100);
}
// 5. Test copy constructor.
{
static_assert(cuda::std::is_trivially_copy_constructible_v<cuda::compute_capability>);
const cuda::compute_capability cc1{cuda::arch_id::sm_100};
cuda::compute_capability cc2{cc1};
assert(cc1.get() == 100);
assert(cc2.get() == 100);
}
// 6. Test assignment operator.
{
static_assert(cuda::std::is_nothrow_copy_assignable_v<cuda::compute_capability>);
const cuda::compute_capability cc1{cuda::arch_id::sm_100};
cuda::compute_capability cc2;
assert(cc1.get() == 100);
assert(cc2.get() == 0);
cc2 = cc1;
assert(cc1.get() == 100);
assert(cc2.get() == 100);
}
// 7. Test get().
{
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.get())>);
static_assert(noexcept(cuda::compute_capability{}.get()));
const cuda::compute_capability cc{cuda::arch_id::sm_100};
assert(cc.get() == 100);
}
// 8. Test major_cap().
{
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.major_cap())>);
static_assert(noexcept(cuda::compute_capability{}.major_cap()));
const cuda::compute_capability cc{cuda::arch_id::sm_100};
assert(cc.major_cap() == 10);
// Test deprecated major().
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.major())>);
static_assert(noexcept(cuda::compute_capability{}.major()));
assert(cc.major() == cc.major_cap());
}
// 9. Test minor_cap().
{
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.minor_cap())>);
static_assert(noexcept(cuda::compute_capability{}.minor_cap()));
const cuda::compute_capability cc{cuda::arch_id::sm_89};
assert(cc.minor_cap() == 9);
// Test deprecated minor().
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.minor())>);
static_assert(noexcept(cuda::compute_capability{}.minor()));
assert(cc.minor() == cc.minor_cap());
}
// 10. operator int()
{
static_assert(noexcept(static_cast<int>(cuda::compute_capability{})));
static_assert(!cuda::std::is_convertible_v<cuda::compute_capability, int>);
const cuda::compute_capability cc{cuda::arch_id::sm_89};
assert(static_cast<int>(cc) == 89);
}
// 11. comparison operators
{
static_assert(
cuda::std::is_same_v<bool, decltype(operator==(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(
cuda::std::is_same_v<bool, decltype(operator!=(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(
cuda::std::is_same_v<bool, decltype(operator<(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(
cuda::std::is_same_v<bool, decltype(operator<=(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(
cuda::std::is_same_v<bool, decltype(operator>(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(
cuda::std::is_same_v<bool, decltype(operator>=(cuda::compute_capability{}, cuda::compute_capability{}))>);
static_assert(noexcept(operator==(cuda::compute_capability{}, cuda::compute_capability{})));
static_assert(noexcept(operator!=(cuda::compute_capability{}, cuda::compute_capability{})));
static_assert(noexcept(operator<(cuda::compute_capability{}, cuda::compute_capability{})));
static_assert(noexcept(operator<=(cuda::compute_capability{}, cuda::compute_capability{})));
static_assert(noexcept(operator>(cuda::compute_capability{}, cuda::compute_capability{})));
static_assert(noexcept(operator>=(cuda::compute_capability{}, cuda::compute_capability{})));
const cuda::compute_capability cc1{127};
const cuda::compute_capability cc2{43};
assert(cc1 == cc1);
assert(cc2 == cc2);
assert(cc1 != cc2);
assert(cc2 != cc1);
assert(!(cc1 < cc1));
assert(!(cc1 < cc2));
assert(!(cc2 < cc2));
assert(cc2 < cc1);
assert(cc1 <= cc1);
assert(!(cc1 <= cc2));
assert(cc2 <= cc2);
assert(cc2 <= cc1);
assert(!(cc1 > cc1));
assert(cc1 > cc2);
assert(!(cc2 > cc2));
assert(!(cc2 > cc1));
assert(cc1 >= cc1);
assert(cc1 > cc2);
assert(cc2 >= cc2);
assert(!(cc2 > cc1));
}
// 12. Test that cuda::compute_capability is a structural type.
#if _CCCL_STD_VER >= 2020
{
[[maybe_unused]] constexpr auto val =
cuda::std::integral_constant<cuda::compute_capability, cuda::compute_capability{100}>{};
}
#endif // _CCCL_STD_VER >= 2020
return true;
}
int main(int, char**)
{
test();
static_assert(test());
NV_IF_TARGET(NV_IS_DEVICE, (test_current();))
return 0;
}

View File

@@ -0,0 +1,58 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#if __cpp_lib_format >= 201907L
# include <format>
#endif // __cpp_lib_format >= 201907L
#include "literal.h"
#if __cpp_lib_format >= 201907L
template <class C>
void test()
{
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{0}) == TEST_STRLIT(C, "0"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{60}) == TEST_STRLIT(C, "60"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{61}) == TEST_STRLIT(C, "61"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{62}) == TEST_STRLIT(C, "62"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{70}) == TEST_STRLIT(C, "70"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{75}) == TEST_STRLIT(C, "75"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{80}) == TEST_STRLIT(C, "80"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{86}) == TEST_STRLIT(C, "86"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{87}) == TEST_STRLIT(C, "87"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{88}) == TEST_STRLIT(C, "88"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{89}) == TEST_STRLIT(C, "89"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{90}) == TEST_STRLIT(C, "90"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{100}) == TEST_STRLIT(C, "100"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{103}) == TEST_STRLIT(C, "103"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{110}) == TEST_STRLIT(C, "110"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{120}) == TEST_STRLIT(C, "120"));
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{121}) == TEST_STRLIT(C, "121"));
}
void test()
{
test<char>();
test<wchar_t>();
}
#endif // __cpp_lib_format >= 201907L
int main(int, char**)
{
#if __cpp_lib_format >= 201907L
NV_IF_TARGET(NV_IS_HOST, (test();))
#endif // __cpp_lib_format >= 201907L
return 0;
}

View File

@@ -0,0 +1,394 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/__driver/driver_api.h>
#include <cuda/__runtime/ensure_current_context.h>
#include <cuda/devices>
#include <cuda/std/__type_traits/is_same.h>
#include <cuda/std/cstddef>
#include <testing.cuh>
namespace
{
template <const auto& Attr, ::cudaDeviceAttr ExpectedAttr, class ExpectedResult>
[[maybe_unused]] auto test_device_attribute()
{
cuda::device_ref dev0(0);
STATIC_REQUIRE(Attr == ExpectedAttr);
STATIC_REQUIRE(::cuda::std::is_same_v<cuda::device_attribute_result_t<Attr>, ExpectedResult>);
auto result = dev0.attribute(Attr);
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(result), ExpectedResult>);
CCCLRT_REQUIRE(result == dev0.attribute<ExpectedAttr>());
CCCLRT_REQUIRE(result == Attr(dev0));
return result;
}
} // namespace
C2H_CCCLRT_TEST("init", "[device]")
{
cuda::device_ref dev{0};
dev.init();
CCCLRT_REQUIRE(cuda::__driver::__isPrimaryCtxActive(cuda::__driver::__deviceGet(0)));
}
C2H_CCCLRT_TEST("Smoke", "[device]")
{
namespace attributes = cuda::device_attributes;
using cuda::device_ref;
SECTION("Compare")
{
CCCLRT_REQUIRE(device_ref{0} == device_ref{0});
CCCLRT_REQUIRE(device_ref{0} == 0);
CCCLRT_REQUIRE(0 == device_ref{0});
if (cuda::devices.size() > 1)
{
CCCLRT_REQUIRE(device_ref{1} != device_ref{0});
CCCLRT_REQUIRE(device_ref{1} != 0);
CCCLRT_REQUIRE(0 != device_ref{1});
}
}
SECTION("Attributes")
{
::test_device_attribute<attributes::max_threads_per_block, ::cudaDevAttrMaxThreadsPerBlock, int>();
::test_device_attribute<attributes::max_block_dim_x, ::cudaDevAttrMaxBlockDimX, int>();
::test_device_attribute<attributes::max_block_dim_y, ::cudaDevAttrMaxBlockDimY, int>();
::test_device_attribute<attributes::max_block_dim_z, ::cudaDevAttrMaxBlockDimZ, int>();
::test_device_attribute<attributes::max_grid_dim_x, ::cudaDevAttrMaxGridDimX, int>();
::test_device_attribute<attributes::max_grid_dim_y, ::cudaDevAttrMaxGridDimY, int>();
::test_device_attribute<attributes::max_grid_dim_z, ::cudaDevAttrMaxGridDimZ, int>();
::test_device_attribute<attributes::max_shared_memory_per_block,
::cudaDevAttrMaxSharedMemoryPerBlock,
cuda::std::size_t>();
::test_device_attribute<attributes::total_constant_memory, ::cudaDevAttrTotalConstantMemory, cuda::std::size_t>();
::test_device_attribute<attributes::warp_size, ::cudaDevAttrWarpSize, int>();
::test_device_attribute<attributes::max_pitch, ::cudaDevAttrMaxPitch, cuda::std::size_t>();
::test_device_attribute<attributes::max_texture_1d_width, ::cudaDevAttrMaxTexture1DWidth, int>();
::test_device_attribute<attributes::max_texture_1d_linear_width, ::cudaDevAttrMaxTexture1DLinearWidth, int>();
::test_device_attribute<attributes::max_texture_1d_mipmapped_width, ::cudaDevAttrMaxTexture1DMipmappedWidth, int>();
::test_device_attribute<attributes::max_texture_2d_width, ::cudaDevAttrMaxTexture2DWidth, int>();
::test_device_attribute<attributes::max_texture_2d_height, ::cudaDevAttrMaxTexture2DHeight, int>();
::test_device_attribute<attributes::max_texture_2d_linear_width, ::cudaDevAttrMaxTexture2DLinearWidth, int>();
::test_device_attribute<attributes::max_texture_2d_linear_height, ::cudaDevAttrMaxTexture2DLinearHeight, int>();
::test_device_attribute<attributes::max_texture_2d_linear_pitch,
::cudaDevAttrMaxTexture2DLinearPitch,
cuda::std::size_t>();
::test_device_attribute<attributes::max_texture_2d_mipmapped_width, ::cudaDevAttrMaxTexture2DMipmappedWidth, int>();
::test_device_attribute<attributes::max_texture_2d_mipmapped_height, ::cudaDevAttrMaxTexture2DMipmappedHeight, int>();
::test_device_attribute<attributes::max_texture_3d_width, ::cudaDevAttrMaxTexture3DWidth, int>();
::test_device_attribute<attributes::max_texture_3d_height, ::cudaDevAttrMaxTexture3DHeight, int>();
::test_device_attribute<attributes::max_texture_3d_depth, ::cudaDevAttrMaxTexture3DDepth, int>();
::test_device_attribute<attributes::max_texture_3d_width_alt, ::cudaDevAttrMaxTexture3DWidthAlt, int>();
::test_device_attribute<attributes::max_texture_3d_height_alt, ::cudaDevAttrMaxTexture3DHeightAlt, int>();
::test_device_attribute<attributes::max_texture_3d_depth_alt, ::cudaDevAttrMaxTexture3DDepthAlt, int>();
::test_device_attribute<attributes::max_texture_cubemap_width, ::cudaDevAttrMaxTextureCubemapWidth, int>();
::test_device_attribute<attributes::max_texture_1d_layered_width, ::cudaDevAttrMaxTexture1DLayeredWidth, int>();
::test_device_attribute<attributes::max_texture_1d_layered_layers, ::cudaDevAttrMaxTexture1DLayeredLayers, int>();
::test_device_attribute<attributes::max_texture_2d_layered_width, ::cudaDevAttrMaxTexture2DLayeredWidth, int>();
::test_device_attribute<attributes::max_texture_2d_layered_height, ::cudaDevAttrMaxTexture2DLayeredHeight, int>();
::test_device_attribute<attributes::max_texture_2d_layered_layers, ::cudaDevAttrMaxTexture2DLayeredLayers, int>();
::test_device_attribute<attributes::max_texture_cubemap_layered_width,
::cudaDevAttrMaxTextureCubemapLayeredWidth,
int>();
::test_device_attribute<attributes::max_texture_cubemap_layered_layers,
::cudaDevAttrMaxTextureCubemapLayeredLayers,
int>();
::test_device_attribute<attributes::max_surface_1d_width, ::cudaDevAttrMaxSurface1DWidth, int>();
::test_device_attribute<attributes::max_surface_2d_width, ::cudaDevAttrMaxSurface2DWidth, int>();
::test_device_attribute<attributes::max_surface_2d_height, ::cudaDevAttrMaxSurface2DHeight, int>();
::test_device_attribute<attributes::max_surface_3d_width, ::cudaDevAttrMaxSurface3DWidth, int>();
::test_device_attribute<attributes::max_surface_3d_height, ::cudaDevAttrMaxSurface3DHeight, int>();
::test_device_attribute<attributes::max_surface_3d_depth, ::cudaDevAttrMaxSurface3DDepth, int>();
::test_device_attribute<attributes::max_surface_1d_layered_width, ::cudaDevAttrMaxSurface1DLayeredWidth, int>();
::test_device_attribute<attributes::max_surface_1d_layered_layers, ::cudaDevAttrMaxSurface1DLayeredLayers, int>();
::test_device_attribute<attributes::max_surface_2d_layered_width, ::cudaDevAttrMaxSurface2DLayeredWidth, int>();
::test_device_attribute<attributes::max_surface_2d_layered_height, ::cudaDevAttrMaxSurface2DLayeredHeight, int>();
::test_device_attribute<attributes::max_surface_2d_layered_layers, ::cudaDevAttrMaxSurface2DLayeredLayers, int>();
::test_device_attribute<attributes::max_surface_cubemap_width, ::cudaDevAttrMaxSurfaceCubemapWidth, int>();
::test_device_attribute<attributes::max_surface_cubemap_layered_width,
::cudaDevAttrMaxSurfaceCubemapLayeredWidth,
int>();
::test_device_attribute<attributes::max_surface_cubemap_layered_layers,
::cudaDevAttrMaxSurfaceCubemapLayeredLayers,
int>();
::test_device_attribute<attributes::max_registers_per_block, ::cudaDevAttrMaxRegistersPerBlock, int>();
::test_device_attribute<attributes::clock_rate, ::cudaDevAttrClockRate, int>();
::test_device_attribute<attributes::texture_alignment, ::cudaDevAttrTextureAlignment, cuda::std::size_t>();
::test_device_attribute<attributes::texture_pitch_alignment, ::cudaDevAttrTexturePitchAlignment, cuda::std::size_t>();
::test_device_attribute<attributes::gpu_overlap, ::cudaDevAttrGpuOverlap, bool>();
::test_device_attribute<attributes::multiprocessor_count, ::cudaDevAttrMultiProcessorCount, int>();
::test_device_attribute<attributes::kernel_exec_timeout, ::cudaDevAttrKernelExecTimeout, bool>();
::test_device_attribute<attributes::integrated, ::cudaDevAttrIntegrated, bool>();
::test_device_attribute<attributes::can_map_host_memory, ::cudaDevAttrCanMapHostMemory, bool>();
::test_device_attribute<attributes::compute_mode, ::cudaDevAttrComputeMode, ::cudaComputeMode>();
::test_device_attribute<attributes::concurrent_kernels, ::cudaDevAttrConcurrentKernels, bool>();
::test_device_attribute<attributes::ecc_enabled, ::cudaDevAttrEccEnabled, bool>();
::test_device_attribute<attributes::pci_bus_id, ::cudaDevAttrPciBusId, int>();
::test_device_attribute<attributes::pci_device_id, ::cudaDevAttrPciDeviceId, int>();
::test_device_attribute<attributes::tcc_driver, ::cudaDevAttrTccDriver, bool>();
::test_device_attribute<attributes::l2_cache_size, ::cudaDevAttrL2CacheSize, cuda::std::size_t>();
::test_device_attribute<attributes::max_threads_per_multiprocessor, ::cudaDevAttrMaxThreadsPerMultiProcessor, int>();
::test_device_attribute<attributes::unified_addressing, ::cudaDevAttrUnifiedAddressing, bool>();
::test_device_attribute<attributes::compute_capability_major, ::cudaDevAttrComputeCapabilityMajor, int>();
::test_device_attribute<attributes::compute_capability_minor, ::cudaDevAttrComputeCapabilityMinor, int>();
::test_device_attribute<attributes::stream_priorities_supported, ::cudaDevAttrStreamPrioritiesSupported, bool>();
::test_device_attribute<attributes::global_l1_cache_supported, ::cudaDevAttrGlobalL1CacheSupported, bool>();
::test_device_attribute<attributes::local_l1_cache_supported, ::cudaDevAttrLocalL1CacheSupported, bool>();
::test_device_attribute<attributes::max_shared_memory_per_multiprocessor,
::cudaDevAttrMaxSharedMemoryPerMultiprocessor,
cuda::std::size_t>();
::test_device_attribute<attributes::max_registers_per_multiprocessor,
::cudaDevAttrMaxRegistersPerMultiprocessor,
int>();
::test_device_attribute<attributes::is_multi_gpu_board, ::cudaDevAttrIsMultiGpuBoard, bool>();
::test_device_attribute<attributes::multi_gpu_board_group_id, ::cudaDevAttrMultiGpuBoardGroupID, int>();
::test_device_attribute<attributes::host_native_atomic_supported, ::cudaDevAttrHostNativeAtomicSupported, bool>();
::test_device_attribute<attributes::single_to_double_precision_perf_ratio,
::cudaDevAttrSingleToDoublePrecisionPerfRatio,
int>();
::test_device_attribute<attributes::pageable_memory_access, ::cudaDevAttrPageableMemoryAccess, bool>();
::test_device_attribute<attributes::concurrent_managed_access, ::cudaDevAttrConcurrentManagedAccess, bool>();
::test_device_attribute<attributes::compute_preemption_supported, ::cudaDevAttrComputePreemptionSupported, bool>();
::test_device_attribute<attributes::can_use_host_pointer_for_registered_mem,
::cudaDevAttrCanUseHostPointerForRegisteredMem,
bool>();
::test_device_attribute<attributes::cooperative_launch, ::cudaDevAttrCooperativeLaunch, bool>();
::test_device_attribute<attributes::can_flush_remote_writes, ::cudaDevAttrCanFlushRemoteWrites, bool>();
::test_device_attribute<attributes::host_register_supported, ::cudaDevAttrHostRegisterSupported, bool>();
::test_device_attribute<attributes::pageable_memory_access_uses_host_page_tables,
::cudaDevAttrPageableMemoryAccessUsesHostPageTables,
bool>();
::test_device_attribute<attributes::direct_managed_mem_access_from_host,
::cudaDevAttrDirectManagedMemAccessFromHost,
bool>();
::test_device_attribute<attributes::max_shared_memory_per_block_optin,
::cudaDevAttrMaxSharedMemoryPerBlockOptin,
cuda::std::size_t>();
::test_device_attribute<attributes::max_blocks_per_multiprocessor, ::cudaDevAttrMaxBlocksPerMultiprocessor, int>();
::test_device_attribute<attributes::max_persisting_l2_cache_size,
::cudaDevAttrMaxPersistingL2CacheSize,
cuda::std::size_t>();
::test_device_attribute<attributes::max_access_policy_window_size,
::cudaDevAttrMaxAccessPolicyWindowSize,
cuda::std::size_t>();
::test_device_attribute<attributes::reserved_shared_memory_per_block,
::cudaDevAttrReservedSharedMemoryPerBlock,
cuda::std::size_t>();
::test_device_attribute<attributes::sparse_cuda_array_supported, ::cudaDevAttrSparseCudaArraySupported, bool>();
::test_device_attribute<attributes::host_register_read_only_supported,
::cudaDevAttrHostRegisterReadOnlySupported,
bool>();
::test_device_attribute<attributes::memory_pools_supported, ::cudaDevAttrMemoryPoolsSupported, bool>();
::test_device_attribute<attributes::gpu_direct_rdma_supported, ::cudaDevAttrGPUDirectRDMASupported, bool>();
::test_device_attribute<attributes::gpu_direct_rdma_flush_writes_options,
::cudaDevAttrGPUDirectRDMAFlushWritesOptions,
::cudaFlushGPUDirectRDMAWritesOptions>();
::test_device_attribute<attributes::gpu_direct_rdma_writes_ordering,
::cudaDevAttrGPUDirectRDMAWritesOrdering,
::cudaGPUDirectRDMAWritesOrdering>();
::test_device_attribute<attributes::memory_pool_supported_handle_types,
::cudaDevAttrMemoryPoolSupportedHandleTypes,
::cudaMemAllocationHandleType>();
::test_device_attribute<attributes::deferred_mapping_cuda_array_supported,
::cudaDevAttrDeferredMappingCudaArraySupported,
bool>();
::test_device_attribute<attributes::ipc_event_support, ::cudaDevAttrIpcEventSupport, bool>();
#if _CCCL_CTK_AT_LEAST(12, 2)
::test_device_attribute<attributes::numa_config, ::cudaDevAttrNumaConfig, ::cudaDeviceNumaConfig>();
::test_device_attribute<attributes::numa_id, ::cudaDevAttrNumaId, int>();
#endif // _CCCL_CTK_AT_LEAST(12, 2)
SECTION("compute_mode")
{
STATIC_REQUIRE(::cudaComputeModeDefault == attributes::compute_mode.default_mode);
STATIC_REQUIRE(::cudaComputeModeProhibited == attributes::compute_mode.prohibited_mode);
STATIC_REQUIRE(::cudaComputeModeExclusiveProcess == attributes::compute_mode.exclusive_process_mode);
auto mode = device_ref(0).attribute(attributes::compute_mode);
CCCLRT_REQUIRE((mode == attributes::compute_mode.default_mode || //
mode == attributes::compute_mode.prohibited_mode || //
mode == attributes::compute_mode.exclusive_process_mode));
}
SECTION("gpu_direct_rdma_flush_writes_options")
{
STATIC_REQUIRE(::cudaFlushGPUDirectRDMAWritesOptionHost == attributes::gpu_direct_rdma_flush_writes_options.host);
STATIC_REQUIRE(
::cudaFlushGPUDirectRDMAWritesOptionMemOps == attributes::gpu_direct_rdma_flush_writes_options.mem_ops);
[[maybe_unused]] auto options = device_ref(0).attribute(attributes::gpu_direct_rdma_flush_writes_options);
#if !_CCCL_COMPILER(MSVC)
CCCLRT_REQUIRE((options == attributes::gpu_direct_rdma_flush_writes_options.host || //
options == attributes::gpu_direct_rdma_flush_writes_options.mem_ops));
#endif
}
SECTION("gpu_direct_rdma_writes_ordering")
{
STATIC_REQUIRE(::cudaGPUDirectRDMAWritesOrderingNone == attributes::gpu_direct_rdma_writes_ordering.none);
STATIC_REQUIRE(::cudaGPUDirectRDMAWritesOrderingOwner == attributes::gpu_direct_rdma_writes_ordering.owner);
STATIC_REQUIRE(
::cudaGPUDirectRDMAWritesOrderingAllDevices == attributes::gpu_direct_rdma_writes_ordering.all_devices);
auto ordering = device_ref(0).attribute(attributes::gpu_direct_rdma_writes_ordering);
CCCLRT_REQUIRE((ordering == attributes::gpu_direct_rdma_writes_ordering.none || //
ordering == attributes::gpu_direct_rdma_writes_ordering.owner || //
ordering == attributes::gpu_direct_rdma_writes_ordering.all_devices));
}
SECTION("memory_pool_supported_handle_types")
{
STATIC_REQUIRE(::cudaMemHandleTypeNone == attributes::memory_pool_supported_handle_types.none);
STATIC_REQUIRE(
::cudaMemHandleTypePosixFileDescriptor == attributes::memory_pool_supported_handle_types.posix_file_descriptor);
STATIC_REQUIRE(::cudaMemHandleTypeWin32 == attributes::memory_pool_supported_handle_types.win32);
STATIC_REQUIRE(::cudaMemHandleTypeWin32Kmt == attributes::memory_pool_supported_handle_types.win32_kmt);
#if _CCCL_CTK_AT_LEAST(12, 4)
STATIC_REQUIRE(::cudaMemHandleTypeFabric == 0x8);
STATIC_REQUIRE(::cudaMemHandleTypeFabric == attributes::memory_pool_supported_handle_types.fabric);
#else // ^^^ _CCCL_CTK_AT_LEAST(12, 4) ^^^ / vvv _CCCL_CTK_BELOW(12, 4) vvv
STATIC_REQUIRE(0x8 == attributes::memory_pool_supported_handle_types.fabric);
#endif // ^^^ _CCCL_CTK_BELOW(12, 4) ^^^
constexpr int all_handle_types =
attributes::memory_pool_supported_handle_types.none
| attributes::memory_pool_supported_handle_types.posix_file_descriptor
| attributes::memory_pool_supported_handle_types.win32
| attributes::memory_pool_supported_handle_types.win32_kmt
| attributes::memory_pool_supported_handle_types.fabric;
auto handle_types = device_ref(0).attribute(attributes::memory_pool_supported_handle_types);
CCCLRT_REQUIRE(static_cast<int>(handle_types) <= static_cast<int>(all_handle_types));
}
#if _CCCL_CTK_AT_LEAST(12, 2)
SECTION("numa_config")
{
STATIC_REQUIRE(::cudaDeviceNumaConfigNone == attributes::numa_config.none);
STATIC_REQUIRE(::cudaDeviceNumaConfigNumaNode == attributes::numa_config.numa_node);
auto config = device_ref(0).attribute(attributes::numa_config);
CCCLRT_REQUIRE((config == attributes::numa_config.none || //
config == attributes::numa_config.numa_node));
}
#endif // _CCCL_CTK_AT_LEAST(12, 2)
SECTION("Compute capability")
{
cuda::compute_capability compute_cap = device_ref(0).attribute(attributes::compute_capability);
int compute_cap_major = device_ref(0).attribute(attributes::compute_capability_major);
int compute_cap_minor = device_ref(0).attribute(attributes::compute_capability_minor);
CCCLRT_REQUIRE(compute_cap.get() == 10 * compute_cap_major + compute_cap_minor);
}
SECTION("Total global memory")
{
auto total_mem = device_ref(0).attribute(attributes::total_global_memory);
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(total_mem), cuda::std::size_t>);
CCCLRT_REQUIRE(total_mem > 0);
}
}
SECTION("Name")
{
const auto name = device_ref(0).name();
CCCLRT_REQUIRE(name.length() != 0);
CCCLRT_REQUIRE(name[0] != 0);
}
}
C2H_CCCLRT_TEST("global devices vector", "[device]")
{
CCCLRT_REQUIRE(cuda::devices.size() > 0);
CCCLRT_REQUIRE(cuda::devices.begin() != cuda::devices.end());
CCCLRT_REQUIRE(cuda::devices.begin() == cuda::devices.begin());
CCCLRT_REQUIRE(cuda::devices.end() == cuda::devices.end());
CCCLRT_REQUIRE(cuda::devices.size() == static_cast<size_t>(cuda::devices.end() - cuda::devices.begin()));
CCCLRT_REQUIRE(0 == cuda::devices[0].get());
CCCLRT_REQUIRE(cuda::device_ref{0} == cuda::devices[0]);
CCCLRT_REQUIRE(0 == (*cuda::devices.begin()).get());
CCCLRT_REQUIRE(cuda::device_ref{0} == *cuda::devices.begin());
CCCLRT_REQUIRE(0 == cuda::devices.begin()->get());
CCCLRT_REQUIRE(0 == cuda::devices.begin()[0].get());
if (cuda::devices.size() > 1)
{
CCCLRT_REQUIRE(1 == cuda::devices[1].get());
CCCLRT_REQUIRE(cuda::device_ref{0} != cuda::devices[1].get());
CCCLRT_REQUIRE(1 == (*std::next(cuda::devices.begin())).get());
CCCLRT_REQUIRE(1 == std::next(cuda::devices.begin())->get());
CCCLRT_REQUIRE(1 == cuda::devices.begin()[1].get());
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>((*std::prev(cuda::devices.end())).get()));
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>(std::prev(cuda::devices.end())->get()));
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>(cuda::devices.end()[-1].get()));
auto peers = cuda::devices[0].peers();
for (auto peer : peers)
{
CCCLRT_REQUIRE(cuda::devices[0].has_peer_access_to(peer));
CCCLRT_REQUIRE(peer.has_peer_access_to(cuda::devices[0]));
}
}
#if _CCCL_HAS_EXCEPTIONS()
try
{
[[maybe_unused]] const cuda::device_ref& dev = cuda::devices[cuda::devices.size()];
CCCLRT_REQUIRE(false); // should not get here
}
catch (const std::out_of_range&)
{
CCCLRT_REQUIRE(true); // expected
}
#endif // _CCCL_HAS_EXCEPTIONS()
}
C2H_CCCLRT_TEST("Device attributes use the explicit device when current device differs", "[device][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
const auto expected_bus_id = cuda::__driver::__deviceGetAttribute(
static_cast<::CUdevice_attribute>(cudaDevAttrPciBusId), cuda::__driver::__deviceGet(explicit_device.get()));
const auto expected_device_id = cuda::__driver::__deviceGetAttribute(
static_cast<::CUdevice_attribute>(cudaDevAttrPciDeviceId), cuda::__driver::__deviceGet(explicit_device.get()));
{
cuda::__ensure_current_context guard(current_device);
CCCLRT_REQUIRE(explicit_device.attribute(cuda::device_attributes::pci_bus_id) == expected_bus_id);
CCCLRT_REQUIRE(explicit_device.attribute(cuda::device_attributes::pci_device_id) == expected_device_id);
}
}
C2H_CCCLRT_TEST("memory location", "[device]")
{
cuda::memory_location loc = cuda::devices[0];
CCCLRT_REQUIRE(loc.type == ::cudaMemLocationTypeDevice);
CCCLRT_REQUIRE(loc.id == 0);
if (cuda::devices.size() > 1)
{
loc = cuda::device_ref{1};
CCCLRT_REQUIRE(loc.type == ::cudaMemLocationTypeDevice);
CCCLRT_REQUIRE(loc.id == 1);
}
}

View File

@@ -0,0 +1,60 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#include <cuda/std/array>
#include <cuda/std/cassert>
#include <cuda/std/cstddef>
#include <cuda/std/type_traits>
#include "test_macros.h"
TEST_FUNC constexpr bool test()
{
// 1. Test signature.
static_assert(cuda::std::is_same_v<bool, decltype(cuda::__is_specific_arch(cuda::arch_id{}))>);
static_assert(noexcept(cuda::__is_specific_arch(cuda::arch_id{})));
// 2. Test values.
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_60));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_61));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_62));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_70));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_75));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_80));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_86));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_87));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_88));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_89));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_90));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_100));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_103));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_110));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_120));
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_121));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_90a));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_100a));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_103a));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_110a));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_120a));
assert(cuda::__is_specific_arch(cuda::arch_id::sm_121a));
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,63 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: enable-tile
// error: return in loop statement is not supported
#include <cuda/devices>
#include <cuda/std/array>
#include <cuda/std/cassert>
#include <cuda/std/cstddef>
#include "test_macros.h"
template <cuda::std::size_t N>
TEST_FUNC constexpr auto make_all_ccs_ref(cuda::std::array<int, N> vs, int scale)
{
cuda::std::array<cuda::compute_capability, N> ret;
for (cuda::std::size_t i = 0; i < N; ++i)
{
ret[i] = cuda::compute_capability{vs[i] / scale};
}
return ret;
}
TEST_FUNC constexpr bool test()
{
// 1. Test signature.
static_assert(cuda::std::__is_cuda_std_array_v<decltype(cuda::__target_compute_capabilities())>);
static_assert(noexcept(cuda::__target_compute_capabilities()));
// 2. Test that all values are present.
const auto all_ccs = cuda::__target_compute_capabilities();
#if defined(__CUDA_ARCH_LIST__)
const auto all_ccs_ref = make_all_ccs_ref(cuda::std::array{__CUDA_ARCH_LIST__}, 10);
#elif defined(NV_TARGET_SM_INTEGER_LIST)
const auto all_ccs_ref = make_all_ccs_ref(cuda::std::array{NV_TARGET_SM_INTEGER_LIST}, 1);
#else
const auto all_ccs_ref = ::cuda::__all_compute_capabilities();
#endif
assert(all_ccs.size() == all_ccs_ref.size());
for (cuda::std::size_t i = 0; i < all_ccs.size(); ++i)
{
assert(all_ccs[i] == all_ccs_ref[i]);
}
return true;
}
int main(int, char**)
{
test();
static_assert(test());
return 0;
}

View File

@@ -0,0 +1,226 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/atomic>
#include <cuda/devices>
#include <cuda/launch>
#include <cuda/stream>
#include <testing.cuh>
#include <utility.cuh>
namespace
{
namespace test
{
cuda::event_ref fn_takes_event_ref(cuda::event_ref ref)
{
return ref;
}
template <class Event>
void test_event_uses_explicit_device_when_current_device_differs()
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
cuda::stream explicit_device_stream{explicit_device};
CCCLRT_REQUIRE(explicit_device_stream.device() == explicit_device);
Event ev = [&]() {
cuda::__ensure_current_context guard(current_device);
return Event(explicit_device);
}();
{
cuda::__ensure_current_context guard(current_device);
ev.record(explicit_device_stream);
ev.sync();
CCCLRT_REQUIRE(ev.is_done());
}
explicit_device_stream.sync();
}
} // namespace test
} // namespace
static_assert(!::cuda::std::is_default_constructible_v<cuda::event_ref>);
static_assert(!::cuda::std::is_default_constructible_v<cuda::event>);
static_assert(!::cuda::std::is_default_constructible_v<cuda::timed_event>);
C2H_CCCLRT_TEST("can construct an event_ref from a cudaEvent_t", "[event]")
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
::cudaEvent_t ev;
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
cuda::event_ref ref(ev);
CCCLRT_REQUIRE(ref.get() == ev);
CCCLRT_REQUIRE(!!ref);
// test implicit conversion from cudaEvent_t:
cuda::event_ref ref2 = ::test::fn_takes_event_ref(ev);
CCCLRT_REQUIRE(ref2.get() == ev);
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
// test an empty event_ref:
cuda::event_ref ref3(::cudaEvent_t{});
CCCLRT_REQUIRE(ref3.get() == ::cudaEvent_t{});
CCCLRT_REQUIRE(!ref3);
}
C2H_CCCLRT_TEST("can copy construct an event_ref and compare for equality", "[event]")
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
::cudaEvent_t ev;
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
const cuda::event_ref ref(ev);
const cuda::event_ref ref2 = ref;
CCCLRT_REQUIRE(ref2 == ref);
CCCLRT_REQUIRE(!(ref != ref2));
CCCLRT_REQUIRE((ref ? true : false)); // test contextual convertibility to bool
CCCLRT_REQUIRE(!!ref);
CCCLRT_REQUIRE(::cudaEvent_t{} != ref);
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
// copy from empty event_ref:
const cuda::event_ref ref3(::cudaEvent_t{});
const cuda::event_ref ref4 = ref3;
CCCLRT_REQUIRE(ref4 == ref3);
CCCLRT_REQUIRE(!(ref3 != ref4));
CCCLRT_REQUIRE(!ref4);
}
C2H_CCCLRT_TEST("can use event_ref to record and wait on an event", "[event]")
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
::cudaEvent_t ev;
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
const cuda::event_ref ref(ev);
test::pinned<int> i(0);
cuda::stream stream{cuda::device_ref{0}};
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
ref.record(stream);
ref.sync();
CCCLRT_REQUIRE(ref.is_done());
CCCLRT_REQUIRE(*i == 42);
stream.sync();
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
}
C2H_CCCLRT_TEST("can construct an event with a stream_ref", "[event]")
{
cuda::stream stream{cuda::device_ref{0}};
cuda::event ev(static_cast<cuda::stream_ref>(stream));
CCCLRT_REQUIRE(ev.get() != ::cudaEvent_t{});
}
C2H_CCCLRT_TEST("can construct an event with a device_ref", "[event]")
{
cuda::device_ref device{0};
cuda::event ev(device);
CCCLRT_REQUIRE(ev.get() != ::cudaEvent_t{});
cuda::stream stream{device};
ev.record(stream);
ev.sync();
CCCLRT_REQUIRE(ev.is_done());
}
C2H_CCCLRT_TEST("event device_ref constructors use the explicit device", "[event][multi_gpu]")
{
::test::test_event_uses_explicit_device_when_current_device_differs<cuda::event>();
::test::test_event_uses_explicit_device_when_current_device_differs<cuda::timed_event>();
}
C2H_CCCLRT_TEST("can wait on an event from another device", "[event][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref event_device{0};
cuda::device_ref waiter_device{1};
cuda::stream event_stream{event_device};
cuda::stream waiter_stream{waiter_device};
cuda::atomic<int> gate = 0;
bool waiter_ran = false;
cuda::host_launch(event_stream, [&gate]() {
while (gate != 1)
;
});
cuda::event ev(event_stream);
{
cuda::__ensure_current_context guard(event_device);
waiter_stream.wait(ev);
cuda::host_launch(waiter_stream, [&waiter_ran]() {
waiter_ran = true;
});
}
CCCLRT_REQUIRE(!waiter_stream.is_done());
CCCLRT_REQUIRE(!waiter_ran);
gate = 1;
waiter_stream.sync();
event_stream.sync();
CCCLRT_REQUIRE(waiter_ran);
}
C2H_CCCLRT_TEST("can wait on an event", "[event]")
{
cuda::stream stream{cuda::device_ref{0}};
::test::pinned<int> i(0);
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
cuda::event ev(stream);
ev.sync();
CCCLRT_REQUIRE(ev.is_done());
CCCLRT_REQUIRE(*i == 42);
stream.sync();
}
C2H_CCCLRT_TEST("can take the difference of two timed_event objects", "[event]")
{
cuda::stream stream{cuda::device_ref{0}};
::test::pinned<int> i(0);
cuda::timed_event start(stream);
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
cuda::timed_event end(stream);
end.sync();
CCCLRT_REQUIRE(end.is_done());
CCCLRT_REQUIRE(*i == 42);
auto elapsed = end - start;
CCCLRT_REQUIRE(elapsed.count() >= 0);
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(elapsed), ::cuda::std::chrono::nanoseconds>);
stream.sync();
}
C2H_CCCLRT_TEST("can observe the event in not ready state", "[event]")
{
::test::pinned<int> i(0);
::cuda::atomic_ref atomic_i(*i);
cuda::stream stream{cuda::device_ref{0}};
::test::launch_kernel_single_thread(stream, ::test::spin_until_80{}, i.get());
cuda::event ev(stream);
CCCLRT_REQUIRE(!ev.is_done());
atomic_i.store(80);
ev.sync();
CCCLRT_REQUIRE(ev.is_done());
}

View File

@@ -0,0 +1,75 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <iostream>
#include <cooperative_groups.h>
#include <host_device.cuh>
struct custom_level : public cuda::hierarchy_level_base<custom_level>
{
using __product_type = unsigned int;
using __allowed_above = cuda::__allowed_levels<cuda::grid_level>;
using __allowed_below = cuda::__allowed_levels<cuda::block_level>;
};
template <typename Level, typename Dims>
struct custom_level_dims : public cuda::hierarchy_level_desc<Level, Dims>
{
int dummy;
constexpr custom_level_dims()
: cuda::hierarchy_level_desc<Level, Dims>() {};
};
struct custom_level_test
{
template <typename DynDims>
TEST_FUNC void operator()(const DynDims& dims) const
{
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
#if !_CCCL_CUDA_COMPILER(CLANG)
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 84 * 1024);
CCCLRT_REQUIRE(custom_level{}.count(cuda::grid, dims) == 42);
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(42 * 512, 2, 2));
CCCLRT_REQUIRE(custom_level{}.dims(cuda::grid, dims) == dim3(42, 1, 1));
#endif // !_CCCL_CUDA_COMPILER(CLANG)
}
void run()
{
// Check extending hierarchy_level_desc with custom info
custom_level_dims<cuda::block_level, cuda::std::extents<int, 64, 1, 1>> custom_block;
custom_block.dummy = 2;
auto custom_dims = cuda::make_hierarchy(cuda::grid_dims<256>(), cuda::cluster_dims<8>(), custom_block);
auto custom_block_back = custom_dims.level(cuda::block);
CCCLRT_REQUIRE(custom_block_back.dummy == 2);
auto custom_dims_fragment = custom_dims.fragment(cuda::gpu_thread, cuda::block);
auto custom_block_back2 = custom_dims_fragment.level(cuda::block);
CCCLRT_REQUIRE(custom_block_back2.dummy == 2);
// Check creating a custom level type works
auto custom_level_dims = cuda::std::extents<cuda::dimensions_index_type, 2, 2, 2>();
auto custom_hierarchy = cuda::make_hierarchy(
cuda::grid_dims(42),
cuda::hierarchy_level_desc<custom_level, decltype(custom_level_dims)>(custom_level_dims),
cuda::block_dims<256>());
static_assert(cuda::gpu_thread.dims(custom_level(), custom_hierarchy) == dim3(512, 2, 2));
static_assert(cuda::gpu_thread.count(custom_level(), custom_hierarchy) == 2048);
test_host_dev(custom_hierarchy, *this);
}
};
C2H_TEST("Custom level", "[hierarchy]")
{
custom_level_test().run();
}

View File

@@ -0,0 +1,567 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/__type_traits/vector_type.h>
#include <cuda/hierarchy>
#include <cuda/launch>
#include <cuda/std/cstddef>
#include <iostream>
#include <cooperative_groups.h>
#include <host_device.cuh>
#include "testing.cuh"
namespace cg = cooperative_groups;
using size_t3 = cuda::vector_type_t<cuda::std::size_t, 3>;
struct basic_test_single_dim
{
static constexpr int block_size = 256;
static constexpr int grid_size = 512;
template <typename DynDims>
TEST_FUNC void operator()(const DynDims& dims) const
{
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
#if !_CCCL_CUDA_COMPILER(CLANG)
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims).x == grid_size * block_size);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == grid_size * block_size);
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::block, dims).x == block_size);
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims).x == grid_size);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, dims) == block_size);
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == grid_size);
#endif // !_CCCL_CUDA_COMPILER(CLANG)
}
void run()
{
auto dims = cuda::make_hierarchy(cuda::block_dims<block_size>(), cuda::grid_dims<grid_size>());
static_assert(cuda::gpu_thread.dims(cuda::grid, dims).x == grid_size * block_size);
static_assert(cuda::gpu_thread.count(cuda::grid, dims) == static_cast<unsigned long>(grid_size) * block_size);
static_assert(
cuda::gpu_thread.static_dims(cuda::grid, dims)[0] == static_cast<unsigned long>(grid_size) * block_size);
static_assert(cuda::gpu_thread.dims(cuda::block, dims).x == block_size);
static_assert(cuda::block.dims(cuda::grid, dims).x == grid_size);
static_assert(cuda::gpu_thread.count(cuda::block, dims) == block_size);
static_assert(cuda::block.count(cuda::grid, dims) == grid_size);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims)[0] == block_size);
auto dims_dyn = cuda::make_hierarchy(cuda::block_dims(block_size), cuda::grid_dims(grid_size));
test_host_dev(dims_dyn, *this);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_dyn)[0] == cuda::std::dynamic_extent);
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims_dyn)[0] == cuda::std::dynamic_extent);
// Test that we can also drop the empty parens in the level constructors:
auto config = cuda::make_hierarchy(cuda::block_dims<block_size>, cuda::grid_dims<grid_size>);
CCCLRT_REQUIRE(dims == config);
}
};
struct basic_test_multi_dim
{
static constexpr int block_size = 256;
template <typename DynDims>
TEST_FUNC void operator()(const DynDims& dims) const
{
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
#if !_CCCL_CUDA_COMPILER(CLANG)
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(32, 12, 4));
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(0) == 32);
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(1) == 12);
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(2) == 4);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 512 * 3);
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::block, dims) == dim3(2, 3, 4));
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(16, 4, 1));
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, dims) == 24);
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 64);
#endif // !_CCCL_CUDA_COMPILER(CLANG)
}
void run()
{
auto dims_multidim = cuda::make_hierarchy(cuda::block_dims<2, 3, 4>(), cuda::grid_dims<16, 4, 1>());
static_assert(cuda::gpu_thread.dims(cuda::grid, dims_multidim) == dim3(32, 12, 4));
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(0) == 32);
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(1) == 12);
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(2) == 4);
static_assert(cuda::gpu_thread.count(cuda::grid, dims_multidim) == 512 * 3);
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims_multidim) == size_t3{32, 12, 4});
static_assert(cuda::gpu_thread.dims(cuda::block, dims_multidim) == dim3(2, 3, 4));
static_assert(cuda::block.dims(cuda::grid, dims_multidim) == dim3(16, 4, 1));
static_assert(cuda::gpu_thread.count(cuda::block, dims_multidim) == 24);
static_assert(cuda::block.count(cuda::grid, dims_multidim) == 64);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_multidim) == size_t3{2, 3, 4});
static_assert(cuda::block.static_dims(cuda::grid, dims_multidim) == size_t3{16, 4, 1});
auto dims_multidim_dyn = cuda::make_hierarchy(cuda::block_dims(dim3(2, 3, 4)), cuda::grid_dims(dim3(16, 4, 1)));
test_host_dev(dims_multidim_dyn, *this);
}
};
struct basic_test_mixed
{
static constexpr int block_size = 256;
template <typename DynDims>
TEST_FUNC void operator()(const DynDims& dims) const
{
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
#if !_CCCL_CUDA_COMPILER(CLANG)
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(2048, 4, 2));
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(0) == 2048);
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(1) == 4);
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(2) == 2);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 16 * 1024);
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(8, 4, 2));
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 64);
#endif // !_CCCL_CUDA_COMPILER(CLANG)
}
void run()
{
auto dims_mixed = cuda::make_hierarchy(cuda::block_dims<block_size>(), cuda::grid_dims(dim3(8, 4, 2)));
test_host_dev(dims_mixed, *this);
static_assert(cuda::gpu_thread.dims(cuda::block, dims_mixed).x == block_size);
static_assert(cuda::gpu_thread.count(cuda::block, dims_mixed) == block_size);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_mixed)[0] == block_size);
// TODO include mixed static and dynamic info on a single level
// Currently bugged in std::extents
}
};
C2H_TEST("Basic", "[hierarchy]")
{
basic_test_single_dim().run();
basic_test_multi_dim().run();
basic_test_mixed().run();
}
struct basic_test_cluster
{
template <typename DynDims>
TEST_FUNC void operator()(const DynDims& dims) const
{
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
#if !_CCCL_CUDA_COMPILER(CLANG)
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(512, 6, 9));
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 27 * 1024);
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(2, 6, 9));
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 108);
CCCLRT_REQUIRE(cuda::cluster.dims(cuda::grid, dims) == dim3(1, 3, 9));
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::cluster, dims) == dim3(512, 2, 1));
#endif // !_CCCL_CUDA_COMPILER(CLANG)
}
void run()
{
SECTION("Static cluster dims")
{
auto dims = cuda::make_hierarchy(cuda::block_dims<256>(), cuda::cluster_dims<8>(), cuda::grid_dims<512>());
static_assert(cuda::gpu_thread.dims(cuda::grid, dims).x == 1024 * 1024);
static_assert(cuda::gpu_thread.count(cuda::grid, dims) == 1024 * 1024);
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims)[0] == 1024 * 1024);
static_assert(cuda::gpu_thread.dims(cuda::block, dims).x == 256);
static_assert(cuda::block.dims(cuda::grid, dims).x == 4 * 1024);
static_assert(cuda::gpu_thread.count(cuda::cluster, dims) == 2 * 1024);
static_assert(cuda::cluster.count(cuda::grid, dims) == 512);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims)[0] == 256);
static_assert(cuda::block.static_dims(cuda::grid, dims)[0] == 4 * 1024);
}
SECTION("Mixed cluster dims")
{
auto dims_mixed = cuda::make_hierarchy(
cuda::block_dims<256>(), cuda::cluster_dims(dim3(2, 2, 1)), cuda::grid_dims(dim3(1, 3, 9)));
test_host_dev(dims_mixed, *this, arch_filter<std::less<int>, 90>);
static_assert(cuda::gpu_thread.dims(cuda::block, dims_mixed).x == 256);
static_assert(cuda::gpu_thread.count(cuda::block, dims_mixed) == 256);
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_mixed)[0] == 256);
static_assert(cuda::block.static_dims(cuda::cluster, dims_mixed)[0] == cuda::std::dynamic_extent);
static_assert(cuda::block.static_dims(cuda::grid, dims_mixed)[0] == cuda::std::dynamic_extent);
}
}
};
C2H_TEST("Cluster dims", "[hierarchy]")
{
basic_test_cluster().run();
}
C2H_TEST("Different constructions", "[hierarchy]")
{
/*
const auto block_size = 512;
const auto cluster_cnt = 8;
const auto grid_size = 256;
[[maybe_unused]] const auto config =
cuda::block_dims<block_size>() & cuda::cluster_dims<cluster_cnt>() &
cuda::grid_dims(grid_size);
[[maybe_unused]] const auto config2 =
cuda::grid_dims(grid_size) & cuda::cluster_dims<cluster_cnt>() &
cuda::block_dims<block_size>();
[[maybe_unused]] const auto config3 =
cuda::cluster_dims<cluster_cnt>() & cuda::grid_dims(grid_size) &
cuda::block_dims<block_size>();
[[maybe_unused]] const auto config4 =
cuda::cluster_dims<cluster_cnt>() & cuda::block_dims<block_size>() &
cuda::grid_dims(grid_size);
[[maybe_unused]] const auto config5 =
cuda::make_config(cuda::block_dims<block_size>(),
cuda::cluster_dims<cluster_cnt>(), cuda::grid_dims(grid_size));
[[maybe_unused]] const auto config6 =
cuda::make_config(cuda::grid_dims(grid_size),
cuda::cluster_dims<cluster_cnt>(), cuda::block_dims<block_size>());
static_assert(std::is_same_v<decltype(config), decltype(config2)>);
static_assert(std::is_same_v<decltype(config), decltype(config3)>);
static_assert(std::is_same_v<decltype(config), decltype(config4)>);
static_assert(std::is_same_v<decltype(config), decltype(config5)>);
static_assert(std::is_same_v<decltype(config), decltype(config6)>);
[[maybe_unused]] const auto conf_weird_order =
cuda::grid_dims(grid_size) & (cuda::cluster_dims<cluster_cnt>() &
cuda::block_dims<block_size>());
static_assert(std::is_same_v<decltype(config), decltype(conf_weird_order)>);
static_assert(config.hierarchy().count(cuda::gpu_thread, cuda::block) == block_size);
static_assert(config.hierarchy().count(cuda::gpu_thread, cuda::cluster) == cluster_cnt *
block_size); static_assert(config.hierarchy().count(cuda::block, cuda::cluster) ==
cluster_cnt); CCCLRT_REQUIRE(config.hierarchy().count() == grid_size * cluster_cnt *
block_size);
static_assert(config.hierarchy().has_level(cuda::block));
static_assert(config.hierarchy().has_level(cuda::cluster));
static_assert(config.hierarchy().has_level(cuda::grid));
static_assert(!config.hierarchy().has_level(cuda::thread));
*/
}
C2H_TEST("Replace level", "[hierarchy]")
{
// GCC 7 and 8 complains here that the hierarchy was not declared constexpr
#if !_CCCL_COMPILER(GCC, <, 9)
const auto dimensions = cuda::make_hierarchy(cuda::block_dims<512>(), cuda::cluster_dims<8>(), cuda::grid_dims(256));
const auto fragment = dimensions.fragment(cuda::block, cuda::grid);
static_assert(!fragment.has_level(cuda::block));
static_assert(!cuda::__has_bottom_unit_or_level_v<cuda::thread_level, decltype(fragment)>);
static_assert(fragment.has_level(cuda::cluster));
static_assert(fragment.has_level(cuda::grid));
static_assert(cuda::__has_bottom_unit_or_level_v<cuda::block_level, decltype(fragment)>);
const auto replaced = cuda::hierarchy_add_level(fragment, cuda::block_dims(256));
static_assert(replaced.has_level(cuda::block));
static_assert(cuda::__has_bottom_unit_or_level_v<cuda::thread_level, decltype(replaced)>);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, replaced) == 256);
#endif // !_CCCL_COMPILER(GCC, <, 9)
}
template <typename Hierarchy>
__global__ void kernel(Hierarchy hierarchy)
{
auto grid = cg::this_grid();
auto block = cg::this_thread_block();
CCCLRT_REQUIRE_DEVICE(grid.thread_rank() == cuda::gpu_thread.rank(cuda::grid));
CCCLRT_REQUIRE_DEVICE(grid.block_rank() == cuda::block.rank(cuda::grid));
CCCLRT_REQUIRE_DEVICE(grid.block_index() == cuda::block.index(cuda::grid));
CCCLRT_REQUIRE_DEVICE(grid.num_threads() == cuda::gpu_thread.count(cuda::grid));
CCCLRT_REQUIRE_DEVICE(grid.num_blocks() == cuda::block.count(cuda::grid));
CCCLRT_REQUIRE_DEVICE(grid.dim_blocks() == cuda::block.dims(cuda::grid));
CCCLRT_REQUIRE_DEVICE(block.thread_rank() == cuda::gpu_thread.rank(cuda::block));
CCCLRT_REQUIRE_DEVICE(block.thread_index() == cuda::gpu_thread.index(cuda::block));
CCCLRT_REQUIRE_DEVICE(block.num_threads() == cuda::gpu_thread.count(cuda::block));
CCCLRT_REQUIRE_DEVICE(block.dim_threads() == cuda::gpu_thread.dims(cuda::block));
CCCLRT_REQUIRE_DEVICE(block.thread_index() == cuda::gpu_thread.index(cuda::block, hierarchy));
const auto grid_index = cuda::gpu_thread.index_as<unsigned long long>(cuda::grid, hierarchy);
CCCLRT_REQUIRE_DEVICE(
grid_index.x
== static_cast<unsigned long long>(grid.block_index().x) * block.dim_threads().x + block.thread_index().x);
CCCLRT_REQUIRE_DEVICE(
grid_index.y
== static_cast<unsigned long long>(grid.block_index().y) * block.dim_threads().y + block.thread_index().y);
CCCLRT_REQUIRE_DEVICE(
grid_index.z
== static_cast<unsigned long long>(grid.block_index().z) * block.dim_threads().z + block.thread_index().z);
CCCLRT_REQUIRE_DEVICE(grid.block_rank() == cuda::block.rank(cuda::grid, hierarchy));
CCCLRT_REQUIRE_DEVICE(block.thread_rank() == cuda::gpu_thread.rank(cuda::block, hierarchy));
CCCLRT_REQUIRE_DEVICE(grid.thread_rank() == cuda::gpu_thread.rank(cuda::grid, hierarchy));
}
C2H_TEST("Dims queries indexing and ambient hierarchy", "[hierarchy]")
{
const auto hierarchies = cuda::std::make_tuple(
cuda::make_hierarchy(cuda::block_dims(dim3(64, 4, 2)), cuda::grid_dims(dim3(12, 6, 3))),
cuda::make_hierarchy(cuda::block_dims(dim3(2, 4, 64)), cuda::grid_dims(dim3(3, 6, 12))),
cuda::make_hierarchy(cuda::block_dims<256>(), cuda::grid_dims<4>()),
cuda::make_hierarchy(cuda::block_dims<16, 2, 4>(), cuda::grid_dims<2, 3, 4>()),
cuda::make_hierarchy(cuda::block_dims(dim3(8, 4, 2)), cuda::grid_dims<4, 5, 6>()),
#if defined(NDEBUG)
cuda::make_hierarchy(cuda::block_dims<32>(), cuda::grid_dims<(1 << 30) - 2>()),
#endif
cuda::make_hierarchy(cuda::block_dims<8, 2, 4>(), cuda::grid_dims(dim3(5, 4, 3))));
apply_each(
[](const auto& hierarchy) {
auto [grid, block] = cuda::get_launch_dimensions(hierarchy);
kernel<<<grid, block>>>(hierarchy);
CUDART(cudaDeviceSynchronize());
},
hierarchies);
}
template <typename Hierarchy>
__global__ void rank_kernel_optimized(Hierarchy hierarchy, unsigned int* out)
{
auto thread_id = cuda::gpu_thread.rank(cuda::block, hierarchy);
out[thread_id] = thread_id;
}
template <typename Hierarchy>
__global__ void rank_kernel(Hierarchy hierarchy, unsigned int* out)
{
auto thread_id = cuda::gpu_thread.rank(cuda::block);
out[thread_id] = thread_id;
}
template <typename Hierarchy>
__global__ void rank_kernel_cg(Hierarchy hierarchy, unsigned int* out)
{
auto thread_id = cg::thread_block::thread_rank();
out[thread_id] = thread_id;
}
// Testcase mostly for generated code comparison
C2H_TEST("On device rank calculation", "[hierarchy]")
{
unsigned int* ptr;
CUDART(cudaMalloc((void**) &ptr, 2 * 1024 * sizeof(unsigned int)));
const auto hierarchy_static = cuda::make_hierarchy(cuda::block_dims<256>(), cuda::grid_dims(dim3(2, 2, 2)));
rank_kernel<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
CUDART(cudaDeviceSynchronize());
rank_kernel_cg<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
CUDART(cudaDeviceSynchronize());
rank_kernel_optimized<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
CUDART(cudaDeviceSynchronize());
CUDART(cudaFree(ptr));
}
template <typename Hierarchy>
__global__ void examples_kernel(Hierarchy hierarchy)
{
{
auto thread_index_in_block = cuda::gpu_thread.index(cuda::block, hierarchy);
CCCLRT_REQUIRE_DEVICE(thread_index_in_block == threadIdx);
auto block_index_in_grid = cuda::block.index(cuda::grid, hierarchy);
CCCLRT_REQUIRE_DEVICE(block_index_in_grid == blockIdx);
}
{
int thread_rank_in_block = cuda::gpu_thread.rank(cuda::block, hierarchy);
int block_rank_in_grid = cuda::block.rank(cuda::grid, hierarchy);
}
{
// Can be called with the instances of level types
int num_threads_in_block = static_cast<int>(cuda::gpu_thread.count(cuda::block));
int num_blocks_in_grid = static_cast<int>(cuda::block.count(cuda::grid));
// Or using the level types as template arguments
int num_threads_in_grid = static_cast<int>(cuda::gpu_thread.count(cuda::grid));
}
{
// Can be called with the instances of level types
int thread_rank_in_block = static_cast<int>(cuda::gpu_thread.rank(cuda::block));
int block_rank_in_grid = static_cast<int>(cuda::block.rank(cuda::grid));
// Or using the level types as template arguments
int thread_rank_in_grid = static_cast<int>(cuda::gpu_thread.rank(cuda::grid));
}
{
// Can be called with the instances of level types
CCCLRT_REQUIRE_DEVICE(cuda::gpu_thread.dims(cuda::block) == blockDim);
CCCLRT_REQUIRE_DEVICE(cuda::block.dims(cuda::grid) == gridDim);
// Or using the level types as template arguments
auto grid_dims_in_threads = cuda::gpu_thread.dims(cuda::grid);
}
{
// Can be called with the instances of level types
CCCLRT_REQUIRE_DEVICE(cuda::gpu_thread.index(cuda::block) == threadIdx);
CCCLRT_REQUIRE_DEVICE(cuda::block.index(cuda::grid) == blockIdx);
// Or using the level types as template arguments
auto thread_index_in_grid = cuda::gpu_thread.index(cuda::grid);
}
}
// Test examples from the inline rst documentation
C2H_TEST("Examples", "[hierarchy]")
{
// GCC 7 and 8 complains here that the hierarchy was not declared constexpr
#if !_CCCL_COMPILER(GCC, <, 9)
{
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
auto fragment = hierarchy.fragment(cuda::block, cuda::grid);
auto new_hierarchy = cuda::hierarchy_add_level(fragment, cuda::block_dims<128>());
static_assert(cuda::gpu_thread.count(cuda::block, new_hierarchy) == 128);
}
{
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
static_assert(cuda::gpu_thread.count(cuda::cluster, hierarchy) == 4 * 8 * 8 * 8);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, hierarchy) == 256 * 4 * 8 * 8 * 8);
CCCLRT_REQUIRE(cuda::cluster.count(cuda::grid, hierarchy) == 256);
}
{
[[maybe_unused]] auto hierarchy =
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
static_assert(cuda::gpu_thread.count(cuda::cluster, hierarchy) == 4 * 8 * 8 * 8);
}
{
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(0) == 4 * 8);
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(1) == 8);
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(2) == 8);
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, hierarchy).extent(0) == 256 * 4 * 8);
CCCLRT_REQUIRE(cuda::cluster.extents(cuda::grid, hierarchy).extent(0) == 256);
}
#endif // !_CCCL_COMPILER(GCC, <, 9)
{
[[maybe_unused]] auto hierarchy =
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
static_assert(decltype(hierarchy.level(cuda::cluster).extents())::static_extent(0) == 4);
}
{
auto partial1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims(256), cuda::cluster_dims<4>());
[[maybe_unused]] auto hierarchy1 = cuda::hierarchy_add_level(partial1, cuda::block_dims<8, 8, 8>());
auto partial2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<8, 8, 8>(), cuda::cluster_dims<4>());
[[maybe_unused]] auto hierarchy2 = cuda::hierarchy_add_level(partial2, cuda::grid_dims(256));
static_assert(cuda::std::is_same_v<decltype(hierarchy1), decltype(hierarchy2)>);
}
{
[[maybe_unused]] auto hierarchy1 =
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
[[maybe_unused]] auto hierarchy2 =
cuda::make_hierarchy(cuda::block_dims<8, 8, 8>(), cuda::cluster_dims<4>(), cuda::grid_dims(256));
static_assert(cuda::std::is_same_v<decltype(hierarchy1), decltype(hierarchy2)>);
}
{
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
auto [grid_dimensions, cluster_dimensions, block_dimensions] = cuda::get_launch_dimensions(hierarchy);
CCCLRT_REQUIRE(grid_dimensions.x == 256 * 4);
CCCLRT_REQUIRE(cluster_dimensions.x == 4);
CCCLRT_REQUIRE(block_dimensions.x == 8);
CCCLRT_REQUIRE(block_dimensions.y == 8);
CCCLRT_REQUIRE(block_dimensions.z == 8);
}
{
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(16), cuda::block_dims<8, 8, 8>());
auto [grid_dimensions, block_dimensions] = cuda::get_launch_dimensions(hierarchy);
examples_kernel<<<grid_dimensions, block_dimensions>>>(hierarchy);
CUDART(cudaGetLastError());
CUDART(cudaDeviceSynchronize());
}
}
C2H_TEST("Trivially constructable", "[hierarchy]")
{
// static_assert(std::is_trivial_v<decltype(cuda::block_dims(256))>);
// static_assert(std::is_trivial_v<decltype(cuda::block_dims<256>())>);
// Hierarchy is not trivially copyable (yet), because tuple is not
// static_assert(std::is_trivially_copyable_v<decltype(cuda::block_dims<256>()
// & cuda::grid_dims<256>())>);
// static_assert(std::is_trivially_copyable_v<decltype(cuda::std::make_tuple(cuda::block_dims<256>(),
// cuda::grid_dims<256>()))>);
}
C2H_TEST("cuda::distribute", "[hierarchy]")
{
unsigned numElements = 50000;
constexpr int threadsPerBlock = 256;
auto config = cuda::distribute<threadsPerBlock>(static_cast<int>(numElements));
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, config) == 256);
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, config) == (numElements + threadsPerBlock - 1) / threadsPerBlock);
}
C2H_TEST("hierarchy merge", "[hierarchy]")
{
SECTION("Non overlapping")
{
auto h1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<2>());
auto h2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<3>());
auto combined = h1.combine(h2);
static_assert(cuda::gpu_thread.count(cuda::grid, combined) == 6);
static_assert(cuda::gpu_thread.count(cuda::block, combined) == 3);
static_assert(cuda::block.count(cuda::grid, combined) == 2);
auto combined_the_other_way = h2.combine(h1);
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_the_other_way)>);
static_assert(cuda::gpu_thread.count(cuda::grid, combined_the_other_way) == 6);
auto dynamic_values = cuda::make_hierarchy(cuda::cluster_dims(4), cuda::block_dims(5));
auto combined_dynamic = dynamic_values.combine(h1);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined_dynamic) == 40);
}
SECTION("Overlapping")
{
auto h1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<2>(), cuda::cluster_dims<3>());
auto h2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<4>(), cuda::cluster_dims<5>());
auto combined = h1.combine(h2);
static_assert(cuda::gpu_thread.count(cuda::grid, combined) == 24);
static_assert(cuda::gpu_thread.count(cuda::block, combined) == 4);
static_assert(cuda::block.count(cuda::grid, combined) == 6);
auto combined_the_other_way = h2.combine(h1);
static_assert(!cuda::std::is_same_v<decltype(combined), decltype(combined_the_other_way)>);
static_assert(cuda::gpu_thread.count(cuda::grid, combined_the_other_way) == 40);
static_assert(cuda::gpu_thread.count(cuda::block, combined_the_other_way) == 4);
static_assert(cuda::block.count(cuda::grid, combined_the_other_way) == 10);
auto ultimate_combination = combined.combine(combined_the_other_way);
static_assert(cuda::std::is_same_v<decltype(combined), decltype(ultimate_combination)>);
static_assert(cuda::gpu_thread.count(cuda::grid, ultimate_combination) == 24);
auto block_level_replacement = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<6>());
auto with_block_replaced = block_level_replacement.combine(combined);
static_assert(cuda::gpu_thread.count(cuda::grid, with_block_replaced) == 36);
static_assert(cuda::gpu_thread.count(cuda::block, with_block_replaced) == 6);
auto grid_cluster_level_replacement =
cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<7>(), cuda::cluster_dims<8>());
auto with_grid_cluster_replaced = grid_cluster_level_replacement.combine(combined);
static_assert(cuda::gpu_thread.count(cuda::grid, with_grid_cluster_replaced) == 7 * 8 * 4);
static_assert(cuda::block.count(cuda::cluster, with_grid_cluster_replaced) == 8);
static_assert(cuda::cluster.count(cuda::grid, with_grid_cluster_replaced) == 7);
}
}

View File

@@ -0,0 +1,301 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda.h>
void test_launch_kernel_replacement(CUlaunchConfig& config, CUfunction kernel, void* args[]);
// This is a replacement for the launch kernel function that is used to test
// the configuration of the launch kernel. It checks if the configuration
// matches the expected configuration and calls the original launch kernel
// function if it does. If the configuration does not match, it will fail the
// test.
#define _CCCLRT_LAUNCH_CONFIG_TEST
#include <cuda/launch>
#include <host_device.cuh>
static CUlaunchConfig expectedConfig;
static bool replacementCalled = false;
void test_launch_kernel_replacement(CUlaunchConfig& config, CUfunction kernel, void* args[])
{
replacementCalled = true;
bool has_cluster = false;
CCCLRT_CHECK(expectedConfig.gridDimX == config.gridDimX);
CCCLRT_CHECK(expectedConfig.gridDimY == config.gridDimY);
CCCLRT_CHECK(expectedConfig.gridDimZ == config.gridDimZ);
CCCLRT_CHECK(expectedConfig.blockDimX == config.blockDimX);
CCCLRT_CHECK(expectedConfig.blockDimY == config.blockDimY);
CCCLRT_CHECK(expectedConfig.blockDimZ == config.blockDimZ);
CCCLRT_CHECK(expectedConfig.sharedMemBytes == config.sharedMemBytes);
CCCLRT_CHECK(expectedConfig.hStream == config.hStream);
CCCLRT_CHECK(expectedConfig.numAttrs == config.numAttrs);
for (unsigned int i = 0; i < expectedConfig.numAttrs; ++i)
{
auto& expectedAttr = expectedConfig.attrs[i];
unsigned int j;
for (j = 0; j < expectedConfig.numAttrs; ++j)
{
auto& actualAttr = config.attrs[j];
if (expectedAttr.id == actualAttr.id)
{
switch (expectedAttr.id)
{
case CU_LAUNCH_ATTRIBUTE_CLUSTER_DIMENSION:
CCCLRT_CHECK(expectedAttr.value.clusterDim.x == actualAttr.value.clusterDim.x);
CCCLRT_CHECK(expectedAttr.value.clusterDim.y == actualAttr.value.clusterDim.y);
CCCLRT_CHECK(expectedAttr.value.clusterDim.z == actualAttr.value.clusterDim.z);
has_cluster = true;
break;
case CU_LAUNCH_ATTRIBUTE_COOPERATIVE:
CCCLRT_CHECK(expectedAttr.value.cooperative == actualAttr.value.cooperative);
break;
case CU_LAUNCH_ATTRIBUTE_PRIORITY:
CCCLRT_CHECK(expectedAttr.value.priority == actualAttr.value.priority);
break;
default:
CCCLRT_CHECK(false);
break;
}
break;
}
}
INFO("Searched attribute is " << expectedAttr.id);
CCCLRT_CHECK(j != expectedConfig.numAttrs);
}
if (!has_cluster || !skip_device_exec(arch_filter<std::less<int>, 90>))
{
return ::cuda::__driver::__launchKernel(config, kernel, args);
}
}
__global__ void empty_kernel(int i) {}
template <bool HasCluster>
auto make_test_dims(const dim3& grid_dims, const dim3& block_dims, const dim3& cluster_dims = dim3())
{
if constexpr (HasCluster)
{
return cuda::make_hierarchy(
cuda::grid_dims(grid_dims), cuda::cluster_dims(cluster_dims), cuda::block_dims(block_dims));
}
else
{
return cuda::make_hierarchy(cuda::grid_dims(grid_dims), cuda::block_dims(block_dims));
}
}
auto add_cluster(const dim3& cluster_dims, CUlaunchAttribute& attr)
{
attr.id = CU_LAUNCH_ATTRIBUTE_CLUSTER_DIMENSION;
attr.value.clusterDim = {cluster_dims.x, cluster_dims.y, cluster_dims.z};
}
template <bool HasCluster, typename... Dims>
auto configuration_test(
::cuda::stream_ref stream, const dim3& grid_dims, const dim3& block_dims, const dim3& cluster_dims = dim3())
{
auto dims = make_test_dims<HasCluster>(grid_dims, block_dims, cluster_dims);
expectedConfig = {};
expectedConfig.hStream = stream.get();
if constexpr (HasCluster)
{
expectedConfig.gridDimX = grid_dims.x * cluster_dims.x;
expectedConfig.gridDimY = grid_dims.y * cluster_dims.y;
expectedConfig.gridDimZ = grid_dims.z * cluster_dims.z;
}
else
{
expectedConfig.gridDimX = grid_dims.x;
expectedConfig.gridDimY = grid_dims.y;
expectedConfig.gridDimZ = grid_dims.z;
}
expectedConfig.blockDimX = block_dims.x;
expectedConfig.blockDimY = block_dims.y;
expectedConfig.blockDimZ = block_dims.z;
SECTION("Simple cooperative launch")
{
CUlaunchAttribute attrs[2];
auto config = cuda::make_config(dims, cuda::cooperative_launch());
expectedConfig.numAttrs = 1 + HasCluster;
expectedConfig.attrs = &attrs[0];
expectedConfig.attrs[0].id = CU_LAUNCH_ATTRIBUTE_COOPERATIVE;
expectedConfig.attrs[0].value.cooperative = 1;
if constexpr (HasCluster)
{
add_cluster(cluster_dims, expectedConfig.attrs[1]);
}
cuda::launch(stream, config, empty_kernel, 0);
}
SECTION("Priority and dynamic smem")
{
CUlaunchAttribute attrs[2];
constexpr int priority = 42;
constexpr int num_ints = 128;
auto config =
cuda::make_config(dims, cuda::launch_priority(priority), cuda::dynamic_shared_memory<int[num_ints]>());
expectedConfig.sharedMemBytes = num_ints * sizeof(int);
expectedConfig.numAttrs = 1 + HasCluster;
expectedConfig.attrs = &attrs[0];
expectedConfig.attrs[0].id = CU_LAUNCH_ATTRIBUTE_PRIORITY;
expectedConfig.attrs[0].value.priority = priority;
if constexpr (HasCluster)
{
add_cluster(cluster_dims, expectedConfig.attrs[1]);
}
cuda::launch(stream, config, empty_kernel, 0);
}
SECTION("Large dynamic smem")
{
// Exceed the default 48kB of shared to check if its properly handled
// TODO move to launch option (available since CUDA 12.4)
struct S
{
int arr[13 * 1024];
};
CUlaunchAttribute attrs[1];
auto config = cuda::make_config(dims, cuda::dynamic_shared_memory<S>(cuda::non_portable));
expectedConfig.sharedMemBytes = sizeof(S);
expectedConfig.numAttrs = HasCluster;
expectedConfig.attrs = &attrs[0];
if constexpr (HasCluster)
{
add_cluster(cluster_dims, expectedConfig.attrs[0]);
}
cuda::launch(stream, config, empty_kernel, 0);
}
stream.sync();
}
C2H_TEST("Launch configuration", "[launch]")
{
cudaStream_t stream;
CUDART(cudaStreamCreate(&stream));
SECTION("No cluster")
{
configuration_test<false>(stream, 8, 64);
}
SECTION("With cluster")
{
configuration_test<true>(stream, 8, 32, 2);
}
CUDART(cudaStreamDestroy(stream));
CCCLRT_CHECK(replacementCalled);
}
C2H_TEST("Hierarchy construction in config", "[launch]")
{
auto config = cuda::make_config(cuda::grid_dims<2>(), cuda::cooperative_launch());
static_assert(cuda::block.count(cuda::grid, config) == 2);
auto config_larger = cuda::make_config(cuda::grid_dims<2>(), cuda::block_dims(256), cuda::cooperative_launch());
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config_larger) == 512);
auto config_no_options = cuda::make_config(cuda::grid_dims(2), cuda::block_dims<128>());
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config_no_options) == 256);
[[maybe_unused]] auto config_no_dims = cuda::make_config(cuda::cooperative_launch());
static_assert(
cuda::std::is_same_v<::cuda::std::remove_cvref_t<decltype(config_no_dims.hierarchy())>, cuda::__empty_hierarchy>);
}
C2H_TEST("Configuration combine", "[launch]")
{
auto grid = cuda::grid_dims<2>;
auto cluster = cuda::cluster_dims<2, 2>;
auto block = cuda::block_dims(256);
SECTION("Combine with no overlap")
{
auto config_part1 = cuda::make_config(grid);
auto config_part2 = cuda::make_config(block, cuda::launch_priority(2));
auto combined = config_part1.combine(config_part2);
[[maybe_unused]] auto combined_other_way = config_part2.combine(config_part1);
[[maybe_unused]] auto combined_with_empty = combined.combine(cuda::make_config());
[[maybe_unused]] auto empty_with_combined = cuda::make_config().combine(combined);
static_assert(
cuda::std::is_same_v<decltype(combined), decltype(cuda::make_config(grid, block, cuda::launch_priority(2)))>);
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_other_way)>);
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_with_empty)>);
static_assert(cuda::std::is_same_v<decltype(combined), decltype(empty_with_combined)>);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined) == 512);
}
SECTION("Combine with overlap")
{
auto config_part1 = make_config(grid, cluster, cuda::launch_priority(2));
auto config_part2 = make_config(cuda::cluster_dims<256>(), block, cuda::launch_priority(42));
auto combined = config_part1.combine(config_part2);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined) == 2048);
CCCLRT_REQUIRE(cuda::std::get<0>(combined.options()).priority == 2);
auto replaced_one_option = cuda::make_config(cuda::launch_priority(3)).combine(combined);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, replaced_one_option) == 2048);
CCCLRT_REQUIRE(cuda::std::get<0>(replaced_one_option.options()).priority == 3);
[[maybe_unused]] auto combined_with_extra_option = combined.combine(cuda::make_config(cuda::cooperative_launch()));
static_assert(
cuda::std::is_same_v<decltype(combined.hierarchy()), decltype(combined_with_extra_option.hierarchy())>);
static_assert(
cuda::std::tuple_size_v<::cuda::std::remove_cvref_t<decltype(combined_with_extra_option.options())>> == 2);
}
}
#if !_CCCL_CUDA_COMPILER(CLANG)
template <typename Config>
TEST_FUNC void test_queries_on_config(const Config& config)
{
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, config) == dim3(1024));
{
auto dims = cuda::gpu_thread.dims_as<int>(cuda::grid, config);
CCCLRT_REQUIRE(dims.x == 1024);
CCCLRT_REQUIRE(dims.y == 1);
CCCLRT_REQUIRE(dims.z == 1);
}
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, config) == 256);
CCCLRT_REQUIRE(cuda::gpu_thread.count_as<int>(cuda::block, config) == 256);
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config) == 1024);
CCCLRT_REQUIRE(cuda::gpu_thread.count_as<int>(cuda::grid, config) == 1024);
CCCLRT_REQUIRE(cuda::block.extents(cuda::grid, config).extent(0) == 4);
CCCLRT_REQUIRE(cuda::block.extents_as<int>(cuda::grid, config).extent(0) == 4);
NV_IF_TARGET(
NV_IS_DEVICE,
(CCCLRT_REQUIRE(cuda::block.rank(cuda::grid, config) == blockIdx.x);
CCCLRT_REQUIRE(cuda::block.rank_as<int>(cuda::grid, config) == blockIdx.x);
CCCLRT_REQUIRE(cuda::gpu_thread.index(cuda::block, config) == threadIdx);
{
auto idx = cuda::gpu_thread.index_as<int>(cuda::block, config);
CCCLRT_REQUIRE(idx.x == static_cast<int>(threadIdx.x));
CCCLRT_REQUIRE(idx.y == static_cast<int>(threadIdx.y));
CCCLRT_REQUIRE(idx.z == static_cast<int>(threadIdx.z));
}));
}
template <typename Config>
__global__ void test_kernel(Config config)
{
test_queries_on_config(config);
}
C2H_TEST("Queries on config", "[launch]")
{
auto config = cuda::make_config(cuda::grid_dims(4), cuda::block_dims<256>(), cuda::cooperative_launch());
test_queries_on_config(config);
test_kernel<<<4, 256>>>(config);
CUDART(cudaGetLastError());
CUDART(cudaDeviceSynchronize());
}
#endif // !_CCCL_CUDA_COMPILER(CLANG)

View File

@@ -0,0 +1,108 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/devices>
#include <cuda/hierarchy>
#include <cuda/launch>
#include <cuda/std/cstddef>
#include <cuda/std/functional>
#include <cuda/std/span>
#include <cuda/std/type_traits>
#include <cuda/stream>
#include <testing.cuh>
#include "test_macros.h"
template <class T, class View>
struct TestKernel
{
template <class Config>
TEST_DEVICE_FUNC void operator()(const Config& config)
{
static_assert(cuda::std::is_same_v<View, decltype(cuda::dynamic_shared_memory(config))>);
static_assert(noexcept(cuda::dynamic_shared_memory(config)));
write_smem(cuda::dynamic_shared_memory(config));
}
TEST_DEVICE_FUNC void write_smem(T& view)
{
view = T{};
CCCLRT_REQUIRE_DEVICE(view == T{});
}
template <cuda::std::size_t N>
TEST_DEVICE_FUNC void write_smem(cuda::std::span<T, N> view)
{
for (cuda::std::size_t i = 0; i < view.size(); ++i)
{
view[i] = T{};
CCCLRT_REQUIRE_DEVICE(view[i] == T{});
}
}
};
template <class T, class View, class Opt>
void test_opt_and_launch(cuda::stream_ref stream, Opt opt)
{
static_assert(cuda::std::is_same_v<T, typename Opt::value_type>);
static_assert(cuda::std::is_same_v<View, typename Opt::view_type>);
const auto config = cuda::make_config(cuda::block_dims<1, 1>(), cuda::grid_dims<1, 1>(), opt);
cuda::launch(stream, config, TestKernel<T, View>{});
stream.sync();
}
template <class T>
void test_ref(cuda::stream_ref stream)
{
static_assert(noexcept(cuda::dynamic_shared_memory<T>()));
test_opt_and_launch<T, T&>(stream, cuda::dynamic_shared_memory<T>());
}
void test_ref(cuda::stream_ref stream)
{
test_ref<int>(stream);
test_ref<float>(stream);
test_ref<double*>(stream);
test_ref<void (*)()>(stream);
}
template <class T, cuda::std::size_t N>
void test_span(cuda::stream_ref stream)
{
static_assert(!noexcept(cuda::dynamic_shared_memory<T[]>(N * 1024 * 1024)));
test_opt_and_launch<T, cuda::std::span<T>>(stream, cuda::dynamic_shared_memory<T[]>(N));
static_assert(noexcept(cuda::dynamic_shared_memory<T[N]>()));
test_opt_and_launch<T, cuda::std::span<T, N>>(stream, cuda::dynamic_shared_memory<T[N]>());
}
void test_span(cuda::stream_ref stream)
{
test_span<int, 1>(stream);
test_span<int, 256>(stream);
test_span<float, 1>(stream);
test_span<float, 256>(stream);
test_span<double*, 1>(stream);
test_span<double*, 256>(stream);
test_span<void (*)(), 1>(stream);
test_span<void (*)(), 256>(stream);
}
C2H_TEST("Dynamic shared memory option", "[launch]")
{
cuda::device_ref device = cuda::devices[0];
cuda::stream stream{device};
test_ref(stream);
test_span(stream);
}

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// XFAIL: enable-tile
// error: indirect call is unsupported in tile code
// ADDITIONAL_COMPILE_FLAGS: --extended-lambda
// UNSUPPORTED: nvrtc
#include <cuda/devices>
#include <cuda/launch>
#include <cuda/stream>
#include "../common/utility.cuh"
#include "test_macros.h"
void test_extended_lambda()
{
cuda::stream stream{cuda::devices[0]};
test::pinned<int> i(0);
auto config = cuda::block_dims<32>() & cuda::grid_dims<1>();
auto assign_42_lambda = [] TEST_DEVICE_FUNC(int* pi) {
*pi = 42;
};
cuda::launch(stream, config, assign_42_lambda, i.get());
stream.sync();
assert(*i == 42);
auto assign_1337_lambda = [] TEST_DEVICE_FUNC(auto config, int* pi) {
static_assert(cuda::gpu_thread.count(cuda::block, config) == 32);
static_assert(cuda::block.count(cuda::grid, config) == 1);
*pi = 1337;
};
cuda::launch(stream, config, assign_1337_lambda, config, i.get());
stream.sync();
assert(*i == 1337);
}
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, test_extended_lambda();)
return 0;
}

View File

@@ -0,0 +1,307 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/__launch/host_launch.h>
#include <cuda/__stream/stream.h>
#include <cuda/atomic>
#include <cuda/devices>
#include <cuda/memory>
#include <cooperative_groups.h>
#include <testing.cuh>
void block_stream(cuda::stream_ref stream, cuda::atomic<int>& atomic)
{
auto block_lambda = [&]() {
while (atomic != 1)
;
};
cuda::host_launch(stream, block_lambda);
}
void unblock_and_wait_stream(cuda::stream_ref stream, cuda::atomic<int>& atomic)
{
CCCLRT_REQUIRE(!stream.is_done());
atomic = 1;
stream.sync();
atomic = 0;
}
bool ordinary_function_run_proof = false;
template <class Ret, class... Args>
Ret ordinary_function(Args...)
{
ordinary_function_run_proof = true;
return (Ret) 0;
}
[[nodiscard]] int nodiscard_ordinary_function()
{
ordinary_function_run_proof = true;
return 0;
}
void launch_local_lambda(cuda::stream_ref stream, int& set, int set_to)
{
auto lambda = [&set, set_to]() {
set = set_to;
};
cuda::host_launch(stream, lambda);
}
template <typename Lambda>
struct lambda_wrapper
{
Lambda lambda;
lambda_wrapper(const Lambda& lambda)
: lambda(lambda)
{}
lambda_wrapper(lambda_wrapper&&) = default;
lambda_wrapper(const lambda_wrapper&) = default;
void operator()()
{
if constexpr (cuda::std::is_same_v<cuda::std::invoke_result_t<Lambda>, void*>)
{
// If lambda returns the address it captured, confirm this object wasn't moved
CCCLRT_REQUIRE(lambda() == this);
}
else
{
lambda();
}
}
// Make sure we fail if const is added to this wrapper anywhere
void operator()() const
{
CCCLRT_REQUIRE(false);
}
};
struct MoveOnlyArg
{
static MoveOnlyArg make()
{
return MoveOnlyArg{};
}
MoveOnlyArg(const MoveOnlyArg& other) = delete;
MoveOnlyArg(MoveOnlyArg&&) = default;
MoveOnlyArg& operator=(const MoveOnlyArg&) = delete;
MoveOnlyArg& operator=(MoveOnlyArg&&) = delete;
private:
MoveOnlyArg() = default;
};
struct MoveOnlyCallable
{
static MoveOnlyCallable make()
{
return MoveOnlyCallable{};
}
MoveOnlyCallable(const MoveOnlyCallable&) = delete;
MoveOnlyCallable(MoveOnlyCallable&&) = default;
MoveOnlyCallable& operator=(const MoveOnlyCallable&) = delete;
MoveOnlyCallable& operator=(MoveOnlyCallable&&) = delete;
void operator()(MoveOnlyArg) {}
private:
MoveOnlyCallable() = default;
};
C2H_CCCLRT_TEST("Host launch", "")
{
cuda::device_ref device{0};
device.init();
cuda::stream stream{device};
SECTION("Ordinary function without arguments returning void")
{
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
cuda::host_launch(stream, ordinary_function<void>);
stream.sync();
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
ordinary_function_run_proof = false;
}
SECTION("Ordinary function without arguments returning int")
{
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
cuda::host_launch(stream, ordinary_function<int>);
stream.sync();
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
ordinary_function_run_proof = false;
}
SECTION("Ordinary function with arguments returning void")
{
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
cuda::host_launch(stream, ordinary_function<int, char, double>, 'c', 1.0);
stream.sync();
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
ordinary_function_run_proof = false;
}
SECTION("Nodiscard ordinary function")
{
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
cuda::host_launch(stream, nodiscard_ordinary_function);
stream.sync();
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
ordinary_function_run_proof = false;
}
cuda::atomic<int> atomic = 0;
int i = 0;
auto set_lambda = [&](int set) {
i = set;
};
SECTION("Can do a host launch")
{
block_stream(stream, atomic);
cuda::host_launch(stream, set_lambda, 2);
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(i == 2);
}
SECTION("Can launch multiple functions")
{
block_stream(stream, atomic);
auto check_lambda = [&]() {
CCCLRT_REQUIRE(i == 4);
};
cuda::host_launch(stream, set_lambda, 3);
cuda::host_launch(stream, set_lambda, 4);
cuda::host_launch(stream, check_lambda);
cuda::host_launch(stream, set_lambda, 5);
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(i == 5);
}
SECTION("Non trivially copyable")
{
std::string s = "hello";
cuda::host_launch(
stream,
[&](auto str_arg) {
CCCLRT_REQUIRE(s == str_arg);
},
s);
stream.sync();
}
SECTION("Confirm no const added to the callable")
{
lambda_wrapper wrapped_lambda([&]() {
i = 21;
});
cuda::host_launch(stream, wrapped_lambda);
stream.sync();
CCCLRT_REQUIRE(i == 21);
}
SECTION("Can launch a local function and return")
{
block_stream(stream, atomic);
launch_local_lambda(stream, i, 42);
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(i == 42);
}
SECTION("Launch by reference")
{
// Grab the pointer to confirm callable was not moved
void* wrapper_ptr = nullptr;
lambda_wrapper another_lambda_setter([&]() {
i = 84;
return wrapper_ptr;
});
wrapper_ptr = static_cast<void*>(&another_lambda_setter);
block_stream(stream, atomic);
cuda::host_launch(stream, cuda::std::ref(another_lambda_setter));
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(i == 84);
}
SECTION("Launch by reference with arguments")
{
i = 10;
int result = 0;
auto lambda = [&result](int j) {
result = j;
};
block_stream(stream, atomic);
cuda::host_launch(stream, cuda::std::ref(lambda), i);
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(result == 10);
}
SECTION("Launch by reference with arguments captured by reference")
{
i = 0;
auto lambda = [](int& j) {
j = 10;
};
block_stream(stream, atomic);
cuda::host_launch(stream, cuda::std::ref(lambda), cuda::std::ref(i));
unblock_and_wait_stream(stream, atomic);
CCCLRT_REQUIRE(i == 10);
}
SECTION("Check that host_launch works with move only callables and arguments")
{
cuda::host_launch(stream, MoveOnlyCallable::make(), MoveOnlyArg::make());
stream.sync();
}
}
C2H_CCCLRT_TEST("Host launch uses the stream device when current device differs", "[launch][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
cuda::stream stream{explicit_device};
int value = 0;
{
cuda::__ensure_current_context guard(current_device);
cuda::host_launch(stream, [&value]() {
value = 42;
});
}
stream.sync();
CCCLRT_REQUIRE(value == 42);
}

View File

@@ -0,0 +1,542 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/atomic>
#include <cuda/devices>
#include <cuda/launch>
#include <cuda/memory>
#include <cuda/stream>
#include <cooperative_groups.h>
#include <testing.cuh>
#include "test_macros.h"
#if !_CCCL_CUDA_COMPILER(CLANG)
__managed__ bool kernel_run_proof = false;
void check_kernel_run(cudaStream_t stream)
{
CUDART(cudaStreamSynchronize(stream));
CCCLRT_CHECK(kernel_run_proof);
kernel_run_proof = false;
}
struct kernel_run_proof_check
{
TEST_DEVICE_FUNC void operator()()
{
CCCLRT_CHECK_DEVICE(kernel_run_proof);
kernel_run_proof = false;
}
};
struct functor_int_argument
{
TEST_DEVICE_FUNC void operator()(int dummy)
{
kernel_run_proof = true;
}
};
template <unsigned int BlockSize>
struct functor_taking_config
{
template <typename Config>
TEST_DEVICE_FUNC void operator()(Config config, int grid_size)
{
static_assert(cuda::gpu_thread.count(cuda::block, config) == BlockSize);
CCCLRT_REQUIRE_DEVICE(cuda::block.count(cuda::grid, config) == grid_size);
kernel_run_proof = true;
}
};
__global__ void kernel_no_arguments()
{
kernel_run_proof = true;
}
__global__ void kernel_int_argument(int dummy)
{
kernel_run_proof = true;
}
template <typename Config, unsigned int BlockSize>
__global__ void kernel_taking_config(Config config, int grid_size)
{
functor_taking_config<BlockSize>()(config, grid_size);
}
struct my_dynamic_smem_t
{
int i;
};
template <typename SmemType>
struct dynamic_smem_single
{
template <typename Config>
TEST_DEVICE_FUNC void operator()(Config config)
{
decltype(auto) dynamic_smem = cuda::dynamic_shared_memory(config);
static_assert(::cuda::std::is_same_v<SmemType&, decltype(dynamic_smem)>);
CCCLRT_REQUIRE_DEVICE(::cuda::device::is_object_from(dynamic_smem, ::cuda::device::address_space::shared));
kernel_run_proof = true;
}
};
template <typename SmemType, size_t Extent>
struct dynamic_smem_span
{
template <typename Config>
TEST_DEVICE_FUNC void operator()(Config config, int size)
{
auto dynamic_smem = cuda::dynamic_shared_memory(config);
static_assert(decltype(dynamic_smem)::extent == Extent);
static_assert(::cuda::std::is_same_v<SmemType&, decltype(dynamic_smem[1])>);
CCCLRT_REQUIRE_DEVICE(dynamic_smem.size() == size);
CCCLRT_REQUIRE_DEVICE(::cuda::device::is_object_from(dynamic_smem[1], ::cuda::device::address_space::shared));
kernel_run_proof = true;
}
};
struct launch_transform_to_int_convertible
{
int value_;
struct int_convertible
{
cudaStream_t stream_;
int value_;
int_convertible(cudaStream_t stream, int value) noexcept
: stream_(stream)
, value_(value)
{
// Check that the constructor runs before the kernel is launched
// Disabled for now because we don't handle it with graphs
// CUDAX_CHECK_FALSE(kernel_run_proof);
}
// Immovable to ensure that launch_transform doesn't copy the returned
// object
int_convertible(int_convertible&&) noexcept = delete;
~int_convertible() noexcept
{
// Check that the destructor runs after the kernel is launched
// Disabled for now because we don't handle it with graphs
// CUDART(cudaStreamSynchronize(stream_));
// CCCLRT_CHECK(kernel_run_proof);
}
// This is the value that will be passed to the kernel
int transformed_argument() const
{
return value_;
}
};
[[nodiscard]] friend int_convertible
transform_launch_argument(::cuda::stream_ref stream, launch_transform_to_int_convertible self) noexcept
{
return int_convertible(stream.get(), self.value_);
}
};
// Needs a separate function for Windows extended lambda
void launch_smoke_test(cudaStream_t dst)
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
// Use raw stream to make sure it can be implicitly converted on call to
// launch
cudaStream_t stream;
CUDART(cudaStreamCreate(&stream));
// Spell out all overloads to make sure they compile, include a check for
// implicit conversions
{
const int grid_size = 4;
constexpr int block_size = 256;
auto dimensions = cuda::make_hierarchy(cuda::grid_dims(grid_size), cuda::block_dims<256>());
auto config = cuda::make_config(dimensions);
// Not taking dims
{
cuda::launch(dst, config, kernel_no_arguments);
check_kernel_run(dst);
const int dummy = 1;
cuda::launch(dst, config, kernel_int_argument, dummy);
check_kernel_run(dst);
cuda::launch(dst, config, kernel_int_argument, 1);
check_kernel_run(dst);
cuda::launch(dst, config, kernel_int_argument, launch_transform_to_int_convertible{1});
check_kernel_run(dst);
cuda::launch(dst, config, kernel_int_argument, 1U);
check_kernel_run(dst);
cuda::launch(dst, config, functor_int_argument(), dummy);
check_kernel_run(dst);
cuda::launch(dst, config, functor_int_argument(), 1);
check_kernel_run(dst);
cuda::launch(dst, config, functor_int_argument(), launch_transform_to_int_convertible{1});
check_kernel_run(dst);
cuda::launch(dst, config, functor_int_argument(), 1U);
check_kernel_run(dst);
}
// Config argument
{
auto functor_instance = functor_taking_config<block_size>();
auto kernel_instance = kernel_taking_config<decltype(config), block_size>;
cuda::launch(dst, config, functor_instance, grid_size);
check_kernel_run(dst);
cuda::launch(dst, config, functor_instance, ::cuda::std::move(grid_size));
check_kernel_run(dst);
cuda::launch(dst, config, functor_instance, launch_transform_to_int_convertible{grid_size});
check_kernel_run(dst);
cuda::launch(dst, config, functor_instance, static_cast<unsigned int>(grid_size));
check_kernel_run(dst);
cuda::launch(dst, config, kernel_instance, grid_size);
check_kernel_run(dst);
cuda::launch(dst, config, kernel_instance, ::cuda::std::move(grid_size));
check_kernel_run(dst);
cuda::launch(dst, config, kernel_instance, launch_transform_to_int_convertible{grid_size});
check_kernel_run(dst);
cuda::launch(dst, config, kernel_instance, static_cast<unsigned int>(grid_size));
check_kernel_run(dst);
}
}
// Dynamic shared memory option
{
auto config = cuda::block_dims<32>() & cuda::grid_dims<1>();
auto test = [&](const auto& input_config) {
// Single element
{
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t>());
cuda::launch(dst, config, dynamic_smem_single<my_dynamic_smem_t>());
check_kernel_run(dst);
}
// Dynamic span
{
const int size = 2;
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t[]>(size));
cuda::launch(dst, config, dynamic_smem_span<my_dynamic_smem_t, ::cuda::std::dynamic_extent>(), size);
check_kernel_run(dst);
}
// Static span
{
constexpr int size = 3;
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t[size]>());
cuda::launch(dst, config, dynamic_smem_span<my_dynamic_smem_t, size>(), size);
check_kernel_run(dst);
}
};
test(config);
test(config.add(cuda::cooperative_launch(), cuda::launch_priority(0)));
}
}
C2H_CCCLRT_TEST("Launch smoke stream", "[launch]")
{
// Use raw stream to make sure it can be implicitly converted on call to
// launch
cudaStream_t stream;
{
::cuda::__ensure_current_context guard(cuda::device_ref{0});
CUDART(cudaStreamCreate(&stream));
}
launch_smoke_test(stream);
{
::cuda::__ensure_current_context guard(cuda::device_ref{0});
CUDART(cudaStreamSynchronize(stream));
CUDART(cudaStreamDestroy(stream));
}
}
template <typename DefaultConfig>
struct kernel_with_default_config
{
DefaultConfig config;
kernel_with_default_config(DefaultConfig c)
: config(c)
{}
DefaultConfig default_config() const
{
return config;
}
template <typename Config, typename ConfigCheckFn>
TEST_DEVICE_FUNC void operator()(Config config, ConfigCheckFn check_fn)
{
check_fn(config);
}
};
struct verify_callable
{
template <typename Config>
TEST_DEVICE_FUNC void operator()(Config config)
{
static_assert(cuda::gpu_thread.count(cuda::block, config) == 256);
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, config) == 4);
cooperative_groups::this_grid().sync();
}
};
C2H_CCCLRT_TEST("Launch with default config", "")
{
cuda::stream stream{cuda::device_ref{0}};
auto grid = cuda::grid_dims(4);
auto block = cuda::block_dims<256>;
SECTION("Combine with empty")
{
kernel_with_default_config kernel{cuda::make_config(block, grid, cuda::cooperative_launch())};
static_assert(cuda::__is_kernel_config<decltype(kernel.default_config())>);
static_assert(cuda::__kernel_has_default_config<decltype(kernel)>);
cuda::launch(stream, cuda::make_config(), kernel, verify_callable{});
stream.sync();
}
SECTION("Combine with no overlap")
{
kernel_with_default_config kernel{cuda::make_config(block)};
cuda::launch(stream, cuda::make_config(grid, cuda::cooperative_launch()), kernel, verify_callable{});
stream.sync();
}
SECTION("Combine with overlap")
{
kernel_with_default_config kernel{cuda::make_config(cuda::block_dims<1>(), cuda::cooperative_launch())};
cuda::launch(stream, cuda::make_config(block, grid, cuda::cooperative_launch()), kernel, verify_callable{});
stream.sync();
}
}
// Regression test: cuda::launch must work when the calling function has
// a __restrict__-qualified pointer parameter. On some nvcc + host compiler
// combos, __restrict__ survives through the type transformation pipeline
// and causes a function pointer conversion failure in __get_kernel_launcher.
struct restrict_assign_functor
{
template <typename Config>
TEST_DEVICE_FUNC void operator()(Config config, int* __restrict__ dst)
{
*dst = 42;
}
};
// The __restrict__ on the function parameter is the trigger: on affected compilers,
// it leaks into the template args of __get_kernel_launcher via cuda::launch.
void launch_with_restrict_param(cuda::stream_ref stream, int* __restrict__ dst)
{
auto config = cuda::make_config(cuda::grid_dims(1), cuda::block_dims<1>());
cuda::launch(stream, config, restrict_assign_functor{}, dst);
}
C2H_CCCLRT_TEST("Launch functor with __restrict__ pointer arg", "[launch]")
{
cuda::stream stream{cuda::device_ref{0}};
test::pinned<int> val{0};
launch_with_restrict_param(stream, val.get());
stream.sync();
CCCLRT_CHECK(*val == 42);
}
C2H_CCCLRT_TEST("Launch uses the stream device when current device differs", "[launch][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
cuda::stream stream{explicit_device};
int* device_value{};
{
cuda::__ensure_current_context guard(explicit_device);
CUDART(cudaMalloc(reinterpret_cast<void**>(&device_value), sizeof(int)));
CUDART(cudaMemsetAsync(device_value, 0, sizeof(int), stream.get()));
}
{
cuda::__ensure_current_context guard(current_device);
auto config = cuda::make_config(cuda::grid_dims(1), cuda::block_dims<1>());
cuda::launch(stream, config, test::assign_42{}, device_value);
}
stream.sync();
int value{};
{
cuda::__ensure_current_context guard(explicit_device);
CUDART(cudaMemcpy(&value, device_value, sizeof(int), cudaMemcpyDeviceToHost));
CUDART(cudaFree(device_value));
}
CCCLRT_CHECK(value == 42);
}
__managed__ cuda::std::size_t launched_nthreads;
__managed__ cuda::std::size_t launched_nblocks;
__managed__ cuda::std::size_t launched_nclusters;
struct LaunchDimsFunctor
{
template <class Config>
TEST_DEVICE_FUNC void operator()(const Config& config)
{
cuda::atomic_ref(launched_nthreads)++;
if (threadIdx.x == 0 && threadIdx.y == 0 && threadIdx.z == 0)
{
cuda::atomic_ref(launched_nblocks)++;
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
({
if (__clusterRelativeBlockRank() == 0)
{
cuda::atomic_ref(launched_nclusters)++;
}
}),
({ cuda::atomic_ref(launched_nclusters)++; }));
}
}
};
template <class GridDesc, class BlockDesc>
void test_launch_dims(cuda::stream_ref stream, GridDesc grid_desc, BlockDesc block_desc)
{
launched_nthreads = 0;
launched_nblocks = 0;
launched_nclusters = 0;
const auto& grid_exts = grid_desc.extents();
const auto& block_exts = block_desc.extents();
const auto config = cuda::make_config(grid_desc, block_desc);
cuda::launch(stream, config, LaunchDimsFunctor{});
stream.sync();
const auto exp_nclusters = cuda::std::size_t{grid_exts.extent(0)} * grid_exts.extent(1) * grid_exts.extent(2);
CCCLRT_CHECK(launched_nclusters == exp_nclusters);
const auto exp_nblocks = exp_nclusters;
CCCLRT_CHECK(launched_nblocks == exp_nblocks);
const auto exp_nthreads = exp_nblocks * block_exts.extent(0) * block_exts.extent(1) * block_exts.extent(2);
CCCLRT_CHECK(launched_nthreads == exp_nthreads);
}
template <class GridDesc, class ClusterDesc, class BlockDesc>
void test_launch_dims(cuda::stream_ref stream, GridDesc grid_desc, ClusterDesc cluster_desc, BlockDesc block_desc)
{
launched_nthreads = 0;
launched_nblocks = 0;
launched_nclusters = 0;
const auto& grid_exts = grid_desc.extents();
const auto& cluster_exts = cluster_desc.extents();
const auto& block_exts = block_desc.extents();
const auto config = cuda::make_config(grid_desc, cluster_desc, block_desc);
cuda::launch(stream, config, LaunchDimsFunctor{});
stream.sync();
const auto exp_nclusters = cuda::std::size_t{grid_exts.extent(0)} * grid_exts.extent(1) * grid_exts.extent(2);
CCCLRT_CHECK(launched_nclusters == exp_nclusters);
const auto exp_nblocks = exp_nclusters * cluster_exts.extent(0) * cluster_exts.extent(1) * cluster_exts.extent(2);
CCCLRT_CHECK(launched_nblocks == exp_nblocks);
const auto exp_nthreads = exp_nblocks * block_exts.extent(0) * block_exts.extent(1) * block_exts.extent(2);
CCCLRT_CHECK(launched_nthreads == exp_nthreads);
}
C2H_TEST("Launch dims", "[launch]")
{
cuda::stream stream{cuda::device_ref{0}};
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::block_dims(dim3{3, 7}));
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::block_dims<3, 7>());
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::block_dims<2, 7, 9>());
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::block_dims(dim3{3, 7}));
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::block_dims<3, 7>());
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::block_dims<2, 7, 9>());
if (cuda::device_attributes::compute_capability_major(stream.device()) >= 9)
{
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims(dim3{3}), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims(dim3{3, 7}));
test_launch_dims(
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims(dim3{3}), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims<3, 7>());
test_launch_dims(
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims<2, 7, 9>());
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims<3>(), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims<1, 5>(), cuda::block_dims(dim3{3, 7}));
test_launch_dims(
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims<3>(), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims<1, 5>(), cuda::block_dims<3, 7>());
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims<2, 7, 9>());
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims(dim3{3}), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims(dim3{3, 7}));
test_launch_dims(
stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims(dim3{3}), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims<3, 7>());
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims<2, 7, 9>());
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims<3>(), cuda::block_dims(dim3{10}));
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims<1, 5>(), cuda::block_dims(dim3{3, 7}));
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims(dim3{2, 7, 9}));
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims<3>(), cuda::block_dims<10>());
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims<1, 5>(), cuda::block_dims<3, 7>());
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims<2, 7, 9>());
}
}
#endif // !_CCCL_CUDA_COMPILER(CLANG)

View File

@@ -0,0 +1,252 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/devices>
#include <cuda/std/type_traits>
#include <cuda/std/utility>
#include <cuda/stream>
#include <testing.cuh>
C2H_CCCLRT_TEST("Can create a stream and launch work into it", "[stream]")
{
cuda::stream str{cuda::device_ref{0}};
::test::pinned<int> i(0);
::test::launch_kernel_single_thread(str, ::test::assign_42{}, i.get());
str.sync();
CCCLRT_REQUIRE(*i == 42);
}
C2H_CCCLRT_TEST("From native handle", "[stream]")
{
cuda::__ensure_current_context guard(cuda::device_ref{0});
cudaStream_t handle;
CUDART(cudaStreamCreate(&handle));
{
auto stream = cuda::stream::from_native_handle(handle);
::test::pinned<int> i(0);
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
stream.sync();
CCCLRT_REQUIRE(*i == 42);
(void) stream.release();
}
CUDART(cudaStreamDestroy(handle));
}
template <typename StreamType>
void add_dependency_test(const StreamType& waiter, const StreamType& waitee)
{
CCCLRT_REQUIRE(waiter != waitee);
auto verify_dependency = [&](const auto& insert_dependency) {
::test::pinned<int> i(0);
::cuda::atomic_ref atomic_i(*i);
::test::launch_kernel_single_thread(waitee, ::test::spin_until_80{}, i.get());
::test::launch_kernel_single_thread(waitee, ::test::assign_42{}, i.get());
insert_dependency();
::test::launch_kernel_single_thread(waiter, ::test::verify_42{}, i.get());
CCCLRT_REQUIRE(atomic_i.load() != 42);
CCCLRT_REQUIRE(!waiter.is_done());
atomic_i.store(80);
waiter.sync();
waitee.sync();
};
SECTION("Stream wait declared event")
{
verify_dependency([&]() {
cuda::event ev(waitee);
waiter.wait(ev);
});
}
SECTION("Stream wait returned event")
{
verify_dependency([&]() {
auto ev = waitee.record_event();
waiter.wait(ev);
});
}
SECTION("Stream wait returned timed event")
{
verify_dependency([&]() {
auto ev = waitee.record_timed_event();
waiter.wait(ev);
});
}
SECTION("Stream wait stream")
{
verify_dependency([&]() {
waiter.wait(waitee);
});
}
}
C2H_CCCLRT_TEST("Can add dependency into a stream", "[stream]")
{
cuda::stream waiter{cuda::device_ref{0}}, waitee{cuda::device_ref{0}};
add_dependency_test<cuda::stream>(waiter, waitee);
add_dependency_test<cuda::stream_ref>(waiter, waitee);
}
C2H_CCCLRT_TEST("Stream priority", "[stream]")
{
cuda::stream stream_default_prio{cuda::device_ref{0}};
CCCLRT_REQUIRE(stream_default_prio.priority() == cuda::stream::default_priority);
auto priority = cuda::stream::default_priority - 1;
cuda::stream stream{cuda::device_ref{0}, priority};
CCCLRT_REQUIRE(stream.priority() == priority);
}
C2H_CCCLRT_TEST("Stream get device", "[stream]")
{
cuda::stream dev0_stream(cuda::device_ref{0});
CCCLRT_REQUIRE(dev0_stream.device() == 0);
cuda::__ensure_current_context guard(cuda::device_ref{*std::prev(cuda::devices.end())});
cudaStream_t stream_handle;
CUDART(cudaStreamCreate(&stream_handle));
auto stream_cudart = cuda::stream::from_native_handle(stream_handle);
CCCLRT_REQUIRE(stream_cudart.device() == *std::prev(cuda::devices.end()));
auto stream_ref_cudart = cuda::stream_ref(stream_handle);
CCCLRT_REQUIRE(stream_ref_cudart.device() == *std::prev(cuda::devices.end()));
}
C2H_CCCLRT_TEST("Stream construction uses the explicit device", "[stream][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
auto stream = [&]() {
cuda::__ensure_current_context guard(current_device);
return cuda::stream{explicit_device};
}();
CCCLRT_REQUIRE(stream.device() == explicit_device);
}
C2H_CCCLRT_TEST("Stream dependency uses the explicit stream device", "[stream][multi_gpu]")
{
if (cuda::devices.size() < 2)
{
return;
}
cuda::device_ref current_device{0};
cuda::device_ref explicit_device{1};
cuda::stream waiter{explicit_device};
cuda::stream waitee{explicit_device};
::test::pinned<int> value(0);
::cuda::atomic_ref atomic_value(*value);
::test::launch_kernel_single_thread(waitee, ::test::spin_until_80{}, value.get());
::test::launch_kernel_single_thread(waitee, ::test::assign_42{}, value.get());
{
cuda::__ensure_current_context guard(current_device);
waiter.wait(waitee);
}
::test::launch_kernel_single_thread(waiter, ::test::verify_42{}, value.get());
CCCLRT_REQUIRE(atomic_value.load() != 42);
CCCLRT_REQUIRE(!waiter.is_done());
atomic_value.store(80);
waiter.sync();
waitee.sync();
}
C2H_CCCLRT_TEST("Stream ID", "[stream]")
{
STATIC_REQUIRE(cuda::std::is_same_v<unsigned long long, cuda::std::underlying_type_t<cuda::stream_id>>);
STATIC_REQUIRE(cuda::std::is_same_v<cuda::stream_id, decltype(cuda::std::declval<cuda::stream_ref>().id())>);
cuda::stream stream1{cuda::device_ref{0}};
cuda::stream stream2{cuda::device_ref{0}};
// Test that id() returns a valid ID
auto id1 = stream1.id();
auto id2 = stream2.id();
// Test that different streams have different IDs
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
CCCLRT_REQUIRE(cuda::std::to_underlying(id1) != cuda::std::to_underlying(id2));
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
CCCLRT_REQUIRE(id1 != id2);
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
// Test that the same stream returns the same ID when called multiple times
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
CCCLRT_REQUIRE(cuda::std::to_underlying(stream1.id()) == cuda::std::to_underlying(id1));
CCCLRT_REQUIRE(cuda::std::to_underlying(stream2.id()) == cuda::std::to_underlying(id2));
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
CCCLRT_REQUIRE(stream1.id() == id1);
CCCLRT_REQUIRE(stream2.id() == id2);
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
{
// Test that stream_ref also supports id()
// NULL stream needs a device to be set
cuda::__ensure_current_context guard(cuda::device_ref{0});
cuda::stream_ref ref1(::cudaStream_t{});
cuda::stream_ref ref2(stream1);
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
CCCLRT_REQUIRE(cuda::std::to_underlying(ref1.id()) != cuda::std::to_underlying(ref2.id()));
CCCLRT_REQUIRE(cuda::std::to_underlying(ref2.id()) == cuda::std::to_underlying(id1));
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
CCCLRT_REQUIRE(ref1.id() != ref2.id());
CCCLRT_REQUIRE(ref2.id() == id1);
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
}
}
C2H_CCCLRT_TEST("Invalid stream", "[stream]")
{
// 1. Test the signature
STATIC_REQUIRE(cuda::std::is_same_v<const cuda::invalid_stream_t, decltype(cuda::invalid_stream)>);
// 2. Test explicit construction of stream_ref from invalid_stream
STATIC_REQUIRE(cuda::std::is_constructible_v<cuda::stream_ref, cuda::invalid_stream_t>);
STATIC_REQUIRE(!cuda::std::is_convertible_v<cuda::invalid_stream_t, cuda::stream_ref>);
{
cuda::stream_ref stream{cuda::invalid_stream};
CCCLRT_REQUIRE(stream.get() == (cudaStream_t) (~0ull)); // NOLINT(performance-no-int-to-ptr)
}
// 3. Test stream_ref comparisons
{
cuda::stream_ref valid_stream{(cudaStream_t) (123ull)}; // NOLINT(performance-no-int-to-ptr)
cuda::stream_ref invalid_stream{cuda::invalid_stream};
CCCLRT_REQUIRE(!(valid_stream == cuda::invalid_stream));
CCCLRT_REQUIRE(invalid_stream == cuda::invalid_stream);
CCCLRT_REQUIRE(!(cuda::invalid_stream == valid_stream));
CCCLRT_REQUIRE(cuda::invalid_stream == invalid_stream);
CCCLRT_REQUIRE(valid_stream != cuda::invalid_stream);
CCCLRT_REQUIRE(!(invalid_stream != cuda::invalid_stream));
CCCLRT_REQUIRE(cuda::invalid_stream != valid_stream);
CCCLRT_REQUIRE(!(cuda::invalid_stream != invalid_stream));
}
}

View File

@@ -0,0 +1,76 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/__driver/driver_api.h>
#include <testing.cuh>
// This test is an exception and shouldn't use C2H_CCCLRT_TEST macro
C2H_TEST("Call each driver api", "[utility]")
{
namespace driver = ::cuda::__driver;
cudaStream_t stream;
// Assumes the ctx stack was empty or had one ctx, should be the case unless some other
// test leaves 2+ ctxs on the stack
// Pushes the primary context if the stack is empty
CUDART(cudaStreamCreate(&stream));
auto ctx = driver::__ctxGetCurrent();
CCCLRT_REQUIRE(ctx != nullptr);
// Confirm pop will leave the stack empty
driver::__ctxPop();
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == nullptr);
// Confirm we can push multiple times
driver::__ctxPush(ctx);
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
driver::__ctxPush(ctx);
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
driver::__ctxPop();
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
// Confirm stream ctx match
auto stream_ctx = driver::__streamGetCtx(stream);
CCCLRT_REQUIRE(ctx == stream_ctx);
CUDART(cudaStreamDestroy(stream));
CCCLRT_REQUIRE(driver::__deviceGet(0) == 0);
// Confirm we can retain the primary ctx that cudart retained first
auto primary_ctx = driver::__primaryCtxRetain(0);
CCCLRT_REQUIRE(ctx == primary_ctx);
driver::__ctxPop();
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == nullptr);
CCCLRT_REQUIRE(driver::__isPrimaryCtxActive(0));
// Confirm we can reset the primary context with double release
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
// Try a third release in case curand retained the primary ctx as well
if (driver::__isPrimaryCtxActive(0))
{
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
}
CCCLRT_REQUIRE(!driver::__isPrimaryCtxActive(0));
// Confirm cudart can recover
CUDART(cudaStreamCreate(&stream));
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
CUDART(driver::__streamDestroyNoThrow(stream));
}

View File

@@ -0,0 +1,50 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/__runtime/ensure_current_context.h>
#include <cuda/devices>
#include <testing.cuh>
namespace driver = cuda::__driver;
void recursive_check_device_setter(int id)
{
int cudart_id;
cuda::__ensure_current_context setter(cuda::device_ref{id});
CCCLRT_REQUIRE(test::count_driver_stack() == cuda::devices.size() - id);
auto ctx = driver::__ctxGetCurrent();
CUDART(cudaGetDevice(&cudart_id));
CCCLRT_REQUIRE(cudart_id == id);
if (id != 0)
{
recursive_check_device_setter(id - 1);
CCCLRT_REQUIRE(test::count_driver_stack() == cuda::devices.size() - id);
CCCLRT_REQUIRE(ctx == driver::__ctxGetCurrent());
CUDART(cudaGetDevice(&cudart_id));
CCCLRT_REQUIRE(cudart_id == id);
}
}
C2H_TEST("ensure current context", "[device]")
{
test::empty_driver_stack();
// If possible use something different than CUDART default 0
int target_device = static_cast<int>(cuda::devices.size() - 1);
SECTION("context setter")
{
recursive_check_device_setter(target_device);
CCCLRT_REQUIRE(test::count_driver_stack() == 0);
}
}

View File

@@ -0,0 +1,30 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// <cuda/std/chrono>
// system_clock
// static time_point from_time_t(time_t t);
#include <cuda/std/chrono>
#include <nv/target>
#include "test_macros.h"
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, ({
using C = ::std::chrono::system_clock;
C::time_point t1 = C::from_time_t(C::to_time_t(C::now()));
unused(t1);
}));
return 0;
}

View File

@@ -0,0 +1,30 @@
//===----------------------------------------------------------------------===//
//
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// <cuda/std/chrono>
// system_clock
// time_t to_time_t(const time_point& t);
#include <cuda/std/chrono>
#include <nv/target>
#include "test_macros.h"
int main(int, char**)
{
NV_IF_TARGET(NV_IS_HOST, ({
using C = ::std::chrono::system_clock;
cuda::std::time_t t1 = C::to_time_t(C::now());
unused(t1);
}));
return 0;
}

View File

@@ -0,0 +1,133 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/cmath>
#include <cuda/std/cassert>
#include <cuda/std/cstddef>
#include <cuda/std/limits>
#include <cuda/std/utility>
#include "test_macros.h"
#if !TEST_COMPILER(NVRTC)
# include <cstdint>
#endif // !TEST_COMPILER(NVRTC)
template <class T, class U>
TEST_FUNC constexpr void test()
{
constexpr T maxv = cuda::std::numeric_limits<T>::max();
// ensure that we return the right type
using Common = ::cuda::std::common_type_t<T, U>;
static_assert(cuda::std::is_same<decltype(cuda::ceil_div(T(0), U(1))), Common>::value);
assert(cuda::ceil_div(T(0), U(1)) == Common(0));
assert(cuda::ceil_div(T(1), U(1)) == Common(1));
assert(cuda::ceil_div(T(126), U(64)) == Common(2));
// ensure that we are resilient against overflow
assert(cuda::ceil_div(maxv, U(1)) == maxv);
assert(cuda::ceil_div(maxv, maxv) == Common(1));
}
template <class T>
TEST_FUNC constexpr void test()
{
// Builtin integer types:
test<T, char>();
test<T, signed char>();
test<T, unsigned char>();
test<T, short>();
test<T, unsigned short>();
test<T, int>();
test<T, unsigned int>();
test<T, long>();
test<T, unsigned long>();
test<T, long long>();
test<T, unsigned long long>();
#if !TEST_COMPILER(NVRTC)
// cstdint types:
test<T, std::size_t>();
test<T, std::ptrdiff_t>();
test<T, std::intptr_t>();
test<T, std::uintptr_t>();
test<T, std::int8_t>();
test<T, std::int16_t>();
test<T, std::int32_t>();
test<T, std::int64_t>();
test<T, std::uint8_t>();
test<T, std::uint16_t>();
test<T, std::uint32_t>();
test<T, std::uint64_t>();
#endif // !TEST_COMPILER(NVRTC)
#if _CCCL_HAS_INT128()
test<T, __int128_t>();
test<T, __uint128_t>();
#endif // _CCCL_HAS_INT128()
}
TEST_FUNC constexpr bool test()
{
// Builtin integer types:
test<char>();
test<signed char>();
test<unsigned char>();
test<short>();
test<unsigned short>();
test<int>();
test<unsigned int>();
test<long>();
test<unsigned long>();
test<long long>();
test<unsigned long long>();
#if !TEST_COMPILER(NVRTC)
// cstdint types:
test<std::size_t>();
test<std::ptrdiff_t>();
test<std::intptr_t>();
test<std::uintptr_t>();
test<std::int8_t>();
test<std::int16_t>();
test<std::int32_t>();
test<std::int64_t>();
test<std::uint8_t>();
test<std::uint16_t>();
test<std::uint32_t>();
test<std::uint64_t>();
#endif // !TEST_COMPILER(NVRTC)
#if _CCCL_HAS_INT128()
test<__int128_t>();
test<__uint128_t>();
#endif // _CCCL_HAS_INT128()
return true;
}
int main(int arg, char** argv)
{
test();
static_assert(test());
return 0;
}

Some files were not shown because too many files have changed in this diff Show More