[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
@@ -0,0 +1,70 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC __noinline__ void test_global_implicit_property(T ap, cudaAccessProperty cp)
|
||||
{
|
||||
// Test implicit conversions
|
||||
cudaAccessProperty v = ap;
|
||||
assert(cp == v);
|
||||
|
||||
// Test default, copy constructor, and copy-assignent
|
||||
cuda::access_property o(ap);
|
||||
cuda::access_property d;
|
||||
d = ap;
|
||||
|
||||
// Test explicit conversion to i64
|
||||
uint64_t x = (uint64_t) o;
|
||||
uint64_t y = (uint64_t) d;
|
||||
assert(x == y);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global()
|
||||
{
|
||||
cuda::access_property o(cuda::access_property::global{});
|
||||
uint64_t x = (uint64_t) o;
|
||||
unused(x);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_shared()
|
||||
{
|
||||
(void) cuda::access_property::shared{};
|
||||
}
|
||||
|
||||
static_assert(sizeof(cuda::access_property::shared) == 1);
|
||||
static_assert(sizeof(cuda::access_property::global) == 1);
|
||||
static_assert(sizeof(cuda::access_property::persisting) == 1);
|
||||
static_assert(sizeof(cuda::access_property::normal) == 1);
|
||||
static_assert(sizeof(cuda::access_property::streaming) == 1);
|
||||
static_assert(sizeof(cuda::access_property) == 8);
|
||||
|
||||
static_assert(alignof(cuda::access_property::shared) == 1);
|
||||
static_assert(alignof(cuda::access_property::global) == 1);
|
||||
static_assert(alignof(cuda::access_property::persisting) == 1);
|
||||
static_assert(alignof(cuda::access_property::normal) == 1);
|
||||
static_assert(alignof(cuda::access_property::streaming) == 1);
|
||||
static_assert(alignof(cuda::access_property) == 8);
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_implicit_property(cuda::access_property::normal{}, cudaAccessProperty::cudaAccessPropertyNormal);
|
||||
test_global_implicit_property(cuda::access_property::streaming{}, cudaAccessProperty::cudaAccessPropertyStreaming);
|
||||
test_global_implicit_property(cuda::access_property::persisting{}, cudaAccessProperty::cudaAccessPropertyPersisting);
|
||||
|
||||
test_global();
|
||||
test_shared();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,54 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
|
||||
// UNSUPPORTED: clang-14, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test_constexpr()
|
||||
{
|
||||
using namespace cuda;
|
||||
access_property a{}; // default constructor
|
||||
access_property b{a}; // copy constructor
|
||||
access_property c{cuda::std::move(a)}; // move constructor
|
||||
// user-declared ctor
|
||||
access_property d1{access_property::global{}};
|
||||
access_property d2{access_property::normal{}};
|
||||
access_property d3{access_property::streaming{}};
|
||||
access_property d4{access_property::persisting{}};
|
||||
auto p1 = static_cast<cudaAccessProperty>(access_property::normal{});
|
||||
auto p2 = static_cast<cudaAccessProperty>(access_property::streaming{});
|
||||
auto p3 = static_cast<cudaAccessProperty>(access_property::persisting{});
|
||||
// fraction ctor
|
||||
access_property e1{access_property::normal{}, 1.0f};
|
||||
access_property e2{access_property::streaming{}, 1.0f};
|
||||
access_property e3{access_property::persisting{}, 1.0f};
|
||||
access_property e4{access_property::normal{}, 1.0f, access_property::streaming{}};
|
||||
access_property e5{access_property::persisting{}, 1.0f, access_property::streaming{}};
|
||||
b = a; // copy assignment
|
||||
b = cuda::std::move(a); // move assignment
|
||||
auto value = static_cast<uint64_t>(a);
|
||||
unused(p1, p2, p3, b, c, d1, d2, d3, d4, e1, e2, e3, e4, e5, value);
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
static_assert(test_constexpr());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
TEST_FUNC __noinline__ void test_access_property_fail()
|
||||
{
|
||||
cuda::access_property o = cuda::access_property::normal{};
|
||||
// Test implicit conversion fails
|
||||
std::uint64_t x;
|
||||
x = o;
|
||||
unused(o);
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_access_property_fail();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,148 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// UNSUPPORTED: pre-sm-80
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
#include <cuda/cmath>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename Prop>
|
||||
TEST_DEVICE_FUNC constexpr cuda::__l2_evict_t to_enum()
|
||||
{
|
||||
if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::normal>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Normal_Demote;
|
||||
}
|
||||
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::streaming>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_First;
|
||||
}
|
||||
else if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::persisting>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Last;
|
||||
}
|
||||
else // if constexpr (cuda::std::is_same_v<Prop, cuda::access_property::global>)
|
||||
{
|
||||
return cuda::__l2_evict_t::_L2_Evict_Unchanged;
|
||||
}
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// test range
|
||||
|
||||
template <typename Primary, typename Secondary = void, int I = 1>
|
||||
TEST_DEVICE_FUNC void test_fraction_constexpr()
|
||||
{
|
||||
if constexpr (I > 16)
|
||||
{
|
||||
return;
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr auto fraction = static_cast<float>(I) * (1.0f / 16.0f);
|
||||
auto policy = cuda::__createpolicy_fraction(to_enum<Primary>(), to_enum<Secondary>(), fraction);
|
||||
if constexpr (cuda::std::is_void_v<Secondary>)
|
||||
{
|
||||
constexpr cuda::access_property property{Primary{}, fraction};
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
else
|
||||
{
|
||||
constexpr cuda::access_property property{Primary{}, fraction, Secondary{}};
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
test_fraction_constexpr<Primary, Secondary, I + 1>();
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void test_fraction()
|
||||
{
|
||||
test_fraction_constexpr<cuda::access_property::normal>();
|
||||
test_fraction_constexpr<cuda::access_property::streaming>();
|
||||
test_fraction_constexpr<cuda::access_property::persisting>();
|
||||
test_fraction_constexpr<cuda::access_property::normal, cuda::access_property::streaming>();
|
||||
test_fraction_constexpr<cuda::access_property::persisting, cuda::access_property::streaming>();
|
||||
}
|
||||
|
||||
//----------------------------------------------------------------------------------------------------------------------
|
||||
// test range
|
||||
|
||||
template <typename Primary, typename Secondary>
|
||||
__global__ void test_range_kernel(void* ptr, uint64_t property, uint32_t primary_size, uint32_t total_size)
|
||||
{
|
||||
auto policy = __createpolicy_range(to_enum<Primary>(), to_enum<Secondary>(), ptr, primary_size, total_size);
|
||||
if (static_cast<uint64_t>(property) != policy)
|
||||
{
|
||||
printf(" primary_size = %u, total_size = %u\n", primary_size, total_size);
|
||||
printf(" primary = %u, secondary = %u\n", (int) to_enum<Primary>(), (int) to_enum<Secondary>());
|
||||
printf(" 0x%llX vs 0x%llX\n", static_cast<unsigned long long>(policy), static_cast<unsigned long long>(property));
|
||||
}
|
||||
assert(static_cast<uint64_t>(property) == policy);
|
||||
}
|
||||
|
||||
template <typename Primary, typename Secondary = void>
|
||||
void test_range_launch(void* ptr, uint32_t primary_size, uint32_t total_size)
|
||||
{
|
||||
cuda::access_property property;
|
||||
if constexpr (cuda::std::is_void_v<Secondary>)
|
||||
{
|
||||
property = cuda::access_property{ptr, primary_size, total_size, Primary{}};
|
||||
}
|
||||
else
|
||||
{
|
||||
property = cuda::access_property{ptr, primary_size, total_size, Primary{}, Secondary{}};
|
||||
}
|
||||
test_range_kernel<Primary, Secondary><<<1, 1>>>(ptr, static_cast<uint64_t>(property), primary_size, total_size);
|
||||
}
|
||||
|
||||
void test_range()
|
||||
{
|
||||
int* ptr = nullptr;
|
||||
ptr++;
|
||||
for (uint32_t total_size = 1, i = 0; i <= 31; i++, total_size <<= 1)
|
||||
{
|
||||
for (uint32_t primary_size = 1, j = 0; j <= i; j++, primary_size <<= 1)
|
||||
{
|
||||
test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::streaming, cuda::access_property::streaming>(
|
||||
ptr, primary_size, total_size);
|
||||
test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(
|
||||
ptr, primary_size, total_size);
|
||||
}
|
||||
}
|
||||
// PTX createpolicy_range and access_property behaviors don't match (for now)
|
||||
// uint32_t primary_size = 0xFFFFFFFF;
|
||||
// uint32_t total_size = 0xFFFFFFFF;
|
||||
// test_range_launch<cuda::access_property::normal>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::persisting>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::global, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::normal, cuda::access_property::streaming>(ptr, primary_size, total_size);
|
||||
// test_range_launch<cuda::access_property::persisting, cuda::access_property::streaming>(ptr, primary_size,
|
||||
// total_size);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (test_range();))
|
||||
NV_IF_TARGET(NV_IS_HOST, (test_fraction<<<1, 1>>>();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,164 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::global>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::persisting>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::normal>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::streaming>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, global> must be pointer size");
|
||||
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property>) == 2 * sizeof(uintptr_t),
|
||||
"annotated_ptr<T,access_property> must be 2 * pointer size");
|
||||
|
||||
// NOTE: we could make these smaller in the future (e.g. 32-bit) but that would be an ABI breaking change:
|
||||
static_assert(sizeof(cuda::annotated_ptr<int, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<char, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
static_assert(sizeof(cuda::annotated_ptr<uintptr_t, cuda::access_property::shared>) == sizeof(uintptr_t),
|
||||
"annotated_ptr<T, shared> must be pointer size");
|
||||
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::global>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::persisting>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::normal>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::streaming>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
|
||||
// NOTE: we could lower the alignment in the future but that would be an ABI breaking change:
|
||||
static_assert(alignof(cuda::annotated_ptr<int, cuda::access_property::shared>) == alignof(int*),
|
||||
"annotated_ptr must align with int*");
|
||||
|
||||
#define N 128
|
||||
|
||||
struct S
|
||||
{
|
||||
int x;
|
||||
TEST_FUNC S& operator=(int o)
|
||||
{
|
||||
this->x = o;
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename In, typename T>
|
||||
TEST_FUNC __noinline__ void test_read_access(In i, T* r)
|
||||
{
|
||||
assert(i);
|
||||
assert(i - i == 0);
|
||||
assert((bool) i);
|
||||
const In o = i;
|
||||
|
||||
// assert(i->x == 0); // FAILS with shmem
|
||||
// assert(o->x == 0); // FAILS with shmem
|
||||
for (int n = 0; n < N; ++n)
|
||||
{
|
||||
assert(i[n].x == n);
|
||||
assert(&i[n] == &i[n]);
|
||||
assert(&i[n] == &r[n]);
|
||||
assert(o[n].x == n);
|
||||
assert(&o[n] == &o[n]);
|
||||
assert(&o[n] == &r[n]);
|
||||
}
|
||||
}
|
||||
|
||||
template <typename In>
|
||||
TEST_FUNC __noinline__ void test_write_access(In i)
|
||||
{
|
||||
assert(i);
|
||||
assert((bool) i);
|
||||
const In o = i;
|
||||
|
||||
for (int n = 0; n < N; ++n)
|
||||
{
|
||||
i[n].x = 2 * n;
|
||||
assert(i[n].x == 2 * n);
|
||||
assert(i[n].x == 2 * n);
|
||||
i[n].x = n;
|
||||
|
||||
o[n].x = 2 * n;
|
||||
assert(o[n].x == 2 * n);
|
||||
assert(o[n].x == 2 * n);
|
||||
o[n].x = n;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void all_tests()
|
||||
{
|
||||
S* arr = global_alloc<S, N>();
|
||||
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::normal>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::streaming>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::persisting>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::global>(arr), arr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property>(arr), arr);
|
||||
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::normal>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::streaming>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::persisting>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::global>(arr));
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property>(arr));
|
||||
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::normal>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::streaming>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::persisting>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::global>(arr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property>(arr));
|
||||
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(S* sarr = shared_alloc<S, N>(); // Allocating shared memory is only supported on device
|
||||
test_read_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<const S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_read_access(cuda::annotated_ptr<const volatile S, cuda::access_property::shared>(sarr), sarr);
|
||||
test_write_access(cuda::annotated_ptr<S, cuda::access_property::shared>(sarr));
|
||||
test_write_access(cuda::annotated_ptr<volatile S, cuda::access_property::shared>(sarr));))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
all_tests();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,157 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
TEST_DEVICE_FUNC void annotated_ptr_timing_dev(int* in, int* out)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
// Retrieve global id
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
|
||||
cuda::annotated_ptr<int, cuda::access_property> in_ann{in, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property> out_ann{out, ap};
|
||||
|
||||
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
|
||||
DPRINTF("&out[i]:%p = &in_ann[i]:%p for i = %d\n", &out_ann[i], &in_ann[i], i);
|
||||
|
||||
out_ann[i] = in_ann[i];
|
||||
};
|
||||
|
||||
__global__ void annotated_ptr_timing(int* in, int* out)
|
||||
{
|
||||
annotated_ptr_timing_dev(in, out);
|
||||
}
|
||||
|
||||
TEST_DEVICE_FUNC void ptr_timing_dev(int* in, int* out)
|
||||
{
|
||||
// Retrieve global id
|
||||
int i = blockIdx.x * blockDim.x + threadIdx.x;
|
||||
DPRINTF("&out[i]:%p = &in[i]:%p for i = %d\n", &out[i], &in[i], i);
|
||||
out[i] = in[i];
|
||||
};
|
||||
|
||||
__global__ void ptr_timing(int* in, int* out)
|
||||
{
|
||||
ptr_timing_dev(in, out);
|
||||
};
|
||||
|
||||
TEST_FUNC __noinline__ void bench()
|
||||
{
|
||||
#ifndef __CUDA_ARCH__
|
||||
static const size_t ARR_SZ = 1 << 22;
|
||||
static const size_t THREAD_CNT = 128;
|
||||
static const size_t BLOCK_CNT = ARR_SZ / THREAD_CNT;
|
||||
const dim3 threads(THREAD_CNT, 1, 1), blocks(BLOCK_CNT, 1, 1);
|
||||
cudaEvent_t start, stop;
|
||||
#else
|
||||
static const size_t ARR_SZ = 1 << 10;
|
||||
#endif
|
||||
int* arr0 = nullptr;
|
||||
int* arr1 = nullptr;
|
||||
float annotated_time = 0.f, pointer_time = 0.f;
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
arr0 = (int*) malloc(ARR_SZ * sizeof(int));
|
||||
arr1 = (int*) malloc(ARR_SZ * sizeof(int));
|
||||
#else
|
||||
assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
#endif
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
ptr_timing_dev(arr0, arr1);
|
||||
#else
|
||||
ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
#endif
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
#ifdef __CUDA_ARCH__
|
||||
ptr_timing_dev(arr0, arr1);
|
||||
#else
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
assert_rt(cudaEventCreate(&start));
|
||||
assert_rt(cudaEventCreate(&stop));
|
||||
assert_rt(cudaEventRecord(start));
|
||||
ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaEventRecord(stop));
|
||||
assert_rt(cudaEventSynchronize(stop));
|
||||
assert_rt(cudaEventElapsedTime(&pointer_time, start, stop));
|
||||
assert_rt(cudaEventDestroy(start));
|
||||
assert_rt(cudaEventDestroy(stop));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE,
|
||||
(annotated_ptr_timing_dev(arr0, arr1);),
|
||||
(assert_rt(cudaDeviceSynchronize()); annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaDeviceSynchronize());))
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(annotated_ptr_timing_dev(arr0, arr1);),
|
||||
(assert_rt(cudaDeviceSynchronize()); assert_rt(cudaEventCreate(&start)); assert_rt(cudaEventCreate(&stop));
|
||||
assert_rt(cudaEventRecord(start));
|
||||
annotated_ptr_timing<<<blocks, threads>>>(arr0, arr1);
|
||||
assert_rt(cudaEventRecord(stop));
|
||||
assert_rt(cudaEventSynchronize(stop));
|
||||
assert_rt(cudaEventElapsedTime(&annotated_time, start, stop));
|
||||
assert_rt(cudaEventDestroy(start));
|
||||
assert_rt(cudaEventDestroy(stop));
|
||||
assert_rt(cudaDeviceSynchronize());
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF("arr1[%d] == %d, should be:%d\n", i, arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}))
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
|
||||
|
||||
printf("array(ms):%f, arrotated_ptr(ms):%f\n", pointer_time, annotated_time);
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (bench();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// error: expression must have a constant value annotated_ptr.h: note #2701-D: attempt to access run-time storage
|
||||
// UNSUPPORTED: clang-14, gcc-12, gcc-11, gcc-10, gcc-9, gcc-8, gcc-7, msvc-19.29
|
||||
// UNSUPPORTED: msvc && nvcc-12.0
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test_public_methods()
|
||||
{
|
||||
using namespace cuda;
|
||||
using annotated_ptr = cuda::annotated_ptr<const int, access_property::persisting>;
|
||||
using annotated_smem_ptr [[maybe_unused]] = cuda::annotated_ptr<const int, access_property::shared>;
|
||||
annotated_ptr a{}; // default constructor
|
||||
annotated_ptr b{a}; // copy constructor
|
||||
annotated_ptr c{cuda::std::move(a)}; // move constructor
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (annotated_smem_ptr d{nullptr};)) // pointer constructor
|
||||
b = a; // copy assignment
|
||||
b = cuda::std::move(a); // move assignment
|
||||
auto diff = a - b;
|
||||
auto pred = static_cast<bool>(a);
|
||||
auto prop = a.__property();
|
||||
unused(c);
|
||||
unused(diff);
|
||||
unused(pred);
|
||||
unused(prop);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test_interleave_values()
|
||||
{
|
||||
using namespace cuda;
|
||||
constexpr auto normal = __l2_interleave(__l2_evict_t::_L2_Evict_Unchanged, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto streaming = __l2_interleave(__l2_evict_t::_L2_Evict_First, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto persisting = __l2_interleave(__l2_evict_t::_L2_Evict_Last, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
constexpr auto normal_demote =
|
||||
__l2_interleave(__l2_evict_t::_L2_Evict_Normal_Demote, __l2_evict_t::_L2_Evict_Unchanged, 1.0f);
|
||||
static_assert(normal == __l2_interleave_normal);
|
||||
static_assert(streaming == __l2_interleave_streaming);
|
||||
static_assert(persisting == __l2_interleave_persisting);
|
||||
static_assert(normal_demote == __l2_interleave_normal_demote);
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
static_assert(test_interleave_values());
|
||||
static_assert(test_public_methods());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_ctor(T* ptr)
|
||||
{
|
||||
// default ctor, cpy and cpy assignment
|
||||
cuda::annotated_ptr<T, P> def;
|
||||
{
|
||||
cuda::annotated_ptr<T, P> temp;
|
||||
temp = def;
|
||||
unused(temp);
|
||||
}
|
||||
cuda::annotated_ptr<T, P> other(def);
|
||||
unused(other);
|
||||
// from ptr
|
||||
cuda::annotated_ptr<T, P> a(ptr);
|
||||
assert(a);
|
||||
|
||||
// cpy ctor & assign to cv
|
||||
cuda::annotated_ptr<const T, P> c(def);
|
||||
cuda::annotated_ptr<volatile T, P> d(def);
|
||||
cuda::annotated_ptr<const volatile T, P> e(def);
|
||||
c = def;
|
||||
d = def;
|
||||
e = def;
|
||||
|
||||
// from c|v to c|v|cv
|
||||
cuda::annotated_ptr<const T, P> f(c);
|
||||
cuda::annotated_ptr<volatile T, P> g(d);
|
||||
cuda::annotated_ptr<const volatile T, P> h(e);
|
||||
f = c;
|
||||
g = d;
|
||||
h = e;
|
||||
unused(f, g, h);
|
||||
|
||||
// to cv
|
||||
cuda::annotated_ptr<const volatile T, P> i(c);
|
||||
cuda::annotated_ptr<const volatile T, P> j(d);
|
||||
i = c;
|
||||
j = d;
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_global_ctor()
|
||||
{
|
||||
T* rp = nullptr;
|
||||
rp++;
|
||||
test_ctor<T, P>(rp);
|
||||
// from ptr + prop
|
||||
P p;
|
||||
cuda::annotated_ptr<T, cuda::access_property> a(rp, p);
|
||||
cuda::annotated_ptr<const T, cuda::access_property> b(rp, p);
|
||||
cuda::annotated_ptr<volatile T, cuda::access_property> c(rp, p);
|
||||
cuda::annotated_ptr<const volatile T, cuda::access_property> d(rp, p);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global_ctors()
|
||||
{
|
||||
test_global_ctor<int, cuda::access_property::normal>();
|
||||
test_global_ctor<int, cuda::access_property::streaming>();
|
||||
test_global_ctor<int, cuda::access_property::persisting>();
|
||||
test_global_ctor<int, cuda::access_property::global>();
|
||||
test_global_ctor<int, cuda::access_property>();
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (__shared__ int smem_value; test_ctor<int, cuda::access_property::shared>(&smem_value);))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_ctors();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_ctor()
|
||||
{
|
||||
// default ctor, cpy and cpy assignment
|
||||
cuda::annotated_ptr<T, P> def;
|
||||
def = def;
|
||||
cuda::annotated_ptr<T, P> other(def);
|
||||
|
||||
// from ptr
|
||||
T* rp = nullptr;
|
||||
cuda::annotated_ptr<T, P> a(rp);
|
||||
assert(!a);
|
||||
|
||||
// cpy ctor & assign to cv
|
||||
cuda::annotated_ptr<const T, P> c(def);
|
||||
cuda::annotated_ptr<volatile T, P> d(def);
|
||||
cuda::annotated_ptr<const volatile T, P> e(def);
|
||||
c = e; // FAIL
|
||||
d = d; // FAIL
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_global_ctor()
|
||||
{
|
||||
test_ctor<T, P>();
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_global_ctors()
|
||||
{
|
||||
test_global_ctor<int, cuda::access_property::normal>();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_global_ctors();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
int* array0 = new int[9];
|
||||
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
|
||||
|
||||
array_anno_ptr = shared_ptr; // fail to compile, as expected
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
int* array0 = new int[9];
|
||||
cuda::annotated_ptr<int, cuda::access_property> array_anno_ptr{array0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property::shared> shared_ptr;
|
||||
|
||||
array_anno_ptr = shared_ptr; // fail to compile, as expected
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
// NVRTC does not do host side testing
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
TEST_FUNC static void fails_from_host()
|
||||
{
|
||||
int a;
|
||||
__nv_associate_access_property(&a, uint64_t{0});
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
// calling from host needs to fail and kill the app
|
||||
fails_from_host();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utils.h"
|
||||
|
||||
template <typename T, typename U>
|
||||
TEST_DEVICE_FUNC __noinline__ void shared_mem_test_dev()
|
||||
{
|
||||
T* smem = shared_alloc<T, 128>();
|
||||
smem[10] = 42;
|
||||
|
||||
cuda::annotated_ptr<U, cuda::access_property::shared> p{smem + 10};
|
||||
|
||||
assert(*p == 42);
|
||||
}
|
||||
|
||||
TEST_DEVICE_FUNC __noinline__ void test_all()
|
||||
{
|
||||
shared_mem_test_dev<int, int>();
|
||||
shared_mem_test_dev<int, const int>();
|
||||
shared_mem_test_dev<int, volatile int>();
|
||||
shared_mem_test_dev<int, const volatile int>();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_all();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
constexpr size_t array_size = 128;
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, array_size>();
|
||||
|
||||
cuda::apply_access_property(arr, array_size * sizeof(T), ap);
|
||||
|
||||
for (size_t i = 0; i < array_size; ++i)
|
||||
{
|
||||
assert(static_cast<size_t>(arr[i]) == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_aligned(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, array_size>();
|
||||
|
||||
cuda::apply_access_property(arr, cuda::aligned_size_t<sizeof(T)>(array_size * sizeof(T)), ap);
|
||||
|
||||
for (size_t i = 0; i < array_size; ++i)
|
||||
{
|
||||
assert(static_cast<size_t>(arr[i]) == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_all()
|
||||
{
|
||||
test<int>(cuda::access_property::normal{});
|
||||
test<int>(cuda::access_property::persisting{});
|
||||
test_aligned<int>(cuda::access_property::normal{});
|
||||
test_aligned<int>(cuda::access_property::persisting{});
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_all();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include "utils.h"
|
||||
#define ARR_SZ 128
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test(P ap)
|
||||
{
|
||||
T* arr = global_alloc<T, ARR_SZ>();
|
||||
|
||||
arr = cuda::associate_access_property(arr, ap);
|
||||
|
||||
for (int i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
assert(arr[i] == i);
|
||||
}
|
||||
|
||||
dealloc<T>(arr);
|
||||
}
|
||||
|
||||
template <typename T, typename P>
|
||||
TEST_FUNC __noinline__ void test_shared(P ap)
|
||||
{
|
||||
T* arr = shared_alloc<T, ARR_SZ>();
|
||||
|
||||
arr = cuda::associate_access_property(arr, ap);
|
||||
|
||||
for (int i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
assert(arr[i] == i);
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC __noinline__ void test_all()
|
||||
{
|
||||
test<int>(cuda::access_property::normal{});
|
||||
test<int>(cuda::access_property::persisting{});
|
||||
test<int>(cuda::access_property::streaming{});
|
||||
test<int>(cuda::access_property::global{});
|
||||
test<int>(cuda::access_property{});
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_shared<int>(cuda::access_property::shared{});))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_all();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,121 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
|
||||
#include "utils.h"
|
||||
|
||||
// TODO: global-shared
|
||||
// TODO: read const
|
||||
TEST_FUNC __noinline__ void test_memcpy_async()
|
||||
{
|
||||
size_t ARR_SZ = 1 << 10;
|
||||
int* arr0 = nullptr;
|
||||
int* arr1 = nullptr;
|
||||
cuda::access_property ap(cuda::access_property::persisting{});
|
||||
cuda::barrier<cuda::thread_scope_system> bar0, bar1, bar2, bar3;
|
||||
init(&bar0, 1);
|
||||
init(&bar1, 1);
|
||||
init(&bar2, 1);
|
||||
init(&bar3, 1);
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(arr0 = (int*) malloc(ARR_SZ * sizeof(int)); arr1 = (int*) malloc(ARR_SZ * sizeof(int));),
|
||||
(assert_rt(cudaMallocManaged((void**) &arr0, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaMallocManaged((void**) &arr1, ARR_SZ * sizeof(int)));
|
||||
assert_rt(cudaDeviceSynchronize());))
|
||||
|
||||
cuda::annotated_ptr<int, cuda::access_property> ann0{arr0, ap};
|
||||
cuda::annotated_ptr<int, cuda::access_property> ann1{arr1, ap};
|
||||
// cuda::annotated_ptr<const int, cuda::access_property> cann0{arr0, ap};
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
arr0[i] = static_cast<int>(i);
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(ann1, ann0, ARR_SZ * sizeof(int), bar0);
|
||||
// cuda::memcpy_async(ann1, cann0, ARR_SZ * sizeof(int), bar0);
|
||||
bar0.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(arr1, ann0, ARR_SZ * sizeof(int), bar1);
|
||||
// cuda::memcpy_async(arr1, cann0, ARR_SZ * sizeof(int), bar1);
|
||||
bar1.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i)
|
||||
{
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == static_cast<int>(i));
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(
|
||||
auto group = cooperative_groups::this_thread_block();
|
||||
|
||||
cuda::memcpy_async(group, ann1, ann0, ARR_SZ * sizeof(int), bar2);
|
||||
// cuda::memcpy_async(group, ann1, cann0, ARR_SZ * sizeof(int), bar2);
|
||||
bar2.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == (int) i);
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}
|
||||
|
||||
cuda::memcpy_async(group, arr1, ann0, ARR_SZ * sizeof(int), bar3);
|
||||
// cuda::memcpy_async(group, arr1, cann0, ARR_SZ * sizeof(int), bar3);
|
||||
bar3.arrive_and_wait();
|
||||
|
||||
for (size_t i = 0; i < ARR_SZ; ++i) {
|
||||
if (arr1[i] != (int) i)
|
||||
{
|
||||
DPRINTF(stderr, "%p:&arr1[i] == %d, should be:%lu\n", &arr1[i], arr1[i], i);
|
||||
assert(arr1[i] == (int) i);
|
||||
}
|
||||
|
||||
arr1[i] = 0;
|
||||
}))
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr0); free(arr1);), (assert_rt(cudaFree(arr0)); assert_rt(cudaFree(arr1));))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_memcpy_async();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_DIAG_SUPPRESS_MSVC(4505)
|
||||
|
||||
#include <cuda/annotated_ptr>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include <nv/target>
|
||||
|
||||
#if defined(DEBUG)
|
||||
# define DPRINTF(...) \
|
||||
{ \
|
||||
printf(__VA_ARGS__); \
|
||||
}
|
||||
#else
|
||||
# define DPRINTF(...) \
|
||||
do \
|
||||
{ \
|
||||
} while (false)
|
||||
#endif
|
||||
|
||||
TEST_FUNC void assert_rt_wrap(cudaError_t code, const char* file, int line)
|
||||
{
|
||||
if (code != cudaSuccess)
|
||||
{
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
NV_IF_ELSE_TARGET(NV_IS_HOST,
|
||||
(printf("assert: %s %s %d\n", cudaGetErrorString(code), file, line);),
|
||||
(printf("assert: error=%d %s %d\n", code, file, line);))
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
assert(code == cudaSuccess);
|
||||
}
|
||||
}
|
||||
#define assert_rt(ret) \
|
||||
{ \
|
||||
assert_rt_wrap((ret), __FILE__, __LINE__); \
|
||||
}
|
||||
|
||||
template <typename T, int N>
|
||||
TEST_FUNC __noinline__ T* global_alloc()
|
||||
{
|
||||
T* arr = nullptr;
|
||||
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_DEVICE, (arr = (T*) malloc(N * sizeof(T));), (assert_rt(cudaMallocManaged((void**) &arr, N * sizeof(T)));))
|
||||
|
||||
for (int i = 0; i < N; ++i)
|
||||
{
|
||||
arr[i] = i;
|
||||
}
|
||||
return arr;
|
||||
}
|
||||
|
||||
template <typename T, int N>
|
||||
TEST_DEVICE_FUNC __noinline__ T* shared_alloc()
|
||||
{
|
||||
__shared__ T data[N];
|
||||
|
||||
for (int i = 0; i < N; ++i)
|
||||
{
|
||||
data[i] = i;
|
||||
}
|
||||
return data;
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC __noinline__ void dealloc(T* arr)
|
||||
{
|
||||
NV_IF_ELSE_TARGET(NV_IS_DEVICE, (free(arr);), assert_rt(cudaFree(arr));)
|
||||
}
|
||||
@@ -0,0 +1,198 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct minimal_comparable_value
|
||||
{
|
||||
int value;
|
||||
};
|
||||
|
||||
TEST_FUNC constexpr bool operator<(minimal_comparable_value lhs, minimal_comparable_value rhs)
|
||||
{
|
||||
return lhs.value < rhs.value;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool operator==(minimal_comparable_value lhs, minimal_comparable_value rhs)
|
||||
{
|
||||
return lhs.value == rhs.value;
|
||||
}
|
||||
|
||||
namespace cuda::std
|
||||
{
|
||||
template <>
|
||||
class numeric_limits<minimal_comparable_value>
|
||||
{
|
||||
public:
|
||||
static constexpr bool is_specialized = true;
|
||||
|
||||
TEST_FUNC static constexpr minimal_comparable_value lowest() noexcept
|
||||
{
|
||||
return {0};
|
||||
}
|
||||
|
||||
TEST_FUNC static constexpr minimal_comparable_value max() noexcept
|
||||
{
|
||||
return {100};
|
||||
}
|
||||
};
|
||||
} // namespace cuda::std
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// --- static_bounds ---
|
||||
|
||||
// Basic static bounds
|
||||
{
|
||||
constexpr auto b = cuda::args::static_bounds<1, 4096>{};
|
||||
static_assert(b.lower() == 1);
|
||||
static_assert(b.upper() == 4096);
|
||||
}
|
||||
|
||||
// Exact static bounds
|
||||
{
|
||||
constexpr auto b = cuda::args::static_bounds<42, 42>{};
|
||||
static_assert(b.lower() == 42);
|
||||
static_assert(b.upper() == 42);
|
||||
}
|
||||
|
||||
// Long type deduced from NTTPs
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<decltype(cuda::args::static_bounds<0L, 1000L>::lower()), long>);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Static bounds preserve their original NTTP types
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds<1.0f, 8.0f>();
|
||||
static_assert(b.lower() == 1.0f);
|
||||
static_assert(b.upper() == 8);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.lower()), float>);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.upper()), float>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- runtime_bounds ---
|
||||
|
||||
// Basic runtime bounds
|
||||
{
|
||||
auto b = cuda::args::runtime_bounds{10, 100};
|
||||
assert(b.lower() == 10);
|
||||
assert(b.upper() == 100);
|
||||
static_assert(cuda::std::is_same_v<decltype(b.lower()), int>);
|
||||
}
|
||||
|
||||
// Default runtime bounds span the element type's numeric_limits range
|
||||
{
|
||||
constexpr cuda::args::runtime_bounds<int> b{};
|
||||
static_assert(b.lower() == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(b.upper() == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
|
||||
// --- argument_bounds factory functions ---
|
||||
|
||||
// Static via factory
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds<1, 8>();
|
||||
static_assert(b.lower() == 1);
|
||||
static_assert(b.upper() == 8);
|
||||
static_assert(cuda::args::__is_static_bounds_cv_v<decltype(b)>);
|
||||
static_assert(!cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
|
||||
}
|
||||
|
||||
// Runtime via factory
|
||||
{
|
||||
auto b = cuda::args::bounds(10, 100);
|
||||
assert(b.lower() == 10);
|
||||
assert(b.upper() == 100);
|
||||
static_assert(!cuda::args::__is_static_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_runtime_bounds_cv_v<decltype(b)>);
|
||||
static_assert(cuda::args::__is_bounds_v<decltype(b)>);
|
||||
}
|
||||
|
||||
// Runtime bounds only require operator< and operator==.
|
||||
{
|
||||
constexpr auto b = cuda::args::bounds(minimal_comparable_value{10}, minimal_comparable_value{20});
|
||||
static_assert(b.lower() == minimal_comparable_value{10});
|
||||
static_assert(b.upper() == minimal_comparable_value{20});
|
||||
}
|
||||
|
||||
// Static and runtime bounds intersection
|
||||
{
|
||||
static_assert(cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<1, 100>>(
|
||||
cuda::args::runtime_bounds<int>{50, 200}));
|
||||
static_assert(!cuda::args::__has_bounds_intersection<int, cuda::args::static_bounds<100, 200>>(
|
||||
cuda::args::runtime_bounds<int>{0, 50}));
|
||||
}
|
||||
|
||||
// Runtime bounds validation with no static bounds only requires operator< and operator==.
|
||||
{
|
||||
minimal_comparable_value values[] = {{10}, {20}};
|
||||
[[maybe_unused]] auto arg = cuda::args::deferred_sequence{
|
||||
cuda::std::span<minimal_comparable_value>{values, 2},
|
||||
cuda::args::bounds(minimal_comparable_value{5}, minimal_comparable_value{50})};
|
||||
}
|
||||
|
||||
// Unsigned no-bounds arguments must not instantiate a pointless `value < 0` comparison.
|
||||
{
|
||||
unsigned int value = 0;
|
||||
[[maybe_unused]] auto arg = cuda::args::deferred{&value};
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Static/runtime bounds intersection only requires operator< and operator==.
|
||||
{
|
||||
using static_bounds_t = cuda::args::static_bounds<minimal_comparable_value{10}, minimal_comparable_value{50}>;
|
||||
|
||||
constexpr auto runtime_bounds = cuda::args::bounds(minimal_comparable_value{20}, minimal_comparable_value{40});
|
||||
static_assert(cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(runtime_bounds));
|
||||
static_assert(!cuda::args::__has_bounds_intersection<minimal_comparable_value, static_bounds_t>(
|
||||
cuda::args::bounds(minimal_comparable_value{60}, minimal_comparable_value{70})));
|
||||
|
||||
cuda::args::__validate_static_element_bounds<minimal_comparable_value, static_bounds_t>(
|
||||
minimal_comparable_value{30});
|
||||
cuda::args::__validate_runtime_element_bounds(minimal_comparable_value{30}, runtime_bounds);
|
||||
|
||||
minimal_comparable_value values[] = {{20}, {30}};
|
||||
[[maybe_unused]] auto arg = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<minimal_comparable_value>{values, 2}, static_bounds_t{}, runtime_bounds};
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Non-bounds type
|
||||
{
|
||||
static_assert(!cuda::args::__is_bounds_v<int>);
|
||||
}
|
||||
|
||||
// Bounds types accepted by argument wrapper template parameters
|
||||
{
|
||||
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::no_bounds>);
|
||||
static_assert(cuda::args::__valid_static_bounds_v<int, cuda::args::static_bounds<1, 8>>);
|
||||
static_assert(!cuda::args::__valid_static_bounds_v<int, cuda::args::runtime_bounds<int>>);
|
||||
static_assert(!cuda::args::__valid_static_bounds_v<int, int>);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/expected>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/mdspan>
|
||||
#include <cuda/std/optional>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/tuple>
|
||||
#include <cuda/std/type_traits>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "test_iterators.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
enum class color
|
||||
{
|
||||
red,
|
||||
green,
|
||||
blue
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct element_type_like
|
||||
{
|
||||
using element_type = _Tp;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct range_like
|
||||
{
|
||||
using iterator = _Tp*;
|
||||
};
|
||||
|
||||
template <class _Tp>
|
||||
struct value_type_like
|
||||
{
|
||||
using value_type = _Tp;
|
||||
};
|
||||
|
||||
struct non_sequence_value
|
||||
{};
|
||||
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
// --- __is_sequence_v ---
|
||||
|
||||
// builtin and class type are not sequences
|
||||
static_assert(!cuda::args::__is_sequence_v<int>);
|
||||
static_assert(!cuda::args::__is_sequence_v<color>);
|
||||
static_assert(!cuda::args::__is_sequence_v<non_sequence_value>);
|
||||
static_assert(!cuda::args::__is_sequence_v<range_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<element_type_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<value_type_like<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::complex<float>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::pair<float, int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::tuple<float, int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::optional<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::std::expected<int, int>>);
|
||||
|
||||
// iterators and pointers can be sequences if they are at least random access
|
||||
static_assert(cuda::args::__is_sequence_v<int*>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int*>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::counting_iterator<int>>);
|
||||
static_assert(!cuda::args::__is_sequence_v<bidirectional_iterator<int*>>);
|
||||
|
||||
// ranges and arrays are sequences
|
||||
static_assert(cuda::args::__is_sequence_v<int[]>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int[]>);
|
||||
static_assert(cuda::args::__is_sequence_v<int[42]>);
|
||||
static_assert(cuda::args::__is_sequence_v<const int[42]>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int, 1>>);
|
||||
static_assert(cuda::args::__is_sequence_v<const cuda::std::span<int, 1>&>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::span<int>>);
|
||||
static_assert(cuda::args::__is_sequence_v<cuda::std::array<int, 3>>);
|
||||
|
||||
// --- __element_type_of_t ---
|
||||
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<const cuda::std::span<int, 1>&>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<int*>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::counting_iterator<int>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::array<int, 3>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<range_like<int>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<element_type_like<int>>, int>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<cuda::args::__element_type_of_t<cuda::std::mdspan<const int, cuda::std::extents<int, 1>>>, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__element_type_of_t<value_type_like<int>>, int>);
|
||||
|
||||
// --- argument_traits: is_deferred ---
|
||||
|
||||
static_assert(!cuda::args::__traits<int>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::immediate<int>>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_deferred);
|
||||
static_assert(!cuda::args::__traits<cuda::args::constant<42>>::is_deferred);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_deferred);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>::is_deferred);
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_deferred);
|
||||
|
||||
// --- argument_traits: is_single_value ---
|
||||
|
||||
static_assert(cuda::args::__traits<int>::is_single_value);
|
||||
static_assert(cuda::args::__traits<int*>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int*>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<cuda::counting_iterator<int>>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::deferred<int*>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
|
||||
// --- argument_traits: value_type ---
|
||||
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<int>::value_type, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type, int>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::value_type,
|
||||
cuda::std::span<int>>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<42>>::value_type, int>);
|
||||
static_assert(cuda::std::is_same_v<cuda::args::__traits<cuda::args::constant<10, float>>::value_type, float>);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::std::is_same_v<
|
||||
cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::value_type,
|
||||
cuda::std::array<int, 3>>);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- argument_traits: lowest / highest ---
|
||||
|
||||
static_assert(cuda::args::__traits<int>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<int>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__traits<const int>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<int&>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__traits<float>::lowest == cuda::std::numeric_limits<float>::lowest());
|
||||
static_assert(cuda::args::__traits<float>::highest == (cuda::std::numeric_limits<float>::max)());
|
||||
static_assert(cuda::args::__traits<const cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int, cuda::args::static_bounds<1, 8>>&>::highest == 8);
|
||||
static_assert(
|
||||
cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>, cuda::args::static_bounds<1, 8>>>::highest
|
||||
== 8);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::lowest == 10.0f);
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<10, float>>::highest == 10.0f);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{3, 1, 2}>>::highest == 3);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// --- Free function bounds on plain values ---
|
||||
|
||||
static_assert(cuda::args::__lowest_(42) == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__highest_(42) == (cuda::std::numeric_limits<int>::max)());
|
||||
static_assert(cuda::args::__lowest_(1.0f) == cuda::std::numeric_limits<float>::lowest());
|
||||
static_assert(cuda::args::__highest_(1.0f) == (cuda::std::numeric_limits<float>::max)());
|
||||
|
||||
// --- Scalar and sequence wrappers expose distinct single-value traits ---
|
||||
|
||||
static_assert(cuda::args::__traits<cuda::args::constant<42>>::is_single_value);
|
||||
static_assert(cuda::args::__traits<cuda::args::immediate<int>>::is_single_value);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,174 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Deferred single value via span<T, 1>
|
||||
{
|
||||
int val = 42;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
assert(cuda::args::__access::__arg(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
|
||||
// Deferred single value with static bounds
|
||||
{
|
||||
int val = 42;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 1000>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 1000);
|
||||
}
|
||||
|
||||
// Deferred single value via pointer
|
||||
{
|
||||
int val = 42;
|
||||
using def_t = cuda::args::deferred<int*, cuda::args::static_bounds<0, 100>>;
|
||||
static_assert(cuda::args::__traits<def_t>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<def_t>::highest == 100);
|
||||
// Also verify construction works
|
||||
auto def = cuda::args::deferred{&val, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def) == &val);
|
||||
}
|
||||
|
||||
// Deferred single value via fancy iterator
|
||||
{
|
||||
auto it = cuda::counting_iterator<int>{42};
|
||||
auto def = cuda::args::deferred{it, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 42);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::is_single_value);
|
||||
}
|
||||
|
||||
// Deferred single value with both bounds, runtime bounds first
|
||||
{
|
||||
int val = 42;
|
||||
auto def =
|
||||
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 256>()};
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 256);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
cuda::args::__access::__runtime_bounds(def) = cuda::args::bounds(5, 90);
|
||||
assert(cuda::args::__highest_(def) == 90);
|
||||
}
|
||||
|
||||
// Deferred sequence via fancy iterator
|
||||
{
|
||||
auto it = cuda::counting_iterator<int>{10};
|
||||
auto def = cuda::args::deferred_sequence{it, cuda::args::bounds<0, 100>()};
|
||||
assert(cuda::args::__unwrap(def)[0] == 10);
|
||||
assert(cuda::args::__unwrap(def)[2] == 12);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 0);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 100);
|
||||
static_assert(!cuda::args::__traits<decltype(def)>::is_single_value);
|
||||
}
|
||||
|
||||
// Deferred sequence with both bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto def = cuda::args::deferred_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 4096>(), cuda::args::bounds(5, 100)};
|
||||
assert(cuda::args::__access::__arg(def).size() == 4);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).lower() == 5);
|
||||
assert(cuda::args::__access::__runtime_bounds(def).upper() == 100);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
}
|
||||
|
||||
// Deferred sequence with both bounds, runtime bounds first
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto def = cuda::args::deferred_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds(5, 100), cuda::args::bounds<1, 4096>()};
|
||||
static_assert(cuda::args::__traits<decltype(def)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(def)>::highest == 4096);
|
||||
assert(cuda::args::__lowest_(def) == 5);
|
||||
assert(cuda::args::__highest_(def) == 100);
|
||||
}
|
||||
|
||||
// Traits: deferred is single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred<cuda::std::span<int, 1>>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
}
|
||||
|
||||
// Traits: deferred with pointer is also single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred<int*>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
}
|
||||
|
||||
// Traits: deferred_sequence is not single value
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::deferred_sequence<cuda::std::span<int>>>;
|
||||
static_assert(traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
}
|
||||
|
||||
// Unwrap: deferred
|
||||
{
|
||||
int val = 99;
|
||||
auto def = cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}};
|
||||
auto& v = cuda::args::__unwrap(def);
|
||||
assert(v[0] == 99);
|
||||
}
|
||||
|
||||
// Unwrap: deferred_sequence
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto def = cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}};
|
||||
const auto& v = cuda::args::__unwrap(def);
|
||||
assert(v.size() == 3);
|
||||
assert(v[1] == 20);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue deferred returns by value
|
||||
{
|
||||
int val = 99;
|
||||
auto v = cuda::args::__unwrap(cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}});
|
||||
assert(v[0] == 99);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue deferred_sequence returns by value
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto v = cuda::args::__unwrap(cuda::args::deferred_sequence{cuda::std::span<int>{arr, 3}});
|
||||
assert(v.size() == 3);
|
||||
assert(v[2] == 30);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
[[maybe_unused]] cuda::args::deferred_sequence<int> invalid_arg{0};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using traits = cuda::args::__traits<cuda::args::deferred_sequence<int>>;
|
||||
|
||||
[[maybe_unused]] constexpr bool invalid_traits = traits::is_deferred;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,181 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct non_sequence_value
|
||||
{
|
||||
int payload;
|
||||
};
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Uniform scalar via CTAD
|
||||
{
|
||||
auto da = cuda::args::immediate{5};
|
||||
assert(cuda::args::__unwrap(da) == 5);
|
||||
assert(cuda::args::__access::__arg(da) == 5);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == (cuda::std::numeric_limits<int>::max)());
|
||||
assert(cuda::args::__lowest_(da) == 5);
|
||||
assert(cuda::args::__highest_(da) == 5);
|
||||
cuda::args::__access::__arg(da) = 6;
|
||||
assert(cuda::args::__unwrap(da) == 6);
|
||||
}
|
||||
|
||||
// Uniform scalar with static bounds
|
||||
{
|
||||
auto da = cuda::args::immediate{5, cuda::args::bounds<1, 8>()};
|
||||
assert(cuda::args::__unwrap(da) == 5);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 8);
|
||||
assert(cuda::args::__lowest_(da) == 5);
|
||||
assert(cuda::args::__highest_(da) == 5);
|
||||
}
|
||||
|
||||
// Non-sequence values are accepted without scalar-only restrictions
|
||||
{
|
||||
auto da = cuda::args::immediate{non_sequence_value{7}};
|
||||
assert(cuda::args::__unwrap(da).payload == 7);
|
||||
}
|
||||
|
||||
// Pointer-like types can still represent a single value when explicitly wrapped that way
|
||||
{
|
||||
int value = 11;
|
||||
auto da = cuda::args::immediate{&value};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::is_single_value);
|
||||
assert(*cuda::args::__unwrap(da) == 11);
|
||||
}
|
||||
|
||||
// Per-segment span with runtime bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds(1L, 100L)};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__access::__arg(da).size() == 4);
|
||||
assert(cuda::args::__access::__runtime_bounds(da).lower() == 1);
|
||||
assert(cuda::args::__access::__runtime_bounds(da).upper() == 100);
|
||||
assert(cuda::args::__lowest_(da) == 1);
|
||||
assert(cuda::args::__highest_(da) == 100);
|
||||
cuda::args::__access::__runtime_bounds(da) = cuda::args::bounds(1, 90);
|
||||
assert(cuda::args::__highest_(da) == 90);
|
||||
}
|
||||
|
||||
// Per-segment span with both bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 256>(), cuda::args::bounds(10, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
|
||||
assert(cuda::args::__lowest_(da) == 10);
|
||||
assert(cuda::args::__highest_(da) == 200);
|
||||
}
|
||||
|
||||
// Per-segment span with both bounds, runtime bounds first
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{arr, 4}, cuda::args::bounds(10, 200), cuda::args::bounds<1, 256>()};
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 256);
|
||||
assert(cuda::args::__lowest_(da) == 10);
|
||||
assert(cuda::args::__highest_(da) == 200);
|
||||
}
|
||||
|
||||
// Per-segment via span
|
||||
{
|
||||
int arr[4] = {1, 2, 3, 4};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__unwrap(da)[0] == 1);
|
||||
assert(cuda::args::__unwrap(da)[3] == 4);
|
||||
}
|
||||
|
||||
// Per-segment with static bounds
|
||||
{
|
||||
int arr[4] = {10, 20, 30, 40};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 4}, cuda::args::bounds<1, 100>()};
|
||||
assert(cuda::args::__unwrap(da).size() == 4);
|
||||
assert(cuda::args::__unwrap(da)[2] == 30);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::lowest == 1);
|
||||
static_assert(cuda::args::__traits<decltype(da)>::highest == 100);
|
||||
}
|
||||
|
||||
// Traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::immediate<int>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, int>);
|
||||
}
|
||||
|
||||
// Sequence traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::span<int>>);
|
||||
}
|
||||
|
||||
// __is_sequence_v on unwrapped types
|
||||
{
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::immediate<int>>::value_type>);
|
||||
static_assert(!cuda::args::__traits<cuda::args::__immediate_sequence<cuda::std::span<int>>>::is_single_value);
|
||||
}
|
||||
|
||||
// Unwrap: scalar
|
||||
{
|
||||
auto da = cuda::args::immediate{7};
|
||||
auto& v = cuda::args::__unwrap(da);
|
||||
assert(v == 7);
|
||||
v = 8;
|
||||
assert(cuda::args::__unwrap(da) == 8);
|
||||
}
|
||||
|
||||
// Unwrap: span
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto da = cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}};
|
||||
const auto& v = cuda::args::__unwrap(da);
|
||||
assert(v.size() == 3);
|
||||
assert(v[1] == 20);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue scalar returns by value
|
||||
{
|
||||
const auto& v = cuda::args::__unwrap(cuda::args::immediate{7});
|
||||
assert(v == 7);
|
||||
}
|
||||
|
||||
// Unwrap: rvalue span returns by value
|
||||
{
|
||||
int arr[3] = {10, 20, 30};
|
||||
auto v = cuda::args::__unwrap(cuda::args::__immediate_sequence{cuda::std::span<int>{arr, 3}});
|
||||
assert(v.size() == 3);
|
||||
assert(v[2] == 30);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,23 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
// A type without a cuda::std::numeric_limits specialization has no meaningful implicit bounds. Default-constructing
|
||||
// runtime_bounds for such a type must be rejected at compile time instead of silently producing a degenerate range.
|
||||
struct unspecialized_type
|
||||
{};
|
||||
|
||||
[[maybe_unused]] cuda::args::runtime_bounds<unspecialized_type> invalid_bounds{};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,197 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct non_sequence_value
|
||||
{
|
||||
int payload;
|
||||
};
|
||||
|
||||
enum class dependent_direction
|
||||
{
|
||||
min,
|
||||
max
|
||||
};
|
||||
|
||||
template <dependent_direction Value>
|
||||
struct dependent_direction_tag
|
||||
{
|
||||
static constexpr auto value = Value;
|
||||
};
|
||||
|
||||
template <class Tag>
|
||||
TEST_FUNC void test_dependent_constant_type()
|
||||
{
|
||||
constexpr auto direction = Tag::value;
|
||||
using constant_t = cuda::args::constant<direction>;
|
||||
|
||||
// Regression: NVCC bug generated a host stub using a cv/ref-qualified constant type while device registration used
|
||||
// the unqualified type, causing cudaErrorInvalidDeviceFunction when launching the kernel.
|
||||
static_assert(cuda::std::is_same_v<typename constant_t::value_type, dependent_direction>);
|
||||
static_assert(cuda::std::is_same_v<constant_t, cuda::args::constant<Tag::value, dependent_direction>>);
|
||||
}
|
||||
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
// Basic value
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
static_assert(cuda::args::__unwrap(sa) == 42);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa)::value_type, int>);
|
||||
}
|
||||
|
||||
// Different types
|
||||
{
|
||||
constexpr auto sa_long = cuda::args::constant<100L>{};
|
||||
static_assert(cuda::args::__unwrap(sa_long) == 100L);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_long)::value_type, long>);
|
||||
|
||||
constexpr auto sa_float = cuda::args::constant<10, float>{};
|
||||
static_assert(cuda::args::__unwrap(sa_float) == 10.0f);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_float)::value_type, float>);
|
||||
static_assert(cuda::std::is_same_v<decltype(cuda::args::__unwrap(sa_float)), float>);
|
||||
}
|
||||
|
||||
// Negative value
|
||||
{
|
||||
constexpr auto sa_neg = cuda::args::constant<-1>{};
|
||||
static_assert(cuda::args::__unwrap(sa_neg) == -1);
|
||||
}
|
||||
|
||||
// Dependent value
|
||||
{
|
||||
test_dependent_constant_type<dependent_direction_tag<dependent_direction::max>>();
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Non-sequence values are accepted without scalar-only restrictions
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<non_sequence_value{7}>{};
|
||||
static_assert(cuda::args::__unwrap(sa).payload == 7);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Array sequence
|
||||
{
|
||||
constexpr auto sa_arr = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[0] == 128);
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[1] == 256);
|
||||
static_assert(cuda::args::__unwrap(sa_arr)[2] == 512);
|
||||
static_assert(cuda::std::is_same_v<decltype(sa_arr)::value_type, cuda::std::array<int, 3>>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Bounds: scalar
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == 42);
|
||||
static_assert(cuda::args::__highest_(sa) == 42);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Bounds: array sequence computes lowest/highest of elements
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{128, 256, 512}>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == 128);
|
||||
static_assert(cuda::args::__highest_(sa) == 512);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Bounds: empty array sequence has unconstrained element bounds
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 0>{}>{};
|
||||
static_assert(cuda::args::__lowest_(sa) == cuda::std::numeric_limits<int>::lowest());
|
||||
static_assert(cuda::args::__highest_(sa) == (cuda::std::numeric_limits<int>::max)());
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::constant<42>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, int>);
|
||||
static_assert(traits::lowest == 42);
|
||||
static_assert(traits::highest == 42);
|
||||
}
|
||||
|
||||
// Traits: explicit constant value type
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::constant<10, float>>;
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, float>);
|
||||
static_assert(cuda::std::is_same_v<traits::element_type, float>);
|
||||
static_assert(traits::lowest == 10.0f);
|
||||
static_assert(traits::highest == 10.0f);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Sequence traits
|
||||
{
|
||||
using traits = cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>;
|
||||
static_assert(traits::is_constant);
|
||||
static_assert(!traits::is_deferred);
|
||||
static_assert(!traits::is_single_value);
|
||||
static_assert(cuda::std::is_same_v<traits::value_type, cuda::std::array<int, 3>>);
|
||||
static_assert(cuda::std::is_same_v<traits::element_type, int>);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// Single value: scalar is single, sequence is not
|
||||
{
|
||||
static_assert(!cuda::args::__is_sequence_v<cuda::args::__traits<cuda::args::constant<42>>::value_type>);
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
static_assert(
|
||||
!cuda::args::__traits<cuda::args::__constant_sequence<cuda::std::array<int, 3>{1, 2, 3}>>::is_single_value);
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
// Unwrap: scalar
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<42>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val == 42);
|
||||
}
|
||||
|
||||
// Unwrap: scalar with explicit value type
|
||||
{
|
||||
constexpr auto sa = cuda::args::constant<10, float>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val == 10.0f);
|
||||
static_assert(cuda::std::is_same_v<decltype(val), const float>);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// Unwrap: sequence
|
||||
{
|
||||
constexpr auto sa = cuda::args::__constant_sequence<cuda::std::array<int, 3>{10, 20, 30}>{};
|
||||
constexpr auto val = cuda::args::__unwrap(sa);
|
||||
static_assert(val[0] == 10);
|
||||
static_assert(val[2] == 30);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using arg_t = cuda::args::immediate<int, cuda::args::runtime_bounds<int>>;
|
||||
|
||||
[[maybe_unused]] arg_t invalid_arg{0};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
using arg_t = cuda::args::immediate<unsigned char, cuda::args::static_bounds<0, 1000>>;
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_highest = cuda::args::__traits<arg_t>::highest;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,18 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_bounds = cuda::args::static_bounds<0, 1L>{};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/argument>
|
||||
|
||||
// Reading the implicit bounds of __traits for an element type without a cuda::std::numeric_limits specialization must
|
||||
// fail to compile rather than silently yielding a value-initialized (and therefore meaningless) bound. This exercises
|
||||
// the __traits_impl primary-template path, which is the bound surface read by generic consumers.
|
||||
struct unspecialized_type
|
||||
{};
|
||||
|
||||
[[maybe_unused]] constexpr auto invalid_lowest = cuda::args::__traits<unspecialized_type>::lowest;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,214 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// Integration test: demonstrates how an algorithm consumes argument wrappers
|
||||
// to make compile-time and runtime resource decisions.
|
||||
// All argument types (plain values, constants, immediate values, deferred values) work uniformly
|
||||
// through the free functions.
|
||||
|
||||
#include <cuda/argument>
|
||||
#include <cuda/std/algorithm>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/span>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
constexpr int shared_memory_capacity = 256;
|
||||
constexpr int default_max_segment_size = 1024;
|
||||
|
||||
enum class algorithm_variant
|
||||
{
|
||||
shared_memory,
|
||||
global_memory
|
||||
};
|
||||
|
||||
// Static scaling: choose algorithm variant at compile time.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr algorithm_variant select_variant(_SegSizeArg)
|
||||
{
|
||||
if constexpr (cuda::args::__traits<_SegSizeArg>::highest <= shared_memory_capacity)
|
||||
{
|
||||
return algorithm_variant::shared_memory;
|
||||
}
|
||||
else
|
||||
{
|
||||
return algorithm_variant::global_memory;
|
||||
}
|
||||
}
|
||||
|
||||
// Dynamic scaling: compute buffer size at runtime, clamped to default.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr int compute_buffer_size(_SegSizeArg __seg_size, int __num_segments)
|
||||
{
|
||||
auto __highest = cuda::std::min(default_max_segment_size, static_cast<int>(cuda::args::__highest_(__seg_size)));
|
||||
return __highest * __num_segments;
|
||||
}
|
||||
|
||||
// Process: use the actual unwrapped value.
|
||||
template <class _SegSizeArg>
|
||||
TEST_FUNC constexpr int process_segments(_SegSizeArg __seg_size)
|
||||
{
|
||||
const auto& __val = cuda::args::__unwrap(__seg_size);
|
||||
|
||||
if constexpr (cuda::args::__traits<_SegSizeArg>::is_single_value)
|
||||
{
|
||||
return static_cast<int>(__val);
|
||||
}
|
||||
else
|
||||
{
|
||||
int __total = 0;
|
||||
for (size_t __i = 0; __i < __val.size(); ++__i)
|
||||
{
|
||||
__total += static_cast<int>(__val[__i]);
|
||||
}
|
||||
return __total;
|
||||
}
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Plain scalar: no bounds, global memory, buffer clamped to default
|
||||
{
|
||||
static_assert(select_variant(100) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(100, 4) == default_max_segment_size * 4);
|
||||
assert(process_segments(100) == 100);
|
||||
}
|
||||
|
||||
#if 0 // FIXME(miscco): This should not work
|
||||
// Plain span: per-segment, no bounds, global memory
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg = cuda::std::span<int>{sizes, 3};
|
||||
assert(select_variant(seg) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg, 3) == default_max_segment_size * 3);
|
||||
assert(process_segments(seg) == 64 + 128 + 96);
|
||||
}
|
||||
#endif
|
||||
|
||||
// constant: scalar, fits in shared memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 128 * 4);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// __constant_sequence: array sequence, highest fits in shared memory
|
||||
{
|
||||
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 256}>{};
|
||||
static_assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 256 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 256);
|
||||
}
|
||||
|
||||
// __constant_sequence: array sequence, highest exceeds shared memory, buffer clamped
|
||||
{
|
||||
constexpr auto seg_sizes = cuda::args::__constant_sequence<cuda::std::array{64, 128, 512}>{};
|
||||
static_assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 512 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 512);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
// immediate: tight static bounds, shared memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 256>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// immediate: wide static bounds, global memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100, cuda::args::bounds<1, 4096>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// immediate: no bounds, global memory, buffer = value
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 100 * 4);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
|
||||
// __immediate_sequence: per-segment span with runtime bounds only
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg_sizes = cuda::args::__immediate_sequence{cuda::std::span<int>{sizes, 3}, cuda::args::bounds(1, 200)};
|
||||
assert(select_variant(seg_sizes) == algorithm_variant::global_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 96);
|
||||
}
|
||||
|
||||
// __immediate_sequence: per-segment span with both bounds
|
||||
{
|
||||
int sizes[3] = {64, 128, 96};
|
||||
auto seg_sizes = cuda::args::__immediate_sequence{
|
||||
cuda::std::span<int>{sizes, 3}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(seg_sizes)>::highest <= shared_memory_capacity);
|
||||
assert(select_variant(seg_sizes) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_sizes, 3) == 200 * 3);
|
||||
assert(process_segments(seg_sizes) == 64 + 128 + 96);
|
||||
}
|
||||
|
||||
// deferred: uniform, bounds for decisions only
|
||||
{
|
||||
int val = 100;
|
||||
auto seg_size =
|
||||
cuda::args::deferred{cuda::std::span<int, 1>{&val, 1}, cuda::args::bounds<1, 256>(), cuda::args::bounds(1, 200)};
|
||||
static_assert(cuda::args::__traits<decltype(seg_size)>::highest <= shared_memory_capacity);
|
||||
assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(compute_buffer_size(seg_size, 4) == 200 * 4);
|
||||
}
|
||||
|
||||
// --- Floating point cases ---
|
||||
|
||||
// Plain float: no bounds
|
||||
{
|
||||
static_assert(select_variant(1.0f) == algorithm_variant::global_memory);
|
||||
assert(process_segments(1.0f) == 1);
|
||||
}
|
||||
|
||||
// constant float using an integer NTTP and explicit value type
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128, float>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
#if TEST_HAS_CLASS_NTTP
|
||||
// constant float (float NTTPs require C++20)
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::constant<128.0f>{};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 128);
|
||||
}
|
||||
|
||||
// immediate float with static bounds
|
||||
{
|
||||
constexpr auto seg_size = cuda::args::immediate{100.0f, cuda::args::bounds<1.0f, 256.0f>()};
|
||||
static_assert(select_variant(seg_size) == algorithm_variant::shared_memory);
|
||||
assert(process_segments(seg_size) == 100);
|
||||
}
|
||||
#endif // TEST_HAS_CLASS_NTTP
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,115 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
// UNSUPPORTED: nvcc-11, nvcc-12
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
// TODO: Add support for new half
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Fetch min
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(T(-5)) == T(-1));
|
||||
printf("%i == %i\n", (int) t.load(), (int) T(-5));
|
||||
NV_IF_TARGET(NV_IS_HOST, (fflush(stdout);))
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(T(-5)) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T, ThreadScope>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
// Fetch max
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST,
|
||||
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestFn<__half, local_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestFn<__half, shared_memory_selector, cuda::thread_scope::thread_scope_thread>()();
|
||||
TestFn<__half, global_memory_selector, cuda::thread_scope::thread_scope_thread>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope ThreadScope,
|
||||
bool Signed = cuda::std::is_signed<T>::value>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Test greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(1);
|
||||
assert(t.fetch_max(2) == T(1));
|
||||
assert(t.load() == T(2));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_max(2) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn<T, Selector, ThreadScope, true>
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Call unsigned tests
|
||||
TestFn<T, Selector, ThreadScope, false>()();
|
||||
// Test greater, but with signed math
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-5);
|
||||
assert(t.fetch_max(-1) == T(-5));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-5);
|
||||
assert(t.fetch_max(-1) == T(-5));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
// Test not greater
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_max(-5) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_max(-5) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFnDispatch
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFn<T, Selector, ThreadScope>()();
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,140 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "atomic_helpers.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope ThreadScope,
|
||||
bool Signed = cuda::std::is_signed<T>::value>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Test lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(5);
|
||||
assert(t.fetch_min(4) == T(5));
|
||||
assert(t.load() == T(4));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(5);
|
||||
assert(t.fetch_min(4) == T(5));
|
||||
assert(t.load() == T(4));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_min(4) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(3);
|
||||
assert(t.fetch_min(4) == T(3));
|
||||
assert(t.load() == T(3));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFn<T, Selector, ThreadScope, true>
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
// Call unsigned tests
|
||||
TestFn<T, Selector, ThreadScope, false>()();
|
||||
// Test lesser, but with signed math
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(-5) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(-5) == T(-1));
|
||||
assert(t.load() == T(-5));
|
||||
}
|
||||
// Test not lesser
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
{
|
||||
using A = cuda::atomic<T>;
|
||||
Selector<volatile A, constructor_initializer> sel;
|
||||
volatile A& t = *sel.construct();
|
||||
t = T(-1);
|
||||
assert(t.fetch_min(4) == T(-1));
|
||||
assert(t.load() == T(-1));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
struct TestFnDispatch
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFn<T, Selector, ThreadScope>()();
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();),
|
||||
NV_PROVIDES_SM_70,
|
||||
(TestEachIntegralType<TestFnDispatch, local_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, local_memory_selector>()();))
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE,
|
||||
(TestEachIntegralType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, shared_memory_selector>()();
|
||||
TestEachIntegralType<TestFnDispatch, global_memory_selector>()();
|
||||
TestEachFloatingPointType<TestFnDispatch, global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,102 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef ATOMIC_HELPERS_H
|
||||
#define ATOMIC_HELPERS_H
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
struct UserAtomicType
|
||||
{
|
||||
int i;
|
||||
|
||||
TEST_FUNC explicit UserAtomicType(int d = 0) noexcept
|
||||
: i(d)
|
||||
{}
|
||||
|
||||
TEST_FUNC friend bool operator==(const UserAtomicType& x, const UserAtomicType& y)
|
||||
{
|
||||
return x.i == y.i;
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachIntegralType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFunctor<char, Selector, Scope>()();
|
||||
TestFunctor<signed char, Selector, Scope>()();
|
||||
TestFunctor<unsigned char, Selector, Scope>()();
|
||||
TestFunctor<short, Selector, Scope>()();
|
||||
TestFunctor<unsigned short, Selector, Scope>()();
|
||||
TestFunctor<int, Selector, Scope>()();
|
||||
TestFunctor<unsigned int, Selector, Scope>()();
|
||||
TestFunctor<long, Selector, Scope>()();
|
||||
TestFunctor<unsigned long, Selector, Scope>()();
|
||||
TestFunctor<long long, Selector, Scope>()();
|
||||
TestFunctor<unsigned long long, Selector, Scope>()();
|
||||
TestFunctor<wchar_t, Selector, Scope>();
|
||||
TestFunctor<char16_t, Selector, Scope>()();
|
||||
TestFunctor<char32_t, Selector, Scope>()();
|
||||
TestFunctor<int8_t, Selector, Scope>()();
|
||||
TestFunctor<uint8_t, Selector, Scope>()();
|
||||
TestFunctor<int16_t, Selector, Scope>()();
|
||||
TestFunctor<uint16_t, Selector, Scope>()();
|
||||
TestFunctor<int32_t, Selector, Scope>()();
|
||||
TestFunctor<uint32_t, Selector, Scope>()();
|
||||
TestFunctor<int64_t, Selector, Scope>()();
|
||||
TestFunctor<uint64_t, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachFloatingPointType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestFunctor<float, Selector, Scope>()();
|
||||
TestFunctor<double, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
template <template <class, template <typename, typename> class, cuda::thread_scope> class TestFunctor,
|
||||
template <typename, typename> class Selector,
|
||||
cuda::thread_scope Scope
|
||||
#if _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
= cuda::thread_scope_system
|
||||
#endif // _CCCL_HOST_COMPILATION() || _CCCL_PTX_ARCH() >= 600
|
||||
>
|
||||
struct TestEachAtomicType
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
TestEachIntegralType<TestFunctor, Selector, Scope>()();
|
||||
TestEachFloatingPointType<TestFunctor, Selector, Scope>()();
|
||||
TestFunctor<UserAtomicType, Selector, Scope>()();
|
||||
TestFunctor<int*, Selector, Scope>()();
|
||||
TestFunctor<const int*, Selector, Scope>()();
|
||||
}
|
||||
};
|
||||
|
||||
#endif // ATOMIC_HELPER_H
|
||||
@@ -0,0 +1,13 @@
|
||||
// -*- C++ -*-
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T store(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.store(in + 1, cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T compare_exchange_weak(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T old = T(7);
|
||||
x.compare_exchange_weak(old, T(42), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T compare_exchange_strong(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T old = T(7);
|
||||
x.compare_exchange_strong(old, T(42), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T exchange(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
T out = x.exchange(T(1), cuda::memory_order_relaxed);
|
||||
return out + x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_add(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_add(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_sub(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_sub(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_and(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_and(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_or(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_or(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_xor(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_xor(T(1), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_min(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_min(T(7), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC T fetch_max(T in)
|
||||
{
|
||||
cuda::atomic<T> x(in);
|
||||
x.fetch_max(T(7), cuda::memory_order_relaxed);
|
||||
return x.load(cuda::memory_order_relaxed);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
TEST_DEVICE_FUNC inline void tests()
|
||||
{
|
||||
const T tid = threadIdx.x;
|
||||
assert(tid + T(1) == store(tid));
|
||||
assert(T(1) + tid == exchange(tid));
|
||||
assert(tid == T(7) ? T(42) : tid == compare_exchange_weak(tid));
|
||||
assert(tid == T(7) ? T(42) : tid == compare_exchange_strong(tid));
|
||||
assert((tid + T(1)) == fetch_add(tid));
|
||||
assert((tid & T(1)) == fetch_and(tid));
|
||||
assert((tid | T(1)) == fetch_or(tid));
|
||||
assert((tid ^ T(1)) == fetch_xor(tid));
|
||||
assert(min(tid, T(7)) == fetch_min(tid));
|
||||
assert(max(tid, T(7)) == fetch_max(tid));
|
||||
assert(T(tid - T(1)) == fetch_sub(tid));
|
||||
}
|
||||
|
||||
int main(int arg, char** argv)
|
||||
{
|
||||
#if !defined(_CCCL_ATOMIC_UNSAFE_AUTOMATIC_STORAGE)
|
||||
NV_IF_ELSE_TARGET(
|
||||
NV_IS_HOST,
|
||||
(cuda_thread_count = 64;),
|
||||
(tests<uint8_t>(); tests<uint16_t>(); tests<uint32_t>(); tests<uint64_t>(); tests<int8_t>(); tests<int16_t>();
|
||||
tests<int32_t>();
|
||||
tests<int64_t>();))
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,119 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#define _LIBCUDACXX_FORCE_PTX_AUTOMATIC_STORAGE_PATH 1 // Force using the PTX is_local atomics path
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
/*
|
||||
Test goals:
|
||||
Pre-load registers with values that will be used to trigger the wrong codepath in local device atomics.
|
||||
|
||||
This test is architecture and driver dependent. It is not possible to reproduce this when compiled to SASS on 12.0, but
|
||||
will repro on 12.8.
|
||||
|
||||
Compiled to SASS is an important point, compiling to PTX will show the failure to initialize the local test flag for
|
||||
isspacep.local to 0, but that might be compiled out by the JIT compiler in the driver
|
||||
*/
|
||||
__global__ void __launch_bounds__(1024) device_test(char* gmem)
|
||||
{
|
||||
constexpr int threads = 1024;
|
||||
|
||||
__shared__ int hidx;
|
||||
__shared__ int histogram[threads];
|
||||
|
||||
cuda::atomic<int, cuda::thread_scope_thread> xatom(0);
|
||||
|
||||
constexpr int passes = 16;
|
||||
constexpr int ops = 32;
|
||||
constexpr int expected = passes * ops;
|
||||
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
hidx = 0;
|
||||
memset(histogram, sizeof(histogram), 0);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
for (xatom = 0; xatom.load() < passes; xatom++)
|
||||
{
|
||||
using A = cuda::atomic_ref<int, cuda::std::thread_scope_block>;
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 0]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 1]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 2]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 3]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 4]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 5]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 6]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 7]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 8]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 9]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 10]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 11]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 12]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 13]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 14]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 15]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 16]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 17]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 18]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 19]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 20]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 21]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 22]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 23]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 24]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 25]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 26]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 27]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 28]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 29]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 30]);
|
||||
A(histogram[A(hidx).fetch_add(1) % threads]).fetch_add(gmem[(xatom.load() * 8) + 31]);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
if (histogram[threadIdx.x] != expected)
|
||||
{
|
||||
printf("[%i] = %i\r\n", threadIdx.x, histogram[threadIdx.x]);
|
||||
}
|
||||
assert(histogram[threadIdx.x] == expected);
|
||||
}
|
||||
|
||||
void launch_kernel()
|
||||
{
|
||||
cudaError_t err;
|
||||
char* inptr = nullptr;
|
||||
CUDA_CALL(err, cudaGetLastError());
|
||||
CUDA_CALL(err, cudaMalloc(&inptr, 1024));
|
||||
CUDA_CALL(err, cudaMemset(inptr, 1, 1024));
|
||||
device_test<<<1, 1024>>>(inptr);
|
||||
CUDA_CALL(err, cudaGetLastError());
|
||||
CUDA_CALL(err, cudaDeviceSynchronize());
|
||||
}
|
||||
|
||||
int main(int arg, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (launch_kernel();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: windows
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Check that atomics on host may be constructed
|
||||
template <class T>
|
||||
TEST_FUNC void do_test()
|
||||
{
|
||||
T v(0);
|
||||
cuda::atomic_ref<T> a(v);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
do_test<__int128_t>();
|
||||
do_test<__uint128_t>();
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,33 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Check that host atomics fail to build
|
||||
template <class T>
|
||||
TEST_FUNC void do_test()
|
||||
{
|
||||
T v(0);
|
||||
cuda::atomic_ref<T> a(v);
|
||||
a.store(1);
|
||||
assert(a++ == 1);
|
||||
assert(a.load() == 2);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, (do_test<__int128_t>(); do_test<__uint128_t>();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
// UNSUPPORTED: windows
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr T combine_literal(uint64_t lower, uint64_t upper)
|
||||
{
|
||||
return T(lower) | (T(upper) << 64);
|
||||
}
|
||||
|
||||
template <template <typename, typename> class Selector, cuda::thread_scope ThreadScope>
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
{
|
||||
using T = __int128_t;
|
||||
using A = cuda::atomic_ref<T, ThreadScope>;
|
||||
Selector<T, constructor_initializer> sel;
|
||||
T& t = *sel.construct();
|
||||
t = T(0);
|
||||
A atom(t);
|
||||
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
|
||||
atom.store(test_v, cuda::std::memory_order_release);
|
||||
assert(atom.load() == test_v);
|
||||
}
|
||||
{
|
||||
using T = __uint128_t;
|
||||
using A = cuda::atomic_ref<T, ThreadScope>;
|
||||
Selector<T, constructor_initializer> sel;
|
||||
T& t = *sel.construct();
|
||||
t = T(0);
|
||||
A atom(t);
|
||||
auto test_v = combine_literal<T>(0x01234567DEADBEEF, 0x1337B33701234567);
|
||||
atom.store(test_v);
|
||||
assert(atom.load() == test_v);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
#if __cccl_ptx_isa >= 840
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70,
|
||||
(test<local_memory_selector, cuda::thread_scope_thread>(); test<shared_memory_selector, cuda::thread_scope_block>();
|
||||
test<global_memory_selector, cuda::thread_scope_block>();
|
||||
test<global_memory_selector, cuda::thread_scope_device>();))
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,99 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
/*
|
||||
Test goals:
|
||||
Interleaved 8b/16b access to a 32b window while there is thread contention.
|
||||
|
||||
for 8b:
|
||||
Launch 1024 threads, fetch_add(1) each window, value at end of kernel should be 0xFF..FF. This checks for corruption
|
||||
caused by interleaved access to different parts of the window.
|
||||
|
||||
for 16b:
|
||||
Launch 1024 threads, fetch_add(1), checking for 0x01FF01FF.
|
||||
*/
|
||||
|
||||
template <class T, int Inc>
|
||||
TEST_FUNC void fetch_add_into_window(T* window, uint16_t* atomHistory)
|
||||
{
|
||||
using Atom = cuda::atomic_ref<T, cuda::thread_scope_block>;
|
||||
|
||||
Atom a(*window);
|
||||
*atomHistory = a.fetch_add(Inc);
|
||||
}
|
||||
|
||||
template <class T>
|
||||
TEST_DEVICE_FUNC void device_do_test(uint32_t expected)
|
||||
{
|
||||
constexpr uint32_t threadCount = 1024;
|
||||
constexpr uint32_t histogramResultCount = 256 * sizeof(T);
|
||||
constexpr uint32_t histogramEntriesPerThread = 4 / sizeof(T);
|
||||
|
||||
__shared__ uint16_t atomHistory[threadCount];
|
||||
__shared__ uint8_t atomHistogram[histogramResultCount];
|
||||
__shared__ uint32_t atomicStorage;
|
||||
|
||||
cuda::atomic_ref<uint32_t, cuda::thread_scope_block> bucket(atomicStorage);
|
||||
|
||||
constexpr uint32_t offsetMask = ((4 / sizeof(T)) - 1);
|
||||
// Access offset is interleaved meaning threads 4, 5, 6, 7 access window 0, 1, 2, 3 and so on.
|
||||
const uint32_t threadOffset = threadIdx.x & offsetMask;
|
||||
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
memset(atomHistogram, 0, histogramResultCount);
|
||||
bucket.store(0);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
T* window = reinterpret_cast<T*>(&atomicStorage) + threadOffset;
|
||||
fetch_add_into_window<T, 1>(window, atomHistory + threadIdx.x);
|
||||
|
||||
__syncthreads();
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// For each thread, add its atomic result into the corresponding bucket
|
||||
for (uint32_t i = 0; i < threadCount; i++)
|
||||
{
|
||||
atomHistogram[atomHistory[i]]++;
|
||||
}
|
||||
// Check that each bucket has exactly (4 / sizeof(T)) entries
|
||||
// This checks that atomic fetch operations were sequential. i.e. 4xfetch_add(1) returns [0, 1, 2, 3]
|
||||
for (uint32_t i = 0; i < histogramResultCount; i++)
|
||||
{
|
||||
assert(atomHistogram[i] == histogramEntriesPerThread);
|
||||
}
|
||||
printf("expected: 0x%X\r\n", expected);
|
||||
printf("result: 0x%X\r\n", bucket.load());
|
||||
assert(bucket.load() == expected);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST,
|
||||
(cuda_thread_count = 1024;),
|
||||
NV_IS_DEVICE,
|
||||
(device_do_test<uint8_t>(0); device_do_test<uint16_t>(0x02000200);));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// XFAIL: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads, pre-sm-60
|
||||
// UNSUPPORTED: windows && pre-sm-70
|
||||
|
||||
// <cuda/atomic>
|
||||
|
||||
// cuda::atomic<key>
|
||||
|
||||
// Original test issue:
|
||||
// https://github.com/NVIDIA/libcudacxx/issues/160
|
||||
|
||||
#include <cuda/atomic>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
template <template <typename, typename> class Selector>
|
||||
struct TestFn
|
||||
{
|
||||
TEST_FUNC void operator()() const
|
||||
{
|
||||
{
|
||||
struct key
|
||||
{
|
||||
int32_t a;
|
||||
int32_t b;
|
||||
};
|
||||
using A = cuda::std::atomic<key>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
cuda::std::atomic_init(&t, key{1, 2});
|
||||
auto r = t.load();
|
||||
auto d = key{5, 5};
|
||||
t.store(r);
|
||||
(void) t.exchange(r);
|
||||
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
}
|
||||
{
|
||||
struct alignas(8) key
|
||||
{
|
||||
int32_t a;
|
||||
int32_t b;
|
||||
};
|
||||
using A = cuda::std::atomic<key>;
|
||||
Selector<A, constructor_initializer> sel;
|
||||
A& t = *sel.construct();
|
||||
cuda::std::atomic_init(&t, key{1, 2});
|
||||
auto r = t.load();
|
||||
auto d = key{5, 5};
|
||||
t.store(r);
|
||||
(void) t.exchange(r);
|
||||
(void) t.compare_exchange_weak(r, d, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
(void) t.compare_exchange_strong(d, r, cuda::memory_order_seq_cst, cuda::memory_order_seq_cst);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(NV_IS_HOST, TestFn<local_memory_selector>()();
|
||||
, NV_PROVIDES_SM_70, TestFn<local_memory_selector>()();)
|
||||
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (TestFn<shared_memory_selector>()(); TestFn<global_memory_selector>()();))
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,87 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
#ifndef TEST_ARRIVE_TX_H_
|
||||
#define TEST_ARRIVE_TX_H_
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/memory>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "concurrent_agents.h"
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
template <typename Barrier>
|
||||
inline TEST_DEVICE_FUNC void mbarrier_complete_tx(Barrier& b, int transaction_count)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90,
|
||||
(
|
||||
if (cuda::device::is_address_from(cuda::device::barrier_native_handle(b), cuda::device::address_space::shared)) {
|
||||
asm volatile(
|
||||
"mbarrier.complete_tx.relaxed.cta.shared::cta.b64 [%0], %1;"
|
||||
:
|
||||
: "r"((unsigned int) __cvta_generic_to_shared(cuda::device::barrier_native_handle(b))), "r"(transaction_count)
|
||||
: "memory");
|
||||
} else { __trap(); }),
|
||||
NV_ANY_TARGET,
|
||||
(
|
||||
// On architectures pre-SM90 (and on host), we drop the transaction count
|
||||
// update. The barriers do not keep track of transaction counts.
|
||||
__trap();));
|
||||
}
|
||||
|
||||
template <bool split_arrive_and_expect>
|
||||
TEST_DEVICE_FUNC void thread(cuda::barrier<cuda::thread_scope_block>& b, int arrives_per_thread)
|
||||
{
|
||||
constexpr int tx_count = 1;
|
||||
typename cuda::barrier<cuda::thread_scope_block>::arrival_token tok;
|
||||
|
||||
if _CCCL_CONSTEXPR_CXX20 (split_arrive_and_expect)
|
||||
{
|
||||
cuda::device::barrier_expect_tx(b, tx_count);
|
||||
tok = b.arrive(arrives_per_thread);
|
||||
}
|
||||
else
|
||||
{
|
||||
tok = cuda::device::barrier_arrive_tx(b, arrives_per_thread, tx_count);
|
||||
}
|
||||
|
||||
// Manually increase the transaction count of the barrier.
|
||||
mbarrier_complete_tx(b, tx_count);
|
||||
|
||||
b.wait(cuda::std::move(tok));
|
||||
}
|
||||
|
||||
template <bool split_arrive_and_expect>
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(
|
||||
// Run all threads, each arriving with arrival count 1
|
||||
using barrier_t = cuda::barrier<cuda::thread_scope_block>;
|
||||
|
||||
shared_memory_selector<barrier_t, constructor_initializer> sel_1;
|
||||
barrier_t* bar_1 = sel_1.construct(blockDim.x);
|
||||
__syncthreads();
|
||||
thread<split_arrive_and_expect>(*bar_1, 1);
|
||||
|
||||
// Run all threads, each arriving with arrival count 2
|
||||
shared_memory_selector<barrier_t, constructor_initializer> sel_2;
|
||||
barrier_t* bar_2 = sel_2.construct(2 * blockDim.x);
|
||||
__syncthreads();
|
||||
thread<split_arrive_and_expect>(*bar_2, 2);));
|
||||
}
|
||||
|
||||
#endif // TEST_ARRIVE_TX_H_
|
||||
@@ -0,0 +1,56 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// UNSUPPORTED: clang && !nvcc
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// When PR #416 is merged, uncomment this line:
|
||||
// cuda_cluster_size = 2;
|
||||
),
|
||||
NV_IS_DEVICE,
|
||||
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
|
||||
|
||||
if (threadIdx.x == 0) { init(&bar, blockDim.x); } namespace cg = cooperative_groups;
|
||||
auto cluster = cg::this_cluster();
|
||||
|
||||
cluster.sync();
|
||||
|
||||
// This test currently fails at this point because support for
|
||||
// clusters has not yet been added.
|
||||
cuda::barrier<cuda::thread_scope_block> * remote_bar;
|
||||
remote_bar = cluster.map_shared_rank(&bar, cluster.block_rank() ^ 1);
|
||||
|
||||
// When PR #416 is merged, this should fail here because the barrier
|
||||
// is in device memory.
|
||||
auto token = cuda::device::barrier_arrive_tx(*remote_bar, 1, 0);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 256;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
TEST_DEVICE_FUNC uint64_t bar_storage;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
|
||||
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
|
||||
|
||||
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
|
||||
|
||||
// Should fail because the barrier is in device memory.
|
||||
[[maybe_unused]] auto token = cuda::device::barrier_arrive_tx(*bar_ptr, 1, 0);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#ifndef __cccl_lib_local_barrier_arrive_tx
|
||||
static_assert(false, "should define __cccl_lib_local_barrier_arrive_tx");
|
||||
#endif // __cccl_lib_local_barrier_arrive_tx
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(__shared__ cuda::barrier<cuda::thread_scope_block> bar;
|
||||
if (threadIdx.x == 0) { init(&bar, blockDim.x); } __syncthreads();
|
||||
|
||||
// barrier_arrive_tx should fail on SM70 and SM80, because it is hidden.
|
||||
auto token = cuda::device::barrier_arrive_tx(bar, 1, 0);
|
||||
|
||||
#ifdef __cccl_lib_local_barrier_arrive_tx
|
||||
static_assert(false, "Fail manually for SM90 and up.");
|
||||
#endif // __cccl_lib_local_barrier_arrive_tx
|
||||
));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 2;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 32;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = false; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,107 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
static constexpr int buf_len = 1024;
|
||||
alignas(128) TEST_GLOBAL_VARIABLE int gmem_buffer[buf_len];
|
||||
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
gmem_buffer[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(16) __shared__ int smem_buffer[buf_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_global_to_shared(smem_buffer, gmem_buffer, sizeof(smem_buffer), bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Update in shared memory
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] += i;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_shared_to_global(gmem_buffer, smem_buffer, sizeof(smem_buffer));
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEAR-DOWN: check that global memory is correct
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
assert(gmem_buffer[i] == 2 * i);
|
||||
}
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512;));
|
||||
|
||||
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test();));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,28 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#ifndef __cccl_lib_experimental_ctk12_cp_async_exposure
|
||||
static_assert(false, "should define __cccl_lib_experimental_ctk12_cp_async_exposure");
|
||||
#endif
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,95 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
// Kernels below are intended to be compiled, but not run. This is to check if
|
||||
// all generated PTX is valid.
|
||||
__global__ void test_bulk_tensor(CUtensorMap* map)
|
||||
{
|
||||
__shared__ int smem;
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
cde::cp_async_bulk_tensor_1d_global_to_shared(&smem, map, 0, bar);
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(&smem, map, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_3d_global_to_shared(&smem, map, 0, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_4d_global_to_shared(&smem, map, 0, 0, 0, 0, bar);
|
||||
cde::cp_async_bulk_tensor_5d_global_to_shared(&smem, map, 0, 0, 0, 0, 0, bar);
|
||||
|
||||
cde::cp_async_bulk_tensor_1d_shared_to_global(map, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(map, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_3d_shared_to_global(map, 0, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_4d_shared_to_global(map, 0, 0, 0, 0, &smem);
|
||||
cde::cp_async_bulk_tensor_5d_shared_to_global(map, 0, 0, 0, 0, 0, &smem);
|
||||
}
|
||||
|
||||
__global__ void test_bulk(void* gmem)
|
||||
{
|
||||
__shared__ int smem;
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
cde::cp_async_bulk_global_to_shared(&smem, gmem, 1024, bar);
|
||||
cde::cp_async_bulk_shared_to_global(gmem, &smem, 1024);
|
||||
}
|
||||
|
||||
__global__ void test_fences_async_group(void*)
|
||||
{
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
|
||||
cde::cp_async_bulk_commit_group();
|
||||
// Wait for up to 8 groups
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
cde::cp_async_bulk_wait_group_read<1>();
|
||||
cde::cp_async_bulk_wait_group_read<2>();
|
||||
cde::cp_async_bulk_wait_group_read<3>();
|
||||
cde::cp_async_bulk_wait_group_read<4>();
|
||||
cde::cp_async_bulk_wait_group_read<5>();
|
||||
cde::cp_async_bulk_wait_group_read<6>();
|
||||
cde::cp_async_bulk_wait_group_read<7>();
|
||||
cde::cp_async_bulk_wait_group_read<8>();
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,221 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
// UNSUPPORTED: clang && !nvcc
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// NVRTC does not support cuda.h (due to import of stdlib.h)
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
constexpr size_t GMEM_WIDTH = 1024; // Width of tensor (in # elements)
|
||||
constexpr size_t GMEM_HEIGHT = 1024; // Height of tensor (in # elements)
|
||||
constexpr size_t gmem_len = GMEM_WIDTH * GMEM_HEIGHT;
|
||||
|
||||
constexpr int SMEM_WIDTH = 32; // Width of shared memory buffer (in # elements)
|
||||
constexpr int SMEM_HEIGHT = 8; // Height of shared memory buffer (in # elements)
|
||||
|
||||
static constexpr int buf_len = SMEM_HEIGHT * SMEM_WIDTH;
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
// We need a type with a size. On NVRTC, cuda.h cannot be imported, so we don't
|
||||
// have access to the definition of CUTensorMap (only to the declaration of CUtensorMap inside
|
||||
// cuda/barrier). So we use this type instead and reinterpret_cast in the
|
||||
// kernel.
|
||||
struct fake_cutensormap
|
||||
{
|
||||
alignas(64) uint64_t opaque[16];
|
||||
};
|
||||
__constant__ fake_cutensormap global_fake_tensor_map;
|
||||
|
||||
TEST_DEVICE_FUNC void test(int base_i, int base_j)
|
||||
{
|
||||
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
|
||||
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < static_cast<int>(gmem_len); i += blockDim.x)
|
||||
{
|
||||
gmem_tensor[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(128) __shared__ int smem_buffer[buf_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// Fastest moving coordinate first.
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(smem_buffer, global_tensor_map, base_j, base_i, bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Check smem
|
||||
for (int i = 0; i < SMEM_HEIGHT; ++i)
|
||||
{
|
||||
for (int j = 0; j < SMEM_HEIGHT; ++j)
|
||||
{
|
||||
const int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
|
||||
const int smem_lin_idx = i * SMEM_WIDTH + j;
|
||||
|
||||
assert(smem_buffer[smem_lin_idx] == gmem_lin_idx);
|
||||
}
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Update smem
|
||||
for (int i = threadIdx.x; i < buf_len; i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] = 2 * smem_buffer[i] + 1;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(global_tensor_map, base_j, base_i, smem_buffer);
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
__threadfence();
|
||||
__syncthreads();
|
||||
|
||||
// TEAR-DOWN: check that global memory is correct
|
||||
for (int i = 0; i < SMEM_HEIGHT; ++i)
|
||||
{
|
||||
for (int j = 0; j < SMEM_HEIGHT; ++j)
|
||||
{
|
||||
int gmem_lin_idx = (base_i + i) * GMEM_WIDTH + base_j + j;
|
||||
|
||||
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
|
||||
}
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# if _CCCL_CTK_BELOW(12, 5)
|
||||
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
|
||||
}
|
||||
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
|
||||
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code =
|
||||
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
|
||||
}
|
||||
# endif // _CCCL_CTK_AT_LEAST(12, 5)
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512;
|
||||
|
||||
int* tensor_ptr = nullptr;
|
||||
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor);
|
||||
assert(code == cudaSuccess && "getsymboladdress failed.");
|
||||
|
||||
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
|
||||
CUtensorMap local_tensor_map{};
|
||||
// rank is the number of dimensions of the array.
|
||||
constexpr uint32_t rank = 2;
|
||||
uint64_t size[rank] = {GMEM_WIDTH, GMEM_HEIGHT};
|
||||
// The stride is the number of bytes to traverse from the first element of one row to the next.
|
||||
// It must be a multiple of 16.
|
||||
uint64_t stride[rank - 1] = {GMEM_WIDTH * sizeof(int)};
|
||||
// The box_size is the size of the shared memory buffer that is used as the
|
||||
// destination of a TMA transfer.
|
||||
uint32_t box_size[rank] = {SMEM_WIDTH, SMEM_HEIGHT};
|
||||
// The distance between elements in units of sizeof(element). A stride of 2
|
||||
// can be used to load only the real component of a complex-valued tensor, for instance.
|
||||
uint32_t elem_stride[rank] = {1, 1};
|
||||
|
||||
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
|
||||
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
|
||||
|
||||
// Create the tensor descriptor.
|
||||
CUresult res = cuTensorMapEncodeTiled(
|
||||
&local_tensor_map, // CUtensorMap *tensorMap,
|
||||
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
|
||||
rank, // cuuint32_t tensorRank,
|
||||
tensor_ptr, // void *globalAddress,
|
||||
size, // const cuuint64_t *globalDim,
|
||||
stride, // const cuuint64_t *globalStrides,
|
||||
box_size, // const cuuint32_t *boxDim,
|
||||
elem_stride, // const cuuint32_t *elementStrides,
|
||||
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
|
||||
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
|
||||
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
|
||||
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
|
||||
|
||||
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
|
||||
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
|
||||
assert(code == cudaSuccess && "memcpytosymbol failed.");));
|
||||
|
||||
NV_DISPATCH_TARGET(NV_IS_DEVICE, (test(0, 0); test(4, 0); test(4, 4);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS{256};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 1> GMEM_DIMS_DEV{256};
|
||||
constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS{32};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> SMEM_DIMS_DEV{32};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 1> TEST_SMEM_COORDS[] = {{0}, {4}, {8}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,69 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS{8, 11};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 2> GMEM_DIMS_DEV{8, 11};
|
||||
constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS{4, 2};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> SMEM_DIMS_DEV{4, 2};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 2> TEST_SMEM_COORDS[] = {
|
||||
{0, 0},
|
||||
{4, 1},
|
||||
{4, 5},
|
||||
{0, 5},
|
||||
};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS{8, 11, 13};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 3> GMEM_DIMS_DEV{8, 11, 13};
|
||||
constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS{4, 2, 4};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> SMEM_DIMS_DEV{4, 2, 4};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 3> TEST_SMEM_COORDS[] = {{0, 0, 0}, {4, 1, 3}, {4, 5, 1}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS{8, 11, 13, 3};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 4> GMEM_DIMS_DEV{8, 11, 13, 3};
|
||||
constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS{4, 2, 4, 1};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> SMEM_DIMS_DEV{4, 2, 4, 1};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 4> TEST_SMEM_COORDS[] = {
|
||||
{0, 0, 0, 0}, {4, 1, 3, 0}, {4, 8, 7, 2}, {4, 5, 1, 1}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
// UNSUPPORTED: nvrtc
|
||||
// XFAIL: clang && !nvcc
|
||||
// NVRTC_SKIP_KERNEL_RUN // This will have effect once PR 433 is merged (line above should be removed.)
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/std/array>
|
||||
|
||||
#include "cp_async_bulk_tensor_generic.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
// Define the size of contiguous tensor in global and shared memory.
|
||||
//
|
||||
// Note that the first dimension is the one with stride 1. This one must be a
|
||||
// multiple of 4 to ensure that each new dimension starts at a 16-byte aligned
|
||||
// offset.
|
||||
//
|
||||
// We have a separate variable for host and device because a constexpr
|
||||
// cuda::std::array cannot be shared between host and device as some of its
|
||||
// member functions take a const reference, which is unsupported by nvcc.
|
||||
constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS{8, 11, 13, 3, 3};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint64_t, 5> GMEM_DIMS_DEV{8, 11, 13, 3, 3};
|
||||
constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS{4, 2, 4, 1, 1};
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> SMEM_DIMS_DEV{4, 2, 4, 1, 1};
|
||||
|
||||
TEST_GLOBAL_VARIABLE constexpr cuda::std::array<uint32_t, 5> TEST_SMEM_COORDS[] = {
|
||||
{0, 0, 0, 0, 0}, {4, 1, 3, 0, 1}, {4, 5, 1, 1, 2}};
|
||||
|
||||
constexpr size_t gmem_len = tensor_len(GMEM_DIMS);
|
||||
constexpr size_t smem_len = tensor_len(SMEM_DIMS);
|
||||
|
||||
TEST_GLOBAL_VARIABLE int gmem_tensor[gmem_len];
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're launching
|
||||
cuda_thread_count = 512; init_tensor_map(gmem_tensor, GMEM_DIMS, SMEM_DIMS);),
|
||||
NV_IS_DEVICE,
|
||||
(for (auto smem_coord : TEST_SMEM_COORDS) {
|
||||
test<smem_len>(smem_coord, SMEM_DIMS_DEV, GMEM_DIMS_DEV, gmem_tensor, gmem_len);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,349 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#ifndef TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
#define TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
|
||||
#include <cuda/barrier>
|
||||
#include <cuda/ptx>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/utility> // cuda::std::move
|
||||
|
||||
namespace ptx = cuda::ptx;
|
||||
|
||||
#include "test_macros.h" // TEST_NV_DIAG_SUPPRESS
|
||||
|
||||
// NVRTC does not support cuda.h (due to import of stdlib.h)
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# include <cstdio>
|
||||
|
||||
# include <cudaTypedefs.h> // PFN_cuTensorMapEncodeTiled, CUtensorMap
|
||||
#endif // ! TEST_COMPILER(NVRTC)
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
using barrier = cuda::barrier<cuda::thread_scope_block>;
|
||||
namespace cde = cuda::device::experimental;
|
||||
|
||||
/*
|
||||
* This header supports the 1d, 2d, ..., 5d test of the TMA PTX wrappers.
|
||||
*
|
||||
* The functions below help convert Nd coordinates into something useful.
|
||||
*
|
||||
*/
|
||||
|
||||
// Compute the total number of elements in a tensor
|
||||
template <class T, size_t num_dims>
|
||||
constexpr TEST_FUNC int tensor_len(cuda::std::array<T, num_dims> dims)
|
||||
{
|
||||
T len = 1;
|
||||
for (T d : dims)
|
||||
{
|
||||
len *= d;
|
||||
}
|
||||
return static_cast<int>(len);
|
||||
}
|
||||
|
||||
// Function to convert:
|
||||
// a linear index into a shared memory tensor
|
||||
// into
|
||||
// a linear index into a global memory tensor.
|
||||
template <size_t num_dims>
|
||||
inline TEST_DEVICE_FUNC int smem_lin_idx_to_gmem_lin_idx(
|
||||
int smem_lin_idx,
|
||||
cuda::std::array<uint32_t, num_dims> smem_coord,
|
||||
cuda::std::array<uint32_t, num_dims> smem_dims,
|
||||
cuda::std::array<uint64_t, num_dims> gmem_dims)
|
||||
{
|
||||
assert(smem_coord.size() == smem_dims.size());
|
||||
assert(smem_coord.size() == gmem_dims.size());
|
||||
|
||||
int gmem_lin_idx = 0;
|
||||
int gmem_stride = 1;
|
||||
for (int i = 0; i < (int) smem_coord.size(); ++i)
|
||||
{
|
||||
int smem_i_idx = smem_lin_idx % smem_dims.begin()[i];
|
||||
gmem_lin_idx += (smem_coord.begin()[i] + smem_i_idx) * gmem_stride;
|
||||
|
||||
smem_lin_idx /= smem_dims.begin()[i];
|
||||
gmem_stride *= gmem_dims.begin()[i];
|
||||
}
|
||||
return gmem_lin_idx;
|
||||
}
|
||||
|
||||
template <size_t num_dims>
|
||||
TEST_DEVICE_FUNC inline void cp_tensor_global_to_shared(
|
||||
CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem, barrier& bar)
|
||||
{
|
||||
switch (indices.size())
|
||||
{
|
||||
case 1:
|
||||
cde::cp_async_bulk_tensor_1d_global_to_shared(smem, tensor_map, indices[0], bar);
|
||||
break;
|
||||
case 2:
|
||||
cde::cp_async_bulk_tensor_2d_global_to_shared(smem, tensor_map, indices[0], indices[1], bar);
|
||||
break;
|
||||
case 3:
|
||||
cde::cp_async_bulk_tensor_3d_global_to_shared(smem, tensor_map, indices[0], indices[1], indices[2], bar);
|
||||
break;
|
||||
case 4:
|
||||
cde::cp_async_bulk_tensor_4d_global_to_shared(
|
||||
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], bar);
|
||||
break;
|
||||
case 5:
|
||||
cde::cp_async_bulk_tensor_5d_global_to_shared(
|
||||
smem, tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], bar);
|
||||
break;
|
||||
default:
|
||||
assert(false && "Wrong number of dimensions.");
|
||||
}
|
||||
}
|
||||
|
||||
template <size_t num_dims>
|
||||
TEST_DEVICE_FUNC inline void
|
||||
cp_tensor_shared_to_global(CUtensorMap* tensor_map, cuda::std::array<uint32_t, num_dims> indices, void* smem)
|
||||
{
|
||||
switch (indices.size())
|
||||
{
|
||||
case 1:
|
||||
cde::cp_async_bulk_tensor_1d_shared_to_global(tensor_map, indices[0], smem);
|
||||
break;
|
||||
case 2:
|
||||
cde::cp_async_bulk_tensor_2d_shared_to_global(tensor_map, indices[0], indices[1], smem);
|
||||
break;
|
||||
case 3:
|
||||
cde::cp_async_bulk_tensor_3d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], smem);
|
||||
break;
|
||||
case 4:
|
||||
cde::cp_async_bulk_tensor_4d_shared_to_global(tensor_map, indices[0], indices[1], indices[2], indices[3], smem);
|
||||
break;
|
||||
case 5:
|
||||
cde::cp_async_bulk_tensor_5d_shared_to_global(
|
||||
tensor_map, indices[0], indices[1], indices[2], indices[3], indices[4], smem);
|
||||
break;
|
||||
default:
|
||||
assert(false && "Wrong number of dimensions.");
|
||||
}
|
||||
}
|
||||
|
||||
// To define a tensor map in constant memory, we need a type with a size. On
|
||||
// NVRTC, cuda.h cannot be imported, so we don't have access to the definition
|
||||
// of CUTensorMap (only to the declaration of CUtensorMap inside cuda/barrier).
|
||||
// So we use this type instead and reinterpret_cast in the kernel.
|
||||
struct fake_cutensormap
|
||||
{
|
||||
alignas(64) uint64_t opaque[16];
|
||||
};
|
||||
__constant__ fake_cutensormap global_fake_tensor_map;
|
||||
|
||||
/*
|
||||
* This test has as primary purpose to make sure that the indices in the mapping
|
||||
* from C++ to PTX didn't get mixed up.
|
||||
*
|
||||
* How does it test this?
|
||||
*
|
||||
* 1. It fills a global memory tensor with linear coordinates 0, 1, ...
|
||||
* 2. It loads a tile into shared memory at some coordinate (x, y, ... )
|
||||
* 3. It checks that the coordinates that were received in shared memory match the expected.
|
||||
* 4. It modifies the coordinates (c = 2 * c + 1)
|
||||
* 5. It writes the tile back to global memory
|
||||
* 6. It checks that all the values in global are properly modified.
|
||||
*/
|
||||
template <size_t smem_len, size_t num_dims>
|
||||
TEST_DEVICE_FUNC void
|
||||
test(cuda::std::array<uint32_t, num_dims> smem_coord,
|
||||
cuda::std::array<uint32_t, num_dims> smem_dims,
|
||||
cuda::std::array<uint64_t, num_dims> gmem_dims,
|
||||
int* gmem_tensor,
|
||||
int gmem_len)
|
||||
{
|
||||
CUtensorMap* global_tensor_map = reinterpret_cast<CUtensorMap*>(&global_fake_tensor_map);
|
||||
|
||||
// SETUP: fill global memory buffer
|
||||
for (int i = threadIdx.x; i < gmem_len; i += blockDim.x)
|
||||
{
|
||||
gmem_tensor[i] = i;
|
||||
}
|
||||
// Ensure that writes to global memory are visible to others, including
|
||||
// those in the async proxy.
|
||||
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
|
||||
// fence.proxy.async.global should suffice, but I am keeping the threadfence
|
||||
// out of an abundance of caution.
|
||||
__threadfence();
|
||||
ptx::fence_proxy_async(ptx::space_global);
|
||||
__syncthreads();
|
||||
|
||||
// TEST: Add i to buffer[i]
|
||||
alignas(128) __shared__ int smem_buffer[smem_len];
|
||||
#if _CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ char barrier_data[sizeof(barrier)];
|
||||
barrier& bar = reinterpret_cast<barrier&>(barrier_data);
|
||||
#else // ^^^ _CCCL_CUDA_COMPILER(CLANG) ^^^ / vvv !_CCCL_CUDA_COMPILER(CLANG)
|
||||
__shared__ barrier bar;
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
init(&bar, blockDim.x);
|
||||
}
|
||||
__syncthreads();
|
||||
|
||||
// Load data:
|
||||
uint64_t token;
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
// Fastest moving coordinate first.
|
||||
cp_tensor_global_to_shared(global_tensor_map, smem_coord, smem_buffer, bar);
|
||||
token = cuda::device::barrier_arrive_tx(bar, 1, sizeof(smem_buffer));
|
||||
}
|
||||
else
|
||||
{
|
||||
token = bar.arrive();
|
||||
}
|
||||
bar.wait(cuda::std::move(token));
|
||||
|
||||
// Check smem
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
|
||||
assert(smem_buffer[i] == gmem_lin_idx);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
|
||||
// Update smem
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
smem_buffer[i] = 2 * smem_buffer[i] + 1;
|
||||
}
|
||||
cde::fence_proxy_async_shared_cta();
|
||||
__syncthreads();
|
||||
|
||||
// Write back to global memory:
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
cp_tensor_shared_to_global(global_tensor_map, smem_coord, smem_buffer);
|
||||
cde::cp_async_bulk_commit_group();
|
||||
cde::cp_async_bulk_wait_group_read<0>();
|
||||
}
|
||||
// ahendriksen: Issuing threadfence and fence.proxy.async.global. The
|
||||
// fence.proxy.async.global should suffice, but I am keeping the threadfence
|
||||
// out of an abundance of caution.
|
||||
__threadfence();
|
||||
ptx::fence_proxy_async(ptx::space_global);
|
||||
__syncthreads();
|
||||
|
||||
// // TEAR-DOWN: check that global memory is correct
|
||||
for (int i = threadIdx.x; i < static_cast<int>(smem_len); i += blockDim.x)
|
||||
{
|
||||
int gmem_lin_idx = smem_lin_idx_to_gmem_lin_idx(i, smem_coord, smem_dims, gmem_dims);
|
||||
|
||||
assert(gmem_tensor[gmem_lin_idx] == 2 * gmem_lin_idx + 1);
|
||||
}
|
||||
__syncthreads();
|
||||
}
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# if _CCCL_CTK_BELOW(12, 5)
|
||||
PFN_cuTensorMapEncodeTiled get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code = cudaGetDriverEntryPoint("cuTensorMapEncodeTiled", &driver_ptr, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled>(driver_ptr);
|
||||
}
|
||||
# else // ^^^ _CCCL_CTK_BELOW(12, 5) ^^^ / vvv _CCCL_CTK_AT_LEAST(12, 5) vvv
|
||||
PFN_cuTensorMapEncodeTiled_v12000 get_cuTensorMapEncodeTiled()
|
||||
{
|
||||
void* driver_ptr = nullptr;
|
||||
cudaDriverEntryPointQueryResult driver_status;
|
||||
auto code =
|
||||
cudaGetDriverEntryPointByVersion("cuTensorMapEncodeTiled", &driver_ptr, 12000, cudaEnableDefault, &driver_status);
|
||||
assert(code == cudaSuccess && "Could not get driver API");
|
||||
return reinterpret_cast<PFN_cuTensorMapEncodeTiled_v12000>(driver_ptr);
|
||||
}
|
||||
# endif // _CCCL_CTK_AT_LEAST(12, 5)
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
template <typename T, size_t num_dims>
|
||||
CUtensorMap map_encode(T* tensor_ptr,
|
||||
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
|
||||
const cuda::std::array<uint32_t, num_dims>& smem_dims)
|
||||
{
|
||||
// https://docs.nvidia.com/cuda/cuda-driver-api/group__CUDA__TENSOR__MEMORY.html
|
||||
CUtensorMap tensor_map{};
|
||||
|
||||
// The stride is the number of bytes to traverse from the first element of one row to the next.
|
||||
// It must be a multiple of 16.
|
||||
// cuTensorMapEncodeTiled requires that the stride array is a valid pointer, so we add one superfluous element
|
||||
// This is necessary for num_dims == 1
|
||||
cuda::std::array<uint64_t, num_dims> stride;
|
||||
uint64_t base_stride = sizeof(T);
|
||||
for (size_t i = 0; i < stride.size() - 1; ++i)
|
||||
{
|
||||
base_stride *= gmem_dims[i];
|
||||
stride[i] = base_stride;
|
||||
}
|
||||
|
||||
// The distance between elements in units of sizeof(element). A stride of 2
|
||||
// can be used to load only the real component of a complex-valued tensor, for instance.
|
||||
cuda::std::array<uint32_t, num_dims> elem_stride; // = {1, .., 1};
|
||||
for (size_t i = 0; i < elem_stride.size(); ++i)
|
||||
{
|
||||
elem_stride[i] = 1;
|
||||
}
|
||||
|
||||
// Get a function pointer to the cuTensorMapEncodeTiled driver API.
|
||||
auto cuTensorMapEncodeTiled = get_cuTensorMapEncodeTiled();
|
||||
|
||||
// Create the tensor descriptor.
|
||||
CUresult res = cuTensorMapEncodeTiled(
|
||||
&tensor_map, // CUtensorMap *tensorMap,
|
||||
CUtensorMapDataType::CU_TENSOR_MAP_DATA_TYPE_INT32,
|
||||
num_dims, // cuuint32_t tensorRank,
|
||||
tensor_ptr, // void *globalAddress,
|
||||
gmem_dims.data(), // const cuuint64_t *globalDim,
|
||||
stride.data(), // const cuuint64_t *globalStrides,
|
||||
smem_dims.data(), // const cuuint32_t *boxDim,
|
||||
elem_stride.data(), // const cuuint32_t *elementStrides,
|
||||
CUtensorMapInterleave::CU_TENSOR_MAP_INTERLEAVE_NONE,
|
||||
CUtensorMapSwizzle::CU_TENSOR_MAP_SWIZZLE_NONE,
|
||||
CUtensorMapL2promotion::CU_TENSOR_MAP_L2_PROMOTION_NONE,
|
||||
CUtensorMapFloatOOBfill::CU_TENSOR_MAP_FLOAT_OOB_FILL_NONE);
|
||||
|
||||
assert(res == CUDA_SUCCESS && "tensormap creation failed.");
|
||||
|
||||
return tensor_map;
|
||||
}
|
||||
|
||||
template <typename T, size_t num_dims>
|
||||
void init_tensor_map(const T& gmem_tensor_symbol,
|
||||
const cuda::std::array<uint64_t, num_dims>& gmem_dims,
|
||||
const cuda::std::array<uint32_t, num_dims>& smem_dims)
|
||||
{
|
||||
// Get pointer to gmem_tensor to create tensor map.
|
||||
int* tensor_ptr = nullptr;
|
||||
auto code = cudaGetSymbolAddress((void**) &tensor_ptr, gmem_tensor_symbol);
|
||||
assert(code == cudaSuccess && "Could not get symbol address.");
|
||||
|
||||
// Create tensor map
|
||||
CUtensorMap local_tensor_map = map_encode(tensor_ptr, gmem_dims, smem_dims);
|
||||
|
||||
// Copy it to device
|
||||
code = cudaMemcpyToSymbol(global_fake_tensor_map, &local_tensor_map, sizeof(CUtensorMap));
|
||||
assert(code == cudaSuccess && "Could not copy symbol to device.");
|
||||
}
|
||||
#endif // ! TEST_COMPILER(NVRTC)
|
||||
|
||||
#endif // TEST_CP_ASYNC_BULK_TENSOR_GENERIC_H_
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 256;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// UNSUPPORTED: no_execute
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
// Suppress warning about barrier in shared memory
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
|
||||
[[maybe_unused]] TEST_GLOBAL_VARIABLE uint64_t bar_storage;
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(cuda::barrier<cuda::thread_scope_block> * bar_ptr;
|
||||
bar_ptr = reinterpret_cast<cuda::barrier<cuda::thread_scope_block>*>(bar_storage);
|
||||
|
||||
if (threadIdx.x == 0) { init(bar_ptr, blockDim.x); } __syncthreads();
|
||||
|
||||
// Should fail because the barrier is in device memory.
|
||||
cuda::device::barrier_expect_tx(*bar_ptr, 1);));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 2;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,34 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
// UNSUPPORTED: pre-sm-90
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
// <cuda/barrier>
|
||||
|
||||
#include "arrive_tx.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_HOST,
|
||||
(
|
||||
// Required by concurrent_agents_launch to know how many we're
|
||||
// launching. This can only be an int, because the nvrtc tests use grep
|
||||
// to figure out how many threads to launch.
|
||||
cuda_thread_count = 32;),
|
||||
NV_IS_DEVICE,
|
||||
(constexpr bool split_arrive_and_expect = true; test<split_arrive_and_expect>();));
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,47 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: pre-sm-70
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
|
||||
template <cuda::thread_scope Sco, template <typename, typename> class BarrierSelector>
|
||||
TEST_FUNC void test()
|
||||
{
|
||||
cuda::barrier<Sco> b(3);
|
||||
|
||||
init(&b, 2);
|
||||
|
||||
auto token = b.arrive();
|
||||
b.arrive_and_wait();
|
||||
b.wait(std::move(token));
|
||||
}
|
||||
|
||||
template <cuda::thread_scope Sco>
|
||||
TEST_FUNC void test_select_barrier()
|
||||
{
|
||||
test<Sco, local_memory_selector>();
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test<Sco, shared_memory_selector>(); test<Sco, global_memory_selector>();))
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
test_select_barrier<cuda::thread_scope_system>();
|
||||
test_select_barrier<cuda::thread_scope_device>();
|
||||
test_select_barrier<cuda::thread_scope_block>();
|
||||
test_select_barrier<cuda::thread_scope_thread>();
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: pre-sm-80
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: asm statement is unsupported in tile code
|
||||
|
||||
#include <cuda/barrier>
|
||||
|
||||
#include "cuda_space_selector.h"
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_NV_DIAG_SUPPRESS(static_var_with_dynamic_init)
|
||||
TEST_NV_DIAG_SUPPRESS(set_but_not_used)
|
||||
|
||||
TEST_DEVICE_FUNC void test()
|
||||
{
|
||||
__shared__ cuda::barrier<cuda::thread_scope_block>* b;
|
||||
shared_memory_selector<cuda::barrier<cuda::thread_scope_block>, constructor_initializer> sel;
|
||||
b = sel.construct(2);
|
||||
|
||||
[[maybe_unused]] uint64_t token;
|
||||
asm volatile("mbarrier.arrive.b64 %0, [%1];" : "=l"(token) : "l"(cuda::device::barrier_native_handle(*b)) : "memory");
|
||||
|
||||
b->arrive_and_wait();
|
||||
}
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
NV_IF_TARGET(NV_PROVIDES_SM_80, test();)
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
using nl = cuda::std::numeric_limits<T>;
|
||||
constexpr T all_ones = static_cast<T>(~T{0});
|
||||
constexpr T half_low = all_ones >> (nl::digits / 2u);
|
||||
constexpr T half_high = static_cast<T>(all_ones << (nl::digits / 2u));
|
||||
static_assert(cuda::bit_reverse(all_ones) == all_ones);
|
||||
static_assert(cuda::bit_reverse(T{0}) == T{0});
|
||||
static_assert(cuda::bit_reverse(half_low) == half_high);
|
||||
static_assert(cuda::bit_reverse(T{0b11001001}) == (T{0b10010011} << (nl::digits - 8u)));
|
||||
static_assert(cuda::bit_reverse(T{T{0b10010011} << (nl::digits - 8u)}) == T{0b11001001});
|
||||
unused(all_ones);
|
||||
unused(half_low);
|
||||
unused(half_high);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
test<unsigned char>();
|
||||
test<unsigned short>();
|
||||
test<unsigned>();
|
||||
test<unsigned long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
test<uint8_t>();
|
||||
test<uint16_t>();
|
||||
test<uint32_t>();
|
||||
test<uint64_t>();
|
||||
test<size_t>();
|
||||
test<uintmax_t>();
|
||||
test<uintptr_t>();
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
assert(test());
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
using T = uint32_t;
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, -1, 1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, -1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 0, 33));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 32, 1));
|
||||
static_assert(cuda::bitfield_insert(T{0}, T{0}, 20, 20));
|
||||
|
||||
static_assert(cuda::bitfield_extract(T{0}, -1, 1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 0, -1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 0, 33));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 32, 1));
|
||||
static_assert(cuda::bitfield_extract(T{0}, 20, 20));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,84 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
using nl = cuda::std::numeric_limits<T>;
|
||||
constexpr T all_ones = static_cast<T>(~T{0});
|
||||
unused(all_ones);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 1) == 1);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 1, 1) == 0b10);
|
||||
assert(cuda::bitfield_insert(T{0b10}, all_ones, 0, 1) == 0b11);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 0) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 0, 1) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, all_ones, 2, 1) == all_ones);
|
||||
assert(cuda::bitfield_insert(all_ones, T{0b1000}, 1, 2) == (all_ones & static_cast<T>(~T{0b110})));
|
||||
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 0, 2) == 0b11);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, 3, 2) == 0b11000);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 3, 2) == 0b10111000);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, T{0b11}, 3, 2) == 0b10111000);
|
||||
assert(cuda::bitfield_insert(T{0}, all_ones, nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, 0, nl::digits) == all_ones);
|
||||
assert(cuda::bitfield_insert(T{0b10100000}, all_ones, nl::digits, 0) == T{0b10100000});
|
||||
|
||||
assert(cuda::bitfield_extract(T{0}, 3, 4) == 0);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 0, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 1, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b1011}, 2, 2) == 0b10);
|
||||
assert(cuda::bitfield_extract(all_ones, 0, 0) == 0);
|
||||
assert(cuda::bitfield_extract(all_ones, 0, 4) == 0b1111);
|
||||
assert(cuda::bitfield_extract(all_ones, 2, 4) == 0b1111);
|
||||
|
||||
assert(cuda::bitfield_extract(T{0b1010010}, 0, 2) == 0b10);
|
||||
assert(cuda::bitfield_extract(T{0b10101100}, 3, 2) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, 3, 3) == 0b100);
|
||||
|
||||
assert(cuda::bitfield_extract(T{all_ones}, nl::digits - 1, 1) == 1);
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, 0, nl::digits) == T{0b10100000});
|
||||
assert(cuda::bitfield_extract(T{0b10100000}, nl::digits, 0) == 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
test<unsigned char>();
|
||||
test<unsigned short>();
|
||||
test<unsigned>();
|
||||
test<unsigned long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
test<uint8_t>();
|
||||
test<uint16_t>();
|
||||
test<uint32_t>();
|
||||
test<uint64_t>();
|
||||
test<size_t>();
|
||||
test<uintmax_t>();
|
||||
test<uintptr_t>();
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
assert(test());
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/bit>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <typename T>
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
using nl = cuda::std::numeric_limits<T>;
|
||||
constexpr T all_ones = static_cast<T>(~T{0});
|
||||
unused(all_ones);
|
||||
assert(cuda::bitmask<T>(0, 0) == 0);
|
||||
assert(cuda::bitmask<T>(0, 1) == 1);
|
||||
assert(cuda::bitmask<T>(1, 0) == 0);
|
||||
assert(cuda::bitmask<T>(1, 1) == 0b10);
|
||||
assert(cuda::bitmask<T>(0, 2) == 0b11);
|
||||
assert(cuda::bitmask<T>(2, 2) == 0b1100);
|
||||
|
||||
assert(cuda::bitmask<T>(0, 2) == 0b11);
|
||||
assert(cuda::bitmask<T>(3, 2) == 0b11000);
|
||||
assert(cuda::bitmask<T>(nl::digits - 1, 1) == (T{1} << (nl::digits - 1u)));
|
||||
assert(cuda::bitmask<T>(0, nl::digits) == all_ones);
|
||||
assert(cuda::bitmask<T>(nl::digits, 0) == 0);
|
||||
return true;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
test<unsigned char>();
|
||||
test<unsigned short>();
|
||||
test<unsigned>();
|
||||
test<unsigned long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
test<uint8_t>();
|
||||
test<uint16_t>();
|
||||
test<uint32_t>();
|
||||
test<uint64_t>();
|
||||
test<size_t>();
|
||||
test<uintmax_t>();
|
||||
test<uintptr_t>();
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
assert(test());
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include <cstdio>
|
||||
|
||||
#include <c2h/catch2_test_helper.h>
|
||||
|
||||
C2H_TEST("libcudacxx can be used", "")
|
||||
{
|
||||
printf("CCCL version: %d.%d.%d\n", CCCL_MAJOR_VERSION, CCCL_MINOR_VERSION, CCCL_PATCH_VERSION);
|
||||
REQUIRE(cuda::std::true_type::value);
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH
|
||||
#define TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH
|
||||
|
||||
#include <cuda/algorithm>
|
||||
#include <cuda/buffer>
|
||||
#include <cuda/devices>
|
||||
#include <cuda/memory_resource>
|
||||
#include <cuda/std/mdspan>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <testing.cuh>
|
||||
#include <utility.cuh>
|
||||
|
||||
inline constexpr uint8_t fill_byte = 1;
|
||||
inline constexpr uint32_t buffer_size = 42;
|
||||
|
||||
using legacy_pinned_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_pinned_memory_resource>;
|
||||
using legacy_managed_resource = cuda::mr::synchronous_resource_adapter<cuda::mr::legacy_managed_memory_resource>;
|
||||
|
||||
template <typename T>
|
||||
auto make_pinned_memory_buffer(cuda::stream_ref stream, std::size_t size, cuda::device_ref device = cuda::devices[0])
|
||||
{
|
||||
legacy_pinned_resource resource{cuda::mr::legacy_pinned_memory_resource{device}};
|
||||
return cuda::make_buffer<T>(stream, resource, size, cuda::no_init);
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
auto make_managed_memory_buffer(cuda::stream_ref stream, std::size_t size, cuda::device_ref device = cuda::devices[0])
|
||||
{
|
||||
legacy_managed_resource resource{cuda::mr::legacy_managed_memory_resource{cudaMemAttachGlobal, device}};
|
||||
return cuda::make_buffer<T>(stream, resource, size, cuda::no_init);
|
||||
}
|
||||
|
||||
inline int get_expected_value(uint8_t pattern_byte)
|
||||
{
|
||||
int result;
|
||||
memset(&result, pattern_byte, sizeof(int));
|
||||
return result;
|
||||
}
|
||||
|
||||
template <typename Result>
|
||||
void check_result_and_erase(cuda::stream_ref stream, Result&& result, uint8_t pattern_byte = fill_byte)
|
||||
{
|
||||
int expected = get_expected_value(pattern_byte);
|
||||
|
||||
stream.sync();
|
||||
for (int& i : result)
|
||||
{
|
||||
CCCLRT_REQUIRE(i == expected);
|
||||
i = 0;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename Layout = cuda::std::layout_right, typename Extents>
|
||||
auto make_buffer_for_mdspan(cuda::stream_ref stream, Extents extents, char value = 0)
|
||||
{
|
||||
auto mapping = typename Layout::template mapping<decltype(extents)>{extents};
|
||||
|
||||
auto buffer = make_pinned_memory_buffer<int>(stream, mapping.required_span_size());
|
||||
|
||||
stream.sync();
|
||||
memset(buffer.data(), value, buffer.size() * sizeof(int));
|
||||
|
||||
return buffer;
|
||||
}
|
||||
|
||||
inline auto create_fake_strided_mdspan()
|
||||
{
|
||||
cuda::std::dextents<size_t, 3> dynamic_extents{1, 2, 3};
|
||||
cuda::std::array<size_t, 3> strides{12, 4, 1};
|
||||
#if _CCCL_CUDACC_BELOW(12, 6)
|
||||
auto map = cuda::std::layout_stride::mapping{dynamic_extents, strides};
|
||||
#else
|
||||
cuda::std::layout_stride::mapping map{dynamic_extents, strides};
|
||||
#endif
|
||||
return cuda::std::mdspan<int, decltype(dynamic_extents), cuda::std::layout_stride>(nullptr, map);
|
||||
};
|
||||
|
||||
#endif // TEST_LIBCUDACXX_CCCLRT_ALGORITHM_COMMON_CUH
|
||||
@@ -0,0 +1,303 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/memory_pool>
|
||||
|
||||
#include "common.cuh"
|
||||
#include "cuda/__algorithm/copy.h"
|
||||
|
||||
C2H_CCCLRT_TEST("1d Copy", "[algorithm]")
|
||||
{
|
||||
cuda::stream _stream{cuda::device_ref{0}};
|
||||
|
||||
SECTION("Device resource")
|
||||
{
|
||||
std::vector<int> host_vector(buffer_size);
|
||||
|
||||
{
|
||||
auto buffer = cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
|
||||
cuda::fill_bytes(_stream, buffer, fill_byte);
|
||||
|
||||
cuda::copy_bytes(_stream, buffer, host_vector);
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
|
||||
cuda::copy_bytes(_stream, buffer, host_vector);
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
}
|
||||
{
|
||||
auto not_yet_const_buffer =
|
||||
cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
|
||||
cuda::fill_bytes(_stream, not_yet_const_buffer, fill_byte);
|
||||
|
||||
const auto& const_buffer = not_yet_const_buffer;
|
||||
|
||||
cuda::copy_bytes(_stream, const_buffer, host_vector);
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
|
||||
cuda::copy_bytes(_stream, const_buffer, cuda::std::span(host_vector));
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
|
||||
cuda::copy_configuration config;
|
||||
config.src_location_hint = cuda::device_ref{0};
|
||||
#if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
config.src_access_order = cuda::source_access_order::stream;
|
||||
#else
|
||||
config.src_access_order = cuda::source_access_order::any;
|
||||
#endif
|
||||
cuda::copy_bytes(_stream, const_buffer, host_vector, config);
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
|
||||
cuda::copy_bytes(_stream, const_buffer.first(0), host_vector);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Host and managed resource")
|
||||
{
|
||||
{
|
||||
auto host_buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
|
||||
auto device_buffer = make_managed_memory_buffer<int>(_stream, buffer_size);
|
||||
|
||||
cuda::fill_bytes(_stream, host_buffer, fill_byte);
|
||||
|
||||
cuda::copy_bytes(_stream, host_buffer, device_buffer);
|
||||
check_result_and_erase(_stream, device_buffer);
|
||||
|
||||
cuda::copy_bytes(_stream, host_buffer, device_buffer);
|
||||
check_result_and_erase(_stream, device_buffer);
|
||||
}
|
||||
|
||||
{
|
||||
auto not_yet_const_host_buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
|
||||
auto device_buffer = make_managed_memory_buffer<int>(_stream, buffer_size);
|
||||
cuda::fill_bytes(_stream, not_yet_const_host_buffer, fill_byte);
|
||||
|
||||
const auto& const_host_buffer = not_yet_const_host_buffer;
|
||||
|
||||
cuda::copy_bytes(_stream, const_host_buffer, device_buffer);
|
||||
check_result_and_erase(_stream, device_buffer);
|
||||
|
||||
cuda::copy_bytes(_stream, const_host_buffer, device_buffer);
|
||||
check_result_and_erase(_stream, device_buffer);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Asymmetric size")
|
||||
{
|
||||
auto host_buffer = make_pinned_memory_buffer<int>(_stream, 1);
|
||||
cuda::fill_bytes(_stream, host_buffer, fill_byte);
|
||||
|
||||
::std::vector<int> vec(buffer_size, 0xbeef);
|
||||
|
||||
cuda::copy_bytes(_stream, host_buffer, vec);
|
||||
_stream.sync();
|
||||
|
||||
CCCLRT_REQUIRE(vec[0] == get_expected_value(fill_byte));
|
||||
CCCLRT_REQUIRE(vec[1] == 0xbeef);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("copy_bytes uses the stream device when current device differs", "[algorithm][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
#if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
if (!explicit_device.attribute(cuda::device_attributes::memory_pools_supported))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::stream stream{explicit_device};
|
||||
|
||||
int expected = get_expected_value(fill_byte);
|
||||
int result{};
|
||||
|
||||
{
|
||||
auto host_src = make_pinned_memory_buffer<int>(stream, 1, explicit_device);
|
||||
auto host_dst = make_pinned_memory_buffer<int>(stream, 1, explicit_device);
|
||||
auto src = cuda::make_device_buffer<int>(stream, explicit_device, 1, cuda::no_init);
|
||||
auto dst = cuda::make_device_buffer<int>(stream, explicit_device, 1, cuda::no_init);
|
||||
|
||||
host_src.get_unsynchronized(0) = expected;
|
||||
host_dst.get_unsynchronized(0) = 0;
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(explicit_device);
|
||||
cuda::copy_bytes(stream, host_src, src);
|
||||
}
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
cuda::copy_bytes(stream, src, dst);
|
||||
}
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(explicit_device);
|
||||
cuda::copy_bytes(stream, dst, host_dst);
|
||||
}
|
||||
|
||||
stream.sync();
|
||||
result = host_dst.get_unsynchronized(0);
|
||||
}
|
||||
stream.sync();
|
||||
|
||||
CCCLRT_REQUIRE(result == expected);
|
||||
#endif // _CCCL_CTK_AT_LEAST(13, 0)
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("copy_bytes can copy between peer device buffers", "[algorithm][multi_gpu]")
|
||||
{
|
||||
// Cross-device copy coverage requires at least two GPUs.
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref source_device{0};
|
||||
auto peers = source_device.peers();
|
||||
// This test exercises direct peer memory access; non-peer topologies have no legal device-to-device path to cover.
|
||||
if (peers.empty())
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref destination_device = peers.front();
|
||||
// Device buffers are allocated from stream-ordered memory pools.
|
||||
if (!source_device.attribute(cuda::device_attributes::memory_pools_supported)
|
||||
|| !destination_device.attribute(cuda::device_attributes::memory_pools_supported))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::stream source_stream{source_device};
|
||||
cuda::stream destination_stream{destination_device};
|
||||
cuda::device_memory_pool source_pool{source_device};
|
||||
cuda::device_memory_pool destination_pool{destination_device};
|
||||
source_pool.enable_access_from(destination_device);
|
||||
CCCLRT_REQUIRE(source_pool.is_accessible_from(destination_device));
|
||||
auto source_resource = source_pool.as_ref();
|
||||
auto destination_resource = destination_pool.as_ref();
|
||||
|
||||
int expected = get_expected_value(fill_byte);
|
||||
int result{};
|
||||
|
||||
{
|
||||
auto host_src = make_pinned_memory_buffer<int>(source_stream, 1, source_device);
|
||||
auto host_dst = make_pinned_memory_buffer<int>(destination_stream, 1, destination_device);
|
||||
auto src = cuda::make_buffer<int>(source_stream, source_resource, 1, cuda::no_init);
|
||||
auto dst = cuda::make_buffer<int>(destination_stream, destination_resource, 1, cuda::no_init);
|
||||
|
||||
host_src.get_unsynchronized(0) = expected;
|
||||
host_dst.get_unsynchronized(0) = 0;
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(source_device);
|
||||
cuda::copy_bytes(source_stream, host_src, src);
|
||||
}
|
||||
source_stream.sync();
|
||||
|
||||
cuda::copy_configuration config;
|
||||
config.src_location_hint = source_device;
|
||||
config.dst_location_hint = destination_device;
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(destination_device);
|
||||
cuda::copy_bytes(destination_stream, src, dst, config);
|
||||
cuda::copy_bytes(destination_stream, dst, host_dst);
|
||||
}
|
||||
destination_stream.sync();
|
||||
result = host_dst.get_unsynchronized(0);
|
||||
}
|
||||
source_stream.sync();
|
||||
destination_stream.sync();
|
||||
|
||||
CCCLRT_REQUIRE(result == expected);
|
||||
}
|
||||
|
||||
template <typename SrcLayout = cuda::std::layout_right,
|
||||
typename DstLayout = SrcLayout,
|
||||
typename SrcExtents,
|
||||
typename DstExtents>
|
||||
void test_mdspan_copy_bytes(
|
||||
cuda::stream_ref stream, SrcExtents src_extents = SrcExtents(), DstExtents dst_extents = DstExtents())
|
||||
{
|
||||
auto src_buffer = make_buffer_for_mdspan<SrcLayout>(stream, src_extents, 1);
|
||||
auto tmp_buffer = make_buffer_for_mdspan<SrcLayout>(stream, src_extents, 0);
|
||||
auto dst_buffer = make_buffer_for_mdspan<DstLayout>(stream, dst_extents, 0);
|
||||
|
||||
cuda::std::mdspan<int, SrcExtents, SrcLayout> src(src_buffer.data(), src_extents);
|
||||
cuda::std::mdspan<int, SrcExtents, SrcLayout> tmp(tmp_buffer.data(), src_extents);
|
||||
cuda::std::mdspan<int, DstExtents, DstLayout> dst(dst_buffer.data(), dst_extents);
|
||||
|
||||
for (int i = 0; i < static_cast<int>(src.extent(1)); i++)
|
||||
{
|
||||
src(0, i) = i;
|
||||
}
|
||||
|
||||
cuda::copy_bytes(stream, std::move(src), tmp);
|
||||
|
||||
cuda::copy_configuration config;
|
||||
#if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
config.src_access_order = cuda::source_access_order::stream;
|
||||
#else
|
||||
config.src_access_order = cuda::source_access_order::any;
|
||||
#endif
|
||||
cuda::copy_bytes(stream, tmp, dst, config);
|
||||
stream.sync();
|
||||
|
||||
for (int i = 0; i < static_cast<int>(dst.extent(1)); i++)
|
||||
{
|
||||
CCCLRT_REQUIRE(dst(0, i) == i);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Mdspan copy", "[algorithm]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
|
||||
SECTION("Different extents")
|
||||
{
|
||||
auto static_extents = cuda::std::extents<size_t, 3, 4>();
|
||||
test_mdspan_copy_bytes(stream, static_extents, static_extents);
|
||||
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, static_extents, static_extents);
|
||||
|
||||
auto dynamic_extents = cuda::std::dextents<size_t, 2>(3, 4);
|
||||
test_mdspan_copy_bytes(stream, dynamic_extents, dynamic_extents);
|
||||
test_mdspan_copy_bytes(stream, static_extents, dynamic_extents);
|
||||
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, static_extents, dynamic_extents);
|
||||
|
||||
auto mixed_extents = cuda::std::extents<int, cuda::std::dynamic_extent, 4>(3);
|
||||
test_mdspan_copy_bytes(stream, dynamic_extents, mixed_extents);
|
||||
test_mdspan_copy_bytes(stream, mixed_extents, static_extents);
|
||||
test_mdspan_copy_bytes<cuda::std::layout_left>(stream, mixed_extents, static_extents);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Non exhaustive mdspan copy_bytes", "[algorithm]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
{
|
||||
auto fake_strided_mdspan = create_fake_strided_mdspan();
|
||||
|
||||
try
|
||||
{
|
||||
cuda::copy_bytes(stream, fake_strided_mdspan, fake_strided_mdspan);
|
||||
}
|
||||
catch (const ::std::invalid_argument& e)
|
||||
{
|
||||
CHECK(e.what() == ::std::string("copy_bytes supports only exhaustive mdspans"));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include "common.cuh"
|
||||
|
||||
C2H_CCCLRT_TEST("Fill", "[algorithm]")
|
||||
{
|
||||
cuda::stream _stream{cuda::device_ref{0}};
|
||||
SECTION("Host memory")
|
||||
{
|
||||
auto buffer = make_pinned_memory_buffer<int>(_stream, buffer_size);
|
||||
|
||||
cuda::fill_bytes(_stream, buffer, fill_byte);
|
||||
|
||||
check_result_and_erase(_stream, buffer);
|
||||
}
|
||||
|
||||
SECTION("Device memory")
|
||||
{
|
||||
auto buffer = cuda::make_device_buffer<int>(_stream, cuda::device_ref{0}, buffer_size, cuda::no_init);
|
||||
cuda::fill_bytes(_stream, buffer, fill_byte);
|
||||
|
||||
auto host_vector = make_pinned_memory_buffer<int>(_stream, buffer_size);
|
||||
cuda::copy_bytes(_stream, buffer, host_vector);
|
||||
|
||||
check_result_and_erase(_stream, host_vector);
|
||||
|
||||
auto span = buffer.first(0);
|
||||
cuda::fill_bytes(_stream, span, fill_byte);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Mdspan Fill", "[algorithm]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
{
|
||||
cuda::std::dextents<size_t, 3> dynamic_extents{1, 2, 3};
|
||||
auto buffer = make_buffer_for_mdspan(stream, dynamic_extents, 0);
|
||||
cuda::std::mdspan<int, decltype(dynamic_extents)> dynamic_mdspan(buffer.data(), dynamic_extents);
|
||||
|
||||
cuda::fill_bytes(stream, dynamic_mdspan, fill_byte);
|
||||
check_result_and_erase(stream, buffer);
|
||||
}
|
||||
{
|
||||
cuda::std::extents<size_t, 2, cuda::std::dynamic_extent, 4> mixed_extents{1};
|
||||
auto buffer = make_buffer_for_mdspan(stream, mixed_extents, 0);
|
||||
cuda::std::mdspan<int, decltype(mixed_extents)> mixed_mdspan(buffer.data(), mixed_extents);
|
||||
|
||||
cuda::fill_bytes(stream, cuda::std::move(mixed_mdspan), fill_byte);
|
||||
check_result_and_erase(stream, buffer);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Non exhaustive mdspan fill_bytes", "[data_manipulation]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
{
|
||||
auto fake_strided_mdspan = create_fake_strided_mdspan();
|
||||
|
||||
try
|
||||
{
|
||||
cuda::fill_bytes(stream, fake_strided_mdspan, fill_byte);
|
||||
}
|
||||
catch (const ::std::invalid_argument& e)
|
||||
{
|
||||
CHECK(e.what() == ::std::string("fill_bytes supports only exhaustive mdspans"));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,103 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __COMMON_HOST_DEVICE_H__
|
||||
#define __COMMON_HOST_DEVICE_H__
|
||||
|
||||
#include <cuda/hierarchy>
|
||||
|
||||
#include "testing.cuh"
|
||||
|
||||
template <typename Dims, typename Lambda>
|
||||
void __global__ lambda_launcher(const Dims dims, const Lambda lambda)
|
||||
{
|
||||
lambda(dims);
|
||||
}
|
||||
|
||||
template <typename Comparator, unsigned int FilterArch>
|
||||
bool arch_filter(const cudaDeviceProp& props)
|
||||
{
|
||||
int act_arch = props.major * 10 + props.minor;
|
||||
if (Comparator()(act_arch, FilterArch))
|
||||
{
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool skip_host_exec(bool (* /* filter */)(const cudaDeviceProp&))
|
||||
{
|
||||
return false;
|
||||
}
|
||||
|
||||
static bool skip_device_exec(bool (*filter)(const cudaDeviceProp&))
|
||||
{
|
||||
cudaDeviceProp props;
|
||||
CUDART(cudaGetDeviceProperties(&props, 0));
|
||||
return filter(props);
|
||||
}
|
||||
|
||||
template <typename Dims, typename Lambda, typename... Filters>
|
||||
void test_host_dev(const Dims& dims, const Lambda& lambda, const Filters&... filters)
|
||||
{
|
||||
SECTION("Host execution")
|
||||
{
|
||||
if ((... && !skip_host_exec(filters)))
|
||||
{
|
||||
// host testing
|
||||
lambda(dims);
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Device execution")
|
||||
{
|
||||
// Asymmetrical but cleaner
|
||||
if ((... || skip_device_exec(filters)))
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cudaLaunchConfig_t config = {};
|
||||
config.gridDim = {0};
|
||||
cudaLaunchAttribute attrs[1];
|
||||
config.attrs = &attrs[0];
|
||||
|
||||
config.blockDim = dim3{cuda::gpu_thread.dims(cuda::block, dims)};
|
||||
config.gridDim = dim3{cuda::block.dims(cuda::grid, dims)};
|
||||
|
||||
if constexpr (Dims::has_level(cuda::cluster))
|
||||
{
|
||||
dim3 cluster_dims{cuda::block.dims(cuda::cluster, dims)};
|
||||
config.attrs[config.numAttrs].id = cudaLaunchAttributeClusterDimension;
|
||||
config.attrs[config.numAttrs].val.clusterDim = {cluster_dims.x, cluster_dims.y, cluster_dims.z};
|
||||
config.numAttrs = 1;
|
||||
}
|
||||
else
|
||||
{
|
||||
config.numAttrs = 0;
|
||||
}
|
||||
|
||||
// device testing
|
||||
CUDART(cudaLaunchKernelEx(&config, lambda_launcher<Dims, Lambda>, dims, lambda));
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
}
|
||||
}
|
||||
|
||||
template <typename Fn, typename Tuple>
|
||||
void apply_each(const Fn& fn, const Tuple& tuple)
|
||||
{
|
||||
cuda::std::apply(
|
||||
[&](const auto&... elems) {
|
||||
(fn(elems), ...);
|
||||
},
|
||||
tuple);
|
||||
}
|
||||
|
||||
#endif // __COMMON_HOST_DEVICE_H__
|
||||
@@ -0,0 +1,120 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__
|
||||
#define __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#include <cuda/__driver/driver_api.h>
|
||||
|
||||
#include <nv/target>
|
||||
|
||||
#include <exception> // IWYU pragma: keep
|
||||
#include <sstream>
|
||||
|
||||
#include "test_macros.h"
|
||||
#include "utility.cuh"
|
||||
#include <c2h/catch2_test_helper.h>
|
||||
|
||||
#define CUDART(call) REQUIRE((call) == cudaSuccess)
|
||||
|
||||
#define CCCLRT_REQUIRE(condition) REQUIRE(condition)
|
||||
|
||||
#define CCCLRT_CHECK(condition) CHECK(condition)
|
||||
|
||||
#define CCCLRT_FAIL(message) FAIL(message)
|
||||
|
||||
#define CCCLRT_CHECK_FALSE(condition) CHECK_FALSE(condition)
|
||||
|
||||
// Explicit device side require macros for clang-cuda
|
||||
#define CCCLRT_REQUIRE_DEVICE(condition) REQUIRE_DEVICE(condition)
|
||||
#define CCCLRT_CHECK_DEVICE(condition) CHECK(condition)
|
||||
#define CCCLRT_FAIL_DEVICE(message) FAIL(message)
|
||||
#define CCCLRT_CHECK_FALSE_DEVICE(condition) CHECK_FALSE(condition)
|
||||
|
||||
TEST_FUNC constexpr bool operator==(const dim3& lhs, const dim3& rhs) noexcept
|
||||
{
|
||||
return (lhs.x == rhs.x) && (lhs.y == rhs.y) && (lhs.z == rhs.z);
|
||||
}
|
||||
|
||||
namespace Catch
|
||||
{
|
||||
template <>
|
||||
struct StringMaker<dim3>
|
||||
{
|
||||
static std::string convert(dim3 const& dims)
|
||||
{
|
||||
std::ostringstream oss;
|
||||
oss << "(" << dims.x << ", " << dims.y << ", " << dims.z << ")";
|
||||
return oss.str();
|
||||
}
|
||||
};
|
||||
} // namespace Catch
|
||||
|
||||
namespace
|
||||
{
|
||||
namespace test
|
||||
{
|
||||
inline int count_driver_stack()
|
||||
{
|
||||
if (::cuda::__driver::__ctxGetCurrent() != nullptr)
|
||||
{
|
||||
auto ctx = ::cuda::__driver::__ctxPop();
|
||||
auto result = 1 + count_driver_stack();
|
||||
::cuda::__driver::__ctxPush(ctx);
|
||||
return result;
|
||||
}
|
||||
else
|
||||
{
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
|
||||
inline void empty_driver_stack()
|
||||
{
|
||||
while (::cuda::__driver::__ctxGetCurrent() != nullptr)
|
||||
{
|
||||
::cuda::__driver::__ctxPop();
|
||||
}
|
||||
}
|
||||
|
||||
inline int cuda_driver_version()
|
||||
{
|
||||
return ::cuda::__driver::__getVersion();
|
||||
}
|
||||
|
||||
// Needs to be a template because we use template catch2 macro
|
||||
template <typename Dummy = void>
|
||||
struct ccclrt_test_fixture
|
||||
{
|
||||
ccclrt_test_fixture()
|
||||
{
|
||||
empty_driver_stack();
|
||||
}
|
||||
~ccclrt_test_fixture()
|
||||
{
|
||||
CCCLRT_CHECK(count_driver_stack() == 0);
|
||||
}
|
||||
};
|
||||
} // namespace test
|
||||
} // namespace
|
||||
|
||||
// Test macro that should be used in all cccl-rt tests
|
||||
// It first empties the driver stack in case some other test has left it non-empty
|
||||
// and then runs the test. At the end it checks if it remained empty, which ensures
|
||||
// we don't accidentally initialize device 0 through CUDART usage and makes sure
|
||||
// our APIs work with empty driver stack.
|
||||
#define C2H_CCCLRT_TEST(NAME, TAGS, ...) C2H_TEST_WITH_FIXTURE(::test::ccclrt_test_fixture, NAME, TAGS, __VA_ARGS__)
|
||||
|
||||
#define C2H_CCCLRT_TEST_LIST(NAME, TAGS, ...) \
|
||||
C2H_TEST_LIST_WITH_FIXTURE(::test::ccclrt_test_fixture, NAME, TAGS, __VA_ARGS__)
|
||||
|
||||
#endif // __LIBCUDACXX_CCCLRT_COMMON_TESTING_H__
|
||||
@@ -0,0 +1,192 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef __COMMON_UTILITY_H__
|
||||
#define __COMMON_UTILITY_H__
|
||||
|
||||
#include <cuda_runtime_api.h>
|
||||
// cuda_runtime_api needs to come first
|
||||
|
||||
#include <cuda/__runtime/api_wrapper.h>
|
||||
#include <cuda/__runtime/ensure_current_context.h>
|
||||
#include <cuda/__stream/stream_ref.h>
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include <new> // IWYU pragma: keep (needed for placement new)
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_DEVICE_FUNC inline void ccclrt_require_impl(
|
||||
bool condition, const char* condition_text, const char* filename, unsigned int linenum, const char* funcname)
|
||||
{
|
||||
if (!condition)
|
||||
{
|
||||
// TODO do warp aggregate prints for easier readability?
|
||||
printf("%s:%u: %s: block: [%d,%d,%d], thread: [%d,%d,%d] Condition `%s` failed.\n",
|
||||
filename,
|
||||
linenum,
|
||||
funcname,
|
||||
blockIdx.x,
|
||||
blockIdx.y,
|
||||
blockIdx.z,
|
||||
threadIdx.x,
|
||||
threadIdx.y,
|
||||
threadIdx.z,
|
||||
condition_text);
|
||||
__trap();
|
||||
}
|
||||
}
|
||||
|
||||
namespace
|
||||
{
|
||||
namespace test
|
||||
{
|
||||
template <typename T1, typename T2>
|
||||
T1& assign(T1& t1, T2&& t2)
|
||||
{
|
||||
t1 = ::cuda::std::forward<T2>(t2);
|
||||
return t1;
|
||||
}
|
||||
|
||||
struct _malloc_pinned
|
||||
{
|
||||
private:
|
||||
void* pv = nullptr;
|
||||
|
||||
public:
|
||||
explicit _malloc_pinned(std::size_t size)
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
_CCCL_TRY_CUDA_API(::cudaMallocHost, "failed to allocate pinned memory", &pv, size);
|
||||
}
|
||||
|
||||
~_malloc_pinned()
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
[[maybe_unused]] auto status = ::cudaFreeHost(pv);
|
||||
}
|
||||
|
||||
template <class T>
|
||||
T* get_as() const noexcept
|
||||
{
|
||||
return static_cast<T*>(pv);
|
||||
}
|
||||
};
|
||||
|
||||
template <class T>
|
||||
struct pinned
|
||||
{
|
||||
private:
|
||||
_malloc_pinned _mem;
|
||||
|
||||
public:
|
||||
explicit pinned(T t)
|
||||
: _mem(sizeof(T))
|
||||
{
|
||||
::new (_mem.get_as<void>()) T(std::move(t));
|
||||
}
|
||||
|
||||
~pinned()
|
||||
{
|
||||
get()->~T();
|
||||
}
|
||||
|
||||
T* get() noexcept
|
||||
{
|
||||
return _mem.get_as<T>();
|
||||
}
|
||||
const T* get() const noexcept
|
||||
{
|
||||
return _mem.get_as<T>();
|
||||
}
|
||||
|
||||
T& operator*() noexcept
|
||||
{
|
||||
return *get();
|
||||
}
|
||||
const T& operator*() const noexcept
|
||||
{
|
||||
return *get();
|
||||
}
|
||||
};
|
||||
|
||||
template <int N>
|
||||
struct assign_n
|
||||
{
|
||||
TEST_DEVICE_FUNC constexpr void operator()(int* pi) const noexcept
|
||||
{
|
||||
*pi = N;
|
||||
}
|
||||
};
|
||||
|
||||
template <int N>
|
||||
struct verify_n
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
|
||||
{
|
||||
// TODO: fix clang CUDA require macro
|
||||
// CCCLRT_REQUIRE(*pi == N);
|
||||
ccclrt_require_impl(*pi == N, "*pi == N", __FILE__, __LINE__, __PRETTY_FUNCTION__);
|
||||
}
|
||||
};
|
||||
|
||||
using assign_42 = assign_n<42>;
|
||||
using verify_42 = verify_n<42>;
|
||||
|
||||
struct atomic_add_one
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
|
||||
{
|
||||
cuda::atomic_ref atomic_pi(*pi);
|
||||
atomic_pi.fetch_add(1);
|
||||
}
|
||||
};
|
||||
|
||||
struct atomic_sub_one
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
|
||||
{
|
||||
cuda::atomic_ref atomic_pi(*pi);
|
||||
atomic_pi.fetch_sub(1);
|
||||
}
|
||||
};
|
||||
|
||||
struct spin_until_80
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()(int* pi) const noexcept
|
||||
{
|
||||
cuda::atomic_ref atomic_pi(*pi);
|
||||
while (atomic_pi.load() != 80)
|
||||
;
|
||||
}
|
||||
};
|
||||
|
||||
struct empty_kernel
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()() const noexcept {}
|
||||
};
|
||||
|
||||
template <class Fn, class... Args>
|
||||
static __global__ void kernel_launcher(Fn fn, Args... args)
|
||||
{
|
||||
fn(args...);
|
||||
}
|
||||
|
||||
template <class Fn, class... Args>
|
||||
void launch_kernel_single_thread(cuda::stream_ref stream, Fn fn, Args... args)
|
||||
{
|
||||
cuda::__ensure_current_context guard(stream);
|
||||
kernel_launcher<<<1, 1, 0, stream.get()>>>(fn, args...);
|
||||
assert(cudaGetLastError() == cudaSuccess);
|
||||
}
|
||||
} // namespace test
|
||||
} // namespace
|
||||
#endif // __COMMON_UTILITY_H__
|
||||
@@ -0,0 +1,65 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstddef>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// 1. Test signature.
|
||||
static_assert(cuda::std::__is_cuda_std_array_v<decltype(cuda::__all_arch_ids())>);
|
||||
static_assert(noexcept(cuda::__all_arch_ids()));
|
||||
|
||||
// 2. Test that all values are present.
|
||||
const auto all_arch_ids = cuda::__all_arch_ids();
|
||||
|
||||
cuda::std::size_t i = 0;
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_50);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_52);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_53);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_60);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_61);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_62);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_70);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_75);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_80);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_86);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_87);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_88);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_89);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_90);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_100);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_103);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_110);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_120);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_121);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_90a);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_100a);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_103a);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_110a);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_120a);
|
||||
assert(all_arch_ids[i++] == cuda::arch_id::sm_121a);
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,151 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_DEVICE_FUNC void test_current()
|
||||
{
|
||||
// 1. Test cuda::device::current_arch_id() signature.
|
||||
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::device::current_arch_id())>);
|
||||
static_assert(noexcept(cuda::device::current_arch_id()));
|
||||
|
||||
// 2. Test cuda::device::current_arch_id() in constexpr context. Unsupported with nvc++ -cuda.
|
||||
#if !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
if constexpr (cuda::device::current_arch_id() == cuda::arch_id{})
|
||||
{
|
||||
// cuda::arch_id{} is an invalid architecture, so this statement should be unrachable.
|
||||
assert(false);
|
||||
}
|
||||
#endif // !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
|
||||
// 3. Test cuda::device::current_arch_id() against the NV_IF_TARGET macros
|
||||
[[maybe_unused]] cuda::arch_id arch = cuda::device::current_arch_id();
|
||||
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_HAS_FEATURE_SM_90a,
|
||||
(assert(arch == cuda::arch_id::sm_90a); return;),
|
||||
NV_HAS_FEATURE_SM_100a,
|
||||
(assert(arch == cuda::arch_id::sm_100a); return;),
|
||||
NV_HAS_FEATURE_SM_103a,
|
||||
(assert(arch == cuda::arch_id::sm_103a); return;),
|
||||
NV_HAS_FEATURE_SM_110a,
|
||||
(assert(arch == cuda::arch_id::sm_110a); return;),
|
||||
NV_HAS_FEATURE_SM_120a,
|
||||
(assert(arch == cuda::arch_id::sm_120a); return;),
|
||||
NV_HAS_FEATURE_SM_121a,
|
||||
(assert(arch == cuda::arch_id::sm_121a); return;))
|
||||
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_EXACTLY_SM_60,
|
||||
(assert(arch == cuda::arch_id::sm_60); return;),
|
||||
NV_IS_EXACTLY_SM_61,
|
||||
(assert(arch == cuda::arch_id::sm_61); return;),
|
||||
NV_IS_EXACTLY_SM_62,
|
||||
(assert(arch == cuda::arch_id::sm_62); return;),
|
||||
NV_IS_EXACTLY_SM_70,
|
||||
(assert(arch == cuda::arch_id::sm_70); return;),
|
||||
NV_IS_EXACTLY_SM_75,
|
||||
(assert(arch == cuda::arch_id::sm_75); return;),
|
||||
NV_IS_EXACTLY_SM_80,
|
||||
(assert(arch == cuda::arch_id::sm_80); return;),
|
||||
NV_IS_EXACTLY_SM_86,
|
||||
(assert(arch == cuda::arch_id::sm_86); return;),
|
||||
NV_IS_EXACTLY_SM_87,
|
||||
(assert(arch == cuda::arch_id::sm_87); return;),
|
||||
NV_IS_EXACTLY_SM_88,
|
||||
(assert(arch == cuda::arch_id::sm_88); return;),
|
||||
NV_IS_EXACTLY_SM_89,
|
||||
(assert(arch == cuda::arch_id::sm_89); return;))
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_EXACTLY_SM_90,
|
||||
(assert(arch == cuda::arch_id::sm_90); return;),
|
||||
NV_IS_EXACTLY_SM_100,
|
||||
(assert(arch == cuda::arch_id::sm_100); return;),
|
||||
NV_IS_EXACTLY_SM_103,
|
||||
(assert(arch == cuda::arch_id::sm_103); return;),
|
||||
NV_IS_EXACTLY_SM_110,
|
||||
(assert(arch == cuda::arch_id::sm_110); return;),
|
||||
NV_IS_EXACTLY_SM_120,
|
||||
(assert(arch == cuda::arch_id::sm_120); return;),
|
||||
NV_IS_EXACTLY_SM_121,
|
||||
(assert(arch == cuda::arch_id::sm_121); return;),
|
||||
NV_ANY_TARGET,
|
||||
(assert(false);) // fail for unknown architecture
|
||||
)
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// 1. Test cuda::arch_id enum values.
|
||||
static_assert(cuda::std::is_scoped_enum_v<cuda::arch_id>);
|
||||
static_assert(cuda::std::is_same_v<cuda::std::underlying_type_t<cuda::arch_id>, int>);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_60) == 60);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_61) == 61);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_62) == 62);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_70) == 70);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_75) == 75);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_80) == 80);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_86) == 86);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_87) == 87);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_88) == 88);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_89) == 89);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_90) == 90);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_100) == 100);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_103) == 103);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_110) == 110);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_120) == 120);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_121) == 121);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_90a) == 90 * 100000);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_100a) == 100 * 100000);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_103a) == 103 * 100000);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_110a) == 110 * 100000);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_120a) == 120 * 100000);
|
||||
static_assert(cuda::std::to_underlying(cuda::arch_id::sm_121a) == 121 * 100000);
|
||||
|
||||
// 2. Test cuda::to_arch_id(cuda::compute_capability).
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::to_arch_id(cuda::compute_capability{}))>);
|
||||
static_assert(noexcept(cuda::to_arch_id(cuda::compute_capability{})));
|
||||
|
||||
cuda::arch_id id_lowest = cuda::to_arch_id(cuda::compute_capability{60});
|
||||
assert(id_lowest == cuda::arch_id::sm_60);
|
||||
cuda::arch_id id_highest = cuda::to_arch_id(cuda::compute_capability{120});
|
||||
assert(id_highest == cuda::arch_id::sm_120);
|
||||
}
|
||||
|
||||
// 3. Test cuda::to_arch_specific_id(cuda::compute_capability).
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<cuda::arch_id, decltype(cuda::to_arch_specific_id(cuda::compute_capability{}))>);
|
||||
static_assert(noexcept(cuda::to_arch_specific_id(cuda::compute_capability{})));
|
||||
|
||||
cuda::arch_id id_lowest = cuda::to_arch_specific_id(cuda::compute_capability{90});
|
||||
assert(id_lowest == cuda::arch_id::sm_90a);
|
||||
cuda::arch_id id_highest = cuda::to_arch_specific_id(cuda::compute_capability{120});
|
||||
assert(id_highest == cuda::arch_id::sm_120a);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_current();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
|
||||
#if __cpp_lib_format >= 201907L
|
||||
# include <format>
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
|
||||
#include "literal.h"
|
||||
|
||||
#if __cpp_lib_format >= 201907L
|
||||
template <class C>
|
||||
void test()
|
||||
{
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_60) == TEST_STRLIT(C, "sm_60"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_61) == TEST_STRLIT(C, "sm_61"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_62) == TEST_STRLIT(C, "sm_62"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_70) == TEST_STRLIT(C, "sm_70"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_75) == TEST_STRLIT(C, "sm_75"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_80) == TEST_STRLIT(C, "sm_80"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_86) == TEST_STRLIT(C, "sm_86"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_87) == TEST_STRLIT(C, "sm_87"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_88) == TEST_STRLIT(C, "sm_88"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_89) == TEST_STRLIT(C, "sm_89"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_90) == TEST_STRLIT(C, "sm_90"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_100) == TEST_STRLIT(C, "sm_100"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_103) == TEST_STRLIT(C, "sm_103"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_110) == TEST_STRLIT(C, "sm_110"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_120) == TEST_STRLIT(C, "sm_120"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_121) == TEST_STRLIT(C, "sm_121"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_90a) == TEST_STRLIT(C, "sm_90a"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_100a) == TEST_STRLIT(C, "sm_100a"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_103a) == TEST_STRLIT(C, "sm_103a"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_110a) == TEST_STRLIT(C, "sm_110a"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_120a) == TEST_STRLIT(C, "sm_120a"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::arch_id::sm_121a) == TEST_STRLIT(C, "sm_121a"));
|
||||
}
|
||||
|
||||
void test()
|
||||
{
|
||||
test<char>();
|
||||
test<wchar_t>();
|
||||
}
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
#if __cpp_lib_format >= 201907L
|
||||
NV_IF_TARGET(NV_IS_HOST, (test();))
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,182 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/cassert>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T>
|
||||
TEST_DEVICE_FUNC T foo(const T& x)
|
||||
{
|
||||
return x;
|
||||
}
|
||||
|
||||
template <cuda::arch_id Arch>
|
||||
__global__ void arch_specific_kernel_mock_do_not_launch()
|
||||
{
|
||||
assert(Arch == cuda::device::current_arch_id());
|
||||
|
||||
// I will try to pack something like this into an API
|
||||
if constexpr (cuda::arch_traits<Arch>().compute_capability != cuda::device::current_arch_traits().compute_capability)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
[[maybe_unused]] __shared__ int array[cuda::arch_traits<Arch>().max_shared_memory_per_block / sizeof(int)];
|
||||
|
||||
// constexpr is useless and I can't use intrinsics here :(
|
||||
if (cuda::device::current_arch_traits().cluster_supported)
|
||||
{
|
||||
[[maybe_unused]] int dummy;
|
||||
asm volatile("mov.u32 %0, %%cluster_ctarank;" : "=r"(dummy));
|
||||
}
|
||||
if (cuda::device::current_arch_traits().redux_intrinisic)
|
||||
{
|
||||
[[maybe_unused]] int dummy1 = 0, dummy2 = 0;
|
||||
asm volatile("redux.sync.add.s32 %0, %1, 0xffffffff;" : "=r"(dummy1) : "r"(dummy2));
|
||||
}
|
||||
if (cuda::device::current_arch_traits().cp_async_supported)
|
||||
{
|
||||
asm volatile("cp.async.commit_group;");
|
||||
}
|
||||
|
||||
// Confirm trait value is defined device code and usable as a reference
|
||||
foo(cuda::arch_traits<Arch>().compute_capability);
|
||||
foo(cuda::device::current_arch_traits().compute_capability);
|
||||
}
|
||||
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_70>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_75>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_80>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_86>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_87>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_88>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_89>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_90>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_100>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_103>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_110>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_120>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_121>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_90a>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_100a>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_103a>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_110a>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_120a>();
|
||||
template __global__ void arch_specific_kernel_mock_do_not_launch<cuda::arch_id::sm_121a>();
|
||||
|
||||
template <int ComputeCapability>
|
||||
void constexpr compare_static_and_dynamic()
|
||||
{
|
||||
constexpr cuda::compute_capability cc{ComputeCapability};
|
||||
constexpr cuda::arch_traits_t static_traits = cuda::arch_traits<cuda::to_arch_id(cc)>();
|
||||
constexpr cuda::arch_traits_t dynamic_traits = cuda::arch_traits_for(cuda::to_arch_id(cc));
|
||||
|
||||
static_assert(static_traits.arch_id == dynamic_traits.arch_id);
|
||||
static_assert(static_traits.max_threads_per_block == dynamic_traits.max_threads_per_block);
|
||||
static_assert(static_traits.max_block_dim_x == dynamic_traits.max_block_dim_x);
|
||||
static_assert(static_traits.max_block_dim_y == dynamic_traits.max_block_dim_y);
|
||||
static_assert(static_traits.max_block_dim_z == dynamic_traits.max_block_dim_z);
|
||||
static_assert(static_traits.max_grid_dim_x == dynamic_traits.max_grid_dim_x);
|
||||
static_assert(static_traits.max_grid_dim_y == dynamic_traits.max_grid_dim_y);
|
||||
static_assert(static_traits.max_grid_dim_z == dynamic_traits.max_grid_dim_z);
|
||||
|
||||
static_assert(static_traits.warp_size == dynamic_traits.warp_size);
|
||||
static_assert(static_traits.total_constant_memory == dynamic_traits.total_constant_memory);
|
||||
static_assert(static_traits.max_resident_grids == dynamic_traits.max_resident_grids);
|
||||
static_assert(static_traits.max_shared_memory_per_block == dynamic_traits.max_shared_memory_per_block);
|
||||
static_assert(static_traits.gpu_overlap == dynamic_traits.gpu_overlap);
|
||||
static_assert(static_traits.can_map_host_memory == dynamic_traits.can_map_host_memory);
|
||||
static_assert(static_traits.concurrent_kernels == dynamic_traits.concurrent_kernels);
|
||||
static_assert(static_traits.stream_priorities_supported == dynamic_traits.stream_priorities_supported);
|
||||
static_assert(static_traits.global_l1_cache_supported == dynamic_traits.global_l1_cache_supported);
|
||||
static_assert(static_traits.local_l1_cache_supported == dynamic_traits.local_l1_cache_supported);
|
||||
static_assert(static_traits.max_registers_per_block == dynamic_traits.max_registers_per_block);
|
||||
static_assert(static_traits.max_registers_per_multiprocessor == dynamic_traits.max_registers_per_multiprocessor);
|
||||
|
||||
static_assert(static_traits.compute_capability == dynamic_traits.compute_capability);
|
||||
static_assert(static_traits.compute_capability_major == dynamic_traits.compute_capability_major);
|
||||
static_assert(static_traits.compute_capability_minor == dynamic_traits.compute_capability_minor);
|
||||
static_assert(static_traits.compute_capability == dynamic_traits.compute_capability);
|
||||
static_assert(
|
||||
static_traits.max_shared_memory_per_multiprocessor == dynamic_traits.max_shared_memory_per_multiprocessor);
|
||||
static_assert(static_traits.max_blocks_per_multiprocessor == dynamic_traits.max_blocks_per_multiprocessor);
|
||||
static_assert(static_traits.max_warps_per_multiprocessor == dynamic_traits.max_warps_per_multiprocessor);
|
||||
static_assert(static_traits.max_threads_per_multiprocessor == dynamic_traits.max_threads_per_multiprocessor);
|
||||
static_assert(static_traits.reserved_shared_memory_per_block == dynamic_traits.reserved_shared_memory_per_block);
|
||||
static_assert(static_traits.max_shared_memory_per_block_optin == dynamic_traits.max_shared_memory_per_block_optin);
|
||||
static_assert(static_traits.cluster_supported == dynamic_traits.cluster_supported);
|
||||
static_assert(static_traits.redux_intrinisic == dynamic_traits.redux_intrinisic);
|
||||
static_assert(static_traits.elect_intrinsic == dynamic_traits.elect_intrinsic);
|
||||
static_assert(static_traits.cp_async_supported == dynamic_traits.cp_async_supported);
|
||||
static_assert(static_traits.tma_supported == dynamic_traits.tma_supported);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Traits", "[device]")
|
||||
{
|
||||
compare_static_and_dynamic<70>();
|
||||
compare_static_and_dynamic<75>();
|
||||
compare_static_and_dynamic<80>();
|
||||
compare_static_and_dynamic<86>();
|
||||
compare_static_and_dynamic<89>();
|
||||
compare_static_and_dynamic<90>();
|
||||
compare_static_and_dynamic<100>();
|
||||
compare_static_and_dynamic<103>();
|
||||
compare_static_and_dynamic<110>();
|
||||
compare_static_and_dynamic<120>();
|
||||
|
||||
// Compare arch traits with attributes
|
||||
for (const cuda::device_ref& dev : cuda::devices)
|
||||
{
|
||||
const auto cc = dev.attribute(cuda::device_attributes::compute_capability);
|
||||
|
||||
const auto traits = cuda::arch_traits_for(cc);
|
||||
|
||||
CCCLRT_REQUIRE(traits.max_threads_per_block == dev.attribute(cuda::device_attributes::max_threads_per_block));
|
||||
CCCLRT_REQUIRE(traits.max_block_dim_x == dev.attribute(cuda::device_attributes::max_block_dim_x));
|
||||
CCCLRT_REQUIRE(traits.max_block_dim_y == dev.attribute(cuda::device_attributes::max_block_dim_y));
|
||||
CCCLRT_REQUIRE(traits.max_block_dim_z == dev.attribute(cuda::device_attributes::max_block_dim_z));
|
||||
CCCLRT_REQUIRE(traits.max_grid_dim_x == dev.attribute(cuda::device_attributes::max_grid_dim_x));
|
||||
CCCLRT_REQUIRE(traits.max_grid_dim_y == dev.attribute(cuda::device_attributes::max_grid_dim_y));
|
||||
CCCLRT_REQUIRE(traits.max_grid_dim_z == dev.attribute(cuda::device_attributes::max_grid_dim_z));
|
||||
|
||||
CCCLRT_REQUIRE(traits.warp_size == dev.attribute(cuda::device_attributes::warp_size));
|
||||
CCCLRT_REQUIRE(traits.total_constant_memory == dev.attribute(cuda::device_attributes::total_constant_memory));
|
||||
CCCLRT_REQUIRE(
|
||||
traits.max_shared_memory_per_block == dev.attribute(cuda::device_attributes::max_shared_memory_per_block));
|
||||
CCCLRT_REQUIRE(traits.gpu_overlap == dev.attribute(cuda::device_attributes::gpu_overlap));
|
||||
CCCLRT_REQUIRE(traits.can_map_host_memory == dev.attribute(cuda::device_attributes::can_map_host_memory));
|
||||
CCCLRT_REQUIRE(traits.concurrent_kernels == dev.attribute(cuda::device_attributes::concurrent_kernels));
|
||||
CCCLRT_REQUIRE(
|
||||
traits.stream_priorities_supported == dev.attribute(cuda::device_attributes::stream_priorities_supported));
|
||||
CCCLRT_REQUIRE(
|
||||
traits.global_l1_cache_supported == dev.attribute(cuda::device_attributes::global_l1_cache_supported));
|
||||
CCCLRT_REQUIRE(traits.local_l1_cache_supported == dev.attribute(cuda::device_attributes::local_l1_cache_supported));
|
||||
CCCLRT_REQUIRE(traits.max_registers_per_block == dev.attribute(cuda::device_attributes::max_registers_per_block));
|
||||
CCCLRT_REQUIRE(traits.max_registers_per_multiprocessor
|
||||
== dev.attribute(cuda::device_attributes::max_registers_per_multiprocessor));
|
||||
CCCLRT_REQUIRE(traits.compute_capability_major == dev.attribute(cuda::device_attributes::compute_capability_major));
|
||||
CCCLRT_REQUIRE(traits.compute_capability_minor == dev.attribute(cuda::device_attributes::compute_capability_minor));
|
||||
CCCLRT_REQUIRE(traits.compute_capability == dev.attribute(cuda::device_attributes::compute_capability));
|
||||
CCCLRT_REQUIRE(traits.max_shared_memory_per_multiprocessor
|
||||
== dev.attribute(cuda::device_attributes::max_shared_memory_per_multiprocessor));
|
||||
CCCLRT_REQUIRE(
|
||||
traits.max_blocks_per_multiprocessor == dev.attribute(cuda::device_attributes::max_blocks_per_multiprocessor));
|
||||
CCCLRT_REQUIRE(
|
||||
traits.max_threads_per_multiprocessor == dev.attribute(cuda::device_attributes::max_threads_per_multiprocessor));
|
||||
CCCLRT_REQUIRE(traits.reserved_shared_memory_per_block
|
||||
== dev.attribute(cuda::device_attributes::reserved_shared_memory_per_block));
|
||||
CCCLRT_REQUIRE(traits.max_shared_memory_per_block_optin
|
||||
== dev.attribute(cuda::device_attributes::max_shared_memory_per_block_optin));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,256 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
// ADDITIONAL_COMPILE_DEFINITIONS: CCCL_IGNORE_DEPRECATED_API
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_DEVICE_FUNC void test_current()
|
||||
{
|
||||
// 1. Test cuda::device::current_compute_capability() signature.
|
||||
static_assert(cuda::std::is_same_v<cuda::compute_capability, decltype(cuda::device::current_compute_capability())>);
|
||||
static_assert(noexcept(cuda::device::current_compute_capability()));
|
||||
|
||||
// 1. Test cuda::device::current_compute_capability() in constexpr context. Unsupported with nvc++ -cuda.
|
||||
#if !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
if constexpr (cuda::device::current_compute_capability() == cuda::compute_capability{})
|
||||
{
|
||||
// cuda::current_compute_capability{} is an invalid compute capability, so this statement should be unrachable.
|
||||
assert(false);
|
||||
}
|
||||
#endif // !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
|
||||
// 2. Test cuda::device::current_compute_capability() against the NV_IF_TARGET macros
|
||||
[[maybe_unused]] cuda::compute_capability cc = cuda::device::current_compute_capability();
|
||||
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_EXACTLY_SM_60,
|
||||
(assert(cc == cuda::compute_capability{60}); return;),
|
||||
NV_IS_EXACTLY_SM_61,
|
||||
(assert(cc == cuda::compute_capability{61}); return;),
|
||||
NV_IS_EXACTLY_SM_62,
|
||||
(assert(cc == cuda::compute_capability{62}); return;),
|
||||
NV_IS_EXACTLY_SM_70,
|
||||
(assert(cc == cuda::compute_capability{70}); return;),
|
||||
NV_IS_EXACTLY_SM_75,
|
||||
(assert(cc == cuda::compute_capability{75}); return;),
|
||||
NV_IS_EXACTLY_SM_80,
|
||||
(assert(cc == cuda::compute_capability{80}); return;),
|
||||
NV_IS_EXACTLY_SM_86,
|
||||
(assert(cc == cuda::compute_capability{86}); return;),
|
||||
NV_IS_EXACTLY_SM_87,
|
||||
(assert(cc == cuda::compute_capability{87}); return;),
|
||||
NV_IS_EXACTLY_SM_88,
|
||||
(assert(cc == cuda::compute_capability{88}); return;),
|
||||
NV_IS_EXACTLY_SM_89,
|
||||
(assert(cc == cuda::compute_capability{89}); return;))
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_IS_EXACTLY_SM_90,
|
||||
(assert(cc == cuda::compute_capability{90}); return;),
|
||||
NV_IS_EXACTLY_SM_100,
|
||||
(assert(cc == cuda::compute_capability{100}); return;),
|
||||
NV_IS_EXACTLY_SM_103,
|
||||
(assert(cc == cuda::compute_capability{103}); return;),
|
||||
NV_IS_EXACTLY_SM_110,
|
||||
(assert(cc == cuda::compute_capability{110}); return;),
|
||||
NV_IS_EXACTLY_SM_120,
|
||||
(assert(cc == cuda::compute_capability{120}); return;),
|
||||
NV_IS_EXACTLY_SM_121,
|
||||
(assert(cc == cuda::compute_capability{121}); return;),
|
||||
NV_ANY_TARGET,
|
||||
(assert(false);) // fail for unknown compute capability
|
||||
)
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// 1. Test default constructor.
|
||||
{
|
||||
static_assert(cuda::std::is_nothrow_default_constructible_v<cuda::compute_capability>);
|
||||
cuda::compute_capability cc;
|
||||
assert(cc.get() == 0);
|
||||
}
|
||||
|
||||
// 2. Test constructor from compute capability in format 10 * major + minor.
|
||||
{
|
||||
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, int>);
|
||||
static_assert(!cuda::std::is_convertible_v<int, cuda::compute_capability>);
|
||||
|
||||
cuda::compute_capability cc{148};
|
||||
assert(cc.get() == 148);
|
||||
}
|
||||
|
||||
// 3. Test constructor from major and minor.
|
||||
{
|
||||
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, int, int>);
|
||||
cuda::compute_capability cc{8, 9};
|
||||
assert(cc.get() == 89);
|
||||
}
|
||||
|
||||
// 4. Test constructor from cuda::arch_id.
|
||||
{
|
||||
static_assert(cuda::std::is_nothrow_constructible_v<cuda::compute_capability, cuda::arch_id>);
|
||||
static_assert(!cuda::std::is_convertible_v<cuda::arch_id, cuda::compute_capability>);
|
||||
|
||||
cuda::compute_capability cc1{cuda::arch_id::sm_100};
|
||||
assert(cc1.get() == 100);
|
||||
cuda::compute_capability cc2{cuda::arch_id::sm_100a};
|
||||
assert(cc2.get() == 100);
|
||||
}
|
||||
|
||||
// 5. Test copy constructor.
|
||||
{
|
||||
static_assert(cuda::std::is_trivially_copy_constructible_v<cuda::compute_capability>);
|
||||
|
||||
const cuda::compute_capability cc1{cuda::arch_id::sm_100};
|
||||
cuda::compute_capability cc2{cc1};
|
||||
assert(cc1.get() == 100);
|
||||
assert(cc2.get() == 100);
|
||||
}
|
||||
|
||||
// 6. Test assignment operator.
|
||||
{
|
||||
static_assert(cuda::std::is_nothrow_copy_assignable_v<cuda::compute_capability>);
|
||||
|
||||
const cuda::compute_capability cc1{cuda::arch_id::sm_100};
|
||||
cuda::compute_capability cc2;
|
||||
assert(cc1.get() == 100);
|
||||
assert(cc2.get() == 0);
|
||||
|
||||
cc2 = cc1;
|
||||
assert(cc1.get() == 100);
|
||||
assert(cc2.get() == 100);
|
||||
}
|
||||
|
||||
// 7. Test get().
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.get())>);
|
||||
static_assert(noexcept(cuda::compute_capability{}.get()));
|
||||
|
||||
const cuda::compute_capability cc{cuda::arch_id::sm_100};
|
||||
assert(cc.get() == 100);
|
||||
}
|
||||
|
||||
// 8. Test major_cap().
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.major_cap())>);
|
||||
static_assert(noexcept(cuda::compute_capability{}.major_cap()));
|
||||
|
||||
const cuda::compute_capability cc{cuda::arch_id::sm_100};
|
||||
assert(cc.major_cap() == 10);
|
||||
|
||||
// Test deprecated major().
|
||||
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.major())>);
|
||||
static_assert(noexcept(cuda::compute_capability{}.major()));
|
||||
|
||||
assert(cc.major() == cc.major_cap());
|
||||
}
|
||||
|
||||
// 9. Test minor_cap().
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.minor_cap())>);
|
||||
static_assert(noexcept(cuda::compute_capability{}.minor_cap()));
|
||||
|
||||
const cuda::compute_capability cc{cuda::arch_id::sm_89};
|
||||
assert(cc.minor_cap() == 9);
|
||||
|
||||
// Test deprecated minor().
|
||||
static_assert(cuda::std::is_same_v<int, decltype(cuda::compute_capability{}.minor())>);
|
||||
static_assert(noexcept(cuda::compute_capability{}.minor()));
|
||||
|
||||
assert(cc.minor() == cc.minor_cap());
|
||||
}
|
||||
|
||||
// 10. operator int()
|
||||
{
|
||||
static_assert(noexcept(static_cast<int>(cuda::compute_capability{})));
|
||||
static_assert(!cuda::std::is_convertible_v<cuda::compute_capability, int>);
|
||||
|
||||
const cuda::compute_capability cc{cuda::arch_id::sm_89};
|
||||
assert(static_cast<int>(cc) == 89);
|
||||
}
|
||||
|
||||
// 11. comparison operators
|
||||
{
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator==(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator!=(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator<(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator<=(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator>(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<bool, decltype(operator>=(cuda::compute_capability{}, cuda::compute_capability{}))>);
|
||||
|
||||
static_assert(noexcept(operator==(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
static_assert(noexcept(operator!=(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
static_assert(noexcept(operator<(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
static_assert(noexcept(operator<=(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
static_assert(noexcept(operator>(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
static_assert(noexcept(operator>=(cuda::compute_capability{}, cuda::compute_capability{})));
|
||||
|
||||
const cuda::compute_capability cc1{127};
|
||||
const cuda::compute_capability cc2{43};
|
||||
|
||||
assert(cc1 == cc1);
|
||||
assert(cc2 == cc2);
|
||||
|
||||
assert(cc1 != cc2);
|
||||
assert(cc2 != cc1);
|
||||
|
||||
assert(!(cc1 < cc1));
|
||||
assert(!(cc1 < cc2));
|
||||
assert(!(cc2 < cc2));
|
||||
assert(cc2 < cc1);
|
||||
|
||||
assert(cc1 <= cc1);
|
||||
assert(!(cc1 <= cc2));
|
||||
assert(cc2 <= cc2);
|
||||
assert(cc2 <= cc1);
|
||||
|
||||
assert(!(cc1 > cc1));
|
||||
assert(cc1 > cc2);
|
||||
assert(!(cc2 > cc2));
|
||||
assert(!(cc2 > cc1));
|
||||
|
||||
assert(cc1 >= cc1);
|
||||
assert(cc1 > cc2);
|
||||
assert(cc2 >= cc2);
|
||||
assert(!(cc2 > cc1));
|
||||
}
|
||||
|
||||
// 12. Test that cuda::compute_capability is a structural type.
|
||||
#if _CCCL_STD_VER >= 2020
|
||||
{
|
||||
[[maybe_unused]] constexpr auto val =
|
||||
cuda::std::integral_constant<cuda::compute_capability, cuda::compute_capability{100}>{};
|
||||
}
|
||||
#endif // _CCCL_STD_VER >= 2020
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
NV_IF_TARGET(NV_IS_DEVICE, (test_current();))
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
|
||||
#if __cpp_lib_format >= 201907L
|
||||
# include <format>
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
|
||||
#include "literal.h"
|
||||
|
||||
#if __cpp_lib_format >= 201907L
|
||||
template <class C>
|
||||
void test()
|
||||
{
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{0}) == TEST_STRLIT(C, "0"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{60}) == TEST_STRLIT(C, "60"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{61}) == TEST_STRLIT(C, "61"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{62}) == TEST_STRLIT(C, "62"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{70}) == TEST_STRLIT(C, "70"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{75}) == TEST_STRLIT(C, "75"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{80}) == TEST_STRLIT(C, "80"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{86}) == TEST_STRLIT(C, "86"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{87}) == TEST_STRLIT(C, "87"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{88}) == TEST_STRLIT(C, "88"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{89}) == TEST_STRLIT(C, "89"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{90}) == TEST_STRLIT(C, "90"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{100}) == TEST_STRLIT(C, "100"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{103}) == TEST_STRLIT(C, "103"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{110}) == TEST_STRLIT(C, "110"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{120}) == TEST_STRLIT(C, "120"));
|
||||
assert(std::format(TEST_STRLIT(C, "{}"), cuda::compute_capability{121}) == TEST_STRLIT(C, "121"));
|
||||
}
|
||||
|
||||
void test()
|
||||
{
|
||||
test<char>();
|
||||
test<wchar_t>();
|
||||
}
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
#if __cpp_lib_format >= 201907L
|
||||
NV_IF_TARGET(NV_IS_HOST, (test();))
|
||||
#endif // __cpp_lib_format >= 201907L
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,394 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/__driver/driver_api.h>
|
||||
#include <cuda/__runtime/ensure_current_context.h>
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/__type_traits/is_same.h>
|
||||
#include <cuda/std/cstddef>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
namespace
|
||||
{
|
||||
template <const auto& Attr, ::cudaDeviceAttr ExpectedAttr, class ExpectedResult>
|
||||
[[maybe_unused]] auto test_device_attribute()
|
||||
{
|
||||
cuda::device_ref dev0(0);
|
||||
STATIC_REQUIRE(Attr == ExpectedAttr);
|
||||
STATIC_REQUIRE(::cuda::std::is_same_v<cuda::device_attribute_result_t<Attr>, ExpectedResult>);
|
||||
|
||||
auto result = dev0.attribute(Attr);
|
||||
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(result), ExpectedResult>);
|
||||
CCCLRT_REQUIRE(result == dev0.attribute<ExpectedAttr>());
|
||||
CCCLRT_REQUIRE(result == Attr(dev0));
|
||||
return result;
|
||||
}
|
||||
} // namespace
|
||||
|
||||
C2H_CCCLRT_TEST("init", "[device]")
|
||||
{
|
||||
cuda::device_ref dev{0};
|
||||
dev.init();
|
||||
CCCLRT_REQUIRE(cuda::__driver::__isPrimaryCtxActive(cuda::__driver::__deviceGet(0)));
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Smoke", "[device]")
|
||||
{
|
||||
namespace attributes = cuda::device_attributes;
|
||||
using cuda::device_ref;
|
||||
|
||||
SECTION("Compare")
|
||||
{
|
||||
CCCLRT_REQUIRE(device_ref{0} == device_ref{0});
|
||||
CCCLRT_REQUIRE(device_ref{0} == 0);
|
||||
CCCLRT_REQUIRE(0 == device_ref{0});
|
||||
if (cuda::devices.size() > 1)
|
||||
{
|
||||
CCCLRT_REQUIRE(device_ref{1} != device_ref{0});
|
||||
CCCLRT_REQUIRE(device_ref{1} != 0);
|
||||
CCCLRT_REQUIRE(0 != device_ref{1});
|
||||
}
|
||||
}
|
||||
|
||||
SECTION("Attributes")
|
||||
{
|
||||
::test_device_attribute<attributes::max_threads_per_block, ::cudaDevAttrMaxThreadsPerBlock, int>();
|
||||
::test_device_attribute<attributes::max_block_dim_x, ::cudaDevAttrMaxBlockDimX, int>();
|
||||
::test_device_attribute<attributes::max_block_dim_y, ::cudaDevAttrMaxBlockDimY, int>();
|
||||
::test_device_attribute<attributes::max_block_dim_z, ::cudaDevAttrMaxBlockDimZ, int>();
|
||||
::test_device_attribute<attributes::max_grid_dim_x, ::cudaDevAttrMaxGridDimX, int>();
|
||||
::test_device_attribute<attributes::max_grid_dim_y, ::cudaDevAttrMaxGridDimY, int>();
|
||||
::test_device_attribute<attributes::max_grid_dim_z, ::cudaDevAttrMaxGridDimZ, int>();
|
||||
::test_device_attribute<attributes::max_shared_memory_per_block,
|
||||
::cudaDevAttrMaxSharedMemoryPerBlock,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::total_constant_memory, ::cudaDevAttrTotalConstantMemory, cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::warp_size, ::cudaDevAttrWarpSize, int>();
|
||||
::test_device_attribute<attributes::max_pitch, ::cudaDevAttrMaxPitch, cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_texture_1d_width, ::cudaDevAttrMaxTexture1DWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_1d_linear_width, ::cudaDevAttrMaxTexture1DLinearWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_1d_mipmapped_width, ::cudaDevAttrMaxTexture1DMipmappedWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_width, ::cudaDevAttrMaxTexture2DWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_height, ::cudaDevAttrMaxTexture2DHeight, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_linear_width, ::cudaDevAttrMaxTexture2DLinearWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_linear_height, ::cudaDevAttrMaxTexture2DLinearHeight, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_linear_pitch,
|
||||
::cudaDevAttrMaxTexture2DLinearPitch,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_texture_2d_mipmapped_width, ::cudaDevAttrMaxTexture2DMipmappedWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_mipmapped_height, ::cudaDevAttrMaxTexture2DMipmappedHeight, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_width, ::cudaDevAttrMaxTexture3DWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_height, ::cudaDevAttrMaxTexture3DHeight, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_depth, ::cudaDevAttrMaxTexture3DDepth, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_width_alt, ::cudaDevAttrMaxTexture3DWidthAlt, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_height_alt, ::cudaDevAttrMaxTexture3DHeightAlt, int>();
|
||||
::test_device_attribute<attributes::max_texture_3d_depth_alt, ::cudaDevAttrMaxTexture3DDepthAlt, int>();
|
||||
::test_device_attribute<attributes::max_texture_cubemap_width, ::cudaDevAttrMaxTextureCubemapWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_1d_layered_width, ::cudaDevAttrMaxTexture1DLayeredWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_1d_layered_layers, ::cudaDevAttrMaxTexture1DLayeredLayers, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_layered_width, ::cudaDevAttrMaxTexture2DLayeredWidth, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_layered_height, ::cudaDevAttrMaxTexture2DLayeredHeight, int>();
|
||||
::test_device_attribute<attributes::max_texture_2d_layered_layers, ::cudaDevAttrMaxTexture2DLayeredLayers, int>();
|
||||
::test_device_attribute<attributes::max_texture_cubemap_layered_width,
|
||||
::cudaDevAttrMaxTextureCubemapLayeredWidth,
|
||||
int>();
|
||||
::test_device_attribute<attributes::max_texture_cubemap_layered_layers,
|
||||
::cudaDevAttrMaxTextureCubemapLayeredLayers,
|
||||
int>();
|
||||
::test_device_attribute<attributes::max_surface_1d_width, ::cudaDevAttrMaxSurface1DWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_2d_width, ::cudaDevAttrMaxSurface2DWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_2d_height, ::cudaDevAttrMaxSurface2DHeight, int>();
|
||||
::test_device_attribute<attributes::max_surface_3d_width, ::cudaDevAttrMaxSurface3DWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_3d_height, ::cudaDevAttrMaxSurface3DHeight, int>();
|
||||
::test_device_attribute<attributes::max_surface_3d_depth, ::cudaDevAttrMaxSurface3DDepth, int>();
|
||||
::test_device_attribute<attributes::max_surface_1d_layered_width, ::cudaDevAttrMaxSurface1DLayeredWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_1d_layered_layers, ::cudaDevAttrMaxSurface1DLayeredLayers, int>();
|
||||
::test_device_attribute<attributes::max_surface_2d_layered_width, ::cudaDevAttrMaxSurface2DLayeredWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_2d_layered_height, ::cudaDevAttrMaxSurface2DLayeredHeight, int>();
|
||||
::test_device_attribute<attributes::max_surface_2d_layered_layers, ::cudaDevAttrMaxSurface2DLayeredLayers, int>();
|
||||
::test_device_attribute<attributes::max_surface_cubemap_width, ::cudaDevAttrMaxSurfaceCubemapWidth, int>();
|
||||
::test_device_attribute<attributes::max_surface_cubemap_layered_width,
|
||||
::cudaDevAttrMaxSurfaceCubemapLayeredWidth,
|
||||
int>();
|
||||
::test_device_attribute<attributes::max_surface_cubemap_layered_layers,
|
||||
::cudaDevAttrMaxSurfaceCubemapLayeredLayers,
|
||||
int>();
|
||||
::test_device_attribute<attributes::max_registers_per_block, ::cudaDevAttrMaxRegistersPerBlock, int>();
|
||||
::test_device_attribute<attributes::clock_rate, ::cudaDevAttrClockRate, int>();
|
||||
::test_device_attribute<attributes::texture_alignment, ::cudaDevAttrTextureAlignment, cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::texture_pitch_alignment, ::cudaDevAttrTexturePitchAlignment, cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::gpu_overlap, ::cudaDevAttrGpuOverlap, bool>();
|
||||
::test_device_attribute<attributes::multiprocessor_count, ::cudaDevAttrMultiProcessorCount, int>();
|
||||
::test_device_attribute<attributes::kernel_exec_timeout, ::cudaDevAttrKernelExecTimeout, bool>();
|
||||
::test_device_attribute<attributes::integrated, ::cudaDevAttrIntegrated, bool>();
|
||||
::test_device_attribute<attributes::can_map_host_memory, ::cudaDevAttrCanMapHostMemory, bool>();
|
||||
::test_device_attribute<attributes::compute_mode, ::cudaDevAttrComputeMode, ::cudaComputeMode>();
|
||||
::test_device_attribute<attributes::concurrent_kernels, ::cudaDevAttrConcurrentKernels, bool>();
|
||||
::test_device_attribute<attributes::ecc_enabled, ::cudaDevAttrEccEnabled, bool>();
|
||||
::test_device_attribute<attributes::pci_bus_id, ::cudaDevAttrPciBusId, int>();
|
||||
::test_device_attribute<attributes::pci_device_id, ::cudaDevAttrPciDeviceId, int>();
|
||||
::test_device_attribute<attributes::tcc_driver, ::cudaDevAttrTccDriver, bool>();
|
||||
::test_device_attribute<attributes::l2_cache_size, ::cudaDevAttrL2CacheSize, cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_threads_per_multiprocessor, ::cudaDevAttrMaxThreadsPerMultiProcessor, int>();
|
||||
::test_device_attribute<attributes::unified_addressing, ::cudaDevAttrUnifiedAddressing, bool>();
|
||||
::test_device_attribute<attributes::compute_capability_major, ::cudaDevAttrComputeCapabilityMajor, int>();
|
||||
::test_device_attribute<attributes::compute_capability_minor, ::cudaDevAttrComputeCapabilityMinor, int>();
|
||||
::test_device_attribute<attributes::stream_priorities_supported, ::cudaDevAttrStreamPrioritiesSupported, bool>();
|
||||
::test_device_attribute<attributes::global_l1_cache_supported, ::cudaDevAttrGlobalL1CacheSupported, bool>();
|
||||
::test_device_attribute<attributes::local_l1_cache_supported, ::cudaDevAttrLocalL1CacheSupported, bool>();
|
||||
::test_device_attribute<attributes::max_shared_memory_per_multiprocessor,
|
||||
::cudaDevAttrMaxSharedMemoryPerMultiprocessor,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_registers_per_multiprocessor,
|
||||
::cudaDevAttrMaxRegistersPerMultiprocessor,
|
||||
int>();
|
||||
::test_device_attribute<attributes::is_multi_gpu_board, ::cudaDevAttrIsMultiGpuBoard, bool>();
|
||||
::test_device_attribute<attributes::multi_gpu_board_group_id, ::cudaDevAttrMultiGpuBoardGroupID, int>();
|
||||
::test_device_attribute<attributes::host_native_atomic_supported, ::cudaDevAttrHostNativeAtomicSupported, bool>();
|
||||
::test_device_attribute<attributes::single_to_double_precision_perf_ratio,
|
||||
::cudaDevAttrSingleToDoublePrecisionPerfRatio,
|
||||
int>();
|
||||
::test_device_attribute<attributes::pageable_memory_access, ::cudaDevAttrPageableMemoryAccess, bool>();
|
||||
::test_device_attribute<attributes::concurrent_managed_access, ::cudaDevAttrConcurrentManagedAccess, bool>();
|
||||
::test_device_attribute<attributes::compute_preemption_supported, ::cudaDevAttrComputePreemptionSupported, bool>();
|
||||
::test_device_attribute<attributes::can_use_host_pointer_for_registered_mem,
|
||||
::cudaDevAttrCanUseHostPointerForRegisteredMem,
|
||||
bool>();
|
||||
::test_device_attribute<attributes::cooperative_launch, ::cudaDevAttrCooperativeLaunch, bool>();
|
||||
::test_device_attribute<attributes::can_flush_remote_writes, ::cudaDevAttrCanFlushRemoteWrites, bool>();
|
||||
::test_device_attribute<attributes::host_register_supported, ::cudaDevAttrHostRegisterSupported, bool>();
|
||||
::test_device_attribute<attributes::pageable_memory_access_uses_host_page_tables,
|
||||
::cudaDevAttrPageableMemoryAccessUsesHostPageTables,
|
||||
bool>();
|
||||
::test_device_attribute<attributes::direct_managed_mem_access_from_host,
|
||||
::cudaDevAttrDirectManagedMemAccessFromHost,
|
||||
bool>();
|
||||
::test_device_attribute<attributes::max_shared_memory_per_block_optin,
|
||||
::cudaDevAttrMaxSharedMemoryPerBlockOptin,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_blocks_per_multiprocessor, ::cudaDevAttrMaxBlocksPerMultiprocessor, int>();
|
||||
::test_device_attribute<attributes::max_persisting_l2_cache_size,
|
||||
::cudaDevAttrMaxPersistingL2CacheSize,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::max_access_policy_window_size,
|
||||
::cudaDevAttrMaxAccessPolicyWindowSize,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::reserved_shared_memory_per_block,
|
||||
::cudaDevAttrReservedSharedMemoryPerBlock,
|
||||
cuda::std::size_t>();
|
||||
::test_device_attribute<attributes::sparse_cuda_array_supported, ::cudaDevAttrSparseCudaArraySupported, bool>();
|
||||
::test_device_attribute<attributes::host_register_read_only_supported,
|
||||
::cudaDevAttrHostRegisterReadOnlySupported,
|
||||
bool>();
|
||||
::test_device_attribute<attributes::memory_pools_supported, ::cudaDevAttrMemoryPoolsSupported, bool>();
|
||||
::test_device_attribute<attributes::gpu_direct_rdma_supported, ::cudaDevAttrGPUDirectRDMASupported, bool>();
|
||||
::test_device_attribute<attributes::gpu_direct_rdma_flush_writes_options,
|
||||
::cudaDevAttrGPUDirectRDMAFlushWritesOptions,
|
||||
::cudaFlushGPUDirectRDMAWritesOptions>();
|
||||
::test_device_attribute<attributes::gpu_direct_rdma_writes_ordering,
|
||||
::cudaDevAttrGPUDirectRDMAWritesOrdering,
|
||||
::cudaGPUDirectRDMAWritesOrdering>();
|
||||
::test_device_attribute<attributes::memory_pool_supported_handle_types,
|
||||
::cudaDevAttrMemoryPoolSupportedHandleTypes,
|
||||
::cudaMemAllocationHandleType>();
|
||||
::test_device_attribute<attributes::deferred_mapping_cuda_array_supported,
|
||||
::cudaDevAttrDeferredMappingCudaArraySupported,
|
||||
bool>();
|
||||
::test_device_attribute<attributes::ipc_event_support, ::cudaDevAttrIpcEventSupport, bool>();
|
||||
|
||||
#if _CCCL_CTK_AT_LEAST(12, 2)
|
||||
::test_device_attribute<attributes::numa_config, ::cudaDevAttrNumaConfig, ::cudaDeviceNumaConfig>();
|
||||
::test_device_attribute<attributes::numa_id, ::cudaDevAttrNumaId, int>();
|
||||
#endif // _CCCL_CTK_AT_LEAST(12, 2)
|
||||
|
||||
SECTION("compute_mode")
|
||||
{
|
||||
STATIC_REQUIRE(::cudaComputeModeDefault == attributes::compute_mode.default_mode);
|
||||
STATIC_REQUIRE(::cudaComputeModeProhibited == attributes::compute_mode.prohibited_mode);
|
||||
STATIC_REQUIRE(::cudaComputeModeExclusiveProcess == attributes::compute_mode.exclusive_process_mode);
|
||||
|
||||
auto mode = device_ref(0).attribute(attributes::compute_mode);
|
||||
CCCLRT_REQUIRE((mode == attributes::compute_mode.default_mode || //
|
||||
mode == attributes::compute_mode.prohibited_mode || //
|
||||
mode == attributes::compute_mode.exclusive_process_mode));
|
||||
}
|
||||
|
||||
SECTION("gpu_direct_rdma_flush_writes_options")
|
||||
{
|
||||
STATIC_REQUIRE(::cudaFlushGPUDirectRDMAWritesOptionHost == attributes::gpu_direct_rdma_flush_writes_options.host);
|
||||
STATIC_REQUIRE(
|
||||
::cudaFlushGPUDirectRDMAWritesOptionMemOps == attributes::gpu_direct_rdma_flush_writes_options.mem_ops);
|
||||
|
||||
[[maybe_unused]] auto options = device_ref(0).attribute(attributes::gpu_direct_rdma_flush_writes_options);
|
||||
#if !_CCCL_COMPILER(MSVC)
|
||||
CCCLRT_REQUIRE((options == attributes::gpu_direct_rdma_flush_writes_options.host || //
|
||||
options == attributes::gpu_direct_rdma_flush_writes_options.mem_ops));
|
||||
#endif
|
||||
}
|
||||
|
||||
SECTION("gpu_direct_rdma_writes_ordering")
|
||||
{
|
||||
STATIC_REQUIRE(::cudaGPUDirectRDMAWritesOrderingNone == attributes::gpu_direct_rdma_writes_ordering.none);
|
||||
STATIC_REQUIRE(::cudaGPUDirectRDMAWritesOrderingOwner == attributes::gpu_direct_rdma_writes_ordering.owner);
|
||||
STATIC_REQUIRE(
|
||||
::cudaGPUDirectRDMAWritesOrderingAllDevices == attributes::gpu_direct_rdma_writes_ordering.all_devices);
|
||||
|
||||
auto ordering = device_ref(0).attribute(attributes::gpu_direct_rdma_writes_ordering);
|
||||
CCCLRT_REQUIRE((ordering == attributes::gpu_direct_rdma_writes_ordering.none || //
|
||||
ordering == attributes::gpu_direct_rdma_writes_ordering.owner || //
|
||||
ordering == attributes::gpu_direct_rdma_writes_ordering.all_devices));
|
||||
}
|
||||
|
||||
SECTION("memory_pool_supported_handle_types")
|
||||
{
|
||||
STATIC_REQUIRE(::cudaMemHandleTypeNone == attributes::memory_pool_supported_handle_types.none);
|
||||
STATIC_REQUIRE(
|
||||
::cudaMemHandleTypePosixFileDescriptor == attributes::memory_pool_supported_handle_types.posix_file_descriptor);
|
||||
STATIC_REQUIRE(::cudaMemHandleTypeWin32 == attributes::memory_pool_supported_handle_types.win32);
|
||||
STATIC_REQUIRE(::cudaMemHandleTypeWin32Kmt == attributes::memory_pool_supported_handle_types.win32_kmt);
|
||||
#if _CCCL_CTK_AT_LEAST(12, 4)
|
||||
STATIC_REQUIRE(::cudaMemHandleTypeFabric == 0x8);
|
||||
STATIC_REQUIRE(::cudaMemHandleTypeFabric == attributes::memory_pool_supported_handle_types.fabric);
|
||||
#else // ^^^ _CCCL_CTK_AT_LEAST(12, 4) ^^^ / vvv _CCCL_CTK_BELOW(12, 4) vvv
|
||||
STATIC_REQUIRE(0x8 == attributes::memory_pool_supported_handle_types.fabric);
|
||||
#endif // ^^^ _CCCL_CTK_BELOW(12, 4) ^^^
|
||||
|
||||
constexpr int all_handle_types =
|
||||
attributes::memory_pool_supported_handle_types.none
|
||||
| attributes::memory_pool_supported_handle_types.posix_file_descriptor
|
||||
| attributes::memory_pool_supported_handle_types.win32
|
||||
| attributes::memory_pool_supported_handle_types.win32_kmt
|
||||
| attributes::memory_pool_supported_handle_types.fabric;
|
||||
auto handle_types = device_ref(0).attribute(attributes::memory_pool_supported_handle_types);
|
||||
CCCLRT_REQUIRE(static_cast<int>(handle_types) <= static_cast<int>(all_handle_types));
|
||||
}
|
||||
|
||||
#if _CCCL_CTK_AT_LEAST(12, 2)
|
||||
SECTION("numa_config")
|
||||
{
|
||||
STATIC_REQUIRE(::cudaDeviceNumaConfigNone == attributes::numa_config.none);
|
||||
STATIC_REQUIRE(::cudaDeviceNumaConfigNumaNode == attributes::numa_config.numa_node);
|
||||
|
||||
auto config = device_ref(0).attribute(attributes::numa_config);
|
||||
CCCLRT_REQUIRE((config == attributes::numa_config.none || //
|
||||
config == attributes::numa_config.numa_node));
|
||||
}
|
||||
#endif // _CCCL_CTK_AT_LEAST(12, 2)
|
||||
|
||||
SECTION("Compute capability")
|
||||
{
|
||||
cuda::compute_capability compute_cap = device_ref(0).attribute(attributes::compute_capability);
|
||||
int compute_cap_major = device_ref(0).attribute(attributes::compute_capability_major);
|
||||
int compute_cap_minor = device_ref(0).attribute(attributes::compute_capability_minor);
|
||||
CCCLRT_REQUIRE(compute_cap.get() == 10 * compute_cap_major + compute_cap_minor);
|
||||
}
|
||||
|
||||
SECTION("Total global memory")
|
||||
{
|
||||
auto total_mem = device_ref(0).attribute(attributes::total_global_memory);
|
||||
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(total_mem), cuda::std::size_t>);
|
||||
CCCLRT_REQUIRE(total_mem > 0);
|
||||
}
|
||||
}
|
||||
SECTION("Name")
|
||||
{
|
||||
const auto name = device_ref(0).name();
|
||||
CCCLRT_REQUIRE(name.length() != 0);
|
||||
CCCLRT_REQUIRE(name[0] != 0);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("global devices vector", "[device]")
|
||||
{
|
||||
CCCLRT_REQUIRE(cuda::devices.size() > 0);
|
||||
CCCLRT_REQUIRE(cuda::devices.begin() != cuda::devices.end());
|
||||
CCCLRT_REQUIRE(cuda::devices.begin() == cuda::devices.begin());
|
||||
CCCLRT_REQUIRE(cuda::devices.end() == cuda::devices.end());
|
||||
CCCLRT_REQUIRE(cuda::devices.size() == static_cast<size_t>(cuda::devices.end() - cuda::devices.begin()));
|
||||
|
||||
CCCLRT_REQUIRE(0 == cuda::devices[0].get());
|
||||
CCCLRT_REQUIRE(cuda::device_ref{0} == cuda::devices[0]);
|
||||
|
||||
CCCLRT_REQUIRE(0 == (*cuda::devices.begin()).get());
|
||||
CCCLRT_REQUIRE(cuda::device_ref{0} == *cuda::devices.begin());
|
||||
|
||||
CCCLRT_REQUIRE(0 == cuda::devices.begin()->get());
|
||||
CCCLRT_REQUIRE(0 == cuda::devices.begin()[0].get());
|
||||
|
||||
if (cuda::devices.size() > 1)
|
||||
{
|
||||
CCCLRT_REQUIRE(1 == cuda::devices[1].get());
|
||||
CCCLRT_REQUIRE(cuda::device_ref{0} != cuda::devices[1].get());
|
||||
|
||||
CCCLRT_REQUIRE(1 == (*std::next(cuda::devices.begin())).get());
|
||||
CCCLRT_REQUIRE(1 == std::next(cuda::devices.begin())->get());
|
||||
CCCLRT_REQUIRE(1 == cuda::devices.begin()[1].get());
|
||||
|
||||
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>((*std::prev(cuda::devices.end())).get()));
|
||||
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>(std::prev(cuda::devices.end())->get()));
|
||||
CCCLRT_REQUIRE(cuda::devices.size() - 1 == static_cast<std::size_t>(cuda::devices.end()[-1].get()));
|
||||
|
||||
auto peers = cuda::devices[0].peers();
|
||||
for (auto peer : peers)
|
||||
{
|
||||
CCCLRT_REQUIRE(cuda::devices[0].has_peer_access_to(peer));
|
||||
CCCLRT_REQUIRE(peer.has_peer_access_to(cuda::devices[0]));
|
||||
}
|
||||
}
|
||||
|
||||
#if _CCCL_HAS_EXCEPTIONS()
|
||||
try
|
||||
{
|
||||
[[maybe_unused]] const cuda::device_ref& dev = cuda::devices[cuda::devices.size()];
|
||||
CCCLRT_REQUIRE(false); // should not get here
|
||||
}
|
||||
catch (const std::out_of_range&)
|
||||
{
|
||||
CCCLRT_REQUIRE(true); // expected
|
||||
}
|
||||
#endif // _CCCL_HAS_EXCEPTIONS()
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Device attributes use the explicit device when current device differs", "[device][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
const auto expected_bus_id = cuda::__driver::__deviceGetAttribute(
|
||||
static_cast<::CUdevice_attribute>(cudaDevAttrPciBusId), cuda::__driver::__deviceGet(explicit_device.get()));
|
||||
const auto expected_device_id = cuda::__driver::__deviceGetAttribute(
|
||||
static_cast<::CUdevice_attribute>(cudaDevAttrPciDeviceId), cuda::__driver::__deviceGet(explicit_device.get()));
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
CCCLRT_REQUIRE(explicit_device.attribute(cuda::device_attributes::pci_bus_id) == expected_bus_id);
|
||||
CCCLRT_REQUIRE(explicit_device.attribute(cuda::device_attributes::pci_device_id) == expected_device_id);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("memory location", "[device]")
|
||||
{
|
||||
cuda::memory_location loc = cuda::devices[0];
|
||||
CCCLRT_REQUIRE(loc.type == ::cudaMemLocationTypeDevice);
|
||||
CCCLRT_REQUIRE(loc.id == 0);
|
||||
|
||||
if (cuda::devices.size() > 1)
|
||||
{
|
||||
loc = cuda::device_ref{1};
|
||||
CCCLRT_REQUIRE(loc.type == ::cudaMemLocationTypeDevice);
|
||||
CCCLRT_REQUIRE(loc.id == 1);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/type_traits>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// 1. Test signature.
|
||||
static_assert(cuda::std::is_same_v<bool, decltype(cuda::__is_specific_arch(cuda::arch_id{}))>);
|
||||
static_assert(noexcept(cuda::__is_specific_arch(cuda::arch_id{})));
|
||||
|
||||
// 2. Test values.
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_60));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_61));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_62));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_70));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_75));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_80));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_86));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_87));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_88));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_89));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_90));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_100));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_103));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_110));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_120));
|
||||
assert(!cuda::__is_specific_arch(cuda::arch_id::sm_121));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_90a));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_100a));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_103a));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_110a));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_120a));
|
||||
assert(cuda::__is_specific_arch(cuda::arch_id::sm_121a));
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,63 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// UNSUPPORTED: enable-tile
|
||||
// error: return in loop statement is not supported
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/array>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstddef>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <cuda::std::size_t N>
|
||||
TEST_FUNC constexpr auto make_all_ccs_ref(cuda::std::array<int, N> vs, int scale)
|
||||
{
|
||||
cuda::std::array<cuda::compute_capability, N> ret;
|
||||
for (cuda::std::size_t i = 0; i < N; ++i)
|
||||
{
|
||||
ret[i] = cuda::compute_capability{vs[i] / scale};
|
||||
}
|
||||
return ret;
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// 1. Test signature.
|
||||
static_assert(cuda::std::__is_cuda_std_array_v<decltype(cuda::__target_compute_capabilities())>);
|
||||
static_assert(noexcept(cuda::__target_compute_capabilities()));
|
||||
|
||||
// 2. Test that all values are present.
|
||||
const auto all_ccs = cuda::__target_compute_capabilities();
|
||||
|
||||
#if defined(__CUDA_ARCH_LIST__)
|
||||
const auto all_ccs_ref = make_all_ccs_ref(cuda::std::array{__CUDA_ARCH_LIST__}, 10);
|
||||
#elif defined(NV_TARGET_SM_INTEGER_LIST)
|
||||
const auto all_ccs_ref = make_all_ccs_ref(cuda::std::array{NV_TARGET_SM_INTEGER_LIST}, 1);
|
||||
#else
|
||||
const auto all_ccs_ref = ::cuda::__all_compute_capabilities();
|
||||
#endif
|
||||
|
||||
assert(all_ccs.size() == all_ccs_ref.size());
|
||||
for (cuda::std::size_t i = 0; i < all_ccs.size(); ++i)
|
||||
{
|
||||
assert(all_ccs[i] == all_ccs_ref[i]);
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,226 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/devices>
|
||||
#include <cuda/launch>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <testing.cuh>
|
||||
#include <utility.cuh>
|
||||
|
||||
namespace
|
||||
{
|
||||
namespace test
|
||||
{
|
||||
cuda::event_ref fn_takes_event_ref(cuda::event_ref ref)
|
||||
{
|
||||
return ref;
|
||||
}
|
||||
|
||||
template <class Event>
|
||||
void test_event_uses_explicit_device_when_current_device_differs()
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
cuda::stream explicit_device_stream{explicit_device};
|
||||
CCCLRT_REQUIRE(explicit_device_stream.device() == explicit_device);
|
||||
|
||||
Event ev = [&]() {
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
return Event(explicit_device);
|
||||
}();
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
ev.record(explicit_device_stream);
|
||||
ev.sync();
|
||||
CCCLRT_REQUIRE(ev.is_done());
|
||||
}
|
||||
|
||||
explicit_device_stream.sync();
|
||||
}
|
||||
} // namespace test
|
||||
} // namespace
|
||||
|
||||
static_assert(!::cuda::std::is_default_constructible_v<cuda::event_ref>);
|
||||
static_assert(!::cuda::std::is_default_constructible_v<cuda::event>);
|
||||
static_assert(!::cuda::std::is_default_constructible_v<cuda::timed_event>);
|
||||
|
||||
C2H_CCCLRT_TEST("can construct an event_ref from a cudaEvent_t", "[event]")
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
::cudaEvent_t ev;
|
||||
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
|
||||
cuda::event_ref ref(ev);
|
||||
CCCLRT_REQUIRE(ref.get() == ev);
|
||||
CCCLRT_REQUIRE(!!ref);
|
||||
// test implicit conversion from cudaEvent_t:
|
||||
cuda::event_ref ref2 = ::test::fn_takes_event_ref(ev);
|
||||
CCCLRT_REQUIRE(ref2.get() == ev);
|
||||
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
|
||||
// test an empty event_ref:
|
||||
cuda::event_ref ref3(::cudaEvent_t{});
|
||||
CCCLRT_REQUIRE(ref3.get() == ::cudaEvent_t{});
|
||||
CCCLRT_REQUIRE(!ref3);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can copy construct an event_ref and compare for equality", "[event]")
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
::cudaEvent_t ev;
|
||||
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
|
||||
const cuda::event_ref ref(ev);
|
||||
const cuda::event_ref ref2 = ref;
|
||||
CCCLRT_REQUIRE(ref2 == ref);
|
||||
CCCLRT_REQUIRE(!(ref != ref2));
|
||||
CCCLRT_REQUIRE((ref ? true : false)); // test contextual convertibility to bool
|
||||
CCCLRT_REQUIRE(!!ref);
|
||||
CCCLRT_REQUIRE(::cudaEvent_t{} != ref);
|
||||
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
|
||||
// copy from empty event_ref:
|
||||
const cuda::event_ref ref3(::cudaEvent_t{});
|
||||
const cuda::event_ref ref4 = ref3;
|
||||
CCCLRT_REQUIRE(ref4 == ref3);
|
||||
CCCLRT_REQUIRE(!(ref3 != ref4));
|
||||
CCCLRT_REQUIRE(!ref4);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can use event_ref to record and wait on an event", "[event]")
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
::cudaEvent_t ev;
|
||||
CCCLRT_REQUIRE(::cudaEventCreate(&ev) == ::cudaSuccess);
|
||||
const cuda::event_ref ref(ev);
|
||||
|
||||
test::pinned<int> i(0);
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
|
||||
ref.record(stream);
|
||||
ref.sync();
|
||||
CCCLRT_REQUIRE(ref.is_done());
|
||||
CCCLRT_REQUIRE(*i == 42);
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(::cudaEventDestroy(ev) == ::cudaSuccess);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can construct an event with a stream_ref", "[event]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
cuda::event ev(static_cast<cuda::stream_ref>(stream));
|
||||
CCCLRT_REQUIRE(ev.get() != ::cudaEvent_t{});
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can construct an event with a device_ref", "[event]")
|
||||
{
|
||||
cuda::device_ref device{0};
|
||||
cuda::event ev(device);
|
||||
CCCLRT_REQUIRE(ev.get() != ::cudaEvent_t{});
|
||||
cuda::stream stream{device};
|
||||
ev.record(stream);
|
||||
ev.sync();
|
||||
CCCLRT_REQUIRE(ev.is_done());
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("event device_ref constructors use the explicit device", "[event][multi_gpu]")
|
||||
{
|
||||
::test::test_event_uses_explicit_device_when_current_device_differs<cuda::event>();
|
||||
::test::test_event_uses_explicit_device_when_current_device_differs<cuda::timed_event>();
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can wait on an event from another device", "[event][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref event_device{0};
|
||||
cuda::device_ref waiter_device{1};
|
||||
|
||||
cuda::stream event_stream{event_device};
|
||||
cuda::stream waiter_stream{waiter_device};
|
||||
|
||||
cuda::atomic<int> gate = 0;
|
||||
bool waiter_ran = false;
|
||||
|
||||
cuda::host_launch(event_stream, [&gate]() {
|
||||
while (gate != 1)
|
||||
;
|
||||
});
|
||||
cuda::event ev(event_stream);
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(event_device);
|
||||
waiter_stream.wait(ev);
|
||||
cuda::host_launch(waiter_stream, [&waiter_ran]() {
|
||||
waiter_ran = true;
|
||||
});
|
||||
}
|
||||
|
||||
CCCLRT_REQUIRE(!waiter_stream.is_done());
|
||||
CCCLRT_REQUIRE(!waiter_ran);
|
||||
|
||||
gate = 1;
|
||||
waiter_stream.sync();
|
||||
event_stream.sync();
|
||||
|
||||
CCCLRT_REQUIRE(waiter_ran);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can wait on an event", "[event]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
::test::pinned<int> i(0);
|
||||
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
|
||||
cuda::event ev(stream);
|
||||
ev.sync();
|
||||
CCCLRT_REQUIRE(ev.is_done());
|
||||
CCCLRT_REQUIRE(*i == 42);
|
||||
stream.sync();
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can take the difference of two timed_event objects", "[event]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
::test::pinned<int> i(0);
|
||||
cuda::timed_event start(stream);
|
||||
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
|
||||
cuda::timed_event end(stream);
|
||||
end.sync();
|
||||
CCCLRT_REQUIRE(end.is_done());
|
||||
CCCLRT_REQUIRE(*i == 42);
|
||||
auto elapsed = end - start;
|
||||
CCCLRT_REQUIRE(elapsed.count() >= 0);
|
||||
STATIC_REQUIRE(::cuda::std::is_same_v<decltype(elapsed), ::cuda::std::chrono::nanoseconds>);
|
||||
stream.sync();
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("can observe the event in not ready state", "[event]")
|
||||
{
|
||||
::test::pinned<int> i(0);
|
||||
::cuda::atomic_ref atomic_i(*i);
|
||||
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
|
||||
::test::launch_kernel_single_thread(stream, ::test::spin_until_80{}, i.get());
|
||||
cuda::event ev(stream);
|
||||
CCCLRT_REQUIRE(!ev.is_done());
|
||||
atomic_i.store(80);
|
||||
ev.sync();
|
||||
CCCLRT_REQUIRE(ev.is_done());
|
||||
}
|
||||
@@ -0,0 +1,75 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <iostream>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
#include <host_device.cuh>
|
||||
|
||||
struct custom_level : public cuda::hierarchy_level_base<custom_level>
|
||||
{
|
||||
using __product_type = unsigned int;
|
||||
using __allowed_above = cuda::__allowed_levels<cuda::grid_level>;
|
||||
using __allowed_below = cuda::__allowed_levels<cuda::block_level>;
|
||||
};
|
||||
|
||||
template <typename Level, typename Dims>
|
||||
struct custom_level_dims : public cuda::hierarchy_level_desc<Level, Dims>
|
||||
{
|
||||
int dummy;
|
||||
constexpr custom_level_dims()
|
||||
: cuda::hierarchy_level_desc<Level, Dims>() {};
|
||||
};
|
||||
|
||||
struct custom_level_test
|
||||
{
|
||||
template <typename DynDims>
|
||||
TEST_FUNC void operator()(const DynDims& dims) const
|
||||
{
|
||||
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 84 * 1024);
|
||||
CCCLRT_REQUIRE(custom_level{}.count(cuda::grid, dims) == 42);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(42 * 512, 2, 2));
|
||||
CCCLRT_REQUIRE(custom_level{}.dims(cuda::grid, dims) == dim3(42, 1, 1));
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
}
|
||||
|
||||
void run()
|
||||
{
|
||||
// Check extending hierarchy_level_desc with custom info
|
||||
custom_level_dims<cuda::block_level, cuda::std::extents<int, 64, 1, 1>> custom_block;
|
||||
custom_block.dummy = 2;
|
||||
auto custom_dims = cuda::make_hierarchy(cuda::grid_dims<256>(), cuda::cluster_dims<8>(), custom_block);
|
||||
auto custom_block_back = custom_dims.level(cuda::block);
|
||||
CCCLRT_REQUIRE(custom_block_back.dummy == 2);
|
||||
|
||||
auto custom_dims_fragment = custom_dims.fragment(cuda::gpu_thread, cuda::block);
|
||||
auto custom_block_back2 = custom_dims_fragment.level(cuda::block);
|
||||
CCCLRT_REQUIRE(custom_block_back2.dummy == 2);
|
||||
|
||||
// Check creating a custom level type works
|
||||
auto custom_level_dims = cuda::std::extents<cuda::dimensions_index_type, 2, 2, 2>();
|
||||
auto custom_hierarchy = cuda::make_hierarchy(
|
||||
cuda::grid_dims(42),
|
||||
cuda::hierarchy_level_desc<custom_level, decltype(custom_level_dims)>(custom_level_dims),
|
||||
cuda::block_dims<256>());
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(custom_level(), custom_hierarchy) == dim3(512, 2, 2));
|
||||
static_assert(cuda::gpu_thread.count(custom_level(), custom_hierarchy) == 2048);
|
||||
|
||||
test_host_dev(custom_hierarchy, *this);
|
||||
}
|
||||
};
|
||||
|
||||
C2H_TEST("Custom level", "[hierarchy]")
|
||||
{
|
||||
custom_level_test().run();
|
||||
}
|
||||
@@ -0,0 +1,567 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/__type_traits/vector_type.h>
|
||||
#include <cuda/hierarchy>
|
||||
#include <cuda/launch>
|
||||
#include <cuda/std/cstddef>
|
||||
|
||||
#include <iostream>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
#include <host_device.cuh>
|
||||
|
||||
#include "testing.cuh"
|
||||
|
||||
namespace cg = cooperative_groups;
|
||||
|
||||
using size_t3 = cuda::vector_type_t<cuda::std::size_t, 3>;
|
||||
|
||||
struct basic_test_single_dim
|
||||
{
|
||||
static constexpr int block_size = 256;
|
||||
static constexpr int grid_size = 512;
|
||||
|
||||
template <typename DynDims>
|
||||
TEST_FUNC void operator()(const DynDims& dims) const
|
||||
{
|
||||
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims).x == grid_size * block_size);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == grid_size * block_size);
|
||||
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::block, dims).x == block_size);
|
||||
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims).x == grid_size);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, dims) == block_size);
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == grid_size);
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
}
|
||||
|
||||
void run()
|
||||
{
|
||||
auto dims = cuda::make_hierarchy(cuda::block_dims<block_size>(), cuda::grid_dims<grid_size>());
|
||||
static_assert(cuda::gpu_thread.dims(cuda::grid, dims).x == grid_size * block_size);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, dims) == static_cast<unsigned long>(grid_size) * block_size);
|
||||
static_assert(
|
||||
cuda::gpu_thread.static_dims(cuda::grid, dims)[0] == static_cast<unsigned long>(grid_size) * block_size);
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(cuda::block, dims).x == block_size);
|
||||
static_assert(cuda::block.dims(cuda::grid, dims).x == grid_size);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, dims) == block_size);
|
||||
static_assert(cuda::block.count(cuda::grid, dims) == grid_size);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims)[0] == block_size);
|
||||
|
||||
auto dims_dyn = cuda::make_hierarchy(cuda::block_dims(block_size), cuda::grid_dims(grid_size));
|
||||
|
||||
test_host_dev(dims_dyn, *this);
|
||||
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_dyn)[0] == cuda::std::dynamic_extent);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims_dyn)[0] == cuda::std::dynamic_extent);
|
||||
|
||||
// Test that we can also drop the empty parens in the level constructors:
|
||||
auto config = cuda::make_hierarchy(cuda::block_dims<block_size>, cuda::grid_dims<grid_size>);
|
||||
CCCLRT_REQUIRE(dims == config);
|
||||
}
|
||||
};
|
||||
|
||||
struct basic_test_multi_dim
|
||||
{
|
||||
static constexpr int block_size = 256;
|
||||
|
||||
template <typename DynDims>
|
||||
TEST_FUNC void operator()(const DynDims& dims) const
|
||||
{
|
||||
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(32, 12, 4));
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(0) == 32);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(1) == 12);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(2) == 4);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 512 * 3);
|
||||
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::block, dims) == dim3(2, 3, 4));
|
||||
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(16, 4, 1));
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, dims) == 24);
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 64);
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
}
|
||||
|
||||
void run()
|
||||
{
|
||||
auto dims_multidim = cuda::make_hierarchy(cuda::block_dims<2, 3, 4>(), cuda::grid_dims<16, 4, 1>());
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(cuda::grid, dims_multidim) == dim3(32, 12, 4));
|
||||
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(0) == 32);
|
||||
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(1) == 12);
|
||||
static_assert(cuda::gpu_thread.extents(cuda::grid, dims_multidim).extent(2) == 4);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, dims_multidim) == 512 * 3);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims_multidim) == size_t3{32, 12, 4});
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(cuda::block, dims_multidim) == dim3(2, 3, 4));
|
||||
static_assert(cuda::block.dims(cuda::grid, dims_multidim) == dim3(16, 4, 1));
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, dims_multidim) == 24);
|
||||
static_assert(cuda::block.count(cuda::grid, dims_multidim) == 64);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_multidim) == size_t3{2, 3, 4});
|
||||
static_assert(cuda::block.static_dims(cuda::grid, dims_multidim) == size_t3{16, 4, 1});
|
||||
|
||||
auto dims_multidim_dyn = cuda::make_hierarchy(cuda::block_dims(dim3(2, 3, 4)), cuda::grid_dims(dim3(16, 4, 1)));
|
||||
|
||||
test_host_dev(dims_multidim_dyn, *this);
|
||||
}
|
||||
};
|
||||
|
||||
struct basic_test_mixed
|
||||
{
|
||||
static constexpr int block_size = 256;
|
||||
|
||||
template <typename DynDims>
|
||||
TEST_FUNC void operator()(const DynDims& dims) const
|
||||
{
|
||||
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(2048, 4, 2));
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(0) == 2048);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(1) == 4);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, dims).extent(2) == 2);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 16 * 1024);
|
||||
|
||||
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(8, 4, 2));
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 64);
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
}
|
||||
|
||||
void run()
|
||||
{
|
||||
auto dims_mixed = cuda::make_hierarchy(cuda::block_dims<block_size>(), cuda::grid_dims(dim3(8, 4, 2)));
|
||||
|
||||
test_host_dev(dims_mixed, *this);
|
||||
static_assert(cuda::gpu_thread.dims(cuda::block, dims_mixed).x == block_size);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, dims_mixed) == block_size);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_mixed)[0] == block_size);
|
||||
|
||||
// TODO include mixed static and dynamic info on a single level
|
||||
// Currently bugged in std::extents
|
||||
}
|
||||
};
|
||||
|
||||
C2H_TEST("Basic", "[hierarchy]")
|
||||
{
|
||||
basic_test_single_dim().run();
|
||||
basic_test_multi_dim().run();
|
||||
basic_test_mixed().run();
|
||||
}
|
||||
|
||||
struct basic_test_cluster
|
||||
{
|
||||
template <typename DynDims>
|
||||
TEST_FUNC void operator()(const DynDims& dims) const
|
||||
{
|
||||
// todo: allow this after fixing CCCLRT_REQUIRE with clang-cuda
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, dims) == dim3(512, 6, 9));
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, dims) == 27 * 1024);
|
||||
|
||||
CCCLRT_REQUIRE(cuda::block.dims(cuda::grid, dims) == dim3(2, 6, 9));
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, dims) == 108);
|
||||
CCCLRT_REQUIRE(cuda::cluster.dims(cuda::grid, dims) == dim3(1, 3, 9));
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::cluster, dims) == dim3(512, 2, 1));
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
}
|
||||
|
||||
void run()
|
||||
{
|
||||
SECTION("Static cluster dims")
|
||||
{
|
||||
auto dims = cuda::make_hierarchy(cuda::block_dims<256>(), cuda::cluster_dims<8>(), cuda::grid_dims<512>());
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(cuda::grid, dims).x == 1024 * 1024);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, dims) == 1024 * 1024);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::grid, dims)[0] == 1024 * 1024);
|
||||
|
||||
static_assert(cuda::gpu_thread.dims(cuda::block, dims).x == 256);
|
||||
static_assert(cuda::block.dims(cuda::grid, dims).x == 4 * 1024);
|
||||
static_assert(cuda::gpu_thread.count(cuda::cluster, dims) == 2 * 1024);
|
||||
static_assert(cuda::cluster.count(cuda::grid, dims) == 512);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims)[0] == 256);
|
||||
static_assert(cuda::block.static_dims(cuda::grid, dims)[0] == 4 * 1024);
|
||||
}
|
||||
SECTION("Mixed cluster dims")
|
||||
{
|
||||
auto dims_mixed = cuda::make_hierarchy(
|
||||
cuda::block_dims<256>(), cuda::cluster_dims(dim3(2, 2, 1)), cuda::grid_dims(dim3(1, 3, 9)));
|
||||
test_host_dev(dims_mixed, *this, arch_filter<std::less<int>, 90>);
|
||||
static_assert(cuda::gpu_thread.dims(cuda::block, dims_mixed).x == 256);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, dims_mixed) == 256);
|
||||
static_assert(cuda::gpu_thread.static_dims(cuda::block, dims_mixed)[0] == 256);
|
||||
static_assert(cuda::block.static_dims(cuda::cluster, dims_mixed)[0] == cuda::std::dynamic_extent);
|
||||
static_assert(cuda::block.static_dims(cuda::grid, dims_mixed)[0] == cuda::std::dynamic_extent);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
C2H_TEST("Cluster dims", "[hierarchy]")
|
||||
{
|
||||
basic_test_cluster().run();
|
||||
}
|
||||
|
||||
C2H_TEST("Different constructions", "[hierarchy]")
|
||||
{
|
||||
/*
|
||||
const auto block_size = 512;
|
||||
const auto cluster_cnt = 8;
|
||||
const auto grid_size = 256;
|
||||
|
||||
[[maybe_unused]] const auto config =
|
||||
cuda::block_dims<block_size>() & cuda::cluster_dims<cluster_cnt>() &
|
||||
cuda::grid_dims(grid_size);
|
||||
[[maybe_unused]] const auto config2 =
|
||||
cuda::grid_dims(grid_size) & cuda::cluster_dims<cluster_cnt>() &
|
||||
cuda::block_dims<block_size>();
|
||||
|
||||
[[maybe_unused]] const auto config3 =
|
||||
cuda::cluster_dims<cluster_cnt>() & cuda::grid_dims(grid_size) &
|
||||
cuda::block_dims<block_size>();
|
||||
[[maybe_unused]] const auto config4 =
|
||||
cuda::cluster_dims<cluster_cnt>() & cuda::block_dims<block_size>() &
|
||||
cuda::grid_dims(grid_size);
|
||||
|
||||
[[maybe_unused]] const auto config5 =
|
||||
cuda::make_config(cuda::block_dims<block_size>(),
|
||||
cuda::cluster_dims<cluster_cnt>(), cuda::grid_dims(grid_size));
|
||||
[[maybe_unused]] const auto config6 =
|
||||
cuda::make_config(cuda::grid_dims(grid_size),
|
||||
cuda::cluster_dims<cluster_cnt>(), cuda::block_dims<block_size>());
|
||||
|
||||
static_assert(std::is_same_v<decltype(config), decltype(config2)>);
|
||||
static_assert(std::is_same_v<decltype(config), decltype(config3)>);
|
||||
static_assert(std::is_same_v<decltype(config), decltype(config4)>);
|
||||
static_assert(std::is_same_v<decltype(config), decltype(config5)>);
|
||||
static_assert(std::is_same_v<decltype(config), decltype(config6)>);
|
||||
|
||||
[[maybe_unused]] const auto conf_weird_order =
|
||||
cuda::grid_dims(grid_size) & (cuda::cluster_dims<cluster_cnt>() &
|
||||
cuda::block_dims<block_size>());
|
||||
static_assert(std::is_same_v<decltype(config), decltype(conf_weird_order)>);
|
||||
|
||||
static_assert(config.hierarchy().count(cuda::gpu_thread, cuda::block) == block_size);
|
||||
static_assert(config.hierarchy().count(cuda::gpu_thread, cuda::cluster) == cluster_cnt *
|
||||
block_size); static_assert(config.hierarchy().count(cuda::block, cuda::cluster) ==
|
||||
cluster_cnt); CCCLRT_REQUIRE(config.hierarchy().count() == grid_size * cluster_cnt *
|
||||
block_size);
|
||||
|
||||
static_assert(config.hierarchy().has_level(cuda::block));
|
||||
static_assert(config.hierarchy().has_level(cuda::cluster));
|
||||
static_assert(config.hierarchy().has_level(cuda::grid));
|
||||
static_assert(!config.hierarchy().has_level(cuda::thread));
|
||||
*/
|
||||
}
|
||||
|
||||
C2H_TEST("Replace level", "[hierarchy]")
|
||||
{
|
||||
// GCC 7 and 8 complains here that the hierarchy was not declared constexpr
|
||||
#if !_CCCL_COMPILER(GCC, <, 9)
|
||||
const auto dimensions = cuda::make_hierarchy(cuda::block_dims<512>(), cuda::cluster_dims<8>(), cuda::grid_dims(256));
|
||||
const auto fragment = dimensions.fragment(cuda::block, cuda::grid);
|
||||
static_assert(!fragment.has_level(cuda::block));
|
||||
static_assert(!cuda::__has_bottom_unit_or_level_v<cuda::thread_level, decltype(fragment)>);
|
||||
static_assert(fragment.has_level(cuda::cluster));
|
||||
static_assert(fragment.has_level(cuda::grid));
|
||||
static_assert(cuda::__has_bottom_unit_or_level_v<cuda::block_level, decltype(fragment)>);
|
||||
|
||||
const auto replaced = cuda::hierarchy_add_level(fragment, cuda::block_dims(256));
|
||||
static_assert(replaced.has_level(cuda::block));
|
||||
static_assert(cuda::__has_bottom_unit_or_level_v<cuda::thread_level, decltype(replaced)>);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, replaced) == 256);
|
||||
#endif // !_CCCL_COMPILER(GCC, <, 9)
|
||||
}
|
||||
|
||||
template <typename Hierarchy>
|
||||
__global__ void kernel(Hierarchy hierarchy)
|
||||
{
|
||||
auto grid = cg::this_grid();
|
||||
auto block = cg::this_thread_block();
|
||||
|
||||
CCCLRT_REQUIRE_DEVICE(grid.thread_rank() == cuda::gpu_thread.rank(cuda::grid));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.block_rank() == cuda::block.rank(cuda::grid));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.block_index() == cuda::block.index(cuda::grid));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.num_threads() == cuda::gpu_thread.count(cuda::grid));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.num_blocks() == cuda::block.count(cuda::grid));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.dim_blocks() == cuda::block.dims(cuda::grid));
|
||||
|
||||
CCCLRT_REQUIRE_DEVICE(block.thread_rank() == cuda::gpu_thread.rank(cuda::block));
|
||||
CCCLRT_REQUIRE_DEVICE(block.thread_index() == cuda::gpu_thread.index(cuda::block));
|
||||
CCCLRT_REQUIRE_DEVICE(block.num_threads() == cuda::gpu_thread.count(cuda::block));
|
||||
CCCLRT_REQUIRE_DEVICE(block.dim_threads() == cuda::gpu_thread.dims(cuda::block));
|
||||
|
||||
CCCLRT_REQUIRE_DEVICE(block.thread_index() == cuda::gpu_thread.index(cuda::block, hierarchy));
|
||||
|
||||
const auto grid_index = cuda::gpu_thread.index_as<unsigned long long>(cuda::grid, hierarchy);
|
||||
CCCLRT_REQUIRE_DEVICE(
|
||||
grid_index.x
|
||||
== static_cast<unsigned long long>(grid.block_index().x) * block.dim_threads().x + block.thread_index().x);
|
||||
CCCLRT_REQUIRE_DEVICE(
|
||||
grid_index.y
|
||||
== static_cast<unsigned long long>(grid.block_index().y) * block.dim_threads().y + block.thread_index().y);
|
||||
CCCLRT_REQUIRE_DEVICE(
|
||||
grid_index.z
|
||||
== static_cast<unsigned long long>(grid.block_index().z) * block.dim_threads().z + block.thread_index().z);
|
||||
|
||||
CCCLRT_REQUIRE_DEVICE(grid.block_rank() == cuda::block.rank(cuda::grid, hierarchy));
|
||||
CCCLRT_REQUIRE_DEVICE(block.thread_rank() == cuda::gpu_thread.rank(cuda::block, hierarchy));
|
||||
CCCLRT_REQUIRE_DEVICE(grid.thread_rank() == cuda::gpu_thread.rank(cuda::grid, hierarchy));
|
||||
}
|
||||
|
||||
C2H_TEST("Dims queries indexing and ambient hierarchy", "[hierarchy]")
|
||||
{
|
||||
const auto hierarchies = cuda::std::make_tuple(
|
||||
cuda::make_hierarchy(cuda::block_dims(dim3(64, 4, 2)), cuda::grid_dims(dim3(12, 6, 3))),
|
||||
cuda::make_hierarchy(cuda::block_dims(dim3(2, 4, 64)), cuda::grid_dims(dim3(3, 6, 12))),
|
||||
cuda::make_hierarchy(cuda::block_dims<256>(), cuda::grid_dims<4>()),
|
||||
cuda::make_hierarchy(cuda::block_dims<16, 2, 4>(), cuda::grid_dims<2, 3, 4>()),
|
||||
cuda::make_hierarchy(cuda::block_dims(dim3(8, 4, 2)), cuda::grid_dims<4, 5, 6>()),
|
||||
#if defined(NDEBUG)
|
||||
cuda::make_hierarchy(cuda::block_dims<32>(), cuda::grid_dims<(1 << 30) - 2>()),
|
||||
#endif
|
||||
cuda::make_hierarchy(cuda::block_dims<8, 2, 4>(), cuda::grid_dims(dim3(5, 4, 3))));
|
||||
|
||||
apply_each(
|
||||
[](const auto& hierarchy) {
|
||||
auto [grid, block] = cuda::get_launch_dimensions(hierarchy);
|
||||
|
||||
kernel<<<grid, block>>>(hierarchy);
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
},
|
||||
hierarchies);
|
||||
}
|
||||
|
||||
template <typename Hierarchy>
|
||||
__global__ void rank_kernel_optimized(Hierarchy hierarchy, unsigned int* out)
|
||||
{
|
||||
auto thread_id = cuda::gpu_thread.rank(cuda::block, hierarchy);
|
||||
out[thread_id] = thread_id;
|
||||
}
|
||||
|
||||
template <typename Hierarchy>
|
||||
__global__ void rank_kernel(Hierarchy hierarchy, unsigned int* out)
|
||||
{
|
||||
auto thread_id = cuda::gpu_thread.rank(cuda::block);
|
||||
out[thread_id] = thread_id;
|
||||
}
|
||||
|
||||
template <typename Hierarchy>
|
||||
__global__ void rank_kernel_cg(Hierarchy hierarchy, unsigned int* out)
|
||||
{
|
||||
auto thread_id = cg::thread_block::thread_rank();
|
||||
out[thread_id] = thread_id;
|
||||
}
|
||||
|
||||
// Testcase mostly for generated code comparison
|
||||
C2H_TEST("On device rank calculation", "[hierarchy]")
|
||||
{
|
||||
unsigned int* ptr;
|
||||
CUDART(cudaMalloc((void**) &ptr, 2 * 1024 * sizeof(unsigned int)));
|
||||
|
||||
const auto hierarchy_static = cuda::make_hierarchy(cuda::block_dims<256>(), cuda::grid_dims(dim3(2, 2, 2)));
|
||||
rank_kernel<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
rank_kernel_cg<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
rank_kernel_optimized<<<dim3(2, 2, 2), 256>>>(hierarchy_static, ptr);
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
CUDART(cudaFree(ptr));
|
||||
}
|
||||
|
||||
template <typename Hierarchy>
|
||||
__global__ void examples_kernel(Hierarchy hierarchy)
|
||||
{
|
||||
{
|
||||
auto thread_index_in_block = cuda::gpu_thread.index(cuda::block, hierarchy);
|
||||
CCCLRT_REQUIRE_DEVICE(thread_index_in_block == threadIdx);
|
||||
auto block_index_in_grid = cuda::block.index(cuda::grid, hierarchy);
|
||||
CCCLRT_REQUIRE_DEVICE(block_index_in_grid == blockIdx);
|
||||
}
|
||||
{
|
||||
int thread_rank_in_block = cuda::gpu_thread.rank(cuda::block, hierarchy);
|
||||
int block_rank_in_grid = cuda::block.rank(cuda::grid, hierarchy);
|
||||
}
|
||||
{
|
||||
// Can be called with the instances of level types
|
||||
int num_threads_in_block = static_cast<int>(cuda::gpu_thread.count(cuda::block));
|
||||
int num_blocks_in_grid = static_cast<int>(cuda::block.count(cuda::grid));
|
||||
|
||||
// Or using the level types as template arguments
|
||||
int num_threads_in_grid = static_cast<int>(cuda::gpu_thread.count(cuda::grid));
|
||||
}
|
||||
{
|
||||
// Can be called with the instances of level types
|
||||
int thread_rank_in_block = static_cast<int>(cuda::gpu_thread.rank(cuda::block));
|
||||
int block_rank_in_grid = static_cast<int>(cuda::block.rank(cuda::grid));
|
||||
|
||||
// Or using the level types as template arguments
|
||||
int thread_rank_in_grid = static_cast<int>(cuda::gpu_thread.rank(cuda::grid));
|
||||
}
|
||||
{
|
||||
// Can be called with the instances of level types
|
||||
CCCLRT_REQUIRE_DEVICE(cuda::gpu_thread.dims(cuda::block) == blockDim);
|
||||
CCCLRT_REQUIRE_DEVICE(cuda::block.dims(cuda::grid) == gridDim);
|
||||
|
||||
// Or using the level types as template arguments
|
||||
auto grid_dims_in_threads = cuda::gpu_thread.dims(cuda::grid);
|
||||
}
|
||||
{
|
||||
// Can be called with the instances of level types
|
||||
CCCLRT_REQUIRE_DEVICE(cuda::gpu_thread.index(cuda::block) == threadIdx);
|
||||
CCCLRT_REQUIRE_DEVICE(cuda::block.index(cuda::grid) == blockIdx);
|
||||
|
||||
// Or using the level types as template arguments
|
||||
auto thread_index_in_grid = cuda::gpu_thread.index(cuda::grid);
|
||||
}
|
||||
}
|
||||
|
||||
// Test examples from the inline rst documentation
|
||||
C2H_TEST("Examples", "[hierarchy]")
|
||||
{
|
||||
// GCC 7 and 8 complains here that the hierarchy was not declared constexpr
|
||||
#if !_CCCL_COMPILER(GCC, <, 9)
|
||||
{
|
||||
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
auto fragment = hierarchy.fragment(cuda::block, cuda::grid);
|
||||
auto new_hierarchy = cuda::hierarchy_add_level(fragment, cuda::block_dims<128>());
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, new_hierarchy) == 128);
|
||||
}
|
||||
{
|
||||
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
static_assert(cuda::gpu_thread.count(cuda::cluster, hierarchy) == 4 * 8 * 8 * 8);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, hierarchy) == 256 * 4 * 8 * 8 * 8);
|
||||
CCCLRT_REQUIRE(cuda::cluster.count(cuda::grid, hierarchy) == 256);
|
||||
}
|
||||
{
|
||||
[[maybe_unused]] auto hierarchy =
|
||||
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
static_assert(cuda::gpu_thread.count(cuda::cluster, hierarchy) == 4 * 8 * 8 * 8);
|
||||
}
|
||||
{
|
||||
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(0) == 4 * 8);
|
||||
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(1) == 8);
|
||||
static_assert(cuda::gpu_thread.extents(cuda::cluster, hierarchy).extent(2) == 8);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.extents(cuda::grid, hierarchy).extent(0) == 256 * 4 * 8);
|
||||
CCCLRT_REQUIRE(cuda::cluster.extents(cuda::grid, hierarchy).extent(0) == 256);
|
||||
}
|
||||
#endif // !_CCCL_COMPILER(GCC, <, 9)
|
||||
{
|
||||
[[maybe_unused]] auto hierarchy =
|
||||
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
static_assert(decltype(hierarchy.level(cuda::cluster).extents())::static_extent(0) == 4);
|
||||
}
|
||||
{
|
||||
auto partial1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims(256), cuda::cluster_dims<4>());
|
||||
[[maybe_unused]] auto hierarchy1 = cuda::hierarchy_add_level(partial1, cuda::block_dims<8, 8, 8>());
|
||||
auto partial2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<8, 8, 8>(), cuda::cluster_dims<4>());
|
||||
[[maybe_unused]] auto hierarchy2 = cuda::hierarchy_add_level(partial2, cuda::grid_dims(256));
|
||||
static_assert(cuda::std::is_same_v<decltype(hierarchy1), decltype(hierarchy2)>);
|
||||
}
|
||||
{
|
||||
[[maybe_unused]] auto hierarchy1 =
|
||||
cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
[[maybe_unused]] auto hierarchy2 =
|
||||
cuda::make_hierarchy(cuda::block_dims<8, 8, 8>(), cuda::cluster_dims<4>(), cuda::grid_dims(256));
|
||||
static_assert(cuda::std::is_same_v<decltype(hierarchy1), decltype(hierarchy2)>);
|
||||
}
|
||||
{
|
||||
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(256), cuda::cluster_dims<4>(), cuda::block_dims<8, 8, 8>());
|
||||
auto [grid_dimensions, cluster_dimensions, block_dimensions] = cuda::get_launch_dimensions(hierarchy);
|
||||
CCCLRT_REQUIRE(grid_dimensions.x == 256 * 4);
|
||||
CCCLRT_REQUIRE(cluster_dimensions.x == 4);
|
||||
CCCLRT_REQUIRE(block_dimensions.x == 8);
|
||||
CCCLRT_REQUIRE(block_dimensions.y == 8);
|
||||
CCCLRT_REQUIRE(block_dimensions.z == 8);
|
||||
}
|
||||
{
|
||||
auto hierarchy = cuda::make_hierarchy(cuda::grid_dims(16), cuda::block_dims<8, 8, 8>());
|
||||
auto [grid_dimensions, block_dimensions] = cuda::get_launch_dimensions(hierarchy);
|
||||
examples_kernel<<<grid_dimensions, block_dimensions>>>(hierarchy);
|
||||
CUDART(cudaGetLastError());
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
}
|
||||
}
|
||||
|
||||
C2H_TEST("Trivially constructable", "[hierarchy]")
|
||||
{
|
||||
// static_assert(std::is_trivial_v<decltype(cuda::block_dims(256))>);
|
||||
// static_assert(std::is_trivial_v<decltype(cuda::block_dims<256>())>);
|
||||
|
||||
// Hierarchy is not trivially copyable (yet), because tuple is not
|
||||
// static_assert(std::is_trivially_copyable_v<decltype(cuda::block_dims<256>()
|
||||
// & cuda::grid_dims<256>())>);
|
||||
// static_assert(std::is_trivially_copyable_v<decltype(cuda::std::make_tuple(cuda::block_dims<256>(),
|
||||
// cuda::grid_dims<256>()))>);
|
||||
}
|
||||
|
||||
C2H_TEST("cuda::distribute", "[hierarchy]")
|
||||
{
|
||||
unsigned numElements = 50000;
|
||||
constexpr int threadsPerBlock = 256;
|
||||
auto config = cuda::distribute<threadsPerBlock>(static_cast<int>(numElements));
|
||||
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, config) == 256);
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, config) == (numElements + threadsPerBlock - 1) / threadsPerBlock);
|
||||
}
|
||||
|
||||
C2H_TEST("hierarchy merge", "[hierarchy]")
|
||||
{
|
||||
SECTION("Non overlapping")
|
||||
{
|
||||
auto h1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<2>());
|
||||
auto h2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<3>());
|
||||
auto combined = h1.combine(h2);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, combined) == 6);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, combined) == 3);
|
||||
static_assert(cuda::block.count(cuda::grid, combined) == 2);
|
||||
auto combined_the_other_way = h2.combine(h1);
|
||||
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_the_other_way)>);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, combined_the_other_way) == 6);
|
||||
|
||||
auto dynamic_values = cuda::make_hierarchy(cuda::cluster_dims(4), cuda::block_dims(5));
|
||||
auto combined_dynamic = dynamic_values.combine(h1);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined_dynamic) == 40);
|
||||
}
|
||||
SECTION("Overlapping")
|
||||
{
|
||||
auto h1 = cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<2>(), cuda::cluster_dims<3>());
|
||||
auto h2 = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<4>(), cuda::cluster_dims<5>());
|
||||
auto combined = h1.combine(h2);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, combined) == 24);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, combined) == 4);
|
||||
static_assert(cuda::block.count(cuda::grid, combined) == 6);
|
||||
|
||||
auto combined_the_other_way = h2.combine(h1);
|
||||
static_assert(!cuda::std::is_same_v<decltype(combined), decltype(combined_the_other_way)>);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, combined_the_other_way) == 40);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, combined_the_other_way) == 4);
|
||||
static_assert(cuda::block.count(cuda::grid, combined_the_other_way) == 10);
|
||||
|
||||
auto ultimate_combination = combined.combine(combined_the_other_way);
|
||||
static_assert(cuda::std::is_same_v<decltype(combined), decltype(ultimate_combination)>);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, ultimate_combination) == 24);
|
||||
|
||||
auto block_level_replacement = cuda::make_hierarchy<cuda::thread_level>(cuda::block_dims<6>());
|
||||
auto with_block_replaced = block_level_replacement.combine(combined);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, with_block_replaced) == 36);
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, with_block_replaced) == 6);
|
||||
|
||||
auto grid_cluster_level_replacement =
|
||||
cuda::make_hierarchy<cuda::block_level>(cuda::grid_dims<7>(), cuda::cluster_dims<8>());
|
||||
auto with_grid_cluster_replaced = grid_cluster_level_replacement.combine(combined);
|
||||
static_assert(cuda::gpu_thread.count(cuda::grid, with_grid_cluster_replaced) == 7 * 8 * 4);
|
||||
static_assert(cuda::block.count(cuda::cluster, with_grid_cluster_replaced) == 8);
|
||||
static_assert(cuda::cluster.count(cuda::grid, with_grid_cluster_replaced) == 7);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,301 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDA Experimental in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda.h>
|
||||
|
||||
void test_launch_kernel_replacement(CUlaunchConfig& config, CUfunction kernel, void* args[]);
|
||||
|
||||
// This is a replacement for the launch kernel function that is used to test
|
||||
// the configuration of the launch kernel. It checks if the configuration
|
||||
// matches the expected configuration and calls the original launch kernel
|
||||
// function if it does. If the configuration does not match, it will fail the
|
||||
// test.
|
||||
#define _CCCLRT_LAUNCH_CONFIG_TEST
|
||||
#include <cuda/launch>
|
||||
|
||||
#include <host_device.cuh>
|
||||
|
||||
static CUlaunchConfig expectedConfig;
|
||||
static bool replacementCalled = false;
|
||||
|
||||
void test_launch_kernel_replacement(CUlaunchConfig& config, CUfunction kernel, void* args[])
|
||||
{
|
||||
replacementCalled = true;
|
||||
bool has_cluster = false;
|
||||
|
||||
CCCLRT_CHECK(expectedConfig.gridDimX == config.gridDimX);
|
||||
CCCLRT_CHECK(expectedConfig.gridDimY == config.gridDimY);
|
||||
CCCLRT_CHECK(expectedConfig.gridDimZ == config.gridDimZ);
|
||||
CCCLRT_CHECK(expectedConfig.blockDimX == config.blockDimX);
|
||||
CCCLRT_CHECK(expectedConfig.blockDimY == config.blockDimY);
|
||||
CCCLRT_CHECK(expectedConfig.blockDimZ == config.blockDimZ);
|
||||
CCCLRT_CHECK(expectedConfig.sharedMemBytes == config.sharedMemBytes);
|
||||
CCCLRT_CHECK(expectedConfig.hStream == config.hStream);
|
||||
CCCLRT_CHECK(expectedConfig.numAttrs == config.numAttrs);
|
||||
|
||||
for (unsigned int i = 0; i < expectedConfig.numAttrs; ++i)
|
||||
{
|
||||
auto& expectedAttr = expectedConfig.attrs[i];
|
||||
unsigned int j;
|
||||
for (j = 0; j < expectedConfig.numAttrs; ++j)
|
||||
{
|
||||
auto& actualAttr = config.attrs[j];
|
||||
if (expectedAttr.id == actualAttr.id)
|
||||
{
|
||||
switch (expectedAttr.id)
|
||||
{
|
||||
case CU_LAUNCH_ATTRIBUTE_CLUSTER_DIMENSION:
|
||||
CCCLRT_CHECK(expectedAttr.value.clusterDim.x == actualAttr.value.clusterDim.x);
|
||||
CCCLRT_CHECK(expectedAttr.value.clusterDim.y == actualAttr.value.clusterDim.y);
|
||||
CCCLRT_CHECK(expectedAttr.value.clusterDim.z == actualAttr.value.clusterDim.z);
|
||||
has_cluster = true;
|
||||
break;
|
||||
case CU_LAUNCH_ATTRIBUTE_COOPERATIVE:
|
||||
CCCLRT_CHECK(expectedAttr.value.cooperative == actualAttr.value.cooperative);
|
||||
break;
|
||||
case CU_LAUNCH_ATTRIBUTE_PRIORITY:
|
||||
CCCLRT_CHECK(expectedAttr.value.priority == actualAttr.value.priority);
|
||||
break;
|
||||
default:
|
||||
CCCLRT_CHECK(false);
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
INFO("Searched attribute is " << expectedAttr.id);
|
||||
CCCLRT_CHECK(j != expectedConfig.numAttrs);
|
||||
}
|
||||
|
||||
if (!has_cluster || !skip_device_exec(arch_filter<std::less<int>, 90>))
|
||||
{
|
||||
return ::cuda::__driver::__launchKernel(config, kernel, args);
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void empty_kernel(int i) {}
|
||||
|
||||
template <bool HasCluster>
|
||||
auto make_test_dims(const dim3& grid_dims, const dim3& block_dims, const dim3& cluster_dims = dim3())
|
||||
{
|
||||
if constexpr (HasCluster)
|
||||
{
|
||||
return cuda::make_hierarchy(
|
||||
cuda::grid_dims(grid_dims), cuda::cluster_dims(cluster_dims), cuda::block_dims(block_dims));
|
||||
}
|
||||
else
|
||||
{
|
||||
return cuda::make_hierarchy(cuda::grid_dims(grid_dims), cuda::block_dims(block_dims));
|
||||
}
|
||||
}
|
||||
|
||||
auto add_cluster(const dim3& cluster_dims, CUlaunchAttribute& attr)
|
||||
{
|
||||
attr.id = CU_LAUNCH_ATTRIBUTE_CLUSTER_DIMENSION;
|
||||
attr.value.clusterDim = {cluster_dims.x, cluster_dims.y, cluster_dims.z};
|
||||
}
|
||||
|
||||
template <bool HasCluster, typename... Dims>
|
||||
auto configuration_test(
|
||||
::cuda::stream_ref stream, const dim3& grid_dims, const dim3& block_dims, const dim3& cluster_dims = dim3())
|
||||
{
|
||||
auto dims = make_test_dims<HasCluster>(grid_dims, block_dims, cluster_dims);
|
||||
expectedConfig = {};
|
||||
expectedConfig.hStream = stream.get();
|
||||
if constexpr (HasCluster)
|
||||
{
|
||||
expectedConfig.gridDimX = grid_dims.x * cluster_dims.x;
|
||||
expectedConfig.gridDimY = grid_dims.y * cluster_dims.y;
|
||||
expectedConfig.gridDimZ = grid_dims.z * cluster_dims.z;
|
||||
}
|
||||
else
|
||||
{
|
||||
expectedConfig.gridDimX = grid_dims.x;
|
||||
expectedConfig.gridDimY = grid_dims.y;
|
||||
expectedConfig.gridDimZ = grid_dims.z;
|
||||
}
|
||||
expectedConfig.blockDimX = block_dims.x;
|
||||
expectedConfig.blockDimY = block_dims.y;
|
||||
expectedConfig.blockDimZ = block_dims.z;
|
||||
|
||||
SECTION("Simple cooperative launch")
|
||||
{
|
||||
CUlaunchAttribute attrs[2];
|
||||
auto config = cuda::make_config(dims, cuda::cooperative_launch());
|
||||
expectedConfig.numAttrs = 1 + HasCluster;
|
||||
expectedConfig.attrs = &attrs[0];
|
||||
expectedConfig.attrs[0].id = CU_LAUNCH_ATTRIBUTE_COOPERATIVE;
|
||||
expectedConfig.attrs[0].value.cooperative = 1;
|
||||
if constexpr (HasCluster)
|
||||
{
|
||||
add_cluster(cluster_dims, expectedConfig.attrs[1]);
|
||||
}
|
||||
cuda::launch(stream, config, empty_kernel, 0);
|
||||
}
|
||||
|
||||
SECTION("Priority and dynamic smem")
|
||||
{
|
||||
CUlaunchAttribute attrs[2];
|
||||
constexpr int priority = 42;
|
||||
constexpr int num_ints = 128;
|
||||
auto config =
|
||||
cuda::make_config(dims, cuda::launch_priority(priority), cuda::dynamic_shared_memory<int[num_ints]>());
|
||||
expectedConfig.sharedMemBytes = num_ints * sizeof(int);
|
||||
expectedConfig.numAttrs = 1 + HasCluster;
|
||||
expectedConfig.attrs = &attrs[0];
|
||||
expectedConfig.attrs[0].id = CU_LAUNCH_ATTRIBUTE_PRIORITY;
|
||||
expectedConfig.attrs[0].value.priority = priority;
|
||||
if constexpr (HasCluster)
|
||||
{
|
||||
add_cluster(cluster_dims, expectedConfig.attrs[1]);
|
||||
}
|
||||
cuda::launch(stream, config, empty_kernel, 0);
|
||||
}
|
||||
|
||||
SECTION("Large dynamic smem")
|
||||
{
|
||||
// Exceed the default 48kB of shared to check if its properly handled
|
||||
// TODO move to launch option (available since CUDA 12.4)
|
||||
struct S
|
||||
{
|
||||
int arr[13 * 1024];
|
||||
};
|
||||
CUlaunchAttribute attrs[1];
|
||||
auto config = cuda::make_config(dims, cuda::dynamic_shared_memory<S>(cuda::non_portable));
|
||||
expectedConfig.sharedMemBytes = sizeof(S);
|
||||
expectedConfig.numAttrs = HasCluster;
|
||||
expectedConfig.attrs = &attrs[0];
|
||||
if constexpr (HasCluster)
|
||||
{
|
||||
add_cluster(cluster_dims, expectedConfig.attrs[0]);
|
||||
}
|
||||
cuda::launch(stream, config, empty_kernel, 0);
|
||||
}
|
||||
stream.sync();
|
||||
}
|
||||
|
||||
C2H_TEST("Launch configuration", "[launch]")
|
||||
{
|
||||
cudaStream_t stream;
|
||||
CUDART(cudaStreamCreate(&stream));
|
||||
SECTION("No cluster")
|
||||
{
|
||||
configuration_test<false>(stream, 8, 64);
|
||||
}
|
||||
SECTION("With cluster")
|
||||
{
|
||||
configuration_test<true>(stream, 8, 32, 2);
|
||||
}
|
||||
|
||||
CUDART(cudaStreamDestroy(stream));
|
||||
CCCLRT_CHECK(replacementCalled);
|
||||
}
|
||||
|
||||
C2H_TEST("Hierarchy construction in config", "[launch]")
|
||||
{
|
||||
auto config = cuda::make_config(cuda::grid_dims<2>(), cuda::cooperative_launch());
|
||||
static_assert(cuda::block.count(cuda::grid, config) == 2);
|
||||
|
||||
auto config_larger = cuda::make_config(cuda::grid_dims<2>(), cuda::block_dims(256), cuda::cooperative_launch());
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config_larger) == 512);
|
||||
|
||||
auto config_no_options = cuda::make_config(cuda::grid_dims(2), cuda::block_dims<128>());
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config_no_options) == 256);
|
||||
|
||||
[[maybe_unused]] auto config_no_dims = cuda::make_config(cuda::cooperative_launch());
|
||||
static_assert(
|
||||
cuda::std::is_same_v<::cuda::std::remove_cvref_t<decltype(config_no_dims.hierarchy())>, cuda::__empty_hierarchy>);
|
||||
}
|
||||
|
||||
C2H_TEST("Configuration combine", "[launch]")
|
||||
{
|
||||
auto grid = cuda::grid_dims<2>;
|
||||
auto cluster = cuda::cluster_dims<2, 2>;
|
||||
auto block = cuda::block_dims(256);
|
||||
SECTION("Combine with no overlap")
|
||||
{
|
||||
auto config_part1 = cuda::make_config(grid);
|
||||
auto config_part2 = cuda::make_config(block, cuda::launch_priority(2));
|
||||
auto combined = config_part1.combine(config_part2);
|
||||
[[maybe_unused]] auto combined_other_way = config_part2.combine(config_part1);
|
||||
[[maybe_unused]] auto combined_with_empty = combined.combine(cuda::make_config());
|
||||
[[maybe_unused]] auto empty_with_combined = cuda::make_config().combine(combined);
|
||||
static_assert(
|
||||
cuda::std::is_same_v<decltype(combined), decltype(cuda::make_config(grid, block, cuda::launch_priority(2)))>);
|
||||
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_other_way)>);
|
||||
static_assert(cuda::std::is_same_v<decltype(combined), decltype(combined_with_empty)>);
|
||||
static_assert(cuda::std::is_same_v<decltype(combined), decltype(empty_with_combined)>);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined) == 512);
|
||||
}
|
||||
SECTION("Combine with overlap")
|
||||
{
|
||||
auto config_part1 = make_config(grid, cluster, cuda::launch_priority(2));
|
||||
auto config_part2 = make_config(cuda::cluster_dims<256>(), block, cuda::launch_priority(42));
|
||||
auto combined = config_part1.combine(config_part2);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, combined) == 2048);
|
||||
CCCLRT_REQUIRE(cuda::std::get<0>(combined.options()).priority == 2);
|
||||
|
||||
auto replaced_one_option = cuda::make_config(cuda::launch_priority(3)).combine(combined);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, replaced_one_option) == 2048);
|
||||
CCCLRT_REQUIRE(cuda::std::get<0>(replaced_one_option.options()).priority == 3);
|
||||
|
||||
[[maybe_unused]] auto combined_with_extra_option = combined.combine(cuda::make_config(cuda::cooperative_launch()));
|
||||
static_assert(
|
||||
cuda::std::is_same_v<decltype(combined.hierarchy()), decltype(combined_with_extra_option.hierarchy())>);
|
||||
static_assert(
|
||||
cuda::std::tuple_size_v<::cuda::std::remove_cvref_t<decltype(combined_with_extra_option.options())>> == 2);
|
||||
}
|
||||
}
|
||||
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
template <typename Config>
|
||||
TEST_FUNC void test_queries_on_config(const Config& config)
|
||||
{
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.dims(cuda::grid, config) == dim3(1024));
|
||||
{
|
||||
auto dims = cuda::gpu_thread.dims_as<int>(cuda::grid, config);
|
||||
CCCLRT_REQUIRE(dims.x == 1024);
|
||||
CCCLRT_REQUIRE(dims.y == 1);
|
||||
CCCLRT_REQUIRE(dims.z == 1);
|
||||
}
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::block, config) == 256);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count_as<int>(cuda::block, config) == 256);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count(cuda::grid, config) == 1024);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.count_as<int>(cuda::grid, config) == 1024);
|
||||
CCCLRT_REQUIRE(cuda::block.extents(cuda::grid, config).extent(0) == 4);
|
||||
CCCLRT_REQUIRE(cuda::block.extents_as<int>(cuda::grid, config).extent(0) == 4);
|
||||
NV_IF_TARGET(
|
||||
NV_IS_DEVICE,
|
||||
(CCCLRT_REQUIRE(cuda::block.rank(cuda::grid, config) == blockIdx.x);
|
||||
CCCLRT_REQUIRE(cuda::block.rank_as<int>(cuda::grid, config) == blockIdx.x);
|
||||
CCCLRT_REQUIRE(cuda::gpu_thread.index(cuda::block, config) == threadIdx);
|
||||
{
|
||||
auto idx = cuda::gpu_thread.index_as<int>(cuda::block, config);
|
||||
CCCLRT_REQUIRE(idx.x == static_cast<int>(threadIdx.x));
|
||||
CCCLRT_REQUIRE(idx.y == static_cast<int>(threadIdx.y));
|
||||
CCCLRT_REQUIRE(idx.z == static_cast<int>(threadIdx.z));
|
||||
}));
|
||||
}
|
||||
|
||||
template <typename Config>
|
||||
__global__ void test_kernel(Config config)
|
||||
{
|
||||
test_queries_on_config(config);
|
||||
}
|
||||
|
||||
C2H_TEST("Queries on config", "[launch]")
|
||||
{
|
||||
auto config = cuda::make_config(cuda::grid_dims(4), cuda::block_dims<256>(), cuda::cooperative_launch());
|
||||
test_queries_on_config(config);
|
||||
test_kernel<<<4, 256>>>(config);
|
||||
CUDART(cudaGetLastError());
|
||||
CUDART(cudaDeviceSynchronize());
|
||||
}
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
@@ -0,0 +1,108 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDA Experimental in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/hierarchy>
|
||||
#include <cuda/launch>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/functional>
|
||||
#include <cuda/std/span>
|
||||
#include <cuda/std/type_traits>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
template <class T, class View>
|
||||
struct TestKernel
|
||||
{
|
||||
template <class Config>
|
||||
TEST_DEVICE_FUNC void operator()(const Config& config)
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<View, decltype(cuda::dynamic_shared_memory(config))>);
|
||||
static_assert(noexcept(cuda::dynamic_shared_memory(config)));
|
||||
|
||||
write_smem(cuda::dynamic_shared_memory(config));
|
||||
}
|
||||
|
||||
TEST_DEVICE_FUNC void write_smem(T& view)
|
||||
{
|
||||
view = T{};
|
||||
CCCLRT_REQUIRE_DEVICE(view == T{});
|
||||
}
|
||||
|
||||
template <cuda::std::size_t N>
|
||||
TEST_DEVICE_FUNC void write_smem(cuda::std::span<T, N> view)
|
||||
{
|
||||
for (cuda::std::size_t i = 0; i < view.size(); ++i)
|
||||
{
|
||||
view[i] = T{};
|
||||
CCCLRT_REQUIRE_DEVICE(view[i] == T{});
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class T, class View, class Opt>
|
||||
void test_opt_and_launch(cuda::stream_ref stream, Opt opt)
|
||||
{
|
||||
static_assert(cuda::std::is_same_v<T, typename Opt::value_type>);
|
||||
static_assert(cuda::std::is_same_v<View, typename Opt::view_type>);
|
||||
|
||||
const auto config = cuda::make_config(cuda::block_dims<1, 1>(), cuda::grid_dims<1, 1>(), opt);
|
||||
cuda::launch(stream, config, TestKernel<T, View>{});
|
||||
stream.sync();
|
||||
}
|
||||
|
||||
template <class T>
|
||||
void test_ref(cuda::stream_ref stream)
|
||||
{
|
||||
static_assert(noexcept(cuda::dynamic_shared_memory<T>()));
|
||||
test_opt_and_launch<T, T&>(stream, cuda::dynamic_shared_memory<T>());
|
||||
}
|
||||
|
||||
void test_ref(cuda::stream_ref stream)
|
||||
{
|
||||
test_ref<int>(stream);
|
||||
test_ref<float>(stream);
|
||||
test_ref<double*>(stream);
|
||||
test_ref<void (*)()>(stream);
|
||||
}
|
||||
|
||||
template <class T, cuda::std::size_t N>
|
||||
void test_span(cuda::stream_ref stream)
|
||||
{
|
||||
static_assert(!noexcept(cuda::dynamic_shared_memory<T[]>(N * 1024 * 1024)));
|
||||
test_opt_and_launch<T, cuda::std::span<T>>(stream, cuda::dynamic_shared_memory<T[]>(N));
|
||||
|
||||
static_assert(noexcept(cuda::dynamic_shared_memory<T[N]>()));
|
||||
test_opt_and_launch<T, cuda::std::span<T, N>>(stream, cuda::dynamic_shared_memory<T[N]>());
|
||||
}
|
||||
|
||||
void test_span(cuda::stream_ref stream)
|
||||
{
|
||||
test_span<int, 1>(stream);
|
||||
test_span<int, 256>(stream);
|
||||
test_span<float, 1>(stream);
|
||||
test_span<float, 256>(stream);
|
||||
test_span<double*, 1>(stream);
|
||||
test_span<double*, 256>(stream);
|
||||
test_span<void (*)(), 1>(stream);
|
||||
test_span<void (*)(), 256>(stream);
|
||||
}
|
||||
|
||||
C2H_TEST("Dynamic shared memory option", "[launch]")
|
||||
{
|
||||
cuda::device_ref device = cuda::devices[0];
|
||||
cuda::stream stream{device};
|
||||
|
||||
test_ref(stream);
|
||||
test_span(stream);
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// XFAIL: enable-tile
|
||||
// error: indirect call is unsupported in tile code
|
||||
|
||||
// ADDITIONAL_COMPILE_FLAGS: --extended-lambda
|
||||
// UNSUPPORTED: nvrtc
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/launch>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "../common/utility.cuh"
|
||||
#include "test_macros.h"
|
||||
|
||||
void test_extended_lambda()
|
||||
{
|
||||
cuda::stream stream{cuda::devices[0]};
|
||||
test::pinned<int> i(0);
|
||||
auto config = cuda::block_dims<32>() & cuda::grid_dims<1>();
|
||||
auto assign_42_lambda = [] TEST_DEVICE_FUNC(int* pi) {
|
||||
*pi = 42;
|
||||
};
|
||||
cuda::launch(stream, config, assign_42_lambda, i.get());
|
||||
stream.sync();
|
||||
assert(*i == 42);
|
||||
|
||||
auto assign_1337_lambda = [] TEST_DEVICE_FUNC(auto config, int* pi) {
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, config) == 32);
|
||||
static_assert(cuda::block.count(cuda::grid, config) == 1);
|
||||
*pi = 1337;
|
||||
};
|
||||
cuda::launch(stream, config, assign_1337_lambda, config, i.get());
|
||||
stream.sync();
|
||||
assert(*i == 1337);
|
||||
}
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, test_extended_lambda();)
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,307 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
#include <cuda/__launch/host_launch.h>
|
||||
#include <cuda/__stream/stream.h>
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/devices>
|
||||
#include <cuda/memory>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
#include <testing.cuh>
|
||||
|
||||
void block_stream(cuda::stream_ref stream, cuda::atomic<int>& atomic)
|
||||
{
|
||||
auto block_lambda = [&]() {
|
||||
while (atomic != 1)
|
||||
;
|
||||
};
|
||||
cuda::host_launch(stream, block_lambda);
|
||||
}
|
||||
|
||||
void unblock_and_wait_stream(cuda::stream_ref stream, cuda::atomic<int>& atomic)
|
||||
{
|
||||
CCCLRT_REQUIRE(!stream.is_done());
|
||||
atomic = 1;
|
||||
stream.sync();
|
||||
atomic = 0;
|
||||
}
|
||||
|
||||
bool ordinary_function_run_proof = false;
|
||||
|
||||
template <class Ret, class... Args>
|
||||
Ret ordinary_function(Args...)
|
||||
{
|
||||
ordinary_function_run_proof = true;
|
||||
return (Ret) 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] int nodiscard_ordinary_function()
|
||||
{
|
||||
ordinary_function_run_proof = true;
|
||||
return 0;
|
||||
}
|
||||
|
||||
void launch_local_lambda(cuda::stream_ref stream, int& set, int set_to)
|
||||
{
|
||||
auto lambda = [&set, set_to]() {
|
||||
set = set_to;
|
||||
};
|
||||
cuda::host_launch(stream, lambda);
|
||||
}
|
||||
|
||||
template <typename Lambda>
|
||||
struct lambda_wrapper
|
||||
{
|
||||
Lambda lambda;
|
||||
|
||||
lambda_wrapper(const Lambda& lambda)
|
||||
: lambda(lambda)
|
||||
{}
|
||||
|
||||
lambda_wrapper(lambda_wrapper&&) = default;
|
||||
lambda_wrapper(const lambda_wrapper&) = default;
|
||||
|
||||
void operator()()
|
||||
{
|
||||
if constexpr (cuda::std::is_same_v<cuda::std::invoke_result_t<Lambda>, void*>)
|
||||
{
|
||||
// If lambda returns the address it captured, confirm this object wasn't moved
|
||||
CCCLRT_REQUIRE(lambda() == this);
|
||||
}
|
||||
else
|
||||
{
|
||||
lambda();
|
||||
}
|
||||
}
|
||||
|
||||
// Make sure we fail if const is added to this wrapper anywhere
|
||||
void operator()() const
|
||||
{
|
||||
CCCLRT_REQUIRE(false);
|
||||
}
|
||||
};
|
||||
|
||||
struct MoveOnlyArg
|
||||
{
|
||||
static MoveOnlyArg make()
|
||||
{
|
||||
return MoveOnlyArg{};
|
||||
}
|
||||
|
||||
MoveOnlyArg(const MoveOnlyArg& other) = delete;
|
||||
MoveOnlyArg(MoveOnlyArg&&) = default;
|
||||
MoveOnlyArg& operator=(const MoveOnlyArg&) = delete;
|
||||
MoveOnlyArg& operator=(MoveOnlyArg&&) = delete;
|
||||
|
||||
private:
|
||||
MoveOnlyArg() = default;
|
||||
};
|
||||
|
||||
struct MoveOnlyCallable
|
||||
{
|
||||
static MoveOnlyCallable make()
|
||||
{
|
||||
return MoveOnlyCallable{};
|
||||
}
|
||||
|
||||
MoveOnlyCallable(const MoveOnlyCallable&) = delete;
|
||||
MoveOnlyCallable(MoveOnlyCallable&&) = default;
|
||||
MoveOnlyCallable& operator=(const MoveOnlyCallable&) = delete;
|
||||
MoveOnlyCallable& operator=(MoveOnlyCallable&&) = delete;
|
||||
|
||||
void operator()(MoveOnlyArg) {}
|
||||
|
||||
private:
|
||||
MoveOnlyCallable() = default;
|
||||
};
|
||||
|
||||
C2H_CCCLRT_TEST("Host launch", "")
|
||||
{
|
||||
cuda::device_ref device{0};
|
||||
device.init();
|
||||
|
||||
cuda::stream stream{device};
|
||||
|
||||
SECTION("Ordinary function without arguments returning void")
|
||||
{
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
|
||||
|
||||
cuda::host_launch(stream, ordinary_function<void>);
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
|
||||
ordinary_function_run_proof = false;
|
||||
}
|
||||
SECTION("Ordinary function without arguments returning int")
|
||||
{
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
|
||||
|
||||
cuda::host_launch(stream, ordinary_function<int>);
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
|
||||
ordinary_function_run_proof = false;
|
||||
}
|
||||
SECTION("Ordinary function with arguments returning void")
|
||||
{
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
|
||||
|
||||
cuda::host_launch(stream, ordinary_function<int, char, double>, 'c', 1.0);
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
|
||||
ordinary_function_run_proof = false;
|
||||
}
|
||||
SECTION("Nodiscard ordinary function")
|
||||
{
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == false);
|
||||
|
||||
cuda::host_launch(stream, nodiscard_ordinary_function);
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(ordinary_function_run_proof == true);
|
||||
ordinary_function_run_proof = false;
|
||||
}
|
||||
|
||||
cuda::atomic<int> atomic = 0;
|
||||
int i = 0;
|
||||
|
||||
auto set_lambda = [&](int set) {
|
||||
i = set;
|
||||
};
|
||||
|
||||
SECTION("Can do a host launch")
|
||||
{
|
||||
block_stream(stream, atomic);
|
||||
|
||||
cuda::host_launch(stream, set_lambda, 2);
|
||||
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(i == 2);
|
||||
}
|
||||
|
||||
SECTION("Can launch multiple functions")
|
||||
{
|
||||
block_stream(stream, atomic);
|
||||
auto check_lambda = [&]() {
|
||||
CCCLRT_REQUIRE(i == 4);
|
||||
};
|
||||
|
||||
cuda::host_launch(stream, set_lambda, 3);
|
||||
cuda::host_launch(stream, set_lambda, 4);
|
||||
cuda::host_launch(stream, check_lambda);
|
||||
cuda::host_launch(stream, set_lambda, 5);
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(i == 5);
|
||||
}
|
||||
|
||||
SECTION("Non trivially copyable")
|
||||
{
|
||||
std::string s = "hello";
|
||||
|
||||
cuda::host_launch(
|
||||
stream,
|
||||
[&](auto str_arg) {
|
||||
CCCLRT_REQUIRE(s == str_arg);
|
||||
},
|
||||
s);
|
||||
stream.sync();
|
||||
}
|
||||
|
||||
SECTION("Confirm no const added to the callable")
|
||||
{
|
||||
lambda_wrapper wrapped_lambda([&]() {
|
||||
i = 21;
|
||||
});
|
||||
|
||||
cuda::host_launch(stream, wrapped_lambda);
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(i == 21);
|
||||
}
|
||||
|
||||
SECTION("Can launch a local function and return")
|
||||
{
|
||||
block_stream(stream, atomic);
|
||||
launch_local_lambda(stream, i, 42);
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(i == 42);
|
||||
}
|
||||
|
||||
SECTION("Launch by reference")
|
||||
{
|
||||
// Grab the pointer to confirm callable was not moved
|
||||
void* wrapper_ptr = nullptr;
|
||||
lambda_wrapper another_lambda_setter([&]() {
|
||||
i = 84;
|
||||
return wrapper_ptr;
|
||||
});
|
||||
wrapper_ptr = static_cast<void*>(&another_lambda_setter);
|
||||
|
||||
block_stream(stream, atomic);
|
||||
cuda::host_launch(stream, cuda::std::ref(another_lambda_setter));
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(i == 84);
|
||||
}
|
||||
|
||||
SECTION("Launch by reference with arguments")
|
||||
{
|
||||
i = 10;
|
||||
int result = 0;
|
||||
auto lambda = [&result](int j) {
|
||||
result = j;
|
||||
};
|
||||
block_stream(stream, atomic);
|
||||
cuda::host_launch(stream, cuda::std::ref(lambda), i);
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(result == 10);
|
||||
}
|
||||
|
||||
SECTION("Launch by reference with arguments captured by reference")
|
||||
{
|
||||
i = 0;
|
||||
auto lambda = [](int& j) {
|
||||
j = 10;
|
||||
};
|
||||
block_stream(stream, atomic);
|
||||
cuda::host_launch(stream, cuda::std::ref(lambda), cuda::std::ref(i));
|
||||
unblock_and_wait_stream(stream, atomic);
|
||||
CCCLRT_REQUIRE(i == 10);
|
||||
}
|
||||
|
||||
SECTION("Check that host_launch works with move only callables and arguments")
|
||||
{
|
||||
cuda::host_launch(stream, MoveOnlyCallable::make(), MoveOnlyArg::make());
|
||||
stream.sync();
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Host launch uses the stream device when current device differs", "[launch][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
cuda::stream stream{explicit_device};
|
||||
int value = 0;
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
cuda::host_launch(stream, [&value]() {
|
||||
value = 42;
|
||||
});
|
||||
}
|
||||
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(value == 42);
|
||||
}
|
||||
@@ -0,0 +1,542 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
#include <cuda/atomic>
|
||||
#include <cuda/devices>
|
||||
#include <cuda/launch>
|
||||
#include <cuda/memory>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <cooperative_groups.h>
|
||||
#include <testing.cuh>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
#if !_CCCL_CUDA_COMPILER(CLANG)
|
||||
|
||||
__managed__ bool kernel_run_proof = false;
|
||||
|
||||
void check_kernel_run(cudaStream_t stream)
|
||||
{
|
||||
CUDART(cudaStreamSynchronize(stream));
|
||||
CCCLRT_CHECK(kernel_run_proof);
|
||||
kernel_run_proof = false;
|
||||
}
|
||||
|
||||
struct kernel_run_proof_check
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()()
|
||||
{
|
||||
CCCLRT_CHECK_DEVICE(kernel_run_proof);
|
||||
kernel_run_proof = false;
|
||||
}
|
||||
};
|
||||
|
||||
struct functor_int_argument
|
||||
{
|
||||
TEST_DEVICE_FUNC void operator()(int dummy)
|
||||
{
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
};
|
||||
|
||||
template <unsigned int BlockSize>
|
||||
struct functor_taking_config
|
||||
{
|
||||
template <typename Config>
|
||||
TEST_DEVICE_FUNC void operator()(Config config, int grid_size)
|
||||
{
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, config) == BlockSize);
|
||||
CCCLRT_REQUIRE_DEVICE(cuda::block.count(cuda::grid, config) == grid_size);
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
};
|
||||
|
||||
__global__ void kernel_no_arguments()
|
||||
{
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
|
||||
__global__ void kernel_int_argument(int dummy)
|
||||
{
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
|
||||
template <typename Config, unsigned int BlockSize>
|
||||
__global__ void kernel_taking_config(Config config, int grid_size)
|
||||
{
|
||||
functor_taking_config<BlockSize>()(config, grid_size);
|
||||
}
|
||||
|
||||
struct my_dynamic_smem_t
|
||||
{
|
||||
int i;
|
||||
};
|
||||
|
||||
template <typename SmemType>
|
||||
struct dynamic_smem_single
|
||||
{
|
||||
template <typename Config>
|
||||
TEST_DEVICE_FUNC void operator()(Config config)
|
||||
{
|
||||
decltype(auto) dynamic_smem = cuda::dynamic_shared_memory(config);
|
||||
static_assert(::cuda::std::is_same_v<SmemType&, decltype(dynamic_smem)>);
|
||||
CCCLRT_REQUIRE_DEVICE(::cuda::device::is_object_from(dynamic_smem, ::cuda::device::address_space::shared));
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename SmemType, size_t Extent>
|
||||
struct dynamic_smem_span
|
||||
{
|
||||
template <typename Config>
|
||||
TEST_DEVICE_FUNC void operator()(Config config, int size)
|
||||
{
|
||||
auto dynamic_smem = cuda::dynamic_shared_memory(config);
|
||||
static_assert(decltype(dynamic_smem)::extent == Extent);
|
||||
static_assert(::cuda::std::is_same_v<SmemType&, decltype(dynamic_smem[1])>);
|
||||
CCCLRT_REQUIRE_DEVICE(dynamic_smem.size() == size);
|
||||
CCCLRT_REQUIRE_DEVICE(::cuda::device::is_object_from(dynamic_smem[1], ::cuda::device::address_space::shared));
|
||||
kernel_run_proof = true;
|
||||
}
|
||||
};
|
||||
|
||||
struct launch_transform_to_int_convertible
|
||||
{
|
||||
int value_;
|
||||
|
||||
struct int_convertible
|
||||
{
|
||||
cudaStream_t stream_;
|
||||
int value_;
|
||||
|
||||
int_convertible(cudaStream_t stream, int value) noexcept
|
||||
: stream_(stream)
|
||||
, value_(value)
|
||||
{
|
||||
// Check that the constructor runs before the kernel is launched
|
||||
// Disabled for now because we don't handle it with graphs
|
||||
// CUDAX_CHECK_FALSE(kernel_run_proof);
|
||||
}
|
||||
|
||||
// Immovable to ensure that launch_transform doesn't copy the returned
|
||||
// object
|
||||
int_convertible(int_convertible&&) noexcept = delete;
|
||||
|
||||
~int_convertible() noexcept
|
||||
{
|
||||
// Check that the destructor runs after the kernel is launched
|
||||
// Disabled for now because we don't handle it with graphs
|
||||
// CUDART(cudaStreamSynchronize(stream_));
|
||||
// CCCLRT_CHECK(kernel_run_proof);
|
||||
}
|
||||
|
||||
// This is the value that will be passed to the kernel
|
||||
int transformed_argument() const
|
||||
{
|
||||
return value_;
|
||||
}
|
||||
};
|
||||
|
||||
[[nodiscard]] friend int_convertible
|
||||
transform_launch_argument(::cuda::stream_ref stream, launch_transform_to_int_convertible self) noexcept
|
||||
{
|
||||
return int_convertible(stream.get(), self.value_);
|
||||
}
|
||||
};
|
||||
|
||||
// Needs a separate function for Windows extended lambda
|
||||
void launch_smoke_test(cudaStream_t dst)
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
// Use raw stream to make sure it can be implicitly converted on call to
|
||||
// launch
|
||||
cudaStream_t stream;
|
||||
|
||||
CUDART(cudaStreamCreate(&stream));
|
||||
// Spell out all overloads to make sure they compile, include a check for
|
||||
// implicit conversions
|
||||
{
|
||||
const int grid_size = 4;
|
||||
constexpr int block_size = 256;
|
||||
auto dimensions = cuda::make_hierarchy(cuda::grid_dims(grid_size), cuda::block_dims<256>());
|
||||
auto config = cuda::make_config(dimensions);
|
||||
|
||||
// Not taking dims
|
||||
{
|
||||
cuda::launch(dst, config, kernel_no_arguments);
|
||||
check_kernel_run(dst);
|
||||
|
||||
const int dummy = 1;
|
||||
cuda::launch(dst, config, kernel_int_argument, dummy);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_int_argument, 1);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_int_argument, launch_transform_to_int_convertible{1});
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_int_argument, 1U);
|
||||
check_kernel_run(dst);
|
||||
|
||||
cuda::launch(dst, config, functor_int_argument(), dummy);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_int_argument(), 1);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_int_argument(), launch_transform_to_int_convertible{1});
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_int_argument(), 1U);
|
||||
check_kernel_run(dst);
|
||||
}
|
||||
|
||||
// Config argument
|
||||
{
|
||||
auto functor_instance = functor_taking_config<block_size>();
|
||||
auto kernel_instance = kernel_taking_config<decltype(config), block_size>;
|
||||
|
||||
cuda::launch(dst, config, functor_instance, grid_size);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_instance, ::cuda::std::move(grid_size));
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_instance, launch_transform_to_int_convertible{grid_size});
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, functor_instance, static_cast<unsigned int>(grid_size));
|
||||
check_kernel_run(dst);
|
||||
|
||||
cuda::launch(dst, config, kernel_instance, grid_size);
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_instance, ::cuda::std::move(grid_size));
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_instance, launch_transform_to_int_convertible{grid_size});
|
||||
check_kernel_run(dst);
|
||||
cuda::launch(dst, config, kernel_instance, static_cast<unsigned int>(grid_size));
|
||||
check_kernel_run(dst);
|
||||
}
|
||||
}
|
||||
|
||||
// Dynamic shared memory option
|
||||
{
|
||||
auto config = cuda::block_dims<32>() & cuda::grid_dims<1>();
|
||||
|
||||
auto test = [&](const auto& input_config) {
|
||||
// Single element
|
||||
{
|
||||
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t>());
|
||||
|
||||
cuda::launch(dst, config, dynamic_smem_single<my_dynamic_smem_t>());
|
||||
check_kernel_run(dst);
|
||||
}
|
||||
|
||||
// Dynamic span
|
||||
{
|
||||
const int size = 2;
|
||||
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t[]>(size));
|
||||
cuda::launch(dst, config, dynamic_smem_span<my_dynamic_smem_t, ::cuda::std::dynamic_extent>(), size);
|
||||
check_kernel_run(dst);
|
||||
}
|
||||
|
||||
// Static span
|
||||
{
|
||||
constexpr int size = 3;
|
||||
auto config = input_config.add(cuda::dynamic_shared_memory<my_dynamic_smem_t[size]>());
|
||||
cuda::launch(dst, config, dynamic_smem_span<my_dynamic_smem_t, size>(), size);
|
||||
check_kernel_run(dst);
|
||||
}
|
||||
};
|
||||
|
||||
test(config);
|
||||
test(config.add(cuda::cooperative_launch(), cuda::launch_priority(0)));
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Launch smoke stream", "[launch]")
|
||||
{
|
||||
// Use raw stream to make sure it can be implicitly converted on call to
|
||||
// launch
|
||||
cudaStream_t stream;
|
||||
|
||||
{
|
||||
::cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
CUDART(cudaStreamCreate(&stream));
|
||||
}
|
||||
|
||||
launch_smoke_test(stream);
|
||||
|
||||
{
|
||||
::cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
CUDART(cudaStreamSynchronize(stream));
|
||||
CUDART(cudaStreamDestroy(stream));
|
||||
}
|
||||
}
|
||||
|
||||
template <typename DefaultConfig>
|
||||
struct kernel_with_default_config
|
||||
{
|
||||
DefaultConfig config;
|
||||
|
||||
kernel_with_default_config(DefaultConfig c)
|
||||
: config(c)
|
||||
{}
|
||||
|
||||
DefaultConfig default_config() const
|
||||
{
|
||||
return config;
|
||||
}
|
||||
|
||||
template <typename Config, typename ConfigCheckFn>
|
||||
TEST_DEVICE_FUNC void operator()(Config config, ConfigCheckFn check_fn)
|
||||
{
|
||||
check_fn(config);
|
||||
}
|
||||
};
|
||||
|
||||
struct verify_callable
|
||||
{
|
||||
template <typename Config>
|
||||
TEST_DEVICE_FUNC void operator()(Config config)
|
||||
{
|
||||
static_assert(cuda::gpu_thread.count(cuda::block, config) == 256);
|
||||
CCCLRT_REQUIRE(cuda::block.count(cuda::grid, config) == 4);
|
||||
cooperative_groups::this_grid().sync();
|
||||
}
|
||||
};
|
||||
|
||||
C2H_CCCLRT_TEST("Launch with default config", "")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
auto grid = cuda::grid_dims(4);
|
||||
auto block = cuda::block_dims<256>;
|
||||
|
||||
SECTION("Combine with empty")
|
||||
{
|
||||
kernel_with_default_config kernel{cuda::make_config(block, grid, cuda::cooperative_launch())};
|
||||
static_assert(cuda::__is_kernel_config<decltype(kernel.default_config())>);
|
||||
static_assert(cuda::__kernel_has_default_config<decltype(kernel)>);
|
||||
|
||||
cuda::launch(stream, cuda::make_config(), kernel, verify_callable{});
|
||||
stream.sync();
|
||||
}
|
||||
SECTION("Combine with no overlap")
|
||||
{
|
||||
kernel_with_default_config kernel{cuda::make_config(block)};
|
||||
cuda::launch(stream, cuda::make_config(grid, cuda::cooperative_launch()), kernel, verify_callable{});
|
||||
stream.sync();
|
||||
}
|
||||
SECTION("Combine with overlap")
|
||||
{
|
||||
kernel_with_default_config kernel{cuda::make_config(cuda::block_dims<1>(), cuda::cooperative_launch())};
|
||||
cuda::launch(stream, cuda::make_config(block, grid, cuda::cooperative_launch()), kernel, verify_callable{});
|
||||
stream.sync();
|
||||
}
|
||||
}
|
||||
|
||||
// Regression test: cuda::launch must work when the calling function has
|
||||
// a __restrict__-qualified pointer parameter. On some nvcc + host compiler
|
||||
// combos, __restrict__ survives through the type transformation pipeline
|
||||
// and causes a function pointer conversion failure in __get_kernel_launcher.
|
||||
struct restrict_assign_functor
|
||||
{
|
||||
template <typename Config>
|
||||
TEST_DEVICE_FUNC void operator()(Config config, int* __restrict__ dst)
|
||||
{
|
||||
*dst = 42;
|
||||
}
|
||||
};
|
||||
|
||||
// The __restrict__ on the function parameter is the trigger: on affected compilers,
|
||||
// it leaks into the template args of __get_kernel_launcher via cuda::launch.
|
||||
void launch_with_restrict_param(cuda::stream_ref stream, int* __restrict__ dst)
|
||||
{
|
||||
auto config = cuda::make_config(cuda::grid_dims(1), cuda::block_dims<1>());
|
||||
cuda::launch(stream, config, restrict_assign_functor{}, dst);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Launch functor with __restrict__ pointer arg", "[launch]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
test::pinned<int> val{0};
|
||||
|
||||
launch_with_restrict_param(stream, val.get());
|
||||
stream.sync();
|
||||
|
||||
CCCLRT_CHECK(*val == 42);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Launch uses the stream device when current device differs", "[launch][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
cuda::stream stream{explicit_device};
|
||||
|
||||
int* device_value{};
|
||||
{
|
||||
cuda::__ensure_current_context guard(explicit_device);
|
||||
CUDART(cudaMalloc(reinterpret_cast<void**>(&device_value), sizeof(int)));
|
||||
CUDART(cudaMemsetAsync(device_value, 0, sizeof(int), stream.get()));
|
||||
}
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
auto config = cuda::make_config(cuda::grid_dims(1), cuda::block_dims<1>());
|
||||
cuda::launch(stream, config, test::assign_42{}, device_value);
|
||||
}
|
||||
|
||||
stream.sync();
|
||||
|
||||
int value{};
|
||||
{
|
||||
cuda::__ensure_current_context guard(explicit_device);
|
||||
CUDART(cudaMemcpy(&value, device_value, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
CUDART(cudaFree(device_value));
|
||||
}
|
||||
|
||||
CCCLRT_CHECK(value == 42);
|
||||
}
|
||||
|
||||
__managed__ cuda::std::size_t launched_nthreads;
|
||||
__managed__ cuda::std::size_t launched_nblocks;
|
||||
__managed__ cuda::std::size_t launched_nclusters;
|
||||
|
||||
struct LaunchDimsFunctor
|
||||
{
|
||||
template <class Config>
|
||||
TEST_DEVICE_FUNC void operator()(const Config& config)
|
||||
{
|
||||
cuda::atomic_ref(launched_nthreads)++;
|
||||
|
||||
if (threadIdx.x == 0 && threadIdx.y == 0 && threadIdx.z == 0)
|
||||
{
|
||||
cuda::atomic_ref(launched_nblocks)++;
|
||||
|
||||
NV_IF_ELSE_TARGET(NV_PROVIDES_SM_90,
|
||||
({
|
||||
if (__clusterRelativeBlockRank() == 0)
|
||||
{
|
||||
cuda::atomic_ref(launched_nclusters)++;
|
||||
}
|
||||
}),
|
||||
({ cuda::atomic_ref(launched_nclusters)++; }));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <class GridDesc, class BlockDesc>
|
||||
void test_launch_dims(cuda::stream_ref stream, GridDesc grid_desc, BlockDesc block_desc)
|
||||
{
|
||||
launched_nthreads = 0;
|
||||
launched_nblocks = 0;
|
||||
launched_nclusters = 0;
|
||||
|
||||
const auto& grid_exts = grid_desc.extents();
|
||||
const auto& block_exts = block_desc.extents();
|
||||
|
||||
const auto config = cuda::make_config(grid_desc, block_desc);
|
||||
cuda::launch(stream, config, LaunchDimsFunctor{});
|
||||
stream.sync();
|
||||
|
||||
const auto exp_nclusters = cuda::std::size_t{grid_exts.extent(0)} * grid_exts.extent(1) * grid_exts.extent(2);
|
||||
CCCLRT_CHECK(launched_nclusters == exp_nclusters);
|
||||
|
||||
const auto exp_nblocks = exp_nclusters;
|
||||
CCCLRT_CHECK(launched_nblocks == exp_nblocks);
|
||||
|
||||
const auto exp_nthreads = exp_nblocks * block_exts.extent(0) * block_exts.extent(1) * block_exts.extent(2);
|
||||
CCCLRT_CHECK(launched_nthreads == exp_nthreads);
|
||||
}
|
||||
|
||||
template <class GridDesc, class ClusterDesc, class BlockDesc>
|
||||
void test_launch_dims(cuda::stream_ref stream, GridDesc grid_desc, ClusterDesc cluster_desc, BlockDesc block_desc)
|
||||
{
|
||||
launched_nthreads = 0;
|
||||
launched_nblocks = 0;
|
||||
launched_nclusters = 0;
|
||||
|
||||
const auto& grid_exts = grid_desc.extents();
|
||||
const auto& cluster_exts = cluster_desc.extents();
|
||||
const auto& block_exts = block_desc.extents();
|
||||
|
||||
const auto config = cuda::make_config(grid_desc, cluster_desc, block_desc);
|
||||
cuda::launch(stream, config, LaunchDimsFunctor{});
|
||||
stream.sync();
|
||||
|
||||
const auto exp_nclusters = cuda::std::size_t{grid_exts.extent(0)} * grid_exts.extent(1) * grid_exts.extent(2);
|
||||
CCCLRT_CHECK(launched_nclusters == exp_nclusters);
|
||||
|
||||
const auto exp_nblocks = exp_nclusters * cluster_exts.extent(0) * cluster_exts.extent(1) * cluster_exts.extent(2);
|
||||
CCCLRT_CHECK(launched_nblocks == exp_nblocks);
|
||||
|
||||
const auto exp_nthreads = exp_nblocks * block_exts.extent(0) * block_exts.extent(1) * block_exts.extent(2);
|
||||
CCCLRT_CHECK(launched_nthreads == exp_nthreads);
|
||||
}
|
||||
|
||||
C2H_TEST("Launch dims", "[launch]")
|
||||
{
|
||||
cuda::stream stream{cuda::device_ref{0}};
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::block_dims<2, 7, 9>());
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::block_dims<2, 7, 9>());
|
||||
|
||||
if (cuda::device_attributes::compute_capability_major(stream.device()) >= 9)
|
||||
{
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims(dim3{3}), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(
|
||||
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims(dim3{3}), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(
|
||||
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims<2, 7, 9>());
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims<3>(), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims<1, 5>(), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(
|
||||
stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2}), cuda::cluster_dims<3>(), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{2, 9}), cuda::cluster_dims<1, 5>(), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(stream, cuda::grid_dims(dim3{3, 4, 5}), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims<2, 7, 9>());
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims(dim3{3}), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(
|
||||
stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims(dim3{3}), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims(dim3{1, 5}), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims(dim3{3, 1, 2}), cuda::block_dims<2, 7, 9>());
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims<3>(), cuda::block_dims(dim3{10}));
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims<1, 5>(), cuda::block_dims(dim3{3, 7}));
|
||||
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims(dim3{2, 7, 9}));
|
||||
|
||||
test_launch_dims(stream, cuda::grid_dims<2>(), cuda::cluster_dims<3>(), cuda::block_dims<10>());
|
||||
test_launch_dims(stream, cuda::grid_dims<2, 9>(), cuda::cluster_dims<1, 5>(), cuda::block_dims<3, 7>());
|
||||
test_launch_dims(stream, cuda::grid_dims<3, 4, 5>(), cuda::cluster_dims<3, 1, 2>(), cuda::block_dims<2, 7, 9>());
|
||||
}
|
||||
}
|
||||
|
||||
#endif // !_CCCL_CUDA_COMPILER(CLANG)
|
||||
@@ -0,0 +1,252 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/devices>
|
||||
#include <cuda/std/type_traits>
|
||||
#include <cuda/std/utility>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
C2H_CCCLRT_TEST("Can create a stream and launch work into it", "[stream]")
|
||||
{
|
||||
cuda::stream str{cuda::device_ref{0}};
|
||||
::test::pinned<int> i(0);
|
||||
::test::launch_kernel_single_thread(str, ::test::assign_42{}, i.get());
|
||||
str.sync();
|
||||
CCCLRT_REQUIRE(*i == 42);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("From native handle", "[stream]")
|
||||
{
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
cudaStream_t handle;
|
||||
CUDART(cudaStreamCreate(&handle));
|
||||
{
|
||||
auto stream = cuda::stream::from_native_handle(handle);
|
||||
|
||||
::test::pinned<int> i(0);
|
||||
::test::launch_kernel_single_thread(stream, ::test::assign_42{}, i.get());
|
||||
stream.sync();
|
||||
CCCLRT_REQUIRE(*i == 42);
|
||||
(void) stream.release();
|
||||
}
|
||||
CUDART(cudaStreamDestroy(handle));
|
||||
}
|
||||
|
||||
template <typename StreamType>
|
||||
void add_dependency_test(const StreamType& waiter, const StreamType& waitee)
|
||||
{
|
||||
CCCLRT_REQUIRE(waiter != waitee);
|
||||
|
||||
auto verify_dependency = [&](const auto& insert_dependency) {
|
||||
::test::pinned<int> i(0);
|
||||
::cuda::atomic_ref atomic_i(*i);
|
||||
|
||||
::test::launch_kernel_single_thread(waitee, ::test::spin_until_80{}, i.get());
|
||||
::test::launch_kernel_single_thread(waitee, ::test::assign_42{}, i.get());
|
||||
insert_dependency();
|
||||
::test::launch_kernel_single_thread(waiter, ::test::verify_42{}, i.get());
|
||||
CCCLRT_REQUIRE(atomic_i.load() != 42);
|
||||
CCCLRT_REQUIRE(!waiter.is_done());
|
||||
atomic_i.store(80);
|
||||
waiter.sync();
|
||||
waitee.sync();
|
||||
};
|
||||
|
||||
SECTION("Stream wait declared event")
|
||||
{
|
||||
verify_dependency([&]() {
|
||||
cuda::event ev(waitee);
|
||||
waiter.wait(ev);
|
||||
});
|
||||
}
|
||||
|
||||
SECTION("Stream wait returned event")
|
||||
{
|
||||
verify_dependency([&]() {
|
||||
auto ev = waitee.record_event();
|
||||
waiter.wait(ev);
|
||||
});
|
||||
}
|
||||
|
||||
SECTION("Stream wait returned timed event")
|
||||
{
|
||||
verify_dependency([&]() {
|
||||
auto ev = waitee.record_timed_event();
|
||||
waiter.wait(ev);
|
||||
});
|
||||
}
|
||||
|
||||
SECTION("Stream wait stream")
|
||||
{
|
||||
verify_dependency([&]() {
|
||||
waiter.wait(waitee);
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Can add dependency into a stream", "[stream]")
|
||||
{
|
||||
cuda::stream waiter{cuda::device_ref{0}}, waitee{cuda::device_ref{0}};
|
||||
|
||||
add_dependency_test<cuda::stream>(waiter, waitee);
|
||||
add_dependency_test<cuda::stream_ref>(waiter, waitee);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Stream priority", "[stream]")
|
||||
{
|
||||
cuda::stream stream_default_prio{cuda::device_ref{0}};
|
||||
CCCLRT_REQUIRE(stream_default_prio.priority() == cuda::stream::default_priority);
|
||||
|
||||
auto priority = cuda::stream::default_priority - 1;
|
||||
cuda::stream stream{cuda::device_ref{0}, priority};
|
||||
CCCLRT_REQUIRE(stream.priority() == priority);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Stream get device", "[stream]")
|
||||
{
|
||||
cuda::stream dev0_stream(cuda::device_ref{0});
|
||||
CCCLRT_REQUIRE(dev0_stream.device() == 0);
|
||||
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{*std::prev(cuda::devices.end())});
|
||||
cudaStream_t stream_handle;
|
||||
CUDART(cudaStreamCreate(&stream_handle));
|
||||
auto stream_cudart = cuda::stream::from_native_handle(stream_handle);
|
||||
CCCLRT_REQUIRE(stream_cudart.device() == *std::prev(cuda::devices.end()));
|
||||
auto stream_ref_cudart = cuda::stream_ref(stream_handle);
|
||||
CCCLRT_REQUIRE(stream_ref_cudart.device() == *std::prev(cuda::devices.end()));
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Stream construction uses the explicit device", "[stream][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
auto stream = [&]() {
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
return cuda::stream{explicit_device};
|
||||
}();
|
||||
|
||||
CCCLRT_REQUIRE(stream.device() == explicit_device);
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Stream dependency uses the explicit stream device", "[stream][multi_gpu]")
|
||||
{
|
||||
if (cuda::devices.size() < 2)
|
||||
{
|
||||
return;
|
||||
}
|
||||
|
||||
cuda::device_ref current_device{0};
|
||||
cuda::device_ref explicit_device{1};
|
||||
|
||||
cuda::stream waiter{explicit_device};
|
||||
cuda::stream waitee{explicit_device};
|
||||
|
||||
::test::pinned<int> value(0);
|
||||
::cuda::atomic_ref atomic_value(*value);
|
||||
|
||||
::test::launch_kernel_single_thread(waitee, ::test::spin_until_80{}, value.get());
|
||||
::test::launch_kernel_single_thread(waitee, ::test::assign_42{}, value.get());
|
||||
|
||||
{
|
||||
cuda::__ensure_current_context guard(current_device);
|
||||
waiter.wait(waitee);
|
||||
}
|
||||
|
||||
::test::launch_kernel_single_thread(waiter, ::test::verify_42{}, value.get());
|
||||
CCCLRT_REQUIRE(atomic_value.load() != 42);
|
||||
CCCLRT_REQUIRE(!waiter.is_done());
|
||||
|
||||
atomic_value.store(80);
|
||||
waiter.sync();
|
||||
waitee.sync();
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Stream ID", "[stream]")
|
||||
{
|
||||
STATIC_REQUIRE(cuda::std::is_same_v<unsigned long long, cuda::std::underlying_type_t<cuda::stream_id>>);
|
||||
STATIC_REQUIRE(cuda::std::is_same_v<cuda::stream_id, decltype(cuda::std::declval<cuda::stream_ref>().id())>);
|
||||
|
||||
cuda::stream stream1{cuda::device_ref{0}};
|
||||
cuda::stream stream2{cuda::device_ref{0}};
|
||||
|
||||
// Test that id() returns a valid ID
|
||||
auto id1 = stream1.id();
|
||||
auto id2 = stream2.id();
|
||||
|
||||
// Test that different streams have different IDs
|
||||
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
|
||||
CCCLRT_REQUIRE(cuda::std::to_underlying(id1) != cuda::std::to_underlying(id2));
|
||||
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
|
||||
CCCLRT_REQUIRE(id1 != id2);
|
||||
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
|
||||
|
||||
// Test that the same stream returns the same ID when called multiple times
|
||||
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
|
||||
CCCLRT_REQUIRE(cuda::std::to_underlying(stream1.id()) == cuda::std::to_underlying(id1));
|
||||
CCCLRT_REQUIRE(cuda::std::to_underlying(stream2.id()) == cuda::std::to_underlying(id2));
|
||||
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
|
||||
CCCLRT_REQUIRE(stream1.id() == id1);
|
||||
CCCLRT_REQUIRE(stream2.id() == id2);
|
||||
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
|
||||
|
||||
{
|
||||
// Test that stream_ref also supports id()
|
||||
// NULL stream needs a device to be set
|
||||
cuda::__ensure_current_context guard(cuda::device_ref{0});
|
||||
cuda::stream_ref ref1(::cudaStream_t{});
|
||||
cuda::stream_ref ref2(stream1);
|
||||
|
||||
#if _CCCL_COMPILER(NVHPC, <, 25, 11)
|
||||
CCCLRT_REQUIRE(cuda::std::to_underlying(ref1.id()) != cuda::std::to_underlying(ref2.id()));
|
||||
CCCLRT_REQUIRE(cuda::std::to_underlying(ref2.id()) == cuda::std::to_underlying(id1));
|
||||
#else // ^^^ _CCCL_COMPILER(NVHPC, <, 25, 11) ^^^ / vvv !_CCCL_COMPILER(NVHPC, <, 25, 11) vvv
|
||||
CCCLRT_REQUIRE(ref1.id() != ref2.id());
|
||||
CCCLRT_REQUIRE(ref2.id() == id1);
|
||||
#endif // ^^^ !_CCCL_COMPILER(NVHPC, <, 25, 11) ^^^
|
||||
}
|
||||
}
|
||||
|
||||
C2H_CCCLRT_TEST("Invalid stream", "[stream]")
|
||||
{
|
||||
// 1. Test the signature
|
||||
STATIC_REQUIRE(cuda::std::is_same_v<const cuda::invalid_stream_t, decltype(cuda::invalid_stream)>);
|
||||
|
||||
// 2. Test explicit construction of stream_ref from invalid_stream
|
||||
STATIC_REQUIRE(cuda::std::is_constructible_v<cuda::stream_ref, cuda::invalid_stream_t>);
|
||||
STATIC_REQUIRE(!cuda::std::is_convertible_v<cuda::invalid_stream_t, cuda::stream_ref>);
|
||||
{
|
||||
cuda::stream_ref stream{cuda::invalid_stream};
|
||||
CCCLRT_REQUIRE(stream.get() == (cudaStream_t) (~0ull)); // NOLINT(performance-no-int-to-ptr)
|
||||
}
|
||||
|
||||
// 3. Test stream_ref comparisons
|
||||
{
|
||||
cuda::stream_ref valid_stream{(cudaStream_t) (123ull)}; // NOLINT(performance-no-int-to-ptr)
|
||||
cuda::stream_ref invalid_stream{cuda::invalid_stream};
|
||||
|
||||
CCCLRT_REQUIRE(!(valid_stream == cuda::invalid_stream));
|
||||
CCCLRT_REQUIRE(invalid_stream == cuda::invalid_stream);
|
||||
CCCLRT_REQUIRE(!(cuda::invalid_stream == valid_stream));
|
||||
CCCLRT_REQUIRE(cuda::invalid_stream == invalid_stream);
|
||||
|
||||
CCCLRT_REQUIRE(valid_stream != cuda::invalid_stream);
|
||||
CCCLRT_REQUIRE(!(invalid_stream != cuda::invalid_stream));
|
||||
CCCLRT_REQUIRE(cuda::invalid_stream != valid_stream);
|
||||
CCCLRT_REQUIRE(!(cuda::invalid_stream != invalid_stream));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,76 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/__driver/driver_api.h>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
// This test is an exception and shouldn't use C2H_CCCLRT_TEST macro
|
||||
C2H_TEST("Call each driver api", "[utility]")
|
||||
{
|
||||
namespace driver = ::cuda::__driver;
|
||||
cudaStream_t stream;
|
||||
// Assumes the ctx stack was empty or had one ctx, should be the case unless some other
|
||||
// test leaves 2+ ctxs on the stack
|
||||
|
||||
// Pushes the primary context if the stack is empty
|
||||
CUDART(cudaStreamCreate(&stream));
|
||||
|
||||
auto ctx = driver::__ctxGetCurrent();
|
||||
CCCLRT_REQUIRE(ctx != nullptr);
|
||||
|
||||
// Confirm pop will leave the stack empty
|
||||
driver::__ctxPop();
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == nullptr);
|
||||
|
||||
// Confirm we can push multiple times
|
||||
driver::__ctxPush(ctx);
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
|
||||
|
||||
driver::__ctxPush(ctx);
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
|
||||
|
||||
driver::__ctxPop();
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
|
||||
|
||||
// Confirm stream ctx match
|
||||
auto stream_ctx = driver::__streamGetCtx(stream);
|
||||
CCCLRT_REQUIRE(ctx == stream_ctx);
|
||||
|
||||
CUDART(cudaStreamDestroy(stream));
|
||||
|
||||
CCCLRT_REQUIRE(driver::__deviceGet(0) == 0);
|
||||
|
||||
// Confirm we can retain the primary ctx that cudart retained first
|
||||
auto primary_ctx = driver::__primaryCtxRetain(0);
|
||||
CCCLRT_REQUIRE(ctx == primary_ctx);
|
||||
|
||||
driver::__ctxPop();
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == nullptr);
|
||||
|
||||
CCCLRT_REQUIRE(driver::__isPrimaryCtxActive(0));
|
||||
// Confirm we can reset the primary context with double release
|
||||
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
|
||||
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
|
||||
|
||||
// Try a third release in case curand retained the primary ctx as well
|
||||
if (driver::__isPrimaryCtxActive(0))
|
||||
{
|
||||
CCCLRT_REQUIRE(driver::__primaryCtxReleaseNoThrow(0) == cudaSuccess);
|
||||
}
|
||||
|
||||
CCCLRT_REQUIRE(!driver::__isPrimaryCtxActive(0));
|
||||
|
||||
// Confirm cudart can recover
|
||||
CUDART(cudaStreamCreate(&stream));
|
||||
CCCLRT_REQUIRE(driver::__ctxGetCurrent() == ctx);
|
||||
|
||||
CUDART(driver::__streamDestroyNoThrow(stream));
|
||||
}
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/__runtime/ensure_current_context.h>
|
||||
#include <cuda/devices>
|
||||
|
||||
#include <testing.cuh>
|
||||
|
||||
namespace driver = cuda::__driver;
|
||||
|
||||
void recursive_check_device_setter(int id)
|
||||
{
|
||||
int cudart_id;
|
||||
cuda::__ensure_current_context setter(cuda::device_ref{id});
|
||||
CCCLRT_REQUIRE(test::count_driver_stack() == cuda::devices.size() - id);
|
||||
auto ctx = driver::__ctxGetCurrent();
|
||||
CUDART(cudaGetDevice(&cudart_id));
|
||||
CCCLRT_REQUIRE(cudart_id == id);
|
||||
|
||||
if (id != 0)
|
||||
{
|
||||
recursive_check_device_setter(id - 1);
|
||||
|
||||
CCCLRT_REQUIRE(test::count_driver_stack() == cuda::devices.size() - id);
|
||||
CCCLRT_REQUIRE(ctx == driver::__ctxGetCurrent());
|
||||
CUDART(cudaGetDevice(&cudart_id));
|
||||
CCCLRT_REQUIRE(cudart_id == id);
|
||||
}
|
||||
}
|
||||
|
||||
C2H_TEST("ensure current context", "[device]")
|
||||
{
|
||||
test::empty_driver_stack();
|
||||
// If possible use something different than CUDART default 0
|
||||
int target_device = static_cast<int>(cuda::devices.size() - 1);
|
||||
|
||||
SECTION("context setter")
|
||||
{
|
||||
recursive_check_device_setter(target_device);
|
||||
|
||||
CCCLRT_REQUIRE(test::count_driver_stack() == 0);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// <cuda/std/chrono>
|
||||
|
||||
// system_clock
|
||||
|
||||
// static time_point from_time_t(time_t t);
|
||||
|
||||
#include <cuda/std/chrono>
|
||||
|
||||
#include <nv/target>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, ({
|
||||
using C = ::std::chrono::system_clock;
|
||||
C::time_point t1 = C::from_time_t(C::to_time_t(C::now()));
|
||||
unused(t1);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,30 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// <cuda/std/chrono>
|
||||
|
||||
// system_clock
|
||||
|
||||
// time_t to_time_t(const time_point& t);
|
||||
|
||||
#include <cuda/std/chrono>
|
||||
|
||||
#include <nv/target>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
int main(int, char**)
|
||||
{
|
||||
NV_IF_TARGET(NV_IS_HOST, ({
|
||||
using C = ::std::chrono::system_clock;
|
||||
cuda::std::time_t t1 = C::to_time_t(C::now());
|
||||
unused(t1);
|
||||
}));
|
||||
return 0;
|
||||
}
|
||||
133
cccl_upstream/libcudacxx/test/libcudacxx/cuda/cmath.pass.cpp
Normal file
133
cccl_upstream/libcudacxx/test/libcudacxx/cuda/cmath.pass.cpp
Normal file
@@ -0,0 +1,133 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/cmath>
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstddef>
|
||||
#include <cuda/std/limits>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "test_macros.h"
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
# include <cstdint>
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
template <class T, class U>
|
||||
TEST_FUNC constexpr void test()
|
||||
{
|
||||
constexpr T maxv = cuda::std::numeric_limits<T>::max();
|
||||
|
||||
// ensure that we return the right type
|
||||
using Common = ::cuda::std::common_type_t<T, U>;
|
||||
static_assert(cuda::std::is_same<decltype(cuda::ceil_div(T(0), U(1))), Common>::value);
|
||||
assert(cuda::ceil_div(T(0), U(1)) == Common(0));
|
||||
assert(cuda::ceil_div(T(1), U(1)) == Common(1));
|
||||
assert(cuda::ceil_div(T(126), U(64)) == Common(2));
|
||||
|
||||
// ensure that we are resilient against overflow
|
||||
assert(cuda::ceil_div(maxv, U(1)) == maxv);
|
||||
assert(cuda::ceil_div(maxv, maxv) == Common(1));
|
||||
}
|
||||
|
||||
template <class T>
|
||||
TEST_FUNC constexpr void test()
|
||||
{
|
||||
// Builtin integer types:
|
||||
test<T, char>();
|
||||
test<T, signed char>();
|
||||
test<T, unsigned char>();
|
||||
|
||||
test<T, short>();
|
||||
test<T, unsigned short>();
|
||||
|
||||
test<T, int>();
|
||||
test<T, unsigned int>();
|
||||
|
||||
test<T, long>();
|
||||
test<T, unsigned long>();
|
||||
|
||||
test<T, long long>();
|
||||
test<T, unsigned long long>();
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
// cstdint types:
|
||||
test<T, std::size_t>();
|
||||
test<T, std::ptrdiff_t>();
|
||||
test<T, std::intptr_t>();
|
||||
test<T, std::uintptr_t>();
|
||||
|
||||
test<T, std::int8_t>();
|
||||
test<T, std::int16_t>();
|
||||
test<T, std::int32_t>();
|
||||
test<T, std::int64_t>();
|
||||
|
||||
test<T, std::uint8_t>();
|
||||
test<T, std::uint16_t>();
|
||||
test<T, std::uint32_t>();
|
||||
test<T, std::uint64_t>();
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<T, __int128_t>();
|
||||
test<T, __uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
}
|
||||
|
||||
TEST_FUNC constexpr bool test()
|
||||
{
|
||||
// Builtin integer types:
|
||||
test<char>();
|
||||
test<signed char>();
|
||||
test<unsigned char>();
|
||||
|
||||
test<short>();
|
||||
test<unsigned short>();
|
||||
|
||||
test<int>();
|
||||
test<unsigned int>();
|
||||
|
||||
test<long>();
|
||||
test<unsigned long>();
|
||||
|
||||
test<long long>();
|
||||
test<unsigned long long>();
|
||||
|
||||
#if !TEST_COMPILER(NVRTC)
|
||||
// cstdint types:
|
||||
test<std::size_t>();
|
||||
test<std::ptrdiff_t>();
|
||||
test<std::intptr_t>();
|
||||
test<std::uintptr_t>();
|
||||
|
||||
test<std::int8_t>();
|
||||
test<std::int16_t>();
|
||||
test<std::int32_t>();
|
||||
test<std::int64_t>();
|
||||
|
||||
test<std::uint8_t>();
|
||||
test<std::uint16_t>();
|
||||
test<std::uint32_t>();
|
||||
test<std::uint64_t>();
|
||||
#endif // !TEST_COMPILER(NVRTC)
|
||||
|
||||
#if _CCCL_HAS_INT128()
|
||||
test<__int128_t>();
|
||||
test<__uint128_t>();
|
||||
#endif // _CCCL_HAS_INT128()
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int arg, char** argv)
|
||||
{
|
||||
test();
|
||||
static_assert(test());
|
||||
return 0;
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user