[CCCL] 瘦身 + 补全: 移除 cudax/python/libcudacxx-tests 冗余文件, 新增 c2h 测试助手 + cmake 构建系统 + 8 个 CUDA thrust examples

变更摘要:
- 删除: cudax/ (783 files, 7.2M) — 实验性组件,竞赛不需要
- 删除: python/ (226 files, 2.0M) — Python 绑定,竞赛不需要
- 删除: libcudacxx/{test,benchmarks,codegen,cmake,share} (4432 files, 31M)
  保留: libcudacxx/include/ (1463 headers, cuda::std 编译依赖)
- 新增: c2h/ (27 files) — CUB Catch2 测试辅助头文件,编译 243 个测试必需
- 新增: cmake/ (29 files) — CCCL 原生 CMake 构建系统
- 新增: thrust/examples/cuda/ (7 files) + cpp_integration/ (1 file)
  async_reduce, custom_temporary_allocation, explicit_cuda_stream,
  global_device_vector, range_view, unwrap_pointer, wrap_pointer, device

结果: cccl_upstream 从 74M→35M (瘦身 53%), 核心内容 100% 保留:
  27/27 tuning headers, 78 benchmarks, 243 tests,
  60 thrust examples, 18 CUB examples, 全部编译头文件
This commit is contained in:
muh-bot
2026-08-03 12:39:26 +00:00
parent a2a5dd8f00
commit 24ef6a91b5
5439 changed files with 0 additions and 719516 deletions

View File

@@ -1,92 +0,0 @@
include(${CMAKE_SOURCE_DIR}/benchmarks/cmake/CCCLBenchmarkRegistry.cmake)
cccl_get_nvbench_helper()
set(benches_root "${CMAKE_CURRENT_LIST_DIR}")
if (NOT CMAKE_BUILD_TYPE STREQUAL "Release")
set(message_type FATAL_ERROR)
if (CCCL_ENABLE_CLANG_TIDY)
# We are here because CI has force-enabled clang-tidy. We must use a debug build for
# this because certain clang-tidy checks (such as out of bounds or clang static
# analyzer) work better when they see assert()'s. In this case we don't actually
# intend to run any of the benchmarks, we just need them to be compilable, so a simple
# warning is enough.
#
# We don't ignore this outright (by making it say, DEBUG or VERBOSE), because it's
# possible that a user may accidentally stumble into enabling the option.
set(message_type WARNING)
endif()
message(${message_type} "libcu++ benchmarks must be built in release mode.")
endif()
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
message(
FATAL_ERROR
"CMAKE_CUDA_ARCHITECTURES must be set to build libcu++ benchmarks."
)
endif()
set(benches_meta_target libcudacxx.all.benches)
add_custom_target(${benches_meta_target})
function(get_recursive_subdirs subdirs)
set(dirs)
file(
GLOB_RECURSE contents
CONFIGURE_DEPENDS
LIST_DIRECTORIES ON
"${CMAKE_CURRENT_LIST_DIR}/bench/*"
)
foreach (test_dir IN LISTS contents)
if (IS_DIRECTORY "${test_dir}")
list(APPEND dirs "${test_dir}")
endif()
endforeach()
set(${subdirs} "${dirs}" PARENT_SCOPE)
endfunction()
create_benchmark_registry()
function(add_bench target_name bench_name bench_src)
set(bench_target ${bench_name})
set(${target_name} ${bench_target} PARENT_SCOPE)
cccl_add_executable(${bench_target} SOURCES "${bench_src}")
target_link_libraries(
${bench_target}
PRIVATE libcudacxx::libcudacxx cccl.nvbench_helper nvbench::main
)
endfunction()
function(add_bench_dir bench_dir)
file(GLOB bench_srcs CONFIGURE_DEPENDS "${bench_dir}/*.cu")
file(RELATIVE_PATH bench_prefix "${benches_root}" "${bench_dir}")
file(TO_CMAKE_PATH "${bench_prefix}" bench_prefix)
string(REPLACE "/" "." bench_prefix "${bench_prefix}")
foreach (bench_src IN LISTS bench_srcs)
# base tuning
get_filename_component(bench_name "${bench_src}" NAME_WLE)
string(PREPEND bench_name "libcudacxx.${bench_prefix}.")
set(base_bench_name "${bench_name}.base")
add_bench(base_bench_target ${base_bench_name} "${bench_src}")
add_dependencies(${benches_meta_target} ${base_bench_target})
target_compile_definitions(${base_bench_target} PRIVATE TUNE_BASE=1)
target_compile_options(
${base_bench_target}
PRIVATE "$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>"
)
# benchmarking
register_cccl_benchmark("${bench_name}" "")
endforeach()
endfunction()
get_recursive_subdirs(subdirs)
foreach (subdir IN LISTS subdirs)
add_bench_dir("${subdir}")
endforeach()

View File

@@ -1,67 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/adjacent_difference.h>
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_difference(cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_difference(
cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin(), ::cuda::std::greater<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,77 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
thrust::device_vector<T> in(elements, thrust::no_init);
thrust::sequence(in.begin(), in.end(), 0);
in[mismatch_point] = in[mismatch_point + 1];
state.add_element_count(elements);
state.add_global_memory_reads<T>(mismatch_point);
state.add_global_memory_writes<T>(0);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
thrust::device_vector<T> in(elements, thrust::no_init);
thrust::sequence(in.begin(), in.end(), 0);
in[mismatch_point] = in[mismatch_point + 1];
state.add_element_count(elements);
state.add_global_memory_reads<T>(mismatch_point);
state.add_global_memory_writes<T>(0);
caching_allocator_t alloc;
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend(), ::cuda::std::greater<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::all_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::any_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,70 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("contiguous")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void random_access(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy(
cuda_policy(alloc, launch),
cuda::counting_iterator<std::size_t>{0},
cuda::counting_iterator{elements},
out.begin()));
});
}
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
.set_name("random_access")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,51 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,67 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::copy_n(cuda_policy(alloc, launch), in.begin(), elements, out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("contiguous")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void random_access(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::copy_n(cuda_policy(alloc, launch), cuda::counting_iterator<std::size_t>{0}, elements, out.begin()));
});
}
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
.set_name("random_access")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,41 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::count(cuda_policy(alloc, launch), in.begin(), in.end(), T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,50 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::count_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,82 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::equal(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_iter")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void range_range(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::equal(
cuda_policy(alloc, launch),
dinput.begin(),
dinput.end(),
cuda::constant_iterator<T>{0},
cuda::constant_iterator<T>{0, elements}));
});
}
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_range")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,68 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::exclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, ::cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,43 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, max_t{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_init_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,40 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::fill(cuda_policy(alloc, launch), output.begin(), output.end(), T{42});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,40 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::fill_n(cuda_policy(alloc, launch), output.begin(), elements, T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,46 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::find(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), val));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::find_if(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::find_if_not(
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::not_fn(cuda::equal_to_value{val})));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,52 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct square_t
{
__device__ void operator()(T& x) const
{
x = x * x;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in(elements, T{1});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
square_t<T> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::for_each(cuda_policy(alloc, launch), in.begin(), in.end(), op);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,52 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct square_t
{
__device__ void operator()(T& x) const
{
x = x * x;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in(elements, T{1});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
square_t<T> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::for_each_n(cuda_policy(alloc, launch), in.begin(), elements, op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,42 +0,0 @@
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
struct generator
{
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
{
return 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::generate(cuda_policy(alloc, launch), output.begin(), output.end(), generator<T>{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,42 +0,0 @@
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
struct generator
{
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
{
return 42;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> output(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::generate_n(cuda_policy(alloc, launch), output.begin(), elements, generator<T>{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,94 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,69 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("range_iter_op_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,85 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
// All-zero is a valid heap; setting one element to 1 forces a violation at
// that child index since its parent is still 0.
template <typename T>
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
{
thrust::fill(d.begin(), d.end(), T{0});
if (violation_point >= 1 && violation_point < d.size())
{
d[violation_point] = T{1};
}
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,85 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/fill.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
// All-zero is a valid heap; setting one element to 1 forces a violation at
// that child index since its parent is still 0.
template <typename T>
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
{
thrust::fill(d.begin(), d.end(), T{0});
if (violation_point >= 1 && violation_point < d.size())
{
d[violation_point] = T{1};
}
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto violation_frac = state.get_float64("ViolationAt");
const auto violation_point = cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
prepare_input(dinput, violation_point);
state.add_global_memory_reads<T>(2 * violation_point);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,50 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <thrust/sequence.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_partitioned(
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), select_op_t{static_cast<T>(mismatch_point)}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,78 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <thrust/sort.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::greater<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,78 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sequence.h>
#include <thrust/sort.h>
#include <cuda/functional>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
thrust::device_vector<T> dinput(elements, thrust::no_init);
thrust::sequence(dinput.begin(), dinput.end(), T{0});
dinput[mismatch_point] = T{-1};
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,66 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/extrema.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,95 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <thrust/merge.h>
#include <thrust/sort.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
const auto entropy = str_to_entropy(state.get_string("Entropy"));
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements, entropy);
thrust::sort(in.begin(), in.begin() + elements_in_lhs);
thrust::sort(in.begin() + elements_in_lhs, in.end());
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::merge(
cuda_policy(alloc, launch),
in.cbegin(),
in.cbegin() + elements_in_lhs,
in.cbegin() + elements_in_lhs,
in.cend(),
out.begin());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"})
.add_int64_axis("InputSizeRatio", {25, 50, 75});
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
const auto entropy = str_to_entropy(state.get_string("Entropy"));
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
thrust::device_vector<T> out(elements);
thrust::device_vector<T> in = generate(elements, entropy);
thrust::sort(in.begin(), in.begin() + elements_in_lhs, ::cuda::std::greater<T>{});
thrust::sort(in.begin() + elements_in_lhs, in.end(), ::cuda::std::greater<T>{});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::merge(
cuda_policy(alloc, launch),
in.cbegin(),
in.cbegin() + elements_in_lhs,
in.cbegin() + elements_in_lhs,
in.cend(),
out.begin(),
::cuda::std::greater<T>{});
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"})
.add_int64_axis("InputSizeRatio", {25, 50, 75});

View File

@@ -1,66 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/extrema.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,82 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::mismatch(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
});
}
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_iter")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
template <typename T>
static void range_range(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::mismatch(
cuda_policy(alloc, launch),
dinput.begin(),
dinput.end(),
cuda::constant_iterator<T>{0},
cuda::constant_iterator<T>{0, elements}));
});
}
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base_range_range")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/functional>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
T val = 1;
// set up input
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto common_prefix = state.get_float64("MismatchAt");
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
thrust::device_vector<T> dinput(elements, thrust::no_init);
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
state.add_global_memory_reads<T>(mismatch_point + 1);
state.add_global_memory_writes<size_t>(1);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::none_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -1,55 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::partition_copy(
cuda_policy(alloc, launch),
input.begin(),
input.end(),
output.begin(),
cuda::std::make_reverse_iterator(output.begin() + elements),
select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -1,41 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::reduce(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,43 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/complex>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements - count);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::remove(cuda_policy(alloc, launch), in.begin(), in.end(), T{42});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,44 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/complex>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements - count);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::remove_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,52 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::remove_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,50 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct is_even
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return static_cast<int>(val) % 2 == 0;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::remove_if(cuda_policy(alloc, launch), in.begin(), in.end(), is_even{});
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,41 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::replace(cuda_policy(alloc, launch), in.begin(), in.end(), 42, 1337);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,42 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::replace_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), 42, 1337));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,52 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == static_cast<T>(42);
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::replace_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), equal_to_42{}, 1337));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,50 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
struct equal_to_42
{
template <class T>
__device__ constexpr bool operator()(const T& val) const noexcept
{
return val == static_cast<T>(42);
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::replace_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}, 1337);
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,42 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/reverse.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::reverse(cuda_policy(alloc, launch), in.begin(), in.end());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,43 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/reverse.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::reverse_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,44 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("MidpointAt");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::rotate(cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});

View File

@@ -1,45 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("MidpointAt");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::rotate_copy(
cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});

View File

@@ -1,43 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("ShiftedTo");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements - midpoint);
state.add_global_memory_writes<T>(elements - midpoint);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::shift_left(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.5, 0.01});

View File

@@ -1,42 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const auto midpoint_float = state.get_float64("ShiftedTo");
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements - midpoint);
state.add_global_memory_writes<T>(elements - midpoint);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::shift_right(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.6, 0.45, 0.01});

View File

@@ -1,81 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/sort.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
thrust::device_vector<T> in = generate(elements, entropy);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"});
struct fake_less
{
template <class T, class U>
[[nodiscard]] _CCCL_API constexpr bool operator()(const T& t, const U& u) const
{
// complex is not less than comparable, so just compare the first element
if constexpr (cuda::std::__is_cpp17_less_than_comparable_v<T, U>)
{
return t < u;
}
else
{
return cuda::std::get<0>(t) < cuda::std::get<0>(u);
}
}
};
template <typename T>
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
thrust::device_vector<T> in = generate(elements, entropy);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end(), fake_less{});
});
}
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(all_types))
.set_name("with_predicate")
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.201"});

View File

@@ -1,48 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of CUDA Experimental in CUDA C++ Core Libraries,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/partition.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
using select_op_t = less_then_t<T>;
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
select_op_t select_op{val};
thrust::device_vector<T> input = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::stable_partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});

View File

@@ -1,72 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/swap.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in1 = generate(elements);
thrust::device_vector<T> in2 = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<T>(2 * elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::swap_ranges(cuda_policy(alloc, launch), in1.begin(), in1.end(), in2.begin());
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_iter_swap(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in1 = generate(elements);
thrust::device_vector<T> in2 = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(2 * elements);
state.add_global_memory_writes<T>(2 * elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
cuda::std::swap_ranges(
cuda_policy(alloc, launch),
cuda::std::reverse_iterator{in1.end()},
cuda::std::reverse_iterator{in1.begin()},
cuda::std::reverse_iterator{in2.end()});
});
}
NVBENCH_BENCH_TYPES(with_iter_swap, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_iter_swap")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,156 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <thrust/iterator/zip_iterator.h>
#include <cuda/functional>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include <nvbench_helper.cuh>
// The benchmarks are inspired by the BabelStream thrust version:
// https://github.com/UoB-HPC/BabelStream/blob/main/src/thrust/ThrustStream.cu
// Modified from BabelStream to also work for integers
constexpr auto startA = 1; // BabelStream: 0.1
constexpr auto startB = 2; // BabelStream: 0.2
constexpr auto startC = 3; // BabelStream: 0.1
constexpr auto startScalar = 4; // BabelStream: 0.4
using element_types = nvbench::type_list<std::int8_t, std::int16_t, float, double, __int128>;
// Different benchmarks use a different number of buffers. H200/B200 can fit 2^31 elements for all benchmarks and types.
// Upstream BabelStream uses 2^25. Allocation failure just skips the benchmark
auto array_size_powers = std::vector<std::int64_t>{25, 31};
template <typename T>
static void mul(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch), c.begin(), c.end(), b.begin(), [=] _CCCL_HOST_DEVICE(const T& ci) {
return ci * scalar;
}));
});
}
NVBENCH_BENCH_TYPES(mul, NVBENCH_TYPE_AXES(element_types))
.set_name("mul")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void add(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(2 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch), a.begin(), a.end(), b.begin(), c.begin(), cuda::std::plus<T>{}));
});
}
NVBENCH_BENCH_TYPES(add, NVBENCH_TYPE_AXES(element_types))
.set_name("add")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void triad(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(2 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch),
b.begin(),
b.end(),
c.begin(),
a.begin(),
[=] _CCCL_HOST_DEVICE(const T& bi, const T& ci) {
return bi + scalar * ci;
}));
});
}
NVBENCH_BENCH_TYPES(triad, NVBENCH_TYPE_AXES(element_types))
.set_name("triad")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);
template <typename T>
static void nstream(nvbench::state& state, nvbench::type_list<T>)
{
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> a(n, startA);
thrust::device_vector<T> b(n, startB);
thrust::device_vector<T> c(n, startC);
state.add_element_count(n);
state.add_global_memory_reads<T>(3 * n);
state.add_global_memory_writes<T>(n);
caching_allocator_t alloc{};
const T scalar = startScalar;
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform(
cuda_policy(alloc, launch),
cuda::make_zip_iterator(a.begin(), b.begin(), c.begin()),
cuda::make_zip_iterator(a.end(), b.end(), c.end()),
a.begin(),
cuda::zip_function{[=] _CCCL_HOST_DEVICE(const T& ai, const T& bi, const T& ci) {
return ai + bi + scalar * ci;
}}));
});
}
NVBENCH_BENCH_TYPES(nstream, NVBENCH_TYPE_AXES(element_types))
.set_name("nstream")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", array_size_powers);

View File

@@ -1,75 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/execution_policy.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include <nvbench_helper.cuh>
template <class InT, class OutT>
struct fib_t
{
__device__ OutT operator()(InT n)
{
OutT t1 = 0;
OutT t2 = 1;
if (n <= 1)
{
return t1;
}
else if (n == 2)
{
return t2;
}
for (InT i = 3; i <= n; ++i)
{
const auto next = t1 + t2;
t1 = t2;
t2 = next;
}
return t2;
}
};
template <typename T>
static void fib(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> input = generate(elements, bit_entropy::_1_000, T{0}, T{42});
thrust::device_vector<T> output(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<nvbench::uint32_t>(elements);
fib_t<T, nvbench::uint32_t> op{};
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(
cuda::std::transform(cuda_policy(alloc, launch), input.cbegin(), input.cend(), output.begin(), op));
});
}
using types = nvbench::type_list<nvbench::uint32_t, nvbench::uint64_t>;
NVBENCH_BENCH_TYPES(fib, NVBENCH_TYPE_AXES(types))
.set_name("fib")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,53 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/transform_scan.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct times_two
{
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
{
return 2 * val;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_exclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, cuda::std::plus<T>{}, times_two<T>{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,79 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/transform_scan.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct times_two
{
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
{
return 2 * val;
}
};
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("basic")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_init(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(elements);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_inclusive_scan(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}, T{42}));
});
}
NVBENCH_BENCH_TYPES(with_init, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_init")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,49 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/iterator>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <typename T>
static void binary(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_reduce(
cuda_policy(alloc, launch),
in.begin(),
in.end(),
cuda::constant_iterator<int>{42},
42,
cuda::std::plus<T>{},
cuda::std::multiplies<T>{}));
});
}
NVBENCH_BENCH_TYPES(binary, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,52 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream>
#include "nvbench_helper.cuh"
template <class T>
struct plus_one
{
template <class U>
[[nodiscard]] __device__ constexpr T operator()(const U val) const noexcept
{
return static_cast<T>(val + 1);
}
};
template <typename T>
static void unary(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in = generate(elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
state.add_global_memory_writes<T>(1);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::transform_reduce(
cuda_policy(alloc, launch), in.begin(), in.end(), 42, cuda::std::plus<T>{}, plus_one<T>{}));
});
}
NVBENCH_BENCH_TYPES(unary, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,88 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/iterator/counting_iterator.h>
#include <thrust/transform.h>
#include <thrust/unique.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream_ref>
#include "nvbench_helper.cuh"
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
template <typename T>
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
{
in.resize(elements);
thrust::transform(
thrust::counting_iterator<std::size_t>(0),
thrust::counting_iterator<std::size_t>(elements),
in.begin(),
[] __device__(std::size_t i) {
// This seems like a clang-tidy bug. Yes we end up converting to double, but the division
// is done entirely in integer land...
return static_cast<T>(i / 2ULL); // NOLINT(bugprone-integer-division)
});
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end(), cuda::std::equal_to<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

View File

@@ -1,90 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <thrust/device_vector.h>
#include <thrust/iterator/counting_iterator.h>
#include <thrust/transform.h>
#include <thrust/unique.h>
#include <cuda/memory_pool>
#include <cuda/std/execution>
#include <cuda/stream_ref>
#include "nvbench_helper.cuh"
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
template <typename T>
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
{
in.resize(elements);
thrust::transform(
thrust::counting_iterator<std::size_t>(0),
thrust::counting_iterator<std::size_t>(elements),
in.begin(),
[] __device__(std::size_t i) {
const auto run = i / 2;
return static_cast<T>(run);
});
}
template <typename T>
static void basic(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique_copy writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
});
}
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("base")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
template <typename T>
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
{
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
thrust::device_vector<T> in;
make_unique_input(in, elements);
thrust::device_vector<T> out(elements, thrust::no_init);
state.add_element_count(elements);
state.add_global_memory_reads<T>(elements);
// unique_copy writes at most elements
state.add_global_memory_writes<T>(elements / 2);
caching_allocator_t alloc{};
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
[&](nvbench::launch& launch) {
do_not_optimize(cuda::std::unique_copy(
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::equal_to<T>{}));
});
}
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
.set_name("with_comp")
.set_type_axes_names({"T{ct}"})
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));

File diff suppressed because it is too large Load Diff

View File

@@ -1,11 +0,0 @@
# Determine if the compiler has GCC-compatible command-line syntax.
if (NOT DEFINED LLVM_COMPILER_IS_GCC_COMPATIBLE)
if (CMAKE_COMPILER_IS_GNUCXX)
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
elseif (MSVC)
set(LLVM_COMPILER_IS_GCC_COMPATIBLE OFF)
elseif ("${CMAKE_CXX_COMPILER_ID}" MATCHES "Clang")
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
endif()
endif()

View File

@@ -1,35 +0,0 @@
# Returns the host triple.
# Invokes config.guess
function(get_host_triple var)
if (MSVC)
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
set(value "x86_64-pc-windows-msvc")
else()
set(value "i686-pc-windows-msvc")
endif()
elseif (MINGW AND NOT MSYS)
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
set(value "x86_64-w64-windows-gnu")
else()
set(value "i686-pc-windows-gnu")
endif()
else(MSVC)
if (CMAKE_HOST_SYSTEM_NAME STREQUAL Windows AND NOT MSYS)
message(WARNING "unable to determine host target triple")
else()
set(config_guess ${LLVM_PATH}/cmake/config.guess)
execute_process(
COMMAND sh ${config_guess}
RESULT_VARIABLE TT_RV
OUTPUT_VARIABLE TT_OUT
OUTPUT_STRIP_TRAILING_WHITESPACE
)
if (NOT TT_RV EQUAL 0)
message(FATAL_ERROR "Failed to execute ${config_guess}")
endif(NOT TT_RV EQUAL 0)
set(value ${TT_OUT})
endif()
endif(MSVC)
set(${var} ${value} PARENT_SCOPE)
endfunction(get_host_triple var)

View File

@@ -1,376 +0,0 @@
function(get_system_libs return_var)
message(AUTHOR_WARNING "get_system_libs no longer needed")
set(${return_var} "" PARENT_SCOPE)
endfunction()
function(link_system_libs target)
message(AUTHOR_WARNING "link_system_libs no longer needed")
endfunction()
# is_llvm_target_library(
# library
# Name of the LLVM library to check
# return_var
# Output variable name
# ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS
# ALL_TARGETS - default looks at the full list of known targets
# INCLUDED_TARGETS - looks only at targets being configured
# OMITTED_TARGETS - looks only at targets that are not being configured
# )
function(is_llvm_target_library library return_var)
cmake_parse_arguments(
ARG
"ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS"
""
""
${ARGN}
)
# Sets variable `return_var' to ON if `library' corresponds to a
# LLVM supported target. To OFF if it doesn't.
set(${return_var} OFF PARENT_SCOPE)
string(TOUPPER "${library}" capitalized_lib)
if (ARG_INCLUDED_TARGETS)
string(TOUPPER "${LLVM_TARGETS_TO_BUILD}" targets)
elseif (ARG_OMITTED_TARGETS)
set(omitted_targets ${LLVM_ALL_TARGETS})
list(REMOVE_ITEM omitted_targets ${LLVM_TARGETS_TO_BUILD})
string(TOUPPER "${omitted_targets}" targets)
else()
string(TOUPPER "${LLVM_ALL_TARGETS}" targets)
endif()
foreach (t ${targets})
if (
capitalized_lib STREQUAL t
OR capitalized_lib STREQUAL "${t}"
OR capitalized_lib STREQUAL "${t}DESC"
OR capitalized_lib STREQUAL "${t}CODEGEN"
OR capitalized_lib STREQUAL "${t}ASMPARSER"
OR capitalized_lib STREQUAL "${t}ASMPRINTER"
OR capitalized_lib STREQUAL "${t}DISASSEMBLER"
OR capitalized_lib STREQUAL "${t}INFO"
OR capitalized_lib STREQUAL "${t}UTILS"
)
set(${return_var} ON PARENT_SCOPE)
break()
endif()
endforeach()
endfunction(is_llvm_target_library)
function(is_llvm_target_specifier library return_var)
is_llvm_target_library(${library} ${return_var} ${ARGN})
string(TOUPPER "${library}" capitalized_lib)
if (NOT ${return_var})
if (
capitalized_lib STREQUAL "ALLTARGETSASMPARSERS"
OR capitalized_lib STREQUAL "ALLTARGETSDESCS"
OR capitalized_lib STREQUAL "ALLTARGETSDISASSEMBLERS"
OR capitalized_lib STREQUAL "ALLTARGETSINFOS"
OR capitalized_lib STREQUAL "NATIVE"
OR capitalized_lib STREQUAL "NATIVECODEGEN"
)
set(${return_var} ON PARENT_SCOPE)
endif()
endif()
endfunction()
macro(llvm_config executable)
cmake_parse_arguments(ARG "USE_SHARED" "" "" ${ARGN})
set(link_components ${ARG_UNPARSED_ARGUMENTS})
if (ARG_USE_SHARED)
# If USE_SHARED is specified, then we link against libLLVM,
# but also against the component libraries below. This is
# done in case libLLVM does not contain all of the components
# the target requires.
#
# Strip LLVM_DYLIB_COMPONENTS out of link_components.
# To do this, we need special handling for "all", since that
# may imply linking to libraries that are not included in
# libLLVM.
if (DEFINED link_components AND DEFINED LLVM_DYLIB_COMPONENTS)
if ("${LLVM_DYLIB_COMPONENTS}" STREQUAL "all")
set(link_components "")
else()
list(REMOVE_ITEM link_components ${LLVM_DYLIB_COMPONENTS})
endif()
endif()
target_link_libraries(${executable} PRIVATE LLVM)
endif()
explicit_llvm_config(${executable} ${link_components})
endmacro(llvm_config)
function(explicit_llvm_config executable)
set(link_components ${ARGN})
llvm_map_components_to_libnames(LIBRARIES ${link_components})
get_target_property(t ${executable} TYPE)
if (t STREQUAL "STATIC_LIBRARY")
target_link_libraries(${executable} INTERFACE ${LIBRARIES})
elseif (
t STREQUAL "EXECUTABLE"
OR t STREQUAL "SHARED_LIBRARY"
OR t STREQUAL "MODULE_LIBRARY"
)
target_link_libraries(${executable} PRIVATE ${LIBRARIES})
else()
# Use plain form for legacy user.
target_link_libraries(${executable} ${LIBRARIES})
endif()
endfunction(explicit_llvm_config)
# This is Deprecated
function(llvm_map_components_to_libraries OUT_VAR)
message(
AUTHOR_WARNING
"Using llvm_map_components_to_libraries() is deprecated. Use llvm_map_components_to_libnames() instead"
)
explicit_map_components_to_libraries(result ${ARGN})
set(${OUT_VAR} ${result} ${sys_result} PARENT_SCOPE)
endfunction(llvm_map_components_to_libraries)
# Expand pseudo-components into real components.
# Does not cover 'native', 'backend', or 'engine' as these require special
# handling. Also does not cover 'all' as we only have a list of the libnames
# available and not a list of the components.
function(llvm_expand_pseudo_components out_components)
set(link_components ${ARGN})
foreach (c ${link_components})
# add codegen, asmprinter, asmparser, disassembler
list(FIND LLVM_TARGETS_TO_BUILD ${c} idx)
if (NOT idx LESS 0)
if (TARGET LLVM${c}CodeGen)
list(APPEND expanded_components "${c}CodeGen")
else()
if (TARGET LLVM${c})
list(APPEND expanded_components "${c}")
else()
message(FATAL_ERROR "Target ${c} is not in the set of libraries.")
endif()
endif()
if (TARGET LLVM${c}AsmPrinter)
list(APPEND expanded_components "${c}AsmPrinter")
endif()
if (TARGET LLVM${c}AsmParser)
list(APPEND expanded_components "${c}AsmParser")
endif()
if (TARGET LLVM${c}Desc)
list(APPEND expanded_components "${c}Desc")
endif()
if (TARGET LLVM${c}Disassembler)
list(APPEND expanded_components "${c}Disassembler")
endif()
if (TARGET LLVM${c}Info)
list(APPEND expanded_components "${c}Info")
endif()
if (TARGET LLVM${c}Utils)
list(APPEND expanded_components "${c}Utils")
endif()
elseif (c STREQUAL "nativecodegen")
if (TARGET LLVM${LLVM_NATIVE_ARCH}CodeGen)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}CodeGen")
endif()
if (TARGET LLVM${LLVM_NATIVE_ARCH}Desc)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Desc")
endif()
if (TARGET LLVM${LLVM_NATIVE_ARCH}Info)
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Info")
endif()
elseif (c STREQUAL "AllTargetsCodeGens")
# Link all the codegens from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}CodeGen)
list(APPEND expanded_components "${t}CodeGen")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsAsmParsers")
# Link all the asm parsers from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}AsmParser)
list(APPEND expanded_components "${t}AsmParser")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsDescs")
# Link all the descs from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Desc)
list(APPEND expanded_components "${t}Desc")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsDisassemblers")
# Link all the disassemblers from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Disassembler)
list(APPEND expanded_components "${t}Disassembler")
endif()
endforeach(t)
elseif (c STREQUAL "AllTargetsInfos")
# Link all the infos from all the targets
foreach (t ${LLVM_TARGETS_TO_BUILD})
if (TARGET LLVM${t}Info)
list(APPEND expanded_components "${t}Info")
endif()
endforeach(t)
else()
list(APPEND expanded_components "${c}")
endif()
endforeach()
set(${out_components} ${expanded_components} PARENT_SCOPE)
endfunction(llvm_expand_pseudo_components out_components)
# This is a variant intended for the final user:
# Map LINK_COMPONENTS to actual libnames.
function(llvm_map_components_to_libnames out_libs)
set(link_components ${ARGN})
if (NOT LLVM_AVAILABLE_LIBS)
# Inside LLVM itself available libs are in a global property.
get_property(LLVM_AVAILABLE_LIBS GLOBAL PROPERTY LLVM_LIBS)
endif()
string(TOUPPER "${LLVM_AVAILABLE_LIBS}" capitalized_libs)
get_property(LLVM_TARGETS_CONFIGURED GLOBAL PROPERTY LLVM_TARGETS_CONFIGURED)
# Generally in our build system we avoid order-dependence. Unfortunately since
# not all targets create the same set of libraries we actually need to ensure
# that all build targets associated with a target are added before we can
# process target dependencies.
if (NOT LLVM_TARGETS_CONFIGURED)
foreach (c ${link_components})
is_llvm_target_specifier(${c} iltl_result ALL_TARGETS)
if (iltl_result)
message(
FATAL_ERROR
"Specified target library before target registration is complete."
)
endif()
endforeach()
endif()
# Expand some keywords:
list(FIND LLVM_TARGETS_TO_BUILD "${LLVM_NATIVE_ARCH}" have_native_backend)
list(FIND link_components "engine" engine_required)
if (NOT engine_required EQUAL -1)
list(FIND LLVM_TARGETS_WITH_JIT "${LLVM_NATIVE_ARCH}" have_jit)
if (NOT have_native_backend EQUAL -1 AND NOT have_jit EQUAL -1)
list(APPEND link_components "jit")
list(APPEND link_components "native")
else()
list(APPEND link_components "interpreter")
endif()
endif()
list(FIND link_components "native" native_required)
if (NOT native_required EQUAL -1)
if (NOT have_native_backend EQUAL -1)
list(APPEND link_components ${LLVM_NATIVE_ARCH})
endif()
endif()
# Translate symbolic component names to real libraries:
llvm_expand_pseudo_components(link_components ${link_components})
foreach (c ${link_components})
get_property(c_rename GLOBAL PROPERTY LLVM_COMPONENT_NAME_${c})
if (c_rename)
set(c ${c_rename})
endif()
if (c STREQUAL "native")
# already processed
elseif (c STREQUAL "backend")
# same case as in `native'.
elseif (c STREQUAL "engine")
# already processed
elseif (c STREQUAL "all")
get_property(all_components GLOBAL PROPERTY LLVM_COMPONENT_LIBS)
list(APPEND expanded_components ${all_components})
else()
# Canonize the component name:
string(TOUPPER "${c}" capitalized)
list(FIND capitalized_libs LLVM${capitalized} lib_idx)
if (lib_idx LESS 0)
# The component is unknown. Maybe is an omitted target?
is_llvm_target_library(${c} iltl_result OMITTED_TARGETS)
if (iltl_result)
# A missing library to a directly referenced omitted target would be bad.
message(
FATAL_ERROR
"Library '${c}' is a direct reference to a target library for an omitted target."
)
else()
# If it is not an omitted target we should assume it is a component
# that hasn't yet been processed by CMake. Missing components will
# cause errors later in the configuration, so we can safely assume
# that this is valid here.
list(APPEND expanded_components LLVM${c})
endif()
else(lib_idx LESS 0)
list(GET LLVM_AVAILABLE_LIBS ${lib_idx} canonical_lib)
list(APPEND expanded_components ${canonical_lib})
endif(lib_idx LESS 0)
endif(c STREQUAL "native")
endforeach(c)
set(${out_libs} ${expanded_components} PARENT_SCOPE)
endfunction()
# Perform a post-order traversal of the dependency graph.
# This duplicates the algorithm used by llvm-config, originally
# in tools/llvm-config/llvm-config.cpp, function ComputeLibsForComponents.
function(expand_topologically name required_libs visited_libs)
list(FIND visited_libs ${name} found)
if (found LESS 0)
list(APPEND visited_libs ${name})
set(visited_libs ${visited_libs} PARENT_SCOPE)
#
get_property(libname GLOBAL PROPERTY LLVM_COMPONENT_NAME_${name})
if (libname)
set(cname LLVM${libname})
elseif (TARGET ${name})
set(cname ${name})
elseif (TARGET LLVM${name})
set(cname LLVM${name})
else()
message(FATAL_ERROR "unknown component ${name}")
endif()
get_property(lib_deps TARGET ${cname} PROPERTY LLVM_LINK_COMPONENTS)
foreach (lib_dep ${lib_deps})
expand_topologically(${lib_dep} "${required_libs}" "${visited_libs}")
set(required_libs ${required_libs} PARENT_SCOPE)
set(visited_libs ${visited_libs} PARENT_SCOPE)
endforeach()
list(APPEND required_libs ${cname})
set(required_libs ${required_libs} PARENT_SCOPE)
endif()
endfunction()
# Expand dependencies while topologically sorting the list of libraries:
function(llvm_expand_dependencies out_libs)
set(expanded_components ${ARGN})
set(required_libs)
set(visited_libs)
foreach (lib ${expanded_components})
expand_topologically(${lib} "${required_libs}" "${visited_libs}")
endforeach()
if (required_libs)
list(REVERSE required_libs)
endif()
set(${out_libs} ${required_libs} PARENT_SCOPE)
endfunction()
function(explicit_map_components_to_libraries out_libs)
llvm_map_components_to_libnames(link_libs ${ARGN})
llvm_expand_dependencies(expanded_components ${link_libs})
# Return just the libraries included in this build:
set(result)
foreach (c ${expanded_components})
if (TARGET ${c})
set(result ${result} ${c})
endif()
endforeach(c)
set(${out_libs} ${result} PARENT_SCOPE)
endfunction(explicit_map_components_to_libraries)

View File

@@ -1,129 +0,0 @@
include(AddFileDependencies)
include(CMakeParseArguments)
function(llvm_replace_compiler_option var old new)
# Replaces a compiler option or switch `old' in `var' by `new'.
# If `old' is not in `var', appends `new' to `var'.
# Example: llvm_replace_compiler_option(CMAKE_CXX_FLAGS_RELEASE "-O3" "-O2")
# If the option already is on the variable, don't add it:
if ("${${var}}" MATCHES "(^| )${new}($| )")
set(n "")
else()
set(n "${new}")
endif()
if ("${${var}}" MATCHES "(^| )${old}($| )")
string(REGEX REPLACE "(^| )${old}($| )" " ${n} " ${var} "${${var}}")
else()
set(${var} "${${var}} ${n}")
endif()
set(${var} "${${var}}" PARENT_SCOPE)
endfunction(llvm_replace_compiler_option)
macro(add_td_sources srcs)
file(GLOB tds *.td)
if (tds)
source_group("TableGen descriptions" FILES ${tds})
set_source_files_properties(${tds} PROPERTIES HEADER_FILE_ONLY ON)
list(APPEND ${srcs} ${tds})
endif()
endmacro(add_td_sources)
function(add_header_files_for_glob hdrs_out glob)
file(GLOB hds ${glob})
set(filtered)
foreach (file ${hds})
# Explicit existence check is necessary to filter dangling symlinks
# out. See https://bugs.gentoo.org/674662.
if (EXISTS ${file})
list(APPEND filtered ${file})
endif()
endforeach()
set(${hdrs_out} ${filtered} PARENT_SCOPE)
endfunction(add_header_files_for_glob)
function(find_all_header_files hdrs_out additional_headerdirs)
add_header_files_for_glob(hds *.h)
list(APPEND all_headers ${hds})
foreach (additional_dir ${additional_headerdirs})
add_header_files_for_glob(hds "${additional_dir}/*.h")
list(APPEND all_headers ${hds})
add_header_files_for_glob(hds "${additional_dir}/*.inc")
list(APPEND all_headers ${hds})
endforeach(additional_dir)
set(${hdrs_out} ${all_headers} PARENT_SCOPE)
endfunction(find_all_header_files)
function(llvm_process_sources OUT_VAR)
cmake_parse_arguments(
ARG
"PARTIAL_SOURCES_INTENDED"
""
"ADDITIONAL_HEADERS;ADDITIONAL_HEADER_DIRS"
${ARGN}
)
set(sources ${ARG_UNPARSED_ARGUMENTS})
if (NOT ARG_PARTIAL_SOURCES_INTENDED)
llvm_check_source_file_list(${sources})
endif()
# This adds .td and .h files to the Visual Studio solution:
add_td_sources(sources)
find_all_header_files(hdrs "${ARG_ADDITIONAL_HEADER_DIRS}")
if (hdrs)
set_source_files_properties(${hdrs} PROPERTIES HEADER_FILE_ONLY ON)
endif()
set_source_files_properties(
${ARG_ADDITIONAL_HEADERS}
PROPERTIES HEADER_FILE_ONLY ON
)
list(APPEND sources ${ARG_ADDITIONAL_HEADERS} ${hdrs})
set(${OUT_VAR} ${sources} PARENT_SCOPE)
endfunction(llvm_process_sources)
function(llvm_check_source_file_list)
cmake_parse_arguments(ARG "" "SOURCE_DIR" "" ${ARGN})
foreach (l ${ARG_UNPARSED_ARGUMENTS})
get_filename_component(fp ${l} REALPATH)
list(APPEND listed ${fp})
endforeach()
if (ARG_SOURCE_DIR)
file(GLOB globbed "${ARG_SOURCE_DIR}/*.c" "${ARG_SOURCE_DIR}/*.cpp")
else()
file(GLOB globbed *.c *.cpp)
endif()
foreach (g ${globbed})
get_filename_component(fn ${g} NAME)
if (ARG_SOURCE_DIR)
set(entry "${g}")
else()
set(entry "${fn}")
endif()
get_filename_component(gp ${g} REALPATH)
# Don't reject hidden files. Some editors create backups in the
# same directory as the file.
if (NOT "${fn}" MATCHES "^\\.")
list(FIND LLVM_OPTIONAL_SOURCES ${entry} idx)
if (idx LESS 0)
list(FIND listed ${gp} idx)
if (idx LESS 0)
if (ARG_SOURCE_DIR)
set(fn_relative "${ARG_SOURCE_DIR}/${fn}")
else()
set(fn_relative "${fn}")
endif()
message(
SEND_ERROR
"Found unknown source file ${fn_relative}
Please update ${CMAKE_CURRENT_LIST_FILE}\n"
)
endif()
endif()
endif()
endforeach()
endfunction(llvm_check_source_file_list)

View File

@@ -1,53 +0,0 @@
# This file defines the `libcudacxx_build_compiler_targets()` function, which
# creates the following interface targets:
#
# libcudacxx.compiler_interface
# - Interface target linked into all targets in the libcudacxx developer build.
# Defines common warning flags, definitions, etc, including those defined in
# the global CCCL targets.
cccl_get_libcudacxx()
function(libcudacxx_build_compiler_targets)
set(cuda_compile_options)
set(cxx_compile_options)
set(cxx_compile_definitions)
# if (CCCL_USE_LIBCXX)
# list(APPEND cxx_compile_options "-stdlib=libc++")
# list(APPEND cxx_compile_definitions "_ALLOW_UNSUPPORTED_LIBCPP=1")
# endif()
# Set test specific flags
list(APPEND cxx_compile_definitions "CCCL_ENABLE_ASSERTIONS")
list(APPEND cxx_compile_definitions "CCCL_IGNORE_DEPRECATED_CPP_DIALECT")
list(
APPEND cxx_compile_definitions
"CCCL_IGNORE_DEPRECATED_DISCARD_MEMORY_HEADER"
)
list(
APPEND cxx_compile_definitions
"CCCL_IGNORE_DEPRECATED_STREAM_REF_HEADER"
)
if (CCCL_ENABLE_TILE)
list(APPEND cuda_compile_options "--enable-tile")
endif()
cccl_build_compiler_interface(
libcudacxx.compiler_flags
"${cuda_compile_options}"
"${cxx_compile_options}"
"${cxx_compile_definitions}"
)
add_library(libcudacxx.compiler_interface INTERFACE)
target_link_libraries(
libcudacxx.compiler_interface
INTERFACE
# order matters here, we need the libcudacxx options to override the cccl options.
cccl.compiler_interface
libcudacxx.compiler_flags
libcudacxx::libcudacxx
)
endfunction()

View File

@@ -1,125 +0,0 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
cccl_get_cudatoolkit()
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.internal_headers)
# Grep all internal headers
file(
GLOB_RECURSE internal_headers
RELATIVE "${libcudacxx_SOURCE_DIR}/include/"
CONFIGURE_DEPENDS
${libcudacxx_SOURCE_DIR}/include/cuda/__*/*.h
${libcudacxx_SOURCE_DIR}/include/cuda/std/__*/*.h
)
# Exclude <cuda/std/__cccl/(prologue|epilogue|visibility).h> from the test
list(
FILTER internal_headers
EXCLUDE
REGEX "__cccl/(prologue|epilogue|visibility)\.h"
)
# headers in `__cuda` are meant to come after the related "cuda" headers so they do not compile on their own
list(FILTER internal_headers EXCLUDE REGEX "__cuda/*")
# generated cuda::ptx headers are not standalone
list(FILTER internal_headers EXCLUDE REGEX "__ptx/instructions/generated")
# don't check nvtx3.h - it's not our header
list(FILTER internal_headers EXCLUDE REGEX ".*/__nvtx/nvtx3.h")
function(libcudacxx_add_internal_header_test_target target_name)
if (NOT ARGN)
return()
endif()
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${ARGN}
)
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
target_link_libraries(
${target_name}
PUBLIC #
libcudacxx.compiler_interface
CUDA::cudart
)
add_dependencies(libcudacxx.test.internal_headers ${target_name})
endfunction()
libcudacxx_add_internal_header_test_target(
libcudacxx.test.internal_headers.base
${internal_headers}
)
# We have fallbacks for some type traits that we want to explicitly test so that they do not bitrot.
set(internal_headers_fallback)
set(internal_headers_fallback_per_header_defines)
foreach (header IN LISTS internal_headers)
# MSVC cannot handle some of the fallbacks.
if ("MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
if (
"${header}" MATCHES "is_base_of"
OR "${header}" MATCHES "is_nothrow_destructible"
OR "${header}" MATCHES "is_polymorphic"
)
continue()
endif()
endif()
file(READ "${libcudacxx_SOURCE_DIR}/include/${header}" header_file)
string(REGEX MATCH "_LIBCUDACXX_[A-Z_]*_FALLBACK" fallback "${header_file}")
if (fallback)
list(APPEND internal_headers_fallback "${header}")
string(
REGEX REPLACE
"([][+.*^$()|?\\\\])"
"\\\\\\1"
header_regex
"${header}"
)
list(
APPEND internal_headers_fallback_per_header_defines
DEFINE
"${fallback}"
"^${header_regex}$"
)
endif()
endforeach()
if (internal_headers_fallback)
cccl_generate_header_tests(
libcudacxx.test.internal_headers.fallback
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${internal_headers_fallback}
PER_HEADER_DEFINES ${internal_headers_fallback_per_header_defines}
)
target_compile_definitions(
libcudacxx.test.internal_headers.fallback
PRIVATE _CCCL_HEADER_TEST
)
target_link_libraries(
libcudacxx.test.internal_headers.fallback
PUBLIC #
libcudacxx.compiler_interface
CUDA::cudart
)
add_dependencies(
libcudacxx.test.internal_headers
libcudacxx.test.internal_headers.fallback
)
endif()

View File

@@ -1,47 +0,0 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.public_headers)
# Grep all public headers
file(
GLOB public_headers
LIST_DIRECTORIES false
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
CONFIGURE_DEPENDS
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
)
# annotated_ptr does not work with clang cuda due to __nv_associate_access_property
if ("Clang" STREQUAL "${CMAKE_CUDA_COMPILER_ID}")
list(REMOVE_ITEM public_headers "annotated_ptr")
endif()
function(libcudacxx_add_public_header_test_target target_name)
if (NOT ARGN)
return()
endif()
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CUDA
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${ARGN}
)
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
add_dependencies(libcudacxx.test.public_headers ${target_name})
endfunction()
libcudacxx_add_public_header_test_target(
libcudacxx.test.public_headers.base
${public_headers}
)

View File

@@ -1,75 +0,0 @@
# For every public header, build a translation unit containing `#include <header>`
# to let the compiler try to figure out warnings in that header if it is not otherwise
# included in tests, and also to verify if the headers are modular enough.
# .inl files are not globbed for, because they are not supposed to be used as public
# entrypoints.
cccl_get_cudatoolkit()
# Meta target for all configs' header builds:
add_custom_target(libcudacxx.test.public_headers_host_only)
add_custom_target(libcudacxx.test.public_headers_host_only_with_ctk)
if (CCCL_ENABLE_TILE) # TODO(miscco): For now only test public headers with tile
return()
endif()
# Grep all public headers
file(
GLOB public_headers_host_only
LIST_DIRECTORIES false
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
CONFIGURE_DEPENDS
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
)
set(public_host_header_cxx_compile_options)
set(public_host_header_cxx_compile_definitions)
# Specifically add libc++ testing if requested to the libcudacxx host suite
if (CCCL_USE_LIBCXX)
list(APPEND public_host_header_cxx_compile_options "-stdlib=libc++")
endif()
function(
libcudacxx_add_public_header_test_host_target
target_name
parent_target
with_ctk
)
cccl_generate_header_tests(
${target_name}
libcudacxx/include
NO_METATARGETS
LANGUAGE CXX
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
HEADERS ${public_headers_host_only}
)
target_compile_definitions(
${target_name}
PRIVATE #
${public_host_header_cxx_compile_definitions}
_CCCL_HEADER_TEST
)
target_compile_options(
${target_name}
PRIVATE ${public_host_header_cxx_compile_options}
)
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
if (with_ctk)
target_link_libraries(${target_name} PUBLIC CUDA::cudart)
endif()
add_dependencies(${parent_target} ${target_name})
endfunction()
libcudacxx_add_public_header_test_host_target(
libcudacxx.test.public_headers_host_only.base
libcudacxx.test.public_headers_host_only
OFF
)
libcudacxx_add_public_header_test_host_target(
libcudacxx.test.public_headers_host_only_with_ctk.base
libcudacxx.test.public_headers_host_only_with_ctk
ON
)

File diff suppressed because it is too large Load Diff

View File

@@ -1,23 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// ignore deprecation warnings
#if defined(__clang__)
# pragma clang diagnostic ignored "-Wdeprecated"
# pragma clang diagnostic ignored "-Wdeprecated-declarations"
#elif defined(_MSC_VER)
# pragma warning (disable: 4996)
#else
# pragma GCC diagnostic ignored "-Wdeprecated"
# pragma GCC diagnostic ignored "-Wdeprecated-declarations"
#endif
// This file tests that the respective header is includable on its own with a cuda compiler
#include <@header@>

View File

@@ -1 +0,0 @@
cccl_add_subdir_helper(libcudacxx)

View File

@@ -1 +0,0 @@
build

View File

@@ -1,51 +0,0 @@
## Codegen adds the following build targets
# libcudacxx.atomics.codegen
# libcudacxx.atomics.codegen.install
## Test targets:
# libcudacxx.test.atomics.codegen.diff
add_executable(codegen EXCLUDE_FROM_ALL codegen.cpp)
target_compile_features(codegen PRIVATE cxx_std_20)
set(
atomic_generated_output
"${libcudacxx_BINARY_DIR}/codegen/cuda_ptx_generated.h"
)
set(
atomic_install_location
"${libcudacxx_SOURCE_DIR}/include/cuda/std/__atomic/functions"
)
add_custom_target(
libcudacxx.atomics.codegen
COMMAND codegen "${atomic_generated_output}"
BYPRODUCTS "${atomic_generated_output}"
)
add_custom_target(
libcudacxx.atomics.codegen.install
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E copy
"${atomic_generated_output}"
"${atomic_install_location}/cuda_ptx_generated.h"
# gersemi: on
DEPENDS libcudacxx.atomics.codegen
BYPRODUCTS "${atomic_install_location}/cuda_ptx_generated.h"
)
add_test(
NAME libcudacxx.test.atomics.codegen.diff
# gersemi: off
COMMAND
"${CMAKE_COMMAND}" -E compare_files
"${atomic_install_location}/cuda_ptx_generated.h"
"${atomic_generated_output}"
# gersemi: on
)
set_tests_properties(
libcudacxx.test.atomics.codegen.diff
PROPERTIES REQUIRED_FILES "${atomic_generated_output}"
)

View File

@@ -1,164 +0,0 @@
#!/usr/bin/env python3
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import argparse
import os
import cccl_paths
docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions")
test = os.path.join(cccl_paths.LIBCUDACXX_TEST_DIR, "libcudacxx", "cuda", "ptx")
src = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "__ptx", "instructions")
ptx_header = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "ptx")
instr_docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions.rst")
def add_docs(ptx_instr, url):
cpp_instr = ptx_instr.replace(".", "_")
underbar = "=" * len(ptx_instr)
(docs / f"{cpp_instr}.rst").write_text(
f""".. _libcudacxx-ptx-instructions-{ptx_instr.replace(".", "-")}:
{ptx_instr}
{underbar}
- PTX ISA:
`{ptx_instr} <{url}>`__
.. include:: generated/{cpp_instr}.rst
"""
)
def add_test(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
dst = test / f"ptx.{ptx_instr}.compile.pass.cpp"
dst.write_text(
f"""//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// UNSUPPORTED: libcpp-has-no-threads
// <cuda/ptx>
#include <cuda/ptx>
#include <cuda/std/utility>
#include "generated/{cpp_instr}.h"
int main(int, char**)
{{
return 0;
}}
"""
)
def add_src(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
(src / f"{cpp_instr}.h").write_text(
f"""// -*- C++ -*-
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA_PTX_{cpp_instr.upper()}_H_
#define _CUDA_PTX_{cpp_instr.upper()}_H_
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__ptx/ptx_dot_variants.h>
#include <cuda/__ptx/ptx_helper_functions.h>
#include <cuda/std/cstdint>
#include <nv/target> // __CUDA_MINIMUM_ARCH__ and friends
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_PTX
#include <cuda/__ptx/instructions/generated/{cpp_instr}.h>
_CCCL_END_NAMESPACE_CUDA_PTX
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA_PTX_{cpp_instr.upper()}_H_
"""
)
def add_ptx_header_include(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
txt = ptx_header.read_text()
# just add as first new include. clang-format will sort it in
idx = txt.index("#include <cuda/__ptx/instructions")
txt = (
txt[:idx]
+ f"""#include <cuda/__ptx/instructions/{cpp_instr}.h>\n"""
+ txt[idx:]
)
ptx_header.write_text(txt)
def add_docs_include(ptx_instr):
cpp_instr = ptx_instr.replace(".", "_")
txt = instr_docs.read_text()
# just add as first new include
idx = txt.index(" instructions/")
txt = txt[:idx] + f" instructions/{cpp_instr}\n" + txt[idx:]
instr_docs.write_text(txt)
if __name__ == "__main__":
parser = argparse.ArgumentParser()
parser.add_argument("ptx_instruction", type=str)
parser.add_argument("url", type=str)
args = parser.parse_args()
ptx_instr = args.ptx_instruction
url = args.url
# Enable using internal urls in the command-line, to be automatically converted to public URLs.
if url.startswith("index.html"):
url = url.replace(
"index.html",
"https://docs.nvidia.com/cuda/parallel-thread-execution/index.html",
)
add_test(ptx_instr)
add_docs(ptx_instr, url)
add_src(ptx_instr)
add_ptx_header_include(ptx_instr)
add_docs_include(ptx_instr)

View File

@@ -1,21 +0,0 @@
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import os
LIBCUDACXX_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
LIBCUDACXX_CMAKE_DIR = os.path.join(LIBCUDACXX_DIR, "cmake")
LIBCUDACXX_CODEGEN_DIR = os.path.join(LIBCUDACXX_DIR, "codegen")
LIBCUDACXX_INCLUDE_DIR = os.path.join(LIBCUDACXX_DIR, "include")
LIBCUDACXX_TEST_DIR = os.path.join(LIBCUDACXX_DIR, "test")
DOCS_DIR = os.path.dirname(LIBCUDACXX_DIR)
DOCS_LIBCUDACXX_DIR = os.path.join(DOCS_DIR, "libcudacxx")

View File

@@ -1,45 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <fstream>
#include <iostream>
#include <ostream>
#include "generators/compare_and_swap.h"
#include "generators/exchange.h"
#include "generators/fence.h"
#include "generators/fetch_ops.h"
#include "generators/header.h"
#include "generators/ld_st.h"
using namespace std::string_literals;
int main(int argc, char** argv)
{
std::fstream filestream;
if (argc == 2)
{
filestream.open(argv[1], filestream.out);
}
std::ostream& stream = filestream.is_open() ? filestream : std::cout;
FormatHeader(stream);
FormatFence(stream);
FormatLoad(stream);
FormatStore(stream);
FormatCompareAndSwap(stream);
FormatExchange(stream);
FormatFetchOps(stream);
FormatTail(stream);
return 0;
}

View File

@@ -1,245 +0,0 @@
#!/usr/bin/env python3
##===----------------------------------------------------------------------===##
##
## Part of libcu++, the C++ Standard Library for your entire system,
## under the Apache License v2.0 with LLVM Exceptions.
## See https://llvm.org/LICENSE.txt for license information.
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
##
##===----------------------------------------------------------------------===##
import datetime
import os
import cccl_paths
PROLOGUE_FILE = os.path.join(
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "prologue.h"
)
EPILOGUE_FILE = os.path.join(
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "epilogue.h"
)
HEADER = f"""\
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) {datetime.datetime.now().year} NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// !!! DO NOT EDIT THIS FILE !!! This file is generated by utils/generate_prologue_epilogue.py.
// NO include guards here (this file is included multiple times)"""
FOOTER = """\
// NO include guards here (this file is included multiple times)
"""
PUSH_POP_MACROS = {
"__declspec modifiers": [
"align",
"allocate",
"allocator",
"appdomain",
"code_seg",
"deprecated",
"dllimport",
"dllexport",
"empty_bases",
"hybrid_patchable",
"jitintrinsic",
"lifetimebound",
"naked",
"noalias",
"noinline",
"noreturn",
"nothrow",
"novtable",
"no_sanitize_address",
"process",
"property",
"restrict",
"safebuffers",
"selectany",
"spectre",
"thread",
"uuid",
],
"[[msvc::attribute]] attributes": [
"msvc",
"flatten",
"forceinline",
"forceinline_calls",
"intrinsic",
"noinline",
"noinline_calls",
"no_tls_guard",
],
"Windows nasty macros": ["min", "max", "interface"],
"sal.h on Windows": ["__valid", "__callback"],
"other macros": ["clang"],
"sys/sysmacros.h on linux": ["major", "minor", "makedev"],
}
def write_section(file, section):
file.write(section)
file.write("\n\n")
def make_prologue(file):
# Write common header.
write_section(file, HEADER)
# Add prologue/epilogue include logic check.
write_section(
file,
"""\
#if defined(_CCCL_PROLOGUE_INCLUDED)
# error \\
"cccl internal error: <cuda/std/__cccl/epilogue.h> must be included before next <cuda/std/__cccl/prologue.h> is reincluded"
#endif
#define _CCCL_PROLOGUE_INCLUDED() 1""",
)
# Add necessary includes.
write_section(
file,
"""\
#include <cuda/std/__cccl/compiler.h>
#include <cuda/std/__cccl/diagnostic.h>
#include <cuda/std/__cccl/dialect.h>""",
)
# Add push macros.
for group_name, macros in PUSH_POP_MACROS.items():
write_section(file, f"// {group_name}")
for macro in macros:
write_section(
file,
f"""\
#if defined({macro})
# pragma push_macro("{macro}")
# undef {macro}
# define _CCCL_POP_MACRO_{macro}
#endif // defined({macro})""",
)
# Add warnings suppressions.
write_section(
file,
'''\
_CCCL_DIAG_PUSH
_CCCL_NV_DIAG_PUSH()
// disable some msvc warnings
// https://github.com/microsoft/STL/blob/master/stl/inc/yvals_core.h#L353
// warning C4100: 'quack': unreferenced formal parameter
// warning C4127: conditional expression is constant
// warning C4180: qualifier applied to function type has no meaning; ignored
// warning C4197: 'purr': top-level volatile in cast is ignored
// warning C4324: 'roar': structure was padded due to alignment specifier
// warning C4455: literal suffix identifiers that do not start with an underscore are reserved
// warning C4503: 'hum': decorated name length exceeded, name was truncated
// warning C4522: 'woof' : multiple assignment operators specified
// warning C4668: 'meow' is not defined as a preprocessor macro, replacing with '0' for '#if/#elif'
// warning C4800: 'boo': forcing value to bool 'true' or 'false' (performance warning)
// warning C4996: 'meow': was declared deprecated
_CCCL_DIAG_SUPPRESS_MSVC(4100 4127 4180 4197 4296 4324 4455 4503 4522 4668 4800 4996)
// Suppress compiler warnings about C++ extensions.
#if _CCCL_COMPILER(GCC, >=, 12)
_CCCL_DIAG_SUPPRESS_GCC("-Wc++20-extensions")
_CCCL_DIAG_SUPPRESS_GCC("-Wc++23-extensions")
#endif // _CCCL_COMPILER(GCC, >=, 12)
#if _CCCL_COMPILER(GCC, >=, 14)
_CCCL_DIAG_SUPPRESS_GCC("-Wc++26-extensions")
#endif // _CCCL_COMPILER(GCC, >=, 14)
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++20-extensions")
#if _CCCL_COMPILER(CLANG, >=, 17)
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++23-extensions")
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++26-extensions")
#else // ^^^ _CCCL_COMPILER(CLANG, >=, 17) ^^^ / vvv _CCCL_COMPILER(CLANG, <, 17) vvv
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++2b-extensions")
#endif // ^^^ _CCCL_COMPILER(CLANG, <, 17) ^^^
// Suppress `if consteval`-related warnings.
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_nonstandard)
_CCCL_DIAG_SUPPRESS_NVHPC(is_constant_evaluated_in_nonconstexpr_context)
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_in_nonconstexpr_function)
_CCCL_DIAG_SUPPRESS_NVCC(3215) // "if consteval" and "if not consteval" are not standard in this mode
_CCCL_DIAG_SUPPRESS_NVCC(3206) // "if consteval" and "if not consteval" are meaningless in a non-constexpr function
_CCCL_DIAG_SUPPRESS_NVCC(3060) // call to __builtin_is_constant_evaluated appearing in a non-constexpr function always
// produces "false"''',
)
# Write the common footer.
file.write(FOOTER)
def make_epilogue(file):
# Write common header.
write_section(file, HEADER)
# Write includes.
write_section(
file,
"""\
#include <cuda/std/__cccl/compiler.h>
#include <cuda/std/__cccl/diagnostic.h>""",
)
# Add prologue/epilogue include logic check.
write_section(
file,
"""\
#if !defined(_CCCL_PROLOGUE_INCLUDED)
# error "cccl internal error: <cuda/std/__cccl/prologue.h> must be included before <cuda/std/__cccl/epilogue.h>"
#endif
#undef _CCCL_PROLOGUE_INCLUDED""",
)
# Pop warning suppressions.
write_section(
file,
"""\
_CCCL_NV_DIAG_POP()
_CCCL_DIAG_POP""",
)
# Add pop macros.
for group_name, macros in PUSH_POP_MACROS.items():
write_section(file, f"// {group_name}")
for macro in macros:
write_section(
file,
f"""\
#if defined({macro})
# error \\
"cccl internal error: macro `{macro}` was redefined between <cuda/std/__cccl/prologue.h> and <cuda/std/__cccl/epilogue.h>"
#elif defined(_CCCL_POP_MACRO_{macro})
# pragma pop_macro("{macro}")
# undef _CCCL_POP_MACRO_{macro}
#endif""",
)
# Write the common footer.
file.write(FOOTER)
if __name__ == "__main__":
with open(PROLOGUE_FILE, "w") as file:
make_prologue(file)
with open(EPILOGUE_FILE, "w") as file:
make_epilogue(file)

View File

@@ -1,201 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef COMPARED_AND_SWAP_H
#define COMPARED_AND_SWAP_H
#include <format>
#include <string>
#include "definitions.h"
inline void FormatCompareAndSwap(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_swap_memory_order_dispatch(_Fn& __cuda_cas, int __success_memorder, int __failure_memorder, _Sco) {
bool __res = false;
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __res = __cuda_cas(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __res = __cuda_cas(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __res = __cuda_cas(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
return __res;
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Type Constraint
// 3 - Memory Order
// 4 - Memory Order function tag
// 5 - Scope Constraint
// 6 - Scope function tag
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b CAS is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90, (),
NV_ANY_TARGET, (__atomic_cas_128b_unsupported_before_SM_90();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
.reg .b128 _v;
mov.b128 _d, {{%3, %4}};
mov.b128 _v, {{%5, %6}};
atom.cas{3}{5}.b128 _d,[%2],_d,_v;
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr), "l"(__cmp.__x),"l"(__cmp.__y), "l"(__op.__x),"l"(__op.__y) : "memory"); return __dst.__x == __cmp.__x && __dst.__y == __cmp.__y; }})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{ asm volatile("atom.cas{3}{5}.{0}{1} %0,[%1],%2,%3;" : "={2}"(__dst) : "l"(__ptr), "{2}"(__cmp), "{2}"(__op) : "memory"); return __dst == __cmp; }})XXX";
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr size_t supported_sizes[] = {
32,
64,
128,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
if (size == 2 && type != Operand::Bit)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
else
{
out << std::format(
asm_intrinsic_format,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_compare_exchange {
_Type* __ptr;
_Type* __exp;
_Type* __des;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE bool operator()(_Atomic_Memorder) {
return __cuda_atomic_compare_exchange(__ptr, *__exp, *__exp, *__des, _Atomic_Memorder{}, _Tag{}, _Sco{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
bool __res = false;
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type volatile* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
bool __res = false;
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
}
)XXX";
}
#endif // COMPARED_AND_SWAP_H

View File

@@ -1,192 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef DEFINITIONS_H
#define DEFINITIONS_H
#include <format>
#include <map>
#include <string>
#include <type_traits>
#include <vector>
enum class Mmio
{
Disabled,
Enabled,
};
inline std::string mmio(Mmio m)
{
static const char* mmio_map[]{
"",
".mmio",
};
return mmio_map[std::underlying_type_t<Mmio>(m)];
}
inline std::string mmio_tag(Mmio m)
{
static const char* mmio_map[]{
"__atomic_cuda_mmio_disable",
"__atomic_cuda_mmio_enable",
};
return mmio_map[std::underlying_type_t<Mmio>(m)];
}
enum class Operand
{
Floating,
Unsigned,
Signed,
Bit,
};
inline std::string operand(Operand op)
{
static std::map op_map = {
std::pair{Operand::Floating, "f"},
std::pair{Operand::Unsigned, "u"},
std::pair{Operand::Signed, "s"},
std::pair{Operand::Bit, "b"},
};
return op_map[op];
}
inline std::string operand_proxy_type(Operand op, size_t sz)
{
if (op == Operand::Floating)
{
if (sz == 32)
{
return {"float"};
}
else
{
return {"double"};
}
}
else if (op == Operand::Signed)
{
return std::format("int{}_t", sz);
}
// Binary and unsigned can be the same proxy_type
return std::format("uint{}_t", sz);
}
inline std::string constraints(Operand op, size_t sz)
{
static std::map constraint_map = {
std::pair{32,
std::map{
std::pair{Operand::Bit, "r"},
std::pair{Operand::Unsigned, "r"},
std::pair{Operand::Signed, "r"},
std::pair{Operand::Floating, "f"},
}},
std::pair{64,
std::map{
std::pair{Operand::Bit, "l"},
std::pair{Operand::Unsigned, "l"},
std::pair{Operand::Signed, "l"},
std::pair{Operand::Floating, "d"},
}},
std::pair{128,
std::map{
std::pair{Operand::Bit, "l"},
std::pair{Operand::Unsigned, "l"},
std::pair{Operand::Signed, "l"},
std::pair{Operand::Floating, "d"},
}},
};
if (sz == 16)
{
return {"h"};
}
else
{
return constraint_map[sz][op];
}
}
enum class Semantic
{
Relaxed,
Release,
Acquire,
Acq_Rel,
Seq_Cst,
Volatile,
};
inline std::string semantic(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, ".relaxed"},
std::pair{Semantic::Release, ".release"},
std::pair{Semantic::Acquire, ".acquire"},
std::pair{Semantic::Acq_Rel, ".acq_rel"},
std::pair{Semantic::Seq_Cst, ".sc"},
std::pair{Semantic::Volatile, ""},
};
return sem_map[sem];
}
inline std::string semantic_tag(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, "__atomic_cuda_relaxed"},
std::pair{Semantic::Release, "__atomic_cuda_release"},
std::pair{Semantic::Acquire, "__atomic_cuda_acquire"},
std::pair{Semantic::Acq_Rel, "__atomic_cuda_acq_rel"},
std::pair{Semantic::Seq_Cst, "__atomic_cuda_seq_cst"},
std::pair{Semantic::Volatile, "__atomic_cuda_volatile"},
};
return sem_map[sem];
}
enum class Scope
{
Thread,
Warp,
CTA,
Cluster,
GPU,
System,
};
inline std::string scope(Scope sco)
{
static std::map sco_map = {
std::pair{Scope::Thread, ""},
std::pair{Scope::Warp, ""},
std::pair{Scope::CTA, ".cta"},
std::pair{Scope::Cluster, ".cluster"},
std::pair{Scope::GPU, ".gpu"},
std::pair{Scope::System, ".sys"},
};
return sco_map[sco];
}
inline std::string scope_tag(Scope sco)
{
static std::map sco_map = {
std::pair{Scope::Thread, "__thread_scope_thread_tag"},
std::pair{Scope::Warp, ""},
std::pair{Scope::CTA, "__thread_scope_block_tag"},
std::pair{Scope::Cluster, "__thread_scope_cluster_tag"},
std::pair{Scope::GPU, "__thread_scope_device_tag"},
std::pair{Scope::System, "__thread_scope_system_tag"},
};
return sco_map[sco];
}
#endif // DEFINITIONS_H

View File

@@ -1,197 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef EXCHANGE_H
#define EXCHANGE_H
#include <format>
#include <string>
#include "definitions.h"
inline void FormatExchange(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_exchange_memory_order_dispatch(_Fn& __cuda_exch, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __cuda_exch(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __cuda_exch(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_exch(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Type Constraint
// 3 - Memory Order
// 4 - Memory Order function tag
// 5 - Scope Constraint
// 6 - Scope function tag
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b exchange is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_90, (),
NV_ANY_TARGET, (__atomic_exchange_128b_unsupported_before_SM_90();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
.reg .b128 _v;
mov.b128 _v, {{%3, %4}};
atom.exch{3}{5}.b128 _d,[%2],_v;
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__old.__x),"=l"(__old.__y) : "l"(__ptr), "l"(__new.__x),"l"(__new.__y) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
{{ asm volatile("atom.exch{3}{5}.{0}{1} %0,[%1],%2;" : "={2}"(__old) : "l"(__ptr), "{2}"(__new) : "memory"); }})XXX";
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr size_t supported_sizes[] = {
32,
64,
128,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
if (size == 2 && type != Operand::Bit)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
else
{
out << std::format(
asm_intrinsic_format,
operand(type),
size,
constraints(type, size),
semantic(sem),
semantic_tag(sem),
scope(sco),
scope_tag(sco));
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_exchange {
_Type* __ptr;
_Type* __old;
_Type* __new;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_exchange(__ptr, *__old, *__new, _Atomic_Memorder{}, _Tag{}, _Sco{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type volatile* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
}
)XXX";
}
#endif // EXCHANGE_H

View File

@@ -1,110 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef FENCE_H
#define FENCE_H
#include <format>
#include <string>
#include "definitions.h"
inline std::string membar_scope(Scope sco)
{
static std::map scope_map{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
return scope_map[sco];
}
inline void FormatFence(std::ostream& out)
{
// Argument ID Reference
// 0 - Membar scope tag
// 1 - Membar scope
constexpr auto intrinsic_membar = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_membar({0})
{{ asm volatile("membar{1};" ::: "memory"); }})XXX";
const std::map membar_scopes{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
for (const auto& sco : membar_scopes)
{
out << std::format(intrinsic_membar, scope_tag(sco.first), sco.second);
}
// Argument ID Reference
// 0 - Fence scope tag
// 1 - Fence scope
// 2 - Fence order tag
// 3 - Fence order
constexpr auto intrinsic_fence = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_fence({0}, {2})
{{ asm volatile("fence{1}{3};" ::: "memory"); }})XXX";
const Scope fence_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
const Semantic fence_semantics[] = {
Semantic::Acq_Rel,
Semantic::Seq_Cst,
};
for (const auto& sco : fence_scopes)
{
for (const auto& sem : fence_semantics)
{
out << std::format(intrinsic_fence, scope_tag(sco), semantic(sem), semantic_tag(sem), scope(sco));
}
}
out << "\n"
<< R"XXX(
template <typename _Sco>
static inline _CCCL_DEVICE void __atomic_thread_fence_cuda(int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); break;
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_fence(_Sco{}, __atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
}
#endif // FENCE_H

View File

@@ -1,219 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef FETCH_OPS_H
#define FETCH_OPS_H
#include <array>
#include <format>
#include <string>
#include "definitions.h"
inline std::string fetch_op_skip_v(std::string fetch_op)
{
if (fetch_op == "add")
{
return "constexpr auto __skip_v = __atomic_ptr_skip_t<_Type>::__skip;";
}
return "constexpr auto __skip_v = 1;";
}
inline void FormatFetchOps(std::ostream& out)
{
const std::vector arithmetic_types = {
Operand::Floating,
Operand::Unsigned,
Operand::Signed,
};
const std::vector minmax_types = {
Operand::Unsigned,
Operand::Signed,
};
const std::vector bitwise_types = {Operand::Bit};
const std::map op_support_map{
std::pair{std::string{"add"}, std::pair{arithmetic_types, std::string{"arithmetic"}}},
std::pair{std::string{"min"}, std::pair{minmax_types, std::string{"minmax"}}},
std::pair{std::string{"max"}, std::pair{minmax_types, std::string{"minmax"}}},
std::pair{std::string{"or"}, std::pair{bitwise_types, std::string{"bitwise"}}},
std::pair{std::string{"xor"}, std::pair{bitwise_types, std::string{"bitwise"}}},
std::pair{std::string{"and"}, std::pair{bitwise_types, std::string{"bitwise"}}},
};
// Memory order dispatcher
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_fetch_memory_order_dispatch(_Fn& __cuda_fetch, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_acquire{}); break;
case __ATOMIC_ACQ_REL: __cuda_fetch(__atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELEASE: __cuda_fetch(__atomic_cuda_release{}); break;
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_fetch(__atomic_cuda_volatile{}); break;
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Atomic Operation
// 1 - Operand Type
// 2 - Operand Size
// 3 - Type Constraint
// 4 - Memory Order
// 5 - Memory Order function tag
// 6 - Scope Constraint
// 7 - Scope function tag
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_fetch_{0}(
_Type* __ptr, _Type& __dst, _Type __op, {5}, __atomic_cuda_operand_{1}{2}, {7})
{{ asm volatile("atom.{0}{4}{6}.{1}{2} %0,[%1],%2;" : "={3}"(__dst) : "l"(__ptr), "{3}"(__op) : "memory"); }})XXX";
// 0 - Atomic Operation
// 1 - Operand type constraint
// 2 - Pointer op skip_v
constexpr auto fetch_bind_invoke = R"XXX(
template <typename _Type, typename _Tag, typename _Sco>
struct __cuda_atomic_bind_fetch_{0} {{
_Type* __ptr;
_Type* __dst;
_Type* __op;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {{
__cuda_atomic_fetch_{0}(__ptr, *__dst, *__op, _Atomic_Memorder{{}}, _Tag{{}}, _Sco{{}});
}}
}};
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
{{
{2}
__op = __op * __skip_v;
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
_Type __dst{{}};
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
return __dst;
}}
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
{{
{2}
__op = __op * __skip_v;
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
_Type __dst{{}};
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
return __dst;
}}
)XXX";
constexpr size_t supported_sizes[] = {
32,
64,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Release,
Semantic::Acq_Rel,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
for (auto& op_kp : op_support_map)
{
const auto& op_name = op_kp.first;
const auto& op_type_kp = op_kp.second;
const auto& type_list = op_type_kp.first;
const auto& deduction = op_type_kp.second;
for (auto type : type_list)
{
for (auto size : supported_sizes)
{
const std::string proxy_type = operand_proxy_type(type, size);
for (auto sco : supported_scopes)
{
for (auto sem : supported_semantics)
{
// There is no atom.add.s64
if (op_name == "add" && type == Operand::Signed && size == 64)
{
continue;
}
out << std::format(
asm_intrinsic_format,
/* 0 */ op_name,
/* 1 */ operand(type),
/* 2 */ size,
/* 3 */ constraints(type, size),
/* 4 */ semantic(sem),
/* 5 */ semantic_tag(sem),
/* 6 */ scope(sco),
/* 7 */ scope_tag(sco));
}
}
}
}
out << "\n" << std::format(fetch_bind_invoke, op_name, deduction, fetch_op_skip_v(op_name));
}
out << R"XXX(
template <class _Type, class _Up, class _Sco>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
{
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
}
template <class _Type, class _Up, class _Sco>
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
{
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
}
)XXX";
}
#endif // FETCH_OPS_H

View File

@@ -1,88 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef HEADER_H
#define HEADER_H
#include <string>
inline void FormatHeader(std::ostream& out)
{
constexpr auto header = R"XXX(//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
// This is an autogenerated file, we want to ensure that it contains exactly the contents we want to generate
// clang-format off
#ifndef _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
#define _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/cassert>
#include <cuda/std/cstdint>
#include <cuda/std/__type_traits/enable_if.h>
#include <cuda/std/__type_traits/is_signed.h>
#include <cuda/std/__type_traits/is_unsigned.h>
#include <cuda/std/__atomic/scopes.h>
#include <cuda/std/__atomic/order.h>
#include <cuda/std/__atomic/functions/common.h>
#include <cuda/std/__atomic/functions/cuda_ptx_generated_helper.h>
#include <cuda/std/__atomic/functions/cuda_local.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_STD
#if _CCCL_CUDA_COMPILATION()
extern "C" _CCCL_DEVICE void __atomic_cas_128b_unsupported_before_SM_90();
extern "C" _CCCL_DEVICE void __atomic_exchange_128b_unsupported_before_SM_90();
extern "C" _CCCL_DEVICE void __atomic_ldst_128b_unsupported_before_SM_70();
)XXX";
out << header;
}
inline void FormatTail(std::ostream& out)
{
constexpr auto tail = R"XXX(
#endif // _CCCL_CUDA_COMPILATION()
_CCCL_END_NAMESPACE_CUDA_STD
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
// clang-format on
)XXX";
out << tail;
}
#endif // HEADER_H

View File

@@ -1,407 +0,0 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef LD_ST_H
#define LD_ST_H
#include <format>
#include <string>
#include "definitions.h"
inline std::string semantic_ld_st(Semantic sem)
{
static std::map sem_map = {
std::pair{Semantic::Relaxed, ".relaxed"},
std::pair{Semantic::Release, ".release"},
std::pair{Semantic::Acquire, ".acquire"},
std::pair{Semantic::Volatile, ".volatile"},
};
return sem_map[sem];
}
inline std::string scope_ld_st(Semantic sem, Scope sco)
{
if (sem == Semantic::Volatile)
{
return "";
}
return scope(sco);
}
inline void FormatLoad(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_load_memory_order_dispatch(_Fn &__cuda_load, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_acquire{}); break;
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Constraint
// 3 - Memory order
// 4 - Memory order semantic
// 5 - Scope tag
// 6 - Scope semantic
// 7 - Mmio tag
// 8 - Mmio semantic
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_load(
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (),
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
)
asm volatile(R"YYY(
{{
.reg .b128 _d;
ld{8}{4}{6}.b128 _d,[%2];
mov.b128 {{%0, %1}}, _d;
}}
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_load(
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{ asm volatile("ld{8}{4}{6}.{0}{1} %0,[%1];" : "={2}"(__dst) : "l"(__ptr) : "memory"); }})XXX";
constexpr size_t supported_sizes[] = {
16,
32,
64,
128,
};
constexpr Operand supported_types[] = {
Operand::Bit,
Operand::Floating,
Operand::Unsigned,
Operand::Signed,
};
constexpr Semantic supported_semantics[] = {
Semantic::Acquire,
Semantic::Relaxed,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
constexpr Mmio mmio_states[] = {
Mmio::Disabled,
Mmio::Enabled,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
for (auto mm : mmio_states)
{
if (size == 16 && type == Operand::Floating)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
else
{
out << std::format(
asm_intrinsic_format,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
struct __cuda_atomic_bind_load {
const _Type* __ptr;
_Type* __dst;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_load(__ptr, *__dst, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type* __ptr, _Type& __dst, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(__ptr);
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type volatile* __ptr, _Type& __dst, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
}
)XXX";
}
inline void FormatStore(std::ostream& out)
{
out << R"XXX(
template <class _Fn, class _Sco>
static inline _CCCL_DEVICE void __cuda_atomic_store_memory_order_dispatch(_Fn &__cuda_store, int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_RELEASE: __cuda_store(__atomic_cuda_release{}); break;
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_relaxed{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_RELEASE: [[fallthrough]];
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_volatile{}); break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
// Argument ID Reference
// 0 - Operand Type
// 1 - Operand Size
// 2 - Constraint
// 3 - Memory order
// 4 - Memory order semantic
// 5 - Scope tag
// 6 - Scope semantic
// 7 - Mmio tag
// 8 - Mmio semantic
constexpr auto asm_intrinsic_format_128 = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_store(
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (),
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
)
asm volatile(R"YYY(
{{
.reg .b128 _v;
mov.b128 _v, {{%1, %2}};
st{8}{4}{6}.b128 [%0],_v;
}}
)YYY" :: "l"(__ptr), "l"(__val.__x),"l"(__val.__y) : "memory");
}})XXX";
constexpr auto asm_intrinsic_format = R"XXX(
template <class _Type>
static inline _CCCL_DEVICE void __cuda_atomic_store(
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
{{ asm volatile("st{8}{4}{6}.{0}{1} [%0],%1;" :: "l"(__ptr), "{2}"(__val) : "memory"); }})XXX";
constexpr size_t supported_sizes[] = {
16,
32,
64,
128,
};
constexpr Operand supported_types[] = {
Operand::Bit,
};
constexpr Semantic supported_semantics[] = {
Semantic::Release,
Semantic::Relaxed,
Semantic::Volatile,
};
constexpr Scope supported_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
constexpr Mmio mmio_states[] = {
Mmio::Disabled,
Mmio::Enabled,
};
for (auto size : supported_sizes)
{
for (auto type : supported_types)
{
for (auto sem : supported_semantics)
{
for (auto sco : supported_scopes)
{
for (auto mm : mmio_states)
{
if (size == 16 && type == Operand::Floating)
{
continue;
}
if (size == 128 && type != Operand::Bit)
{
continue;
}
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
{
continue;
}
if (size == 128)
{
out << std::format(
asm_intrinsic_format_128,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
else
{
out << std::format(
asm_intrinsic_format,
/* 0 */ operand(type),
/* 1 */ size,
/* 2 */ constraints(type, size),
/* 3 */ semantic_tag(sem),
/* 4 */ semantic_ld_st(sem),
/* 5 */ scope_tag(sco),
/* 6 */ scope_ld_st(sem, sco),
/* 7 */ mmio_tag(mm),
/* 8 */ mmio(mm));
}
}
}
}
}
}
out << "\n"
<< R"XXX(
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
struct __cuda_atomic_bind_store {
_Type* __ptr;
_Type* __val;
template <typename _Atomic_Memorder>
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
__cuda_atomic_store(__ptr, *__val, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
}
};
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_store_cuda(_Type* __ptr, _Type& __val, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
}
template <class _Type, class _Sco>
static inline _CCCL_DEVICE void __atomic_store_cuda(volatile _Type* __ptr, _Type& __val, int __memorder, _Sco)
{
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
}
)XXX";
}
#endif // LD_ST_H

View File

@@ -1,34 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB entry point for CUDA C++ Core Libraries pretty printers.
Requires Python 3.12 or newer.
"""
from __future__ import annotations
import sys
from pathlib import Path
import gdb
_SCRIPT_DIRECTORY = str(Path(__file__).resolve().parent)
if _SCRIPT_DIRECTORY not in sys.path:
sys.path.insert(0, _SCRIPT_DIRECTORY)
import buffer # noqa: E402
import memory_resource # noqa: E402
import std_array # noqa: E402
_PRINTERS = (memory_resource, buffer, std_array)
def register() -> None:
"""Register every CCCL GDB pretty printer."""
for printer in _PRINTERS:
printer.register(gdb)
register()

View File

@@ -1,127 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for cuda::buffer."""
from __future__ import annotations
from collections.abc import Iterator
from types import ModuleType
import memory_resource
import gdb
import gdb.printing
# GDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
# parser does not necessarily import the enum constants. This is the value of
# cudaMemcpyDefault from the CUDA Runtime API.
_CUDA_MEMCPY_DEFAULT = 4
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_cuda_buffer(value_type: gdb.Type) -> bool:
# strip_typedefs resolves aliases that can hide accessibility properties.
value_type = value_type.strip_typedefs().unqualified()
template_name = _template_name(value_type)
return (
template_name.startswith("cuda::")
and template_name.rsplit("::", 1)[-1] == "buffer"
)
class BufferPrinter:
"""Expose cuda::buffer metadata and elements to GDB."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
self.type = value.type.strip_typedefs().unqualified()
self.type_name = memory_resource.public_type_name(self.type)
self.value_type = self.type.template_argument(0)
storage = value["__buf_"]
self.memory_resource = storage["__mr_"]
self.stream = int(storage["__stream_"]["__stream"])
self.size = int(storage["__count_"])
self.alignment = int(storage["__alignment_"])
raw_address = int(storage["__buf_"])
self.data_address = (raw_address + self.alignment - 1) & ~(self.alignment - 1)
host_accessible = "host_accessible" in self.type_name
device_accessible = "device_accessible" in self.type_name
if host_accessible and device_accessible:
self.accessibility = "host/device"
elif device_accessible:
self.accessibility = "device"
elif host_accessible:
self.accessibility = "host"
else:
self.accessibility = "unknown"
self.host_copy: gdb.Value | None = None
self._copy_to_host()
def __del__(self) -> None:
self.clear()
def clear(self) -> None:
"""Release state staged in the inferior for synthetic children."""
if self.host_copy is None:
return
try:
gdb.parse_and_eval(f"(void)free((void*){int(self.host_copy):#x})")
except gdb.error:
pass
self.host_copy = None
def _copy_to_host(self) -> None:
if self.size == 0:
return
byte_count = self.size * self.value_type.sizeof
self.host_copy = gdb.parse_and_eval(f"(void*)malloc({byte_count})")
host_address = int(self.host_copy)
status = gdb.parse_and_eval(
"(int)cudaMemcpy((void*)"
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
f"{_CUDA_MEMCPY_DEFAULT})"
)
if int(status) != 0:
self.clear()
def children(self) -> Iterator[tuple[str, gdb.Value]]:
if self.host_copy is None:
return
pointer = self.host_copy.cast(self.value_type.pointer())
for index in range(self.size):
yield f"[{index}]", (pointer + index).dereference()
def to_string(self) -> str:
resource = memory_resource.memory_resource_description(self.memory_resource)
return (
f"{self.type_name} mr={resource}, stream={self.stream:#x}, "
f"size={self.size}, align={self.alignment}, "
f"data={self.data_address:#x} ({self.accessibility})"
)
class BufferPrinterLookup(gdb.printing.PrettyPrinter):
"""Select the cuda::buffer printer by its public class name."""
def __init__(self) -> None:
super().__init__("cuda::buffer")
def __call__(self, value: gdb.Value) -> BufferPrinter | None:
if _is_cuda_buffer(value.type):
return BufferPrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register the cuda::buffer printer with GDB."""
gdb.printing.register_pretty_printer(objfile, BufferPrinterLookup(), replace=True)

View File

@@ -1,76 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for CUDA type-erased memory resources."""
from __future__ import annotations
import re
from types import ModuleType
import gdb
import gdb.printing
_ABI_NAMESPACE_PATTERN = re.compile(r"::__(?:\d+|version_bump_ver\d+_)(?=::)")
_RESOURCE_NAMES = frozenset(
{"any_resource", "any_synchronous_resource", "basic_any_resource"}
)
def public_type_name(value_type: gdb.Type) -> str:
"""Return a type name without CUDA ABI inline namespaces."""
return _ABI_NAMESPACE_PATTERN.sub("", str(value_type))
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_memory_resource(value_type: gdb.Type) -> bool:
value_type = value_type.strip_typedefs().unqualified()
type_name = public_type_name(value_type)
template_name = _template_name(value_type)
return (
type_name.startswith("cuda::mr::")
and template_name.rsplit("::", 1)[-1] in _RESOURCE_NAMES
)
def memory_resource_description(value: gdb.Value) -> str:
value_type = value.type.strip_typedefs().unqualified()
type_name = public_type_name(value_type)
try:
address = int(value.address)
except (gdb.error, TypeError):
return type_name
return f"{type_name} @ {address:#x}"
class MemoryResourcePrinter:
"""Summarize a CUDA type-erased memory resource."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
def to_string(self) -> str:
return memory_resource_description(self.value)
class MemoryResourcePrinterLookup(gdb.printing.PrettyPrinter):
"""Select printers for public CUDA type-erased resource types."""
def __init__(self) -> None:
super().__init__("cuda::mr::any_resource")
def __call__(self, value: gdb.Value) -> MemoryResourcePrinter | None:
if _is_memory_resource(value.type):
return MemoryResourcePrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register CUDA memory-resource formatters with GDB."""
gdb.printing.register_pretty_printer(
objfile, MemoryResourcePrinterLookup(), replace=True
)

View File

@@ -1,63 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""GDB pretty printer for cuda::std::array."""
from __future__ import annotations
from collections.abc import Iterator
from types import ModuleType
import memory_resource
import gdb
import gdb.printing
def _template_name(value_type: gdb.Type) -> str:
return str(value_type).split("<", 1)[0]
def _is_cuda_array(value_type: gdb.Type) -> bool:
value_type = value_type.strip_typedefs().unqualified()
template_name = _template_name(value_type)
return (
template_name.startswith("cuda::std::")
and template_name.rsplit("::", 1)[-1] == "array"
)
class ArrayPrinter:
"""Expose cuda::std::array metadata and elements to GDB."""
def __init__(self, value: gdb.Value) -> None:
self.value = value
self.type = value.type.strip_typedefs().unqualified()
self.type_name = memory_resource.public_type_name(self.type)
self.size = int(self.type.template_argument(1))
def children(self) -> Iterator[tuple[str, gdb.Value]]:
elems = self.value["__elems_"]
for index in range(self.size):
yield f"[{index}]", elems[index]
def to_string(self) -> str:
return self.type_name
class ArrayPrinterLookup(gdb.printing.PrettyPrinter):
"""Select the cuda::std::array printer by its public class name."""
def __init__(self) -> None:
super().__init__("cuda::std::array")
def __call__(self, value: gdb.Value) -> ArrayPrinter | None:
if _is_cuda_array(value.type):
return ArrayPrinter(value)
return None
def register(objfile: ModuleType) -> None:
"""Register the cuda::std::array printer with GDB."""
gdb.printing.register_pretty_printer(objfile, ArrayPrinterLookup(), replace=True)

View File

@@ -1,28 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB entry point for CUDA C++ Core Libraries pretty printers.
Requires Python 3.12 or newer.
"""
from __future__ import annotations
import buffer
import memory_resource
import std_array
import lldb
_CATEGORY = "cccl"
_FORMATTERS = (memory_resource, buffer, std_array)
InternalDict = dict[str, object]
def __lldb_init_module(debugger: lldb.SBDebugger, _internal_dict: InternalDict) -> None:
debugger.HandleCommand(f"type category define {_CATEGORY}")
for formatter in _FORMATTERS:
module = f"{__name__}.{formatter.__name__.rsplit('.', 1)[-1]}"
formatter.register(debugger, _CATEGORY, module)
debugger.HandleCommand(f"type category enable {_CATEGORY}")

View File

@@ -1,219 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for cuda::buffer."""
from __future__ import annotations
import re
from typing import NamedTuple
import memory_resource
import lldb
_BUFFER_PATTERN = re.compile(r"^cuda::buffer<.+>$")
# LLDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
# parser does not necessarily import the enum constants. This is the value of
# cudaMemcpyDefault from the CUDA Runtime API.
_CUDA_MEMCPY_DEFAULT = 4
InternalDict = dict[str, object]
class BufferInfo(NamedTuple):
size: int
data_address: int
value_type: lldb.SBType
accessibility: str
memory_resource: lldb.SBValue
stream: lldb.SBValue
alignment: lldb.SBValue
def is_cuda_buffer(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _BUFFER_PATTERN.fullmatch(type_name) is not None
def _buffer_info(value: lldb.SBValue) -> BufferInfo | None:
value = value.GetNonSyntheticValue()
storage = value.GetChildMemberWithName("__buf_")
if not storage.IsValid():
return None
count = storage.GetChildMemberWithName("__count_")
memory_resource = storage.GetChildMemberWithName("__mr_")
stream_ref = storage.GetChildMemberWithName("__stream_")
stream = stream_ref.GetChildMemberWithName("__stream")
alignment = storage.GetChildMemberWithName("__alignment_")
allocation = storage.GetChildMemberWithName("__buf_")
if not all(
child.IsValid()
for child in (count, memory_resource, stream, alignment, allocation)
):
return None
# A source-level alias can hide the accessibility properties from
# GetTypeName(), so use the canonical public type for all property checks.
buffer_type = value.GetType().GetCanonicalType().GetUnqualifiedType()
value_type = buffer_type.GetTemplateArgumentType(0)
if not value_type.IsValid():
return None
type_name = buffer_type.GetDisplayTypeName() or ""
host_accessible = "host_accessible" in type_name
device_accessible = "device_accessible" in type_name
if host_accessible and device_accessible:
accessibility = "host/device"
elif device_accessible:
accessibility = "device"
elif host_accessible:
accessibility = "host"
else:
accessibility = "unknown"
size = count.GetValueAsUnsigned(0)
align = alignment.GetValueAsUnsigned(1)
raw_address = allocation.GetValueAsUnsigned(0)
data_address = (raw_address + align - 1) & ~(align - 1)
return BufferInfo(
size,
data_address,
value_type,
accessibility,
memory_resource,
stream,
alignment,
)
def buffer_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str | None:
info = _buffer_info(value)
if info is None:
return None
resource = memory_resource.memory_resource_description(info.memory_resource)
stream = info.stream.GetValueAsUnsigned(0)
alignment = info.alignment.GetValueAsUnsigned(0)
return (
f"mr={resource}, stream={stream:#x}, size={info.size}, align={alignment}, "
f"data={info.data_address:#x} ({info.accessibility})"
)
class BufferSyntheticProvider:
"""Expose cuda::buffer elements as LLDB synthetic children."""
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
self.value = value.GetNonSyntheticValue()
self.host_copy = lldb.SBValue()
self.clear()
self.update()
def __del__(self) -> None:
self.clear()
def _evaluate(self, expression: str) -> lldb.SBValue:
frame = self.value.GetFrame()
if not frame.IsValid():
return lldb.SBValue()
options = lldb.SBExpressionOptions()
options.SetIgnoreBreakpoints(True)
options.SetUnwindOnError(True)
return frame.EvaluateExpression(expression, options)
def clear(self) -> None:
"""Release the staged copy and reset all cached buffer information."""
if self.host_copy.IsValid():
address = self.host_copy.GetValueAsUnsigned(0)
if address:
self._evaluate(f"(void)free((void*){address:#x})")
self.host_copy = lldb.SBValue()
self.size = 0
self.data_address = 0
self.value_type = lldb.SBType()
self.value_size = 0
def _copy_to_host(self) -> bool:
if self.size == 0:
return True
byte_count = self.size * self.value_size
self.host_copy = self._evaluate(f"(void*)malloc({byte_count})")
if not self.host_copy.IsValid() or self.host_copy.GetError().Fail():
return False
host_address = self.host_copy.GetValueAsUnsigned(0)
result = self._evaluate(
"(int)cudaMemcpy((void*)"
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
f"{_CUDA_MEMCPY_DEFAULT})"
)
if (
not result.IsValid()
or result.GetError().Fail()
or result.GetValueAsSigned(-1) != 0
):
self.clear()
return False
return True
def update(self) -> bool:
self.clear()
info = _buffer_info(self.value)
if info is None:
return False
self.size = info.size
self.data_address = info.data_address
self.value_type = info.value_type
self.value_size = self.value_type.GetByteSize()
self._copy_to_host()
return True
def num_children(self) -> int:
return self.size
def has_children(self) -> bool:
return self.size != 0
def get_type_name(self) -> str:
# STL element access can preserve an alloc_traits::value_type typedef.
# Report the canonical display name so LLDB shows cuda::buffer instead.
return (
self.value.GetType()
.GetCanonicalType()
.GetUnqualifiedType()
.GetDisplayTypeName()
or ""
)
def get_child_index(self, name: str) -> int:
if name.startswith("[") and name.endswith("]"):
try:
return int(name[1:-1])
except ValueError:
pass
return -1
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
if index < 0:
return None
if index >= self.size:
return None
offset = index * self.value_size
return self.host_copy.CreateChildAtOffset(f"[{index}]", offset, self.value_type)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register the cuda::buffer formatter in an LLDB category."""
debugger.HandleCommand(
f"type summary add --category {category} --expand --python-function {module}.buffer_summary "
f"--recognizer-function {module}.is_cuda_buffer"
)
debugger.HandleCommand(
f"type synthetic add --category {category} --python-class {module}.BufferSyntheticProvider "
f"--recognizer-function {module}.is_cuda_buffer"
)

View File

@@ -1,50 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for CUDA type-erased memory resources."""
from __future__ import annotations
import re
import lldb
_RESOURCE_PATTERN = re.compile(
r"^cuda::mr::(?:basic_any_resource|any_resource|any_synchronous_resource)<.+>$"
)
InternalDict = dict[str, object]
def is_memory_resource(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _RESOURCE_PATTERN.fullmatch(type_name) is not None
def memory_resource_description(value: lldb.SBValue) -> str:
"""Describe a type-erased resource using only public type information."""
type_name = (
value.GetType().GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName()
)
if not type_name:
type_name = "type-erased resource"
address = value.GetLoadAddress()
if address == lldb.LLDB_INVALID_ADDRESS:
return type_name
return f"{type_name} @ {address:#x}"
def memory_resource_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str:
return memory_resource_description(value)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register CUDA memory-resource formatters in an LLDB category."""
debugger.HandleCommand(
f"type summary add --category {category} --python-function "
f"{module}.memory_resource_summary --recognizer-function "
f"{module}.is_memory_resource"
)

View File

@@ -1,78 +0,0 @@
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
#
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
"""LLDB pretty printer for cuda::std::array."""
from __future__ import annotations
import re
import lldb
_ARRAY_PATTERN = re.compile(r"^cuda::std::array<.+,\s*(\d+)>$")
InternalDict = dict[str, object]
def is_cuda_array(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
type_name = (
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
)
return _ARRAY_PATTERN.fullmatch(type_name) is not None
class ArraySyntheticProvider:
"""Expose cuda::std::array elements as LLDB synthetic children."""
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
self.value = value.GetNonSyntheticValue()
self.update()
def update(self) -> bool:
type_name = (
self.value.GetType()
.GetCanonicalType()
.GetUnqualifiedType()
.GetDisplayTypeName()
or ""
)
self.type_name = type_name
match = _ARRAY_PATTERN.fullmatch(type_name)
self.elems = self.value.GetChildMemberWithName("__elems_")
self.size = 0
if not self.elems.IsValid() or not match:
return False
self.size = int(match.group(1))
return True
def num_children(self) -> int:
return self.size
def has_children(self) -> bool:
return self.size != 0
def get_type_name(self) -> str:
return self.type_name
def get_child_index(self, name: str) -> int:
if name.startswith("[") and name.endswith("]"):
try:
return int(name[1:-1])
except ValueError:
pass
return -1
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
if index < 0:
return None
if index >= self.size:
return None
return self.elems.GetChildAtIndex(index)
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
"""Register the cuda::std::array formatter in an LLDB category."""
debugger.HandleCommand(
f"type synthetic add --category {category} --python-class {module}.ArraySyntheticProvider "
f"--recognizer-function {module}.is_cuda_array"
)

View File

@@ -1,68 +0,0 @@
# Byte-compiled / optimized / DLL files
__pycache__/
*.py[cod]
# C extensions
*.so
# Distribution / packaging
.Python
env/
build/
develop-eggs/
dist/
downloads/
eggs/
#lib/ # We actually have things checked in to lib/
lib64/
parts/
sdist/
var/
*.egg-info/
.installed.cfg
*.egg
# PyInstaller
# Usually these files are written by a python script from a template
# before PyInstaller builds the exe, so as to inject date/other infos into it.
*.manifest
*.spec
!*.spec/
# Installer logs
pip-log.txt
pip-delete-this-directory.txt
# Unit test / coverage reports
htmlcov/
.tox/
.coverage
.cache
nosetests.xml
coverage.xml
# Translations
*.mo
*.pot
# Django stuff:
*.log
# Sphinx documentation
docs/_build/
# PyBuilder
target/
# MSVC libraries test harness
env.lst
keep.lst
# Editor by-products
.vscode/
# Random things
build-*
# Perforce files
.p4config

View File

@@ -1,102 +0,0 @@
find_package(Python COMPONENTS Interpreter)
if (NOT Python_Interpreter_FOUND)
message(
FATAL_ERROR
"Failed to find python interpreter, which is required for running tests and building a libcu++ static library."
)
endif()
# Determine the host triple to avoid invoking `${CXX} -dumpmachine`.
include(GetHostTriple)
get_host_triple(LLVM_INFERRED_HOST_TRIPLE)
set(
LLVM_HOST_TRIPLE
"${LLVM_INFERRED_HOST_TRIPLE}"
CACHE STRING
"Host on which LLVM binaries will run"
)
# By default, we target the host, but this can be overridden at CMake
# invocation time.
set(
LLVM_DEFAULT_TARGET_TRIPLE
"${LLVM_HOST_TRIPLE}"
CACHE STRING
"Default target for which LLVM will generate code."
)
set(TARGET_TRIPLE "${LLVM_DEFAULT_TARGET_TRIPLE}")
message(STATUS "LLVM host triple: ${LLVM_HOST_TRIPLE}")
message(STATUS "LLVM default target triple: ${LLVM_DEFAULT_TARGET_TRIPLE}")
set(LIT_EXTRA_ARGS "" CACHE STRING "Use for additional options (e.g. -j12)")
find_program(LLVM_DEFAULT_EXTERNAL_LIT lit)
set(LLVM_LIT_ARGS "-sv ${LIT_EXTRA_ARGS}")
# Libcudacxx's main lit tests
add_subdirectory(libcudacxx)
add_subdirectory(cmake)
# Set appropriate warning levels for MSVC/sane
if ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "NVIDIA")
# CUDA 11.5 and down do not support '-use-local-env'
if (MSVC)
set(
headertest_warning_levels_device
-Xcompiler=/W4
-Xcompiler=/WX
-Wno-deprecated-gpu-targets
)
if ("${CMAKE_CUDA_COMPILER_VERSION}" GREATER_EQUAL "11.6.0")
list(APPEND headertest_warning_levels_device --use-local-env)
endif()
else()
set(
headertest_warning_levels_device
-Wall
-Werror
all-warnings
-Wno-deprecated-gpu-targets
)
endif()
if (
CCCL_ENABLE_TILE
AND "${CMAKE_CUDA_COMPILER_VERSION}" VERSION_LESS_EQUAL "13.2"
)
message(
FATAL_ERROR
"tile programs require NVCC 13.3 or later; found NVCC ${CMAKE_CUDA_COMPILER_VERSION}"
)
endif()
# Set warnings for Clang as device compiler
elseif ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "Clang")
set(
headertest_warning_levels_device
-Wall
-Werror
-Wno-unknown-cuda-version
-Xclang=-fcuda-allow-variadic-functions
)
# If the CMAKE_CUDA_COMPILER is unknown, try to use gcc style warnings
else()
set(headertest_warning_levels_device -Wall -Werror)
endif()
# Set raw host/device warnings
if (MSVC)
set(headertest_warning_levels_host /W4 /WX)
else()
set(headertest_warning_levels_host -Wall -Werror)
endif()
# Enable building the nvrtcc project if NVRTC is enabled
if (LIBCUDACXX_TEST_WITH_NVRTC)
add_subdirectory(utils/nvidia/nvrtc)
endif()
add_subdirectory(nvtarget)
add_subdirectory(atomic_codegen)
add_subdirectory(simd_codegen)
add_subdirectory(debugging)

View File

@@ -1,150 +0,0 @@
This file is a partial list of people who have contributed to the LLVM/libc++
project. If you have contributed a patch or made some other contribution to
LLVM/libc++, please submit a patch to this file to add yourself, and it will be
done!
The list is sorted by surname and formatted to allow easy grepping and
beautification by scripts. The fields are: name (N), email (E), web-address
(W), PGP key ID and fingerprint (P), description (D), and snail-mail address
(S).
N: Saleem Abdulrasool
E: compnerd@compnerd.org
D: Minor patches and Linux fixes.
N: Dan Albert
E: danalbert@google.com
D: Android support and test runner improvements.
N: Dimitry Andric
E: dimitry@andric.com
D: Visibility fixes, minor FreeBSD portability patches.
N: Holger Arnold
E: holgerar@gmail.com
D: Minor fix.
N: Ruben Van Boxem
E: vanboxem dot ruben at gmail dot com
D: Initial Windows patches.
N: David Chisnall
E: theraven at theravensnest dot org
D: FreeBSD and Solaris ports, libcxxrt support, some atomics work.
N: Marshall Clow
E: mclow.lists@gmail.com
E: marshall@idio.com
D: C++14 support, patches and bug fixes.
N: Jonathan B Coe
E: jbcoe@me.com
D: Implementation of propagate_const.
N: Glen Joseph Fernandes
E: glenjofe@gmail.com
D: Implementation of to_address.
N: Eric Fiselier
E: eric@efcs.ca
D: LFTS support, patches and bug fixes.
N: Bill Fisher
E: william.w.fisher@gmail.com
D: Regex bug fixes.
N: Matthew Dempsky
E: matthew@dempsky.org
D: Minor patches and bug fixes.
N: Google Inc.
D: Copyright owner and contributor of the CityHash algorithm
N: Howard Hinnant
E: hhinnant@apple.com
D: Architect and primary author of libc++
N: Hyeon-bin Jeong
E: tuhertz@gmail.com
D: Minor patches and bug fixes.
N: Argyrios Kyrtzidis
E: kyrtzidis@apple.com
D: Bug fixes.
N: Bruce Mitchener, Jr.
E: bruce.mitchener@gmail.com
D: Emscripten-related changes.
N: Michel Morin
E: mimomorin@gmail.com
D: Minor patches to is_convertible.
N: Andrew Morrow
E: andrew.c.morrow@gmail.com
D: Minor patches and Linux fixes.
N: Michael Park
E: mcypark@gmail.com
D: Implementation of <variant>.
N: Arvid Picciani
E: aep at exys dot org
D: Minor patches and musl port.
N: Bjorn Reese
E: breese@users.sourceforge.net
D: Initial regex prototype
N: Nico Rieck
E: nico.rieck@gmail.com
D: Windows fixes
N: Jon Roelofs
E: jroelofS@jroelofs.com
D: Remote testing, Newlib port, baremetal/single-threaded support.
N: Jonathan Sauer
D: Minor patches, mostly related to constexpr
N: Craig Silverstein
E: csilvers@google.com
D: Implemented Cityhash as the string hash function on 64-bit machines
N: Richard Smith
D: Minor patches.
N: Joerg Sonnenberger
E: joerg@NetBSD.org
D: NetBSD port.
N: Stephan Tolksdorf
E: st@quanttec.com
D: Minor <atomic> fix
N: Michael van der Westhuizen
E: r1mikey at gmail dot com
N: Larisse Voufo
D: Minor patches.
N: Klaas de Vries
E: klaas at klaasgaaf dot nl
D: Minor bug fix.
N: Zhang Xiongpang
E: zhangxiongpang@gmail.com
D: Minor patches and bug fixes.
N: Xing Xue
E: xingxue@ca.ibm.com
D: AIX port
N: Zhihao Yuan
E: lichray@gmail.com
D: Standard compatibility fixes.
N: Jeffrey Yasskin
E: jyasskin@gmail.com
E: jyasskin@google.com
D: Linux fixes.

View File

@@ -1,311 +0,0 @@
==============================================================================
The LLVM Project is under the Apache License v2.0 with LLVM Exceptions:
==============================================================================
Apache License
Version 2.0, January 2004
http://www.apache.org/licenses/
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
1. Definitions.
"License" shall mean the terms and conditions for use, reproduction,
and distribution as defined by Sections 1 through 9 of this document.
"Licensor" shall mean the copyright owner or entity authorized by
the copyright owner that is granting the License.
"Legal Entity" shall mean the union of the acting entity and all
other entities that control, are controlled by, or are under common
control with that entity. For the purposes of this definition,
"control" means (i) the power, direct or indirect, to cause the
direction or management of such entity, whether by contract or
otherwise, or (ii) ownership of fifty percent (50%) or more of the
outstanding shares, or (iii) beneficial ownership of such entity.
"You" (or "Your") shall mean an individual or Legal Entity
exercising permissions granted by this License.
"Source" form shall mean the preferred form for making modifications,
including but not limited to software source code, documentation
source, and configuration files.
"Object" form shall mean any form resulting from mechanical
transformation or translation of a Source form, including but
not limited to compiled object code, generated documentation,
and conversions to other media types.
"Work" shall mean the work of authorship, whether in Source or
Object form, made available under the License, as indicated by a
copyright notice that is included in or attached to the work
(an example is provided in the Appendix below).
"Derivative Works" shall mean any work, whether in Source or Object
form, that is based on (or derived from) the Work and for which the
editorial revisions, annotations, elaborations, or other modifications
represent, as a whole, an original work of authorship. For the purposes
of this License, Derivative Works shall not include works that remain
separable from, or merely link (or bind by name) to the interfaces of,
the Work and Derivative Works thereof.
"Contribution" shall mean any work of authorship, including
the original version of the Work and any modifications or additions
to that Work or Derivative Works thereof, that is intentionally
submitted to Licensor for inclusion in the Work by the copyright owner
or by an individual or Legal Entity authorized to submit on behalf of
the copyright owner. For the purposes of this definition, "submitted"
means any form of electronic, verbal, or written communication sent
to the Licensor or its representatives, including but not limited to
communication on electronic mailing lists, source code control systems,
and issue tracking systems that are managed by, or on behalf of, the
Licensor for the purpose of discussing and improving the Work, but
excluding communication that is conspicuously marked or otherwise
designated in writing by the copyright owner as "Not a Contribution."
"Contributor" shall mean Licensor and any individual or Legal Entity
on behalf of whom a Contribution has been received by Licensor and
subsequently incorporated within the Work.
2. Grant of Copyright License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
copyright license to reproduce, prepare Derivative Works of,
publicly display, publicly perform, sublicense, and distribute the
Work and such Derivative Works in Source or Object form.
3. Grant of Patent License. Subject to the terms and conditions of
this License, each Contributor hereby grants to You a perpetual,
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
(except as stated in this section) patent license to make, have made,
use, offer to sell, sell, import, and otherwise transfer the Work,
where such license applies only to those patent claims licensable
by such Contributor that are necessarily infringed by their
Contribution(s) alone or by combination of their Contribution(s)
with the Work to which such Contribution(s) was submitted. If You
institute patent litigation against any entity (including a
cross-claim or counterclaim in a lawsuit) alleging that the Work
or a Contribution incorporated within the Work constitutes direct
or contributory patent infringement, then any patent licenses
granted to You under this License for that Work shall terminate
as of the date such litigation is filed.
4. Redistribution. You may reproduce and distribute copies of the
Work or Derivative Works thereof in any medium, with or without
modifications, and in Source or Object form, provided that You
meet the following conditions:
(a) You must give any other recipients of the Work or
Derivative Works a copy of this License; and
(b) You must cause any modified files to carry prominent notices
stating that You changed the files; and
(c) You must retain, in the Source form of any Derivative Works
that You distribute, all copyright, patent, trademark, and
attribution notices from the Source form of the Work,
excluding those notices that do not pertain to any part of
the Derivative Works; and
(d) If the Work includes a "NOTICE" text file as part of its
distribution, then any Derivative Works that You distribute must
include a readable copy of the attribution notices contained
within such NOTICE file, excluding those notices that do not
pertain to any part of the Derivative Works, in at least one
of the following places: within a NOTICE text file distributed
as part of the Derivative Works; within the Source form or
documentation, if provided along with the Derivative Works; or,
within a display generated by the Derivative Works, if and
wherever such third-party notices normally appear. The contents
of the NOTICE file are for informational purposes only and
do not modify the License. You may add Your own attribution
notices within Derivative Works that You distribute, alongside
or as an addendum to the NOTICE text from the Work, provided
that such additional attribution notices cannot be construed
as modifying the License.
You may add Your own copyright statement to Your modifications and
may provide additional or different license terms and conditions
for use, reproduction, or distribution of Your modifications, or
for any such Derivative Works as a whole, provided Your use,
reproduction, and distribution of the Work otherwise complies with
the conditions stated in this License.
5. Submission of Contributions. Unless You explicitly state otherwise,
any Contribution intentionally submitted for inclusion in the Work
by You to the Licensor shall be under the terms and conditions of
this License, without any additional terms or conditions.
Notwithstanding the above, nothing herein shall supersede or modify
the terms of any separate license agreement you may have executed
with Licensor regarding such Contributions.
6. Trademarks. This License does not grant permission to use the trade
names, trademarks, service marks, or product names of the Licensor,
except as required for reasonable and customary use in describing the
origin of the Work and reproducing the content of the NOTICE file.
7. Disclaimer of Warranty. Unless required by applicable law or
agreed to in writing, Licensor provides the Work (and each
Contributor provides its Contributions) on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
implied, including, without limitation, any warranties or conditions
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
PARTICULAR PURPOSE. You are solely responsible for determining the
appropriateness of using or redistributing the Work and assume any
risks associated with Your exercise of permissions under this License.
8. Limitation of Liability. In no event and under no legal theory,
whether in tort (including negligence), contract, or otherwise,
unless required by applicable law (such as deliberate and grossly
negligent acts) or agreed to in writing, shall any Contributor be
liable to You for damages, including any direct, indirect, special,
incidental, or consequential damages of any character arising as a
result of this License or out of the use or inability to use the
Work (including but not limited to damages for loss of goodwill,
work stoppage, computer failure or malfunction, or any and all
other commercial damages or losses), even if such Contributor
has been advised of the possibility of such damages.
9. Accepting Warranty or Additional Liability. While redistributing
the Work or Derivative Works thereof, You may choose to offer,
and charge a fee for, acceptance of support, warranty, indemnity,
or other liability obligations and/or rights consistent with this
License. However, in accepting such obligations, You may act only
on Your own behalf and on Your sole responsibility, not on behalf
of any other Contributor, and only if You agree to indemnify,
defend, and hold each Contributor harmless for any liability
incurred by, or claims asserted against, such Contributor by reason
of your accepting any such warranty or additional liability.
END OF TERMS AND CONDITIONS
APPENDIX: How to apply the Apache License to your work.
To apply the Apache License to your work, attach the following
boilerplate notice, with the fields enclosed by brackets "[]"
replaced with your own identifying information. (Don't include
the brackets!) The text should be enclosed in the appropriate
comment syntax for the file format. We also recommend that a
file or class name and description of purpose be included on the
same "printed page" as the copyright notice for easier
identification within third-party archives.
Copyright [yyyy] [name of copyright owner]
Licensed under the Apache License, Version 2.0 (the "License");
you may not use this file except in compliance with the License.
You may obtain a copy of the License at
http://www.apache.org/licenses/LICENSE-2.0
Unless required by applicable law or agreed to in writing, software
distributed under the License is distributed on an "AS IS" BASIS,
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
See the License for the specific language governing permissions and
limitations under the License.
---- LLVM Exceptions to the Apache 2.0 License ----
As an exception, if, as a result of your compiling your source code, portions
of this Software are embedded into an Object form of such source code, you
may redistribute such embedded portions in such Object form without complying
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
In addition, if you combine or link compiled forms of this Software with
software that is licensed under the GPLv2 ("Combined Software") and if a
court of competent jurisdiction determines that the patent provision (Section
3), the indemnity provision (Section 9) or other Section of the License
conflicts with the conditions of the GPLv2, you may retroactively and
prospectively choose to deem waived or otherwise exclude such Section(s) of
the License, but only in their entirety and only with respect to the Combined
Software.
==============================================================================
Software from third parties included in the LLVM Project:
==============================================================================
The LLVM Project contains third party software which is under different license
terms. All such code will be identified clearly using at least one of two
mechanisms:
1) It will be in a separate directory tree with its own `LICENSE.txt` or
`LICENSE` file at the top containing the specific license and restrictions
which apply to that software, or
2) It will contain specific license and restriction terms at the top of every
file.
==============================================================================
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
==============================================================================
The libc++ library is dual licensed under both the University of Illinois
"BSD-Like" license and the MIT license. As a user of this code you may choose
to use it under either license. As a contributor, you agree to allow your code
to be used under both.
Full text of the relevant licenses is included below.
==============================================================================
University of Illinois/NCSA
Open Source License
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
All rights reserved.
Developed by:
LLVM Team
University of Illinois at Urbana-Champaign
http://llvm.org
Permission is hereby granted, free of charge, to any person obtaining a copy of
this software and associated documentation files (the "Software"), to deal with
the Software without restriction, including without limitation the rights to
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
of the Software, and to permit persons to whom the Software is furnished to do
so, subject to the following conditions:
* Redistributions of source code must retain the above copyright notice,
this list of conditions and the following disclaimers.
* Redistributions in binary form must reproduce the above copyright notice,
this list of conditions and the following disclaimers in the
documentation and/or other materials provided with the distribution.
* Neither the names of the LLVM Team, University of Illinois at
Urbana-Champaign, nor the names of its contributors may be used to
endorse or promote products derived from this Software without specific
prior written permission.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
SOFTWARE.
==============================================================================
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
Permission is hereby granted, free of charge, to any person obtaining a copy
of this software and associated documentation files (the "Software"), to deal
in the Software without restriction, including without limitation the rights
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
copies of the Software, and to permit persons to whom the Software is
furnished to do so, subject to the following conditions:
The above copyright notice and this permission notice shall be included in
all copies or substantial portions of the Software.
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
THE SOFTWARE.

View File

@@ -1,29 +0,0 @@
//===---------------------------------------------------------------------===//
// Notes relating to various libc++ tasks
//===---------------------------------------------------------------------===//
This file contains notes about various libc++ tasks and processes.
//===---------------------------------------------------------------------===//
// Post-Release TODO
//===---------------------------------------------------------------------===//
These notes contain a list of things that must be done after branching for
an LLVM release.
1. Update _LIBCUDACXX_VERSION in `__config`
2. Update the __cccl_version file.
3. Update the version number in `docs/conf.py`
4. Create ABI lists for the previous release under `lib/abi`
//===---------------------------------------------------------------------===//
// Adding a new header TODO
//===---------------------------------------------------------------------===//
These notes contain a list of things that must be done upon adding a new header
to libc++.
1. Add a test under `test/libcxx` that the header defines `_LIBCUDACXX_VERSION`.
2. Update `test/libcxx/double_include.sh.cpp` to include the new header.
3. Create a submodule in `include/module.modulemap` for the new header.
4. Update the include/CMakeLists.txt file to include the new header.

Some files were not shown because too many files have changed in this diff Show More