[CCCL] 瘦身 + 补全: 移除 cudax/python/libcudacxx-tests 冗余文件, 新增 c2h 测试助手 + cmake 构建系统 + 8 个 CUDA thrust examples
变更摘要:
- 删除: cudax/ (783 files, 7.2M) — 实验性组件,竞赛不需要
- 删除: python/ (226 files, 2.0M) — Python 绑定,竞赛不需要
- 删除: libcudacxx/{test,benchmarks,codegen,cmake,share} (4432 files, 31M)
保留: libcudacxx/include/ (1463 headers, cuda::std 编译依赖)
- 新增: c2h/ (27 files) — CUB Catch2 测试辅助头文件,编译 243 个测试必需
- 新增: cmake/ (29 files) — CCCL 原生 CMake 构建系统
- 新增: thrust/examples/cuda/ (7 files) + cpp_integration/ (1 file)
async_reduce, custom_temporary_allocation, explicit_cuda_stream,
global_device_vector, range_view, unwrap_pointer, wrap_pointer, device
结果: cccl_upstream 从 74M→35M (瘦身 53%), 核心内容 100% 保留:
27/27 tuning headers, 78 benchmarks, 243 tests,
60 thrust examples, 18 CUB examples, 全部编译头文件
This commit is contained in:
@@ -1,92 +0,0 @@
|
||||
include(${CMAKE_SOURCE_DIR}/benchmarks/cmake/CCCLBenchmarkRegistry.cmake)
|
||||
|
||||
cccl_get_nvbench_helper()
|
||||
|
||||
set(benches_root "${CMAKE_CURRENT_LIST_DIR}")
|
||||
|
||||
if (NOT CMAKE_BUILD_TYPE STREQUAL "Release")
|
||||
set(message_type FATAL_ERROR)
|
||||
if (CCCL_ENABLE_CLANG_TIDY)
|
||||
# We are here because CI has force-enabled clang-tidy. We must use a debug build for
|
||||
# this because certain clang-tidy checks (such as out of bounds or clang static
|
||||
# analyzer) work better when they see assert()'s. In this case we don't actually
|
||||
# intend to run any of the benchmarks, we just need them to be compilable, so a simple
|
||||
# warning is enough.
|
||||
#
|
||||
# We don't ignore this outright (by making it say, DEBUG or VERBOSE), because it's
|
||||
# possible that a user may accidentally stumble into enabling the option.
|
||||
set(message_type WARNING)
|
||||
endif()
|
||||
message(${message_type} "libcu++ benchmarks must be built in release mode.")
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"CMAKE_CUDA_ARCHITECTURES must be set to build libcu++ benchmarks."
|
||||
)
|
||||
endif()
|
||||
|
||||
set(benches_meta_target libcudacxx.all.benches)
|
||||
add_custom_target(${benches_meta_target})
|
||||
|
||||
function(get_recursive_subdirs subdirs)
|
||||
set(dirs)
|
||||
file(
|
||||
GLOB_RECURSE contents
|
||||
CONFIGURE_DEPENDS
|
||||
LIST_DIRECTORIES ON
|
||||
"${CMAKE_CURRENT_LIST_DIR}/bench/*"
|
||||
)
|
||||
|
||||
foreach (test_dir IN LISTS contents)
|
||||
if (IS_DIRECTORY "${test_dir}")
|
||||
list(APPEND dirs "${test_dir}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(${subdirs} "${dirs}" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
create_benchmark_registry()
|
||||
|
||||
function(add_bench target_name bench_name bench_src)
|
||||
set(bench_target ${bench_name})
|
||||
set(${target_name} ${bench_target} PARENT_SCOPE)
|
||||
|
||||
cccl_add_executable(${bench_target} SOURCES "${bench_src}")
|
||||
target_link_libraries(
|
||||
${bench_target}
|
||||
PRIVATE libcudacxx::libcudacxx cccl.nvbench_helper nvbench::main
|
||||
)
|
||||
endfunction()
|
||||
|
||||
function(add_bench_dir bench_dir)
|
||||
file(GLOB bench_srcs CONFIGURE_DEPENDS "${bench_dir}/*.cu")
|
||||
file(RELATIVE_PATH bench_prefix "${benches_root}" "${bench_dir}")
|
||||
file(TO_CMAKE_PATH "${bench_prefix}" bench_prefix)
|
||||
string(REPLACE "/" "." bench_prefix "${bench_prefix}")
|
||||
|
||||
foreach (bench_src IN LISTS bench_srcs)
|
||||
# base tuning
|
||||
get_filename_component(bench_name "${bench_src}" NAME_WLE)
|
||||
string(PREPEND bench_name "libcudacxx.${bench_prefix}.")
|
||||
|
||||
set(base_bench_name "${bench_name}.base")
|
||||
add_bench(base_bench_target ${base_bench_name} "${bench_src}")
|
||||
add_dependencies(${benches_meta_target} ${base_bench_target})
|
||||
target_compile_definitions(${base_bench_target} PRIVATE TUNE_BASE=1)
|
||||
target_compile_options(
|
||||
${base_bench_target}
|
||||
PRIVATE "$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>"
|
||||
)
|
||||
# benchmarking
|
||||
register_cccl_benchmark("${bench_name}" "")
|
||||
endforeach()
|
||||
endfunction()
|
||||
|
||||
get_recursive_subdirs(subdirs)
|
||||
|
||||
foreach (subdir IN LISTS subdirs)
|
||||
add_bench_dir("${subdir}")
|
||||
endforeach()
|
||||
@@ -1,67 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/adjacent_difference.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(
|
||||
cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,77 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::all_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::any_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,70 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::counting_iterator<std::size_t>{0},
|
||||
cuda::counting_iterator{elements},
|
||||
out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,51 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,67 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_n(cuda_policy(alloc, launch), in.begin(), elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::copy_n(cuda_policy(alloc, launch), cuda::counting_iterator<std::size_t>{0}, elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,41 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count(cuda_policy(alloc, launch), in.begin(), in.end(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,50 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,82 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::equal(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::equal(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,68 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,43 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,40 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::fill(cuda_policy(alloc, launch), output.begin(), output.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,40 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::fill_n(cuda_policy(alloc, launch), output.begin(), elements, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,46 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), val));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::find_if(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find_if_not(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::not_fn(cuda::equal_to_value{val})));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,52 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::for_each(cuda_policy(alloc, launch), in.begin(), in.end(), op);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,52 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::for_each_n(cuda_policy(alloc, launch), in.begin(), elements, op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,42 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate(cuda_policy(alloc, launch), output.begin(), output.end(), generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,42 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate_n(cuda_policy(alloc, launch), output.begin(), elements, generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,94 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,69 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,85 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,85 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,50 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_partitioned(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), select_op_t{static_cast<T>(mismatch_point)}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,78 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::greater<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,78 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,66 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,95 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/merge.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs);
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end());
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs, ::cuda::std::greater<T>{});
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end(), ::cuda::std::greater<T>{});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin(),
|
||||
::cuda::std::greater<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
@@ -1,66 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,82 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::mismatch(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::mismatch(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::none_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -1,55 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition_copy(
|
||||
cuda_policy(alloc, launch),
|
||||
input.begin(),
|
||||
input.end(),
|
||||
output.begin(),
|
||||
cuda::std::make_reverse_iterator(output.begin() + elements),
|
||||
select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -1,41 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reduce(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,43 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove(cuda_policy(alloc, launch), in.begin(), in.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,44 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::remove_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,52 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::remove_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,50 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove_if(cuda_policy(alloc, launch), in.begin(), in.end(), is_even{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,41 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace(cuda_policy(alloc, launch), in.begin(), in.end(), 42, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,42 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::replace_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), 42, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,52 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::replace_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), equal_to_42{}, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,50 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,42 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::reverse(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,43 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reverse_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,44 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::rotate(cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -1,45 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::rotate_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -1,43 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_left(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -1,42 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_right(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.6, 0.45, 0.01});
|
||||
@@ -1,81 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
|
||||
struct fake_less
|
||||
{
|
||||
template <class T, class U>
|
||||
[[nodiscard]] _CCCL_API constexpr bool operator()(const T& t, const U& u) const
|
||||
{
|
||||
// complex is not less than comparable, so just compare the first element
|
||||
if constexpr (cuda::std::__is_cpp17_less_than_comparable_v<T, U>)
|
||||
{
|
||||
return t < u;
|
||||
}
|
||||
else
|
||||
{
|
||||
return cuda::std::get<0>(t) < cuda::std::get<0>(u);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end(), fake_less{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(all_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
@@ -1,48 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDA Experimental in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::stable_partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -1,72 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/swap.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(cuda_policy(alloc, launch), in1.begin(), in1.end(), in2.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_iter_swap(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::std::reverse_iterator{in1.end()},
|
||||
cuda::std::reverse_iterator{in1.begin()},
|
||||
cuda::std::reverse_iterator{in2.end()});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_iter_swap, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_iter_swap")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,156 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/iterator/zip_iterator.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
// The benchmarks are inspired by the BabelStream thrust version:
|
||||
// https://github.com/UoB-HPC/BabelStream/blob/main/src/thrust/ThrustStream.cu
|
||||
|
||||
// Modified from BabelStream to also work for integers
|
||||
constexpr auto startA = 1; // BabelStream: 0.1
|
||||
constexpr auto startB = 2; // BabelStream: 0.2
|
||||
constexpr auto startC = 3; // BabelStream: 0.1
|
||||
constexpr auto startScalar = 4; // BabelStream: 0.4
|
||||
|
||||
using element_types = nvbench::type_list<std::int8_t, std::int16_t, float, double, __int128>;
|
||||
// Different benchmarks use a different number of buffers. H200/B200 can fit 2^31 elements for all benchmarks and types.
|
||||
// Upstream BabelStream uses 2^25. Allocation failure just skips the benchmark
|
||||
auto array_size_powers = std::vector<std::int64_t>{25, 31};
|
||||
|
||||
template <typename T>
|
||||
static void mul(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), c.begin(), c.end(), b.begin(), [=] _CCCL_HOST_DEVICE(const T& ci) {
|
||||
return ci * scalar;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(mul, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("mul")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void add(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), a.begin(), a.end(), b.begin(), c.begin(), cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(add, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("add")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void triad(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
b.begin(),
|
||||
b.end(),
|
||||
c.begin(),
|
||||
a.begin(),
|
||||
[=] _CCCL_HOST_DEVICE(const T& bi, const T& ci) {
|
||||
return bi + scalar * ci;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(triad, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("triad")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void nstream(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(3 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::make_zip_iterator(a.begin(), b.begin(), c.begin()),
|
||||
cuda::make_zip_iterator(a.end(), b.end(), c.end()),
|
||||
a.begin(),
|
||||
cuda::zip_function{[=] _CCCL_HOST_DEVICE(const T& ai, const T& bi, const T& ci) {
|
||||
return ai + bi + scalar * ci;
|
||||
}}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(nstream, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("nstream")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
@@ -1,75 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
template <class InT, class OutT>
|
||||
struct fib_t
|
||||
{
|
||||
__device__ OutT operator()(InT n)
|
||||
{
|
||||
OutT t1 = 0;
|
||||
OutT t2 = 1;
|
||||
|
||||
if (n <= 1)
|
||||
{
|
||||
return t1;
|
||||
}
|
||||
else if (n == 2)
|
||||
{
|
||||
return t2;
|
||||
}
|
||||
for (InT i = 3; i <= n; ++i)
|
||||
{
|
||||
const auto next = t1 + t2;
|
||||
t1 = t2;
|
||||
t2 = next;
|
||||
}
|
||||
|
||||
return t2;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void fib(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> input = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<nvbench::uint32_t>(elements);
|
||||
|
||||
fib_t<T, nvbench::uint32_t> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::transform(cuda_policy(alloc, launch), input.cbegin(), input.cend(), output.begin(), op));
|
||||
});
|
||||
}
|
||||
|
||||
using types = nvbench::type_list<nvbench::uint32_t, nvbench::uint64_t>;
|
||||
|
||||
NVBENCH_BENCH_TYPES(fib, NVBENCH_TYPE_AXES(types))
|
||||
.set_name("fib")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,53 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,79 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("basic")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,49 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void binary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch),
|
||||
in.begin(),
|
||||
in.end(),
|
||||
cuda::constant_iterator<int>{42},
|
||||
42,
|
||||
cuda::std::plus<T>{},
|
||||
cuda::std::multiplies<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(binary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,52 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct plus_one
|
||||
{
|
||||
template <class U>
|
||||
[[nodiscard]] __device__ constexpr T operator()(const U val) const noexcept
|
||||
{
|
||||
return static_cast<T>(val + 1);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void unary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), 42, cuda::std::plus<T>{}, plus_one<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(unary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,88 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
// This seems like a clang-tidy bug. Yes we end up converting to double, but the division
|
||||
// is done entirely in integer land...
|
||||
return static_cast<T>(i / 2ULL); // NOLINT(bugprone-integer-division)
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -1,90 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
const auto run = i / 2;
|
||||
return static_cast<T>(run);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,11 +0,0 @@
|
||||
# Determine if the compiler has GCC-compatible command-line syntax.
|
||||
|
||||
if (NOT DEFINED LLVM_COMPILER_IS_GCC_COMPATIBLE)
|
||||
if (CMAKE_COMPILER_IS_GNUCXX)
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
|
||||
elseif (MSVC)
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE OFF)
|
||||
elseif ("${CMAKE_CXX_COMPILER_ID}" MATCHES "Clang")
|
||||
set(LLVM_COMPILER_IS_GCC_COMPATIBLE ON)
|
||||
endif()
|
||||
endif()
|
||||
@@ -1,35 +0,0 @@
|
||||
# Returns the host triple.
|
||||
# Invokes config.guess
|
||||
|
||||
function(get_host_triple var)
|
||||
if (MSVC)
|
||||
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
set(value "x86_64-pc-windows-msvc")
|
||||
else()
|
||||
set(value "i686-pc-windows-msvc")
|
||||
endif()
|
||||
elseif (MINGW AND NOT MSYS)
|
||||
if (CMAKE_SIZEOF_VOID_P EQUAL 8)
|
||||
set(value "x86_64-w64-windows-gnu")
|
||||
else()
|
||||
set(value "i686-pc-windows-gnu")
|
||||
endif()
|
||||
else(MSVC)
|
||||
if (CMAKE_HOST_SYSTEM_NAME STREQUAL Windows AND NOT MSYS)
|
||||
message(WARNING "unable to determine host target triple")
|
||||
else()
|
||||
set(config_guess ${LLVM_PATH}/cmake/config.guess)
|
||||
execute_process(
|
||||
COMMAND sh ${config_guess}
|
||||
RESULT_VARIABLE TT_RV
|
||||
OUTPUT_VARIABLE TT_OUT
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
)
|
||||
if (NOT TT_RV EQUAL 0)
|
||||
message(FATAL_ERROR "Failed to execute ${config_guess}")
|
||||
endif(NOT TT_RV EQUAL 0)
|
||||
set(value ${TT_OUT})
|
||||
endif()
|
||||
endif(MSVC)
|
||||
set(${var} ${value} PARENT_SCOPE)
|
||||
endfunction(get_host_triple var)
|
||||
@@ -1,376 +0,0 @@
|
||||
function(get_system_libs return_var)
|
||||
message(AUTHOR_WARNING "get_system_libs no longer needed")
|
||||
set(${return_var} "" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
function(link_system_libs target)
|
||||
message(AUTHOR_WARNING "link_system_libs no longer needed")
|
||||
endfunction()
|
||||
|
||||
# is_llvm_target_library(
|
||||
# library
|
||||
# Name of the LLVM library to check
|
||||
# return_var
|
||||
# Output variable name
|
||||
# ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS
|
||||
# ALL_TARGETS - default looks at the full list of known targets
|
||||
# INCLUDED_TARGETS - looks only at targets being configured
|
||||
# OMITTED_TARGETS - looks only at targets that are not being configured
|
||||
# )
|
||||
function(is_llvm_target_library library return_var)
|
||||
cmake_parse_arguments(
|
||||
ARG
|
||||
"ALL_TARGETS;INCLUDED_TARGETS;OMITTED_TARGETS"
|
||||
""
|
||||
""
|
||||
${ARGN}
|
||||
)
|
||||
# Sets variable `return_var' to ON if `library' corresponds to a
|
||||
# LLVM supported target. To OFF if it doesn't.
|
||||
set(${return_var} OFF PARENT_SCOPE)
|
||||
string(TOUPPER "${library}" capitalized_lib)
|
||||
if (ARG_INCLUDED_TARGETS)
|
||||
string(TOUPPER "${LLVM_TARGETS_TO_BUILD}" targets)
|
||||
elseif (ARG_OMITTED_TARGETS)
|
||||
set(omitted_targets ${LLVM_ALL_TARGETS})
|
||||
list(REMOVE_ITEM omitted_targets ${LLVM_TARGETS_TO_BUILD})
|
||||
string(TOUPPER "${omitted_targets}" targets)
|
||||
else()
|
||||
string(TOUPPER "${LLVM_ALL_TARGETS}" targets)
|
||||
endif()
|
||||
foreach (t ${targets})
|
||||
if (
|
||||
capitalized_lib STREQUAL t
|
||||
OR capitalized_lib STREQUAL "${t}"
|
||||
OR capitalized_lib STREQUAL "${t}DESC"
|
||||
OR capitalized_lib STREQUAL "${t}CODEGEN"
|
||||
OR capitalized_lib STREQUAL "${t}ASMPARSER"
|
||||
OR capitalized_lib STREQUAL "${t}ASMPRINTER"
|
||||
OR capitalized_lib STREQUAL "${t}DISASSEMBLER"
|
||||
OR capitalized_lib STREQUAL "${t}INFO"
|
||||
OR capitalized_lib STREQUAL "${t}UTILS"
|
||||
)
|
||||
set(${return_var} ON PARENT_SCOPE)
|
||||
break()
|
||||
endif()
|
||||
endforeach()
|
||||
endfunction(is_llvm_target_library)
|
||||
|
||||
function(is_llvm_target_specifier library return_var)
|
||||
is_llvm_target_library(${library} ${return_var} ${ARGN})
|
||||
string(TOUPPER "${library}" capitalized_lib)
|
||||
if (NOT ${return_var})
|
||||
if (
|
||||
capitalized_lib STREQUAL "ALLTARGETSASMPARSERS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSDESCS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSDISASSEMBLERS"
|
||||
OR capitalized_lib STREQUAL "ALLTARGETSINFOS"
|
||||
OR capitalized_lib STREQUAL "NATIVE"
|
||||
OR capitalized_lib STREQUAL "NATIVECODEGEN"
|
||||
)
|
||||
set(${return_var} ON PARENT_SCOPE)
|
||||
endif()
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
macro(llvm_config executable)
|
||||
cmake_parse_arguments(ARG "USE_SHARED" "" "" ${ARGN})
|
||||
set(link_components ${ARG_UNPARSED_ARGUMENTS})
|
||||
|
||||
if (ARG_USE_SHARED)
|
||||
# If USE_SHARED is specified, then we link against libLLVM,
|
||||
# but also against the component libraries below. This is
|
||||
# done in case libLLVM does not contain all of the components
|
||||
# the target requires.
|
||||
#
|
||||
# Strip LLVM_DYLIB_COMPONENTS out of link_components.
|
||||
# To do this, we need special handling for "all", since that
|
||||
# may imply linking to libraries that are not included in
|
||||
# libLLVM.
|
||||
|
||||
if (DEFINED link_components AND DEFINED LLVM_DYLIB_COMPONENTS)
|
||||
if ("${LLVM_DYLIB_COMPONENTS}" STREQUAL "all")
|
||||
set(link_components "")
|
||||
else()
|
||||
list(REMOVE_ITEM link_components ${LLVM_DYLIB_COMPONENTS})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
target_link_libraries(${executable} PRIVATE LLVM)
|
||||
endif()
|
||||
|
||||
explicit_llvm_config(${executable} ${link_components})
|
||||
endmacro(llvm_config)
|
||||
|
||||
function(explicit_llvm_config executable)
|
||||
set(link_components ${ARGN})
|
||||
|
||||
llvm_map_components_to_libnames(LIBRARIES ${link_components})
|
||||
get_target_property(t ${executable} TYPE)
|
||||
if (t STREQUAL "STATIC_LIBRARY")
|
||||
target_link_libraries(${executable} INTERFACE ${LIBRARIES})
|
||||
elseif (
|
||||
t STREQUAL "EXECUTABLE"
|
||||
OR t STREQUAL "SHARED_LIBRARY"
|
||||
OR t STREQUAL "MODULE_LIBRARY"
|
||||
)
|
||||
target_link_libraries(${executable} PRIVATE ${LIBRARIES})
|
||||
else()
|
||||
# Use plain form for legacy user.
|
||||
target_link_libraries(${executable} ${LIBRARIES})
|
||||
endif()
|
||||
endfunction(explicit_llvm_config)
|
||||
|
||||
# This is Deprecated
|
||||
function(llvm_map_components_to_libraries OUT_VAR)
|
||||
message(
|
||||
AUTHOR_WARNING
|
||||
"Using llvm_map_components_to_libraries() is deprecated. Use llvm_map_components_to_libnames() instead"
|
||||
)
|
||||
explicit_map_components_to_libraries(result ${ARGN})
|
||||
set(${OUT_VAR} ${result} ${sys_result} PARENT_SCOPE)
|
||||
endfunction(llvm_map_components_to_libraries)
|
||||
|
||||
# Expand pseudo-components into real components.
|
||||
# Does not cover 'native', 'backend', or 'engine' as these require special
|
||||
# handling. Also does not cover 'all' as we only have a list of the libnames
|
||||
# available and not a list of the components.
|
||||
function(llvm_expand_pseudo_components out_components)
|
||||
set(link_components ${ARGN})
|
||||
foreach (c ${link_components})
|
||||
# add codegen, asmprinter, asmparser, disassembler
|
||||
list(FIND LLVM_TARGETS_TO_BUILD ${c} idx)
|
||||
if (NOT idx LESS 0)
|
||||
if (TARGET LLVM${c}CodeGen)
|
||||
list(APPEND expanded_components "${c}CodeGen")
|
||||
else()
|
||||
if (TARGET LLVM${c})
|
||||
list(APPEND expanded_components "${c}")
|
||||
else()
|
||||
message(FATAL_ERROR "Target ${c} is not in the set of libraries.")
|
||||
endif()
|
||||
endif()
|
||||
if (TARGET LLVM${c}AsmPrinter)
|
||||
list(APPEND expanded_components "${c}AsmPrinter")
|
||||
endif()
|
||||
if (TARGET LLVM${c}AsmParser)
|
||||
list(APPEND expanded_components "${c}AsmParser")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Desc)
|
||||
list(APPEND expanded_components "${c}Desc")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Disassembler)
|
||||
list(APPEND expanded_components "${c}Disassembler")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Info)
|
||||
list(APPEND expanded_components "${c}Info")
|
||||
endif()
|
||||
if (TARGET LLVM${c}Utils)
|
||||
list(APPEND expanded_components "${c}Utils")
|
||||
endif()
|
||||
elseif (c STREQUAL "nativecodegen")
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}CodeGen)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}CodeGen")
|
||||
endif()
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}Desc)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Desc")
|
||||
endif()
|
||||
if (TARGET LLVM${LLVM_NATIVE_ARCH}Info)
|
||||
list(APPEND expanded_components "${LLVM_NATIVE_ARCH}Info")
|
||||
endif()
|
||||
elseif (c STREQUAL "AllTargetsCodeGens")
|
||||
# Link all the codegens from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}CodeGen)
|
||||
list(APPEND expanded_components "${t}CodeGen")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsAsmParsers")
|
||||
# Link all the asm parsers from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}AsmParser)
|
||||
list(APPEND expanded_components "${t}AsmParser")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsDescs")
|
||||
# Link all the descs from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Desc)
|
||||
list(APPEND expanded_components "${t}Desc")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsDisassemblers")
|
||||
# Link all the disassemblers from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Disassembler)
|
||||
list(APPEND expanded_components "${t}Disassembler")
|
||||
endif()
|
||||
endforeach(t)
|
||||
elseif (c STREQUAL "AllTargetsInfos")
|
||||
# Link all the infos from all the targets
|
||||
foreach (t ${LLVM_TARGETS_TO_BUILD})
|
||||
if (TARGET LLVM${t}Info)
|
||||
list(APPEND expanded_components "${t}Info")
|
||||
endif()
|
||||
endforeach(t)
|
||||
else()
|
||||
list(APPEND expanded_components "${c}")
|
||||
endif()
|
||||
endforeach()
|
||||
set(${out_components} ${expanded_components} PARENT_SCOPE)
|
||||
endfunction(llvm_expand_pseudo_components out_components)
|
||||
|
||||
# This is a variant intended for the final user:
|
||||
# Map LINK_COMPONENTS to actual libnames.
|
||||
function(llvm_map_components_to_libnames out_libs)
|
||||
set(link_components ${ARGN})
|
||||
if (NOT LLVM_AVAILABLE_LIBS)
|
||||
# Inside LLVM itself available libs are in a global property.
|
||||
get_property(LLVM_AVAILABLE_LIBS GLOBAL PROPERTY LLVM_LIBS)
|
||||
endif()
|
||||
string(TOUPPER "${LLVM_AVAILABLE_LIBS}" capitalized_libs)
|
||||
|
||||
get_property(LLVM_TARGETS_CONFIGURED GLOBAL PROPERTY LLVM_TARGETS_CONFIGURED)
|
||||
|
||||
# Generally in our build system we avoid order-dependence. Unfortunately since
|
||||
# not all targets create the same set of libraries we actually need to ensure
|
||||
# that all build targets associated with a target are added before we can
|
||||
# process target dependencies.
|
||||
if (NOT LLVM_TARGETS_CONFIGURED)
|
||||
foreach (c ${link_components})
|
||||
is_llvm_target_specifier(${c} iltl_result ALL_TARGETS)
|
||||
if (iltl_result)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Specified target library before target registration is complete."
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
endif()
|
||||
|
||||
# Expand some keywords:
|
||||
list(FIND LLVM_TARGETS_TO_BUILD "${LLVM_NATIVE_ARCH}" have_native_backend)
|
||||
list(FIND link_components "engine" engine_required)
|
||||
if (NOT engine_required EQUAL -1)
|
||||
list(FIND LLVM_TARGETS_WITH_JIT "${LLVM_NATIVE_ARCH}" have_jit)
|
||||
if (NOT have_native_backend EQUAL -1 AND NOT have_jit EQUAL -1)
|
||||
list(APPEND link_components "jit")
|
||||
list(APPEND link_components "native")
|
||||
else()
|
||||
list(APPEND link_components "interpreter")
|
||||
endif()
|
||||
endif()
|
||||
list(FIND link_components "native" native_required)
|
||||
if (NOT native_required EQUAL -1)
|
||||
if (NOT have_native_backend EQUAL -1)
|
||||
list(APPEND link_components ${LLVM_NATIVE_ARCH})
|
||||
endif()
|
||||
endif()
|
||||
|
||||
# Translate symbolic component names to real libraries:
|
||||
llvm_expand_pseudo_components(link_components ${link_components})
|
||||
foreach (c ${link_components})
|
||||
get_property(c_rename GLOBAL PROPERTY LLVM_COMPONENT_NAME_${c})
|
||||
if (c_rename)
|
||||
set(c ${c_rename})
|
||||
endif()
|
||||
if (c STREQUAL "native")
|
||||
# already processed
|
||||
elseif (c STREQUAL "backend")
|
||||
# same case as in `native'.
|
||||
elseif (c STREQUAL "engine")
|
||||
# already processed
|
||||
elseif (c STREQUAL "all")
|
||||
get_property(all_components GLOBAL PROPERTY LLVM_COMPONENT_LIBS)
|
||||
list(APPEND expanded_components ${all_components})
|
||||
else()
|
||||
# Canonize the component name:
|
||||
string(TOUPPER "${c}" capitalized)
|
||||
list(FIND capitalized_libs LLVM${capitalized} lib_idx)
|
||||
if (lib_idx LESS 0)
|
||||
# The component is unknown. Maybe is an omitted target?
|
||||
is_llvm_target_library(${c} iltl_result OMITTED_TARGETS)
|
||||
if (iltl_result)
|
||||
# A missing library to a directly referenced omitted target would be bad.
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Library '${c}' is a direct reference to a target library for an omitted target."
|
||||
)
|
||||
else()
|
||||
# If it is not an omitted target we should assume it is a component
|
||||
# that hasn't yet been processed by CMake. Missing components will
|
||||
# cause errors later in the configuration, so we can safely assume
|
||||
# that this is valid here.
|
||||
list(APPEND expanded_components LLVM${c})
|
||||
endif()
|
||||
else(lib_idx LESS 0)
|
||||
list(GET LLVM_AVAILABLE_LIBS ${lib_idx} canonical_lib)
|
||||
list(APPEND expanded_components ${canonical_lib})
|
||||
endif(lib_idx LESS 0)
|
||||
endif(c STREQUAL "native")
|
||||
endforeach(c)
|
||||
|
||||
set(${out_libs} ${expanded_components} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
# Perform a post-order traversal of the dependency graph.
|
||||
# This duplicates the algorithm used by llvm-config, originally
|
||||
# in tools/llvm-config/llvm-config.cpp, function ComputeLibsForComponents.
|
||||
function(expand_topologically name required_libs visited_libs)
|
||||
list(FIND visited_libs ${name} found)
|
||||
if (found LESS 0)
|
||||
list(APPEND visited_libs ${name})
|
||||
set(visited_libs ${visited_libs} PARENT_SCOPE)
|
||||
|
||||
#
|
||||
get_property(libname GLOBAL PROPERTY LLVM_COMPONENT_NAME_${name})
|
||||
if (libname)
|
||||
set(cname LLVM${libname})
|
||||
elseif (TARGET ${name})
|
||||
set(cname ${name})
|
||||
elseif (TARGET LLVM${name})
|
||||
set(cname LLVM${name})
|
||||
else()
|
||||
message(FATAL_ERROR "unknown component ${name}")
|
||||
endif()
|
||||
|
||||
get_property(lib_deps TARGET ${cname} PROPERTY LLVM_LINK_COMPONENTS)
|
||||
foreach (lib_dep ${lib_deps})
|
||||
expand_topologically(${lib_dep} "${required_libs}" "${visited_libs}")
|
||||
set(required_libs ${required_libs} PARENT_SCOPE)
|
||||
set(visited_libs ${visited_libs} PARENT_SCOPE)
|
||||
endforeach()
|
||||
|
||||
list(APPEND required_libs ${cname})
|
||||
set(required_libs ${required_libs} PARENT_SCOPE)
|
||||
endif()
|
||||
endfunction()
|
||||
|
||||
# Expand dependencies while topologically sorting the list of libraries:
|
||||
function(llvm_expand_dependencies out_libs)
|
||||
set(expanded_components ${ARGN})
|
||||
|
||||
set(required_libs)
|
||||
set(visited_libs)
|
||||
foreach (lib ${expanded_components})
|
||||
expand_topologically(${lib} "${required_libs}" "${visited_libs}")
|
||||
endforeach()
|
||||
|
||||
if (required_libs)
|
||||
list(REVERSE required_libs)
|
||||
endif()
|
||||
set(${out_libs} ${required_libs} PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
function(explicit_map_components_to_libraries out_libs)
|
||||
llvm_map_components_to_libnames(link_libs ${ARGN})
|
||||
llvm_expand_dependencies(expanded_components ${link_libs})
|
||||
# Return just the libraries included in this build:
|
||||
set(result)
|
||||
foreach (c ${expanded_components})
|
||||
if (TARGET ${c})
|
||||
set(result ${result} ${c})
|
||||
endif()
|
||||
endforeach(c)
|
||||
set(${out_libs} ${result} PARENT_SCOPE)
|
||||
endfunction(explicit_map_components_to_libraries)
|
||||
@@ -1,129 +0,0 @@
|
||||
include(AddFileDependencies)
|
||||
include(CMakeParseArguments)
|
||||
|
||||
function(llvm_replace_compiler_option var old new)
|
||||
# Replaces a compiler option or switch `old' in `var' by `new'.
|
||||
# If `old' is not in `var', appends `new' to `var'.
|
||||
# Example: llvm_replace_compiler_option(CMAKE_CXX_FLAGS_RELEASE "-O3" "-O2")
|
||||
# If the option already is on the variable, don't add it:
|
||||
if ("${${var}}" MATCHES "(^| )${new}($| )")
|
||||
set(n "")
|
||||
else()
|
||||
set(n "${new}")
|
||||
endif()
|
||||
if ("${${var}}" MATCHES "(^| )${old}($| )")
|
||||
string(REGEX REPLACE "(^| )${old}($| )" " ${n} " ${var} "${${var}}")
|
||||
else()
|
||||
set(${var} "${${var}} ${n}")
|
||||
endif()
|
||||
set(${var} "${${var}}" PARENT_SCOPE)
|
||||
endfunction(llvm_replace_compiler_option)
|
||||
|
||||
macro(add_td_sources srcs)
|
||||
file(GLOB tds *.td)
|
||||
if (tds)
|
||||
source_group("TableGen descriptions" FILES ${tds})
|
||||
set_source_files_properties(${tds} PROPERTIES HEADER_FILE_ONLY ON)
|
||||
list(APPEND ${srcs} ${tds})
|
||||
endif()
|
||||
endmacro(add_td_sources)
|
||||
|
||||
function(add_header_files_for_glob hdrs_out glob)
|
||||
file(GLOB hds ${glob})
|
||||
set(filtered)
|
||||
foreach (file ${hds})
|
||||
# Explicit existence check is necessary to filter dangling symlinks
|
||||
# out. See https://bugs.gentoo.org/674662.
|
||||
if (EXISTS ${file})
|
||||
list(APPEND filtered ${file})
|
||||
endif()
|
||||
endforeach()
|
||||
set(${hdrs_out} ${filtered} PARENT_SCOPE)
|
||||
endfunction(add_header_files_for_glob)
|
||||
|
||||
function(find_all_header_files hdrs_out additional_headerdirs)
|
||||
add_header_files_for_glob(hds *.h)
|
||||
list(APPEND all_headers ${hds})
|
||||
|
||||
foreach (additional_dir ${additional_headerdirs})
|
||||
add_header_files_for_glob(hds "${additional_dir}/*.h")
|
||||
list(APPEND all_headers ${hds})
|
||||
add_header_files_for_glob(hds "${additional_dir}/*.inc")
|
||||
list(APPEND all_headers ${hds})
|
||||
endforeach(additional_dir)
|
||||
|
||||
set(${hdrs_out} ${all_headers} PARENT_SCOPE)
|
||||
endfunction(find_all_header_files)
|
||||
|
||||
function(llvm_process_sources OUT_VAR)
|
||||
cmake_parse_arguments(
|
||||
ARG
|
||||
"PARTIAL_SOURCES_INTENDED"
|
||||
""
|
||||
"ADDITIONAL_HEADERS;ADDITIONAL_HEADER_DIRS"
|
||||
${ARGN}
|
||||
)
|
||||
set(sources ${ARG_UNPARSED_ARGUMENTS})
|
||||
if (NOT ARG_PARTIAL_SOURCES_INTENDED)
|
||||
llvm_check_source_file_list(${sources})
|
||||
endif()
|
||||
|
||||
# This adds .td and .h files to the Visual Studio solution:
|
||||
add_td_sources(sources)
|
||||
find_all_header_files(hdrs "${ARG_ADDITIONAL_HEADER_DIRS}")
|
||||
if (hdrs)
|
||||
set_source_files_properties(${hdrs} PROPERTIES HEADER_FILE_ONLY ON)
|
||||
endif()
|
||||
set_source_files_properties(
|
||||
${ARG_ADDITIONAL_HEADERS}
|
||||
PROPERTIES HEADER_FILE_ONLY ON
|
||||
)
|
||||
list(APPEND sources ${ARG_ADDITIONAL_HEADERS} ${hdrs})
|
||||
|
||||
set(${OUT_VAR} ${sources} PARENT_SCOPE)
|
||||
endfunction(llvm_process_sources)
|
||||
|
||||
function(llvm_check_source_file_list)
|
||||
cmake_parse_arguments(ARG "" "SOURCE_DIR" "" ${ARGN})
|
||||
foreach (l ${ARG_UNPARSED_ARGUMENTS})
|
||||
get_filename_component(fp ${l} REALPATH)
|
||||
list(APPEND listed ${fp})
|
||||
endforeach()
|
||||
|
||||
if (ARG_SOURCE_DIR)
|
||||
file(GLOB globbed "${ARG_SOURCE_DIR}/*.c" "${ARG_SOURCE_DIR}/*.cpp")
|
||||
else()
|
||||
file(GLOB globbed *.c *.cpp)
|
||||
endif()
|
||||
|
||||
foreach (g ${globbed})
|
||||
get_filename_component(fn ${g} NAME)
|
||||
if (ARG_SOURCE_DIR)
|
||||
set(entry "${g}")
|
||||
else()
|
||||
set(entry "${fn}")
|
||||
endif()
|
||||
get_filename_component(gp ${g} REALPATH)
|
||||
|
||||
# Don't reject hidden files. Some editors create backups in the
|
||||
# same directory as the file.
|
||||
if (NOT "${fn}" MATCHES "^\\.")
|
||||
list(FIND LLVM_OPTIONAL_SOURCES ${entry} idx)
|
||||
if (idx LESS 0)
|
||||
list(FIND listed ${gp} idx)
|
||||
if (idx LESS 0)
|
||||
if (ARG_SOURCE_DIR)
|
||||
set(fn_relative "${ARG_SOURCE_DIR}/${fn}")
|
||||
else()
|
||||
set(fn_relative "${fn}")
|
||||
endif()
|
||||
message(
|
||||
SEND_ERROR
|
||||
"Found unknown source file ${fn_relative}
|
||||
Please update ${CMAKE_CURRENT_LIST_FILE}\n"
|
||||
)
|
||||
endif()
|
||||
endif()
|
||||
endif()
|
||||
endforeach()
|
||||
endfunction(llvm_check_source_file_list)
|
||||
@@ -1,53 +0,0 @@
|
||||
# This file defines the `libcudacxx_build_compiler_targets()` function, which
|
||||
# creates the following interface targets:
|
||||
#
|
||||
# libcudacxx.compiler_interface
|
||||
# - Interface target linked into all targets in the libcudacxx developer build.
|
||||
# Defines common warning flags, definitions, etc, including those defined in
|
||||
# the global CCCL targets.
|
||||
|
||||
cccl_get_libcudacxx()
|
||||
|
||||
function(libcudacxx_build_compiler_targets)
|
||||
set(cuda_compile_options)
|
||||
set(cxx_compile_options)
|
||||
set(cxx_compile_definitions)
|
||||
|
||||
# if (CCCL_USE_LIBCXX)
|
||||
# list(APPEND cxx_compile_options "-stdlib=libc++")
|
||||
# list(APPEND cxx_compile_definitions "_ALLOW_UNSUPPORTED_LIBCPP=1")
|
||||
# endif()
|
||||
|
||||
# Set test specific flags
|
||||
list(APPEND cxx_compile_definitions "CCCL_ENABLE_ASSERTIONS")
|
||||
list(APPEND cxx_compile_definitions "CCCL_IGNORE_DEPRECATED_CPP_DIALECT")
|
||||
list(
|
||||
APPEND cxx_compile_definitions
|
||||
"CCCL_IGNORE_DEPRECATED_DISCARD_MEMORY_HEADER"
|
||||
)
|
||||
list(
|
||||
APPEND cxx_compile_definitions
|
||||
"CCCL_IGNORE_DEPRECATED_STREAM_REF_HEADER"
|
||||
)
|
||||
|
||||
if (CCCL_ENABLE_TILE)
|
||||
list(APPEND cuda_compile_options "--enable-tile")
|
||||
endif()
|
||||
|
||||
cccl_build_compiler_interface(
|
||||
libcudacxx.compiler_flags
|
||||
"${cuda_compile_options}"
|
||||
"${cxx_compile_options}"
|
||||
"${cxx_compile_definitions}"
|
||||
)
|
||||
|
||||
add_library(libcudacxx.compiler_interface INTERFACE)
|
||||
target_link_libraries(
|
||||
libcudacxx.compiler_interface
|
||||
INTERFACE
|
||||
# order matters here, we need the libcudacxx options to override the cccl options.
|
||||
cccl.compiler_interface
|
||||
libcudacxx.compiler_flags
|
||||
libcudacxx::libcudacxx
|
||||
)
|
||||
endfunction()
|
||||
@@ -1,125 +0,0 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.internal_headers)
|
||||
|
||||
# Grep all internal headers
|
||||
file(
|
||||
GLOB_RECURSE internal_headers
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include/"
|
||||
CONFIGURE_DEPENDS
|
||||
${libcudacxx_SOURCE_DIR}/include/cuda/__*/*.h
|
||||
${libcudacxx_SOURCE_DIR}/include/cuda/std/__*/*.h
|
||||
)
|
||||
|
||||
# Exclude <cuda/std/__cccl/(prologue|epilogue|visibility).h> from the test
|
||||
list(
|
||||
FILTER internal_headers
|
||||
EXCLUDE
|
||||
REGEX "__cccl/(prologue|epilogue|visibility)\.h"
|
||||
)
|
||||
|
||||
# headers in `__cuda` are meant to come after the related "cuda" headers so they do not compile on their own
|
||||
list(FILTER internal_headers EXCLUDE REGEX "__cuda/*")
|
||||
|
||||
# generated cuda::ptx headers are not standalone
|
||||
list(FILTER internal_headers EXCLUDE REGEX "__ptx/instructions/generated")
|
||||
|
||||
# don't check nvtx3.h - it's not our header
|
||||
list(FILTER internal_headers EXCLUDE REGEX ".*/__nvtx/nvtx3.h")
|
||||
|
||||
function(libcudacxx_add_internal_header_test_target target_name)
|
||||
if (NOT ARGN)
|
||||
return()
|
||||
endif()
|
||||
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${ARGN}
|
||||
)
|
||||
|
||||
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
|
||||
target_link_libraries(
|
||||
${target_name}
|
||||
PUBLIC #
|
||||
libcudacxx.compiler_interface
|
||||
CUDA::cudart
|
||||
)
|
||||
add_dependencies(libcudacxx.test.internal_headers ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_internal_header_test_target(
|
||||
libcudacxx.test.internal_headers.base
|
||||
${internal_headers}
|
||||
)
|
||||
|
||||
# We have fallbacks for some type traits that we want to explicitly test so that they do not bitrot.
|
||||
set(internal_headers_fallback)
|
||||
set(internal_headers_fallback_per_header_defines)
|
||||
foreach (header IN LISTS internal_headers)
|
||||
# MSVC cannot handle some of the fallbacks.
|
||||
if ("MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
||||
if (
|
||||
"${header}" MATCHES "is_base_of"
|
||||
OR "${header}" MATCHES "is_nothrow_destructible"
|
||||
OR "${header}" MATCHES "is_polymorphic"
|
||||
)
|
||||
continue()
|
||||
endif()
|
||||
endif()
|
||||
|
||||
file(READ "${libcudacxx_SOURCE_DIR}/include/${header}" header_file)
|
||||
string(REGEX MATCH "_LIBCUDACXX_[A-Z_]*_FALLBACK" fallback "${header_file}")
|
||||
if (fallback)
|
||||
list(APPEND internal_headers_fallback "${header}")
|
||||
string(
|
||||
REGEX REPLACE
|
||||
"([][+.*^$()|?\\\\])"
|
||||
"\\\\\\1"
|
||||
header_regex
|
||||
"${header}"
|
||||
)
|
||||
list(
|
||||
APPEND internal_headers_fallback_per_header_defines
|
||||
DEFINE
|
||||
"${fallback}"
|
||||
"^${header_regex}$"
|
||||
)
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
if (internal_headers_fallback)
|
||||
cccl_generate_header_tests(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${internal_headers_fallback}
|
||||
PER_HEADER_DEFINES ${internal_headers_fallback_per_header_defines}
|
||||
)
|
||||
target_compile_definitions(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
PRIVATE _CCCL_HEADER_TEST
|
||||
)
|
||||
target_link_libraries(
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
PUBLIC #
|
||||
libcudacxx.compiler_interface
|
||||
CUDA::cudart
|
||||
)
|
||||
add_dependencies(
|
||||
libcudacxx.test.internal_headers
|
||||
libcudacxx.test.internal_headers.fallback
|
||||
)
|
||||
endif()
|
||||
@@ -1,47 +0,0 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.public_headers)
|
||||
|
||||
# Grep all public headers
|
||||
file(
|
||||
GLOB public_headers
|
||||
LIST_DIRECTORIES false
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
|
||||
CONFIGURE_DEPENDS
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
|
||||
)
|
||||
|
||||
# annotated_ptr does not work with clang cuda due to __nv_associate_access_property
|
||||
if ("Clang" STREQUAL "${CMAKE_CUDA_COMPILER_ID}")
|
||||
list(REMOVE_ITEM public_headers "annotated_ptr")
|
||||
endif()
|
||||
|
||||
function(libcudacxx_add_public_header_test_target target_name)
|
||||
if (NOT ARGN)
|
||||
return()
|
||||
endif()
|
||||
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CUDA
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${ARGN}
|
||||
)
|
||||
|
||||
target_compile_definitions(${target_name} PRIVATE _CCCL_HEADER_TEST)
|
||||
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
|
||||
add_dependencies(libcudacxx.test.public_headers ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_public_header_test_target(
|
||||
libcudacxx.test.public_headers.base
|
||||
${public_headers}
|
||||
)
|
||||
@@ -1,75 +0,0 @@
|
||||
# For every public header, build a translation unit containing `#include <header>`
|
||||
# to let the compiler try to figure out warnings in that header if it is not otherwise
|
||||
# included in tests, and also to verify if the headers are modular enough.
|
||||
# .inl files are not globbed for, because they are not supposed to be used as public
|
||||
# entrypoints.
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
|
||||
# Meta target for all configs' header builds:
|
||||
add_custom_target(libcudacxx.test.public_headers_host_only)
|
||||
add_custom_target(libcudacxx.test.public_headers_host_only_with_ctk)
|
||||
|
||||
if (CCCL_ENABLE_TILE) # TODO(miscco): For now only test public headers with tile
|
||||
return()
|
||||
endif()
|
||||
|
||||
# Grep all public headers
|
||||
file(
|
||||
GLOB public_headers_host_only
|
||||
LIST_DIRECTORIES false
|
||||
RELATIVE "${libcudacxx_SOURCE_DIR}/include"
|
||||
CONFIGURE_DEPENDS
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/*"
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/*"
|
||||
)
|
||||
|
||||
set(public_host_header_cxx_compile_options)
|
||||
set(public_host_header_cxx_compile_definitions)
|
||||
|
||||
# Specifically add libc++ testing if requested to the libcudacxx host suite
|
||||
if (CCCL_USE_LIBCXX)
|
||||
list(APPEND public_host_header_cxx_compile_options "-stdlib=libc++")
|
||||
endif()
|
||||
|
||||
function(
|
||||
libcudacxx_add_public_header_test_host_target
|
||||
target_name
|
||||
parent_target
|
||||
with_ctk
|
||||
)
|
||||
cccl_generate_header_tests(
|
||||
${target_name}
|
||||
libcudacxx/include
|
||||
NO_METATARGETS
|
||||
LANGUAGE CXX
|
||||
HEADER_TEMPLATE "${libcudacxx_SOURCE_DIR}/cmake/header_test.cpp.in"
|
||||
HEADERS ${public_headers_host_only}
|
||||
)
|
||||
target_compile_definitions(
|
||||
${target_name}
|
||||
PRIVATE #
|
||||
${public_host_header_cxx_compile_definitions}
|
||||
_CCCL_HEADER_TEST
|
||||
)
|
||||
target_compile_options(
|
||||
${target_name}
|
||||
PRIVATE ${public_host_header_cxx_compile_options}
|
||||
)
|
||||
target_link_libraries(${target_name} PUBLIC libcudacxx.compiler_interface)
|
||||
if (with_ctk)
|
||||
target_link_libraries(${target_name} PUBLIC CUDA::cudart)
|
||||
endif()
|
||||
add_dependencies(${parent_target} ${target_name})
|
||||
endfunction()
|
||||
|
||||
libcudacxx_add_public_header_test_host_target(
|
||||
libcudacxx.test.public_headers_host_only.base
|
||||
libcudacxx.test.public_headers_host_only
|
||||
OFF
|
||||
)
|
||||
libcudacxx_add_public_header_test_host_target(
|
||||
libcudacxx.test.public_headers_host_only_with_ctk.base
|
||||
libcudacxx.test.public_headers_host_only_with_ctk
|
||||
ON
|
||||
)
|
||||
1569
cccl_upstream/libcudacxx/cmake/config.guess
vendored
1569
cccl_upstream/libcudacxx/cmake/config.guess
vendored
File diff suppressed because it is too large
Load Diff
@@ -1,23 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// ignore deprecation warnings
|
||||
#if defined(__clang__)
|
||||
# pragma clang diagnostic ignored "-Wdeprecated"
|
||||
# pragma clang diagnostic ignored "-Wdeprecated-declarations"
|
||||
#elif defined(_MSC_VER)
|
||||
# pragma warning (disable: 4996)
|
||||
#else
|
||||
# pragma GCC diagnostic ignored "-Wdeprecated"
|
||||
# pragma GCC diagnostic ignored "-Wdeprecated-declarations"
|
||||
#endif
|
||||
|
||||
// This file tests that the respective header is includable on its own with a cuda compiler
|
||||
#include <@header@>
|
||||
@@ -1 +0,0 @@
|
||||
cccl_add_subdir_helper(libcudacxx)
|
||||
1
cccl_upstream/libcudacxx/codegen/.gitignore
vendored
1
cccl_upstream/libcudacxx/codegen/.gitignore
vendored
@@ -1 +0,0 @@
|
||||
build
|
||||
@@ -1,51 +0,0 @@
|
||||
## Codegen adds the following build targets
|
||||
# libcudacxx.atomics.codegen
|
||||
# libcudacxx.atomics.codegen.install
|
||||
## Test targets:
|
||||
# libcudacxx.test.atomics.codegen.diff
|
||||
|
||||
add_executable(codegen EXCLUDE_FROM_ALL codegen.cpp)
|
||||
|
||||
target_compile_features(codegen PRIVATE cxx_std_20)
|
||||
|
||||
set(
|
||||
atomic_generated_output
|
||||
"${libcudacxx_BINARY_DIR}/codegen/cuda_ptx_generated.h"
|
||||
)
|
||||
set(
|
||||
atomic_install_location
|
||||
"${libcudacxx_SOURCE_DIR}/include/cuda/std/__atomic/functions"
|
||||
)
|
||||
|
||||
add_custom_target(
|
||||
libcudacxx.atomics.codegen
|
||||
COMMAND codegen "${atomic_generated_output}"
|
||||
BYPRODUCTS "${atomic_generated_output}"
|
||||
)
|
||||
|
||||
add_custom_target(
|
||||
libcudacxx.atomics.codegen.install
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E copy
|
||||
"${atomic_generated_output}"
|
||||
"${atomic_install_location}/cuda_ptx_generated.h"
|
||||
# gersemi: on
|
||||
DEPENDS libcudacxx.atomics.codegen
|
||||
BYPRODUCTS "${atomic_install_location}/cuda_ptx_generated.h"
|
||||
)
|
||||
|
||||
add_test(
|
||||
NAME libcudacxx.test.atomics.codegen.diff
|
||||
# gersemi: off
|
||||
COMMAND
|
||||
"${CMAKE_COMMAND}" -E compare_files
|
||||
"${atomic_install_location}/cuda_ptx_generated.h"
|
||||
"${atomic_generated_output}"
|
||||
# gersemi: on
|
||||
)
|
||||
|
||||
set_tests_properties(
|
||||
libcudacxx.test.atomics.codegen.diff
|
||||
PROPERTIES REQUIRED_FILES "${atomic_generated_output}"
|
||||
)
|
||||
@@ -1,164 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import argparse
|
||||
import os
|
||||
|
||||
import cccl_paths
|
||||
|
||||
docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions")
|
||||
test = os.path.join(cccl_paths.LIBCUDACXX_TEST_DIR, "libcudacxx", "cuda", "ptx")
|
||||
src = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "__ptx", "instructions")
|
||||
ptx_header = os.path.join(cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "ptx")
|
||||
instr_docs = os.path.join(cccl_paths.DOCS_LIBCUDACXX_DIR, "ptx", "instructions.rst")
|
||||
|
||||
|
||||
def add_docs(ptx_instr, url):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
underbar = "=" * len(ptx_instr)
|
||||
|
||||
(docs / f"{cpp_instr}.rst").write_text(
|
||||
f""".. _libcudacxx-ptx-instructions-{ptx_instr.replace(".", "-")}:
|
||||
|
||||
{ptx_instr}
|
||||
{underbar}
|
||||
|
||||
- PTX ISA:
|
||||
`{ptx_instr} <{url}>`__
|
||||
|
||||
.. include:: generated/{cpp_instr}.rst
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_test(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
dst = test / f"ptx.{ptx_instr}.compile.pass.cpp"
|
||||
dst.write_text(
|
||||
f"""//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
// UNSUPPORTED: libcpp-has-no-threads
|
||||
|
||||
// <cuda/ptx>
|
||||
|
||||
#include <cuda/ptx>
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include "generated/{cpp_instr}.h"
|
||||
|
||||
int main(int, char**)
|
||||
{{
|
||||
return 0;
|
||||
}}
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_src(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
(src / f"{cpp_instr}.h").write_text(
|
||||
f"""// -*- C++ -*-
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
#define _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/__ptx/ptx_dot_variants.h>
|
||||
#include <cuda/__ptx/ptx_helper_functions.h>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <nv/target> // __CUDA_MINIMUM_ARCH__ and friends
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_PTX
|
||||
|
||||
#include <cuda/__ptx/instructions/generated/{cpp_instr}.h>
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_PTX
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA_PTX_{cpp_instr.upper()}_H_
|
||||
"""
|
||||
)
|
||||
|
||||
|
||||
def add_ptx_header_include(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
txt = ptx_header.read_text()
|
||||
# just add as first new include. clang-format will sort it in
|
||||
idx = txt.index("#include <cuda/__ptx/instructions")
|
||||
txt = (
|
||||
txt[:idx]
|
||||
+ f"""#include <cuda/__ptx/instructions/{cpp_instr}.h>\n"""
|
||||
+ txt[idx:]
|
||||
)
|
||||
ptx_header.write_text(txt)
|
||||
|
||||
|
||||
def add_docs_include(ptx_instr):
|
||||
cpp_instr = ptx_instr.replace(".", "_")
|
||||
txt = instr_docs.read_text()
|
||||
# just add as first new include
|
||||
idx = txt.index(" instructions/")
|
||||
txt = txt[:idx] + f" instructions/{cpp_instr}\n" + txt[idx:]
|
||||
instr_docs.write_text(txt)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
parser = argparse.ArgumentParser()
|
||||
parser.add_argument("ptx_instruction", type=str)
|
||||
parser.add_argument("url", type=str)
|
||||
|
||||
args = parser.parse_args()
|
||||
|
||||
ptx_instr = args.ptx_instruction
|
||||
url = args.url
|
||||
|
||||
# Enable using internal urls in the command-line, to be automatically converted to public URLs.
|
||||
if url.startswith("index.html"):
|
||||
url = url.replace(
|
||||
"index.html",
|
||||
"https://docs.nvidia.com/cuda/parallel-thread-execution/index.html",
|
||||
)
|
||||
|
||||
add_test(ptx_instr)
|
||||
add_docs(ptx_instr, url)
|
||||
add_src(ptx_instr)
|
||||
add_ptx_header_include(ptx_instr)
|
||||
add_docs_include(ptx_instr)
|
||||
@@ -1,21 +0,0 @@
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import os
|
||||
|
||||
LIBCUDACXX_DIR = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
||||
|
||||
LIBCUDACXX_CMAKE_DIR = os.path.join(LIBCUDACXX_DIR, "cmake")
|
||||
LIBCUDACXX_CODEGEN_DIR = os.path.join(LIBCUDACXX_DIR, "codegen")
|
||||
LIBCUDACXX_INCLUDE_DIR = os.path.join(LIBCUDACXX_DIR, "include")
|
||||
LIBCUDACXX_TEST_DIR = os.path.join(LIBCUDACXX_DIR, "test")
|
||||
|
||||
DOCS_DIR = os.path.dirname(LIBCUDACXX_DIR)
|
||||
DOCS_LIBCUDACXX_DIR = os.path.join(DOCS_DIR, "libcudacxx")
|
||||
@@ -1,45 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <ostream>
|
||||
|
||||
#include "generators/compare_and_swap.h"
|
||||
#include "generators/exchange.h"
|
||||
#include "generators/fence.h"
|
||||
#include "generators/fetch_ops.h"
|
||||
#include "generators/header.h"
|
||||
#include "generators/ld_st.h"
|
||||
|
||||
using namespace std::string_literals;
|
||||
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
std::fstream filestream;
|
||||
|
||||
if (argc == 2)
|
||||
{
|
||||
filestream.open(argv[1], filestream.out);
|
||||
}
|
||||
|
||||
std::ostream& stream = filestream.is_open() ? filestream : std::cout;
|
||||
|
||||
FormatHeader(stream);
|
||||
FormatFence(stream);
|
||||
FormatLoad(stream);
|
||||
FormatStore(stream);
|
||||
FormatCompareAndSwap(stream);
|
||||
FormatExchange(stream);
|
||||
FormatFetchOps(stream);
|
||||
FormatTail(stream);
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -1,245 +0,0 @@
|
||||
#!/usr/bin/env python3
|
||||
|
||||
##===----------------------------------------------------------------------===##
|
||||
##
|
||||
## Part of libcu++, the C++ Standard Library for your entire system,
|
||||
## under the Apache License v2.0 with LLVM Exceptions.
|
||||
## See https://llvm.org/LICENSE.txt for license information.
|
||||
## SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
## SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
##
|
||||
##===----------------------------------------------------------------------===##
|
||||
|
||||
import datetime
|
||||
import os
|
||||
|
||||
import cccl_paths
|
||||
|
||||
PROLOGUE_FILE = os.path.join(
|
||||
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "prologue.h"
|
||||
)
|
||||
EPILOGUE_FILE = os.path.join(
|
||||
cccl_paths.LIBCUDACXX_INCLUDE_DIR, "cuda", "std", "__cccl", "epilogue.h"
|
||||
)
|
||||
|
||||
HEADER = f"""\
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) {datetime.datetime.now().year} NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// !!! DO NOT EDIT THIS FILE !!! This file is generated by utils/generate_prologue_epilogue.py.
|
||||
|
||||
// NO include guards here (this file is included multiple times)"""
|
||||
|
||||
FOOTER = """\
|
||||
// NO include guards here (this file is included multiple times)
|
||||
"""
|
||||
|
||||
PUSH_POP_MACROS = {
|
||||
"__declspec modifiers": [
|
||||
"align",
|
||||
"allocate",
|
||||
"allocator",
|
||||
"appdomain",
|
||||
"code_seg",
|
||||
"deprecated",
|
||||
"dllimport",
|
||||
"dllexport",
|
||||
"empty_bases",
|
||||
"hybrid_patchable",
|
||||
"jitintrinsic",
|
||||
"lifetimebound",
|
||||
"naked",
|
||||
"noalias",
|
||||
"noinline",
|
||||
"noreturn",
|
||||
"nothrow",
|
||||
"novtable",
|
||||
"no_sanitize_address",
|
||||
"process",
|
||||
"property",
|
||||
"restrict",
|
||||
"safebuffers",
|
||||
"selectany",
|
||||
"spectre",
|
||||
"thread",
|
||||
"uuid",
|
||||
],
|
||||
"[[msvc::attribute]] attributes": [
|
||||
"msvc",
|
||||
"flatten",
|
||||
"forceinline",
|
||||
"forceinline_calls",
|
||||
"intrinsic",
|
||||
"noinline",
|
||||
"noinline_calls",
|
||||
"no_tls_guard",
|
||||
],
|
||||
"Windows nasty macros": ["min", "max", "interface"],
|
||||
"sal.h on Windows": ["__valid", "__callback"],
|
||||
"other macros": ["clang"],
|
||||
"sys/sysmacros.h on linux": ["major", "minor", "makedev"],
|
||||
}
|
||||
|
||||
|
||||
def write_section(file, section):
|
||||
file.write(section)
|
||||
file.write("\n\n")
|
||||
|
||||
|
||||
def make_prologue(file):
|
||||
# Write common header.
|
||||
write_section(file, HEADER)
|
||||
|
||||
# Add prologue/epilogue include logic check.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#if defined(_CCCL_PROLOGUE_INCLUDED)
|
||||
# error \\
|
||||
"cccl internal error: <cuda/std/__cccl/epilogue.h> must be included before next <cuda/std/__cccl/prologue.h> is reincluded"
|
||||
#endif
|
||||
#define _CCCL_PROLOGUE_INCLUDED() 1""",
|
||||
)
|
||||
|
||||
# Add necessary includes.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#include <cuda/std/__cccl/compiler.h>
|
||||
#include <cuda/std/__cccl/diagnostic.h>
|
||||
#include <cuda/std/__cccl/dialect.h>""",
|
||||
)
|
||||
|
||||
# Add push macros.
|
||||
for group_name, macros in PUSH_POP_MACROS.items():
|
||||
write_section(file, f"// {group_name}")
|
||||
for macro in macros:
|
||||
write_section(
|
||||
file,
|
||||
f"""\
|
||||
#if defined({macro})
|
||||
# pragma push_macro("{macro}")
|
||||
# undef {macro}
|
||||
# define _CCCL_POP_MACRO_{macro}
|
||||
#endif // defined({macro})""",
|
||||
)
|
||||
|
||||
# Add warnings suppressions.
|
||||
write_section(
|
||||
file,
|
||||
'''\
|
||||
_CCCL_DIAG_PUSH
|
||||
_CCCL_NV_DIAG_PUSH()
|
||||
|
||||
// disable some msvc warnings
|
||||
// https://github.com/microsoft/STL/blob/master/stl/inc/yvals_core.h#L353
|
||||
// warning C4100: 'quack': unreferenced formal parameter
|
||||
// warning C4127: conditional expression is constant
|
||||
// warning C4180: qualifier applied to function type has no meaning; ignored
|
||||
// warning C4197: 'purr': top-level volatile in cast is ignored
|
||||
// warning C4324: 'roar': structure was padded due to alignment specifier
|
||||
// warning C4455: literal suffix identifiers that do not start with an underscore are reserved
|
||||
// warning C4503: 'hum': decorated name length exceeded, name was truncated
|
||||
// warning C4522: 'woof' : multiple assignment operators specified
|
||||
// warning C4668: 'meow' is not defined as a preprocessor macro, replacing with '0' for '#if/#elif'
|
||||
// warning C4800: 'boo': forcing value to bool 'true' or 'false' (performance warning)
|
||||
// warning C4996: 'meow': was declared deprecated
|
||||
_CCCL_DIAG_SUPPRESS_MSVC(4100 4127 4180 4197 4296 4324 4455 4503 4522 4668 4800 4996)
|
||||
|
||||
// Suppress compiler warnings about C++ extensions.
|
||||
|
||||
#if _CCCL_COMPILER(GCC, >=, 12)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++20-extensions")
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++23-extensions")
|
||||
#endif // _CCCL_COMPILER(GCC, >=, 12)
|
||||
#if _CCCL_COMPILER(GCC, >=, 14)
|
||||
_CCCL_DIAG_SUPPRESS_GCC("-Wc++26-extensions")
|
||||
#endif // _CCCL_COMPILER(GCC, >=, 14)
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++20-extensions")
|
||||
#if _CCCL_COMPILER(CLANG, >=, 17)
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++23-extensions")
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++26-extensions")
|
||||
#else // ^^^ _CCCL_COMPILER(CLANG, >=, 17) ^^^ / vvv _CCCL_COMPILER(CLANG, <, 17) vvv
|
||||
_CCCL_DIAG_SUPPRESS_CLANG("-Wc++2b-extensions")
|
||||
#endif // ^^^ _CCCL_COMPILER(CLANG, <, 17) ^^^
|
||||
|
||||
// Suppress `if consteval`-related warnings.
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_nonstandard)
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(is_constant_evaluated_in_nonconstexpr_context)
|
||||
_CCCL_DIAG_SUPPRESS_NVHPC(if_consteval_in_nonconstexpr_function)
|
||||
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3215) // "if consteval" and "if not consteval" are not standard in this mode
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3206) // "if consteval" and "if not consteval" are meaningless in a non-constexpr function
|
||||
_CCCL_DIAG_SUPPRESS_NVCC(3060) // call to __builtin_is_constant_evaluated appearing in a non-constexpr function always
|
||||
// produces "false"''',
|
||||
)
|
||||
|
||||
# Write the common footer.
|
||||
file.write(FOOTER)
|
||||
|
||||
|
||||
def make_epilogue(file):
|
||||
# Write common header.
|
||||
write_section(file, HEADER)
|
||||
|
||||
# Write includes.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#include <cuda/std/__cccl/compiler.h>
|
||||
#include <cuda/std/__cccl/diagnostic.h>""",
|
||||
)
|
||||
|
||||
# Add prologue/epilogue include logic check.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
#if !defined(_CCCL_PROLOGUE_INCLUDED)
|
||||
# error "cccl internal error: <cuda/std/__cccl/prologue.h> must be included before <cuda/std/__cccl/epilogue.h>"
|
||||
#endif
|
||||
#undef _CCCL_PROLOGUE_INCLUDED""",
|
||||
)
|
||||
|
||||
# Pop warning suppressions.
|
||||
write_section(
|
||||
file,
|
||||
"""\
|
||||
_CCCL_NV_DIAG_POP()
|
||||
_CCCL_DIAG_POP""",
|
||||
)
|
||||
|
||||
# Add pop macros.
|
||||
for group_name, macros in PUSH_POP_MACROS.items():
|
||||
write_section(file, f"// {group_name}")
|
||||
for macro in macros:
|
||||
write_section(
|
||||
file,
|
||||
f"""\
|
||||
#if defined({macro})
|
||||
# error \\
|
||||
"cccl internal error: macro `{macro}` was redefined between <cuda/std/__cccl/prologue.h> and <cuda/std/__cccl/epilogue.h>"
|
||||
#elif defined(_CCCL_POP_MACRO_{macro})
|
||||
# pragma pop_macro("{macro}")
|
||||
# undef _CCCL_POP_MACRO_{macro}
|
||||
#endif""",
|
||||
)
|
||||
|
||||
# Write the common footer.
|
||||
file.write(FOOTER)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
with open(PROLOGUE_FILE, "w") as file:
|
||||
make_prologue(file)
|
||||
|
||||
with open(EPILOGUE_FILE, "w") as file:
|
||||
make_epilogue(file)
|
||||
@@ -1,201 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef COMPARED_AND_SWAP_H
|
||||
#define COMPARED_AND_SWAP_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline void FormatCompareAndSwap(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_swap_memory_order_dispatch(_Fn& __cuda_cas, int __success_memorder, int __failure_memorder, _Sco) {
|
||||
bool __res = false;
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __res = __cuda_cas(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __res = __cuda_cas(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__stronger_order_cuda(__success_memorder, __failure_memorder)) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __res = __cuda_cas(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __res = __cuda_cas(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __res = __cuda_cas(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
return __res;
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Type Constraint
|
||||
// 3 - Memory Order
|
||||
// 4 - Memory Order function tag
|
||||
// 5 - Scope Constraint
|
||||
// 6 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
|
||||
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b CAS is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90, (),
|
||||
NV_ANY_TARGET, (__atomic_cas_128b_unsupported_before_SM_90();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
.reg .b128 _v;
|
||||
mov.b128 _d, {{%3, %4}};
|
||||
mov.b128 _v, {{%5, %6}};
|
||||
atom.cas{3}{5}.b128 _d,[%2],_d,_v;
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr), "l"(__cmp.__x),"l"(__cmp.__y), "l"(__op.__x),"l"(__op.__y) : "memory"); return __dst.__x == __cmp.__x && __dst.__y == __cmp.__y; }})XXX";
|
||||
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE bool __cuda_atomic_compare_exchange(
|
||||
_Type* __ptr, _Type& __dst, _Type __cmp, _Type __op, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{ asm volatile("atom.cas{3}{5}.{0}{1} %0,[%1],%2,%3;" : "={2}"(__dst) : "l"(__ptr), "{2}"(__cmp), "{2}"(__op) : "memory"); return __dst == __cmp; }})XXX";
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
if (size == 2 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_compare_exchange {
|
||||
_Type* __ptr;
|
||||
_Type* __exp;
|
||||
_Type* __des;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE bool operator()(_Atomic_Memorder) {
|
||||
return __cuda_atomic_compare_exchange(__ptr, *__exp, *__exp, *__des, _Atomic_Memorder{}, _Tag{}, _Sco{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
|
||||
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
|
||||
bool __res = false;
|
||||
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
|
||||
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
|
||||
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE bool __atomic_compare_exchange_cuda(_Type volatile* __ptr, _Type* __exp, _Type __des, bool, int __success_memorder, int __failure_memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __exp_proxy = reinterpret_cast<__proxy_t*>(__exp);
|
||||
__proxy_t* __des_proxy = reinterpret_cast<__proxy_t*>(&__des);
|
||||
bool __res = false;
|
||||
if (__cuda_compare_exchange_weak_if_local(__ptr_proxy, __exp_proxy, __des_proxy, &__res)) {return __res;}
|
||||
__cuda_atomic_bind_compare_exchange<__proxy_t, __proxy_tag, _Sco> __bound_compare_swap{__ptr_proxy, __exp_proxy, __des_proxy};
|
||||
return __cuda_atomic_compare_swap_memory_order_dispatch(__bound_compare_swap, __success_memorder, __failure_memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // COMPARED_AND_SWAP_H
|
||||
@@ -1,192 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef DEFINITIONS_H
|
||||
#define DEFINITIONS_H
|
||||
|
||||
#include <format>
|
||||
#include <map>
|
||||
#include <string>
|
||||
#include <type_traits>
|
||||
#include <vector>
|
||||
|
||||
enum class Mmio
|
||||
{
|
||||
Disabled,
|
||||
Enabled,
|
||||
};
|
||||
|
||||
inline std::string mmio(Mmio m)
|
||||
{
|
||||
static const char* mmio_map[]{
|
||||
"",
|
||||
".mmio",
|
||||
};
|
||||
return mmio_map[std::underlying_type_t<Mmio>(m)];
|
||||
}
|
||||
|
||||
inline std::string mmio_tag(Mmio m)
|
||||
{
|
||||
static const char* mmio_map[]{
|
||||
"__atomic_cuda_mmio_disable",
|
||||
"__atomic_cuda_mmio_enable",
|
||||
};
|
||||
return mmio_map[std::underlying_type_t<Mmio>(m)];
|
||||
}
|
||||
|
||||
enum class Operand
|
||||
{
|
||||
Floating,
|
||||
Unsigned,
|
||||
Signed,
|
||||
Bit,
|
||||
};
|
||||
|
||||
inline std::string operand(Operand op)
|
||||
{
|
||||
static std::map op_map = {
|
||||
std::pair{Operand::Floating, "f"},
|
||||
std::pair{Operand::Unsigned, "u"},
|
||||
std::pair{Operand::Signed, "s"},
|
||||
std::pair{Operand::Bit, "b"},
|
||||
};
|
||||
return op_map[op];
|
||||
}
|
||||
|
||||
inline std::string operand_proxy_type(Operand op, size_t sz)
|
||||
{
|
||||
if (op == Operand::Floating)
|
||||
{
|
||||
if (sz == 32)
|
||||
{
|
||||
return {"float"};
|
||||
}
|
||||
else
|
||||
{
|
||||
return {"double"};
|
||||
}
|
||||
}
|
||||
else if (op == Operand::Signed)
|
||||
{
|
||||
return std::format("int{}_t", sz);
|
||||
}
|
||||
// Binary and unsigned can be the same proxy_type
|
||||
return std::format("uint{}_t", sz);
|
||||
}
|
||||
|
||||
inline std::string constraints(Operand op, size_t sz)
|
||||
{
|
||||
static std::map constraint_map = {
|
||||
std::pair{32,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "r"},
|
||||
std::pair{Operand::Unsigned, "r"},
|
||||
std::pair{Operand::Signed, "r"},
|
||||
std::pair{Operand::Floating, "f"},
|
||||
}},
|
||||
std::pair{64,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "l"},
|
||||
std::pair{Operand::Unsigned, "l"},
|
||||
std::pair{Operand::Signed, "l"},
|
||||
std::pair{Operand::Floating, "d"},
|
||||
}},
|
||||
std::pair{128,
|
||||
std::map{
|
||||
std::pair{Operand::Bit, "l"},
|
||||
std::pair{Operand::Unsigned, "l"},
|
||||
std::pair{Operand::Signed, "l"},
|
||||
std::pair{Operand::Floating, "d"},
|
||||
}},
|
||||
};
|
||||
|
||||
if (sz == 16)
|
||||
{
|
||||
return {"h"};
|
||||
}
|
||||
else
|
||||
{
|
||||
return constraint_map[sz][op];
|
||||
}
|
||||
}
|
||||
|
||||
enum class Semantic
|
||||
{
|
||||
Relaxed,
|
||||
Release,
|
||||
Acquire,
|
||||
Acq_Rel,
|
||||
Seq_Cst,
|
||||
Volatile,
|
||||
};
|
||||
|
||||
inline std::string semantic(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, ".relaxed"},
|
||||
std::pair{Semantic::Release, ".release"},
|
||||
std::pair{Semantic::Acquire, ".acquire"},
|
||||
std::pair{Semantic::Acq_Rel, ".acq_rel"},
|
||||
std::pair{Semantic::Seq_Cst, ".sc"},
|
||||
std::pair{Semantic::Volatile, ""},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
inline std::string semantic_tag(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, "__atomic_cuda_relaxed"},
|
||||
std::pair{Semantic::Release, "__atomic_cuda_release"},
|
||||
std::pair{Semantic::Acquire, "__atomic_cuda_acquire"},
|
||||
std::pair{Semantic::Acq_Rel, "__atomic_cuda_acq_rel"},
|
||||
std::pair{Semantic::Seq_Cst, "__atomic_cuda_seq_cst"},
|
||||
std::pair{Semantic::Volatile, "__atomic_cuda_volatile"},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
enum class Scope
|
||||
{
|
||||
Thread,
|
||||
Warp,
|
||||
CTA,
|
||||
Cluster,
|
||||
GPU,
|
||||
System,
|
||||
};
|
||||
|
||||
inline std::string scope(Scope sco)
|
||||
{
|
||||
static std::map sco_map = {
|
||||
std::pair{Scope::Thread, ""},
|
||||
std::pair{Scope::Warp, ""},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
std::pair{Scope::Cluster, ".cluster"},
|
||||
std::pair{Scope::GPU, ".gpu"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
};
|
||||
return sco_map[sco];
|
||||
}
|
||||
|
||||
inline std::string scope_tag(Scope sco)
|
||||
{
|
||||
static std::map sco_map = {
|
||||
std::pair{Scope::Thread, "__thread_scope_thread_tag"},
|
||||
std::pair{Scope::Warp, ""},
|
||||
std::pair{Scope::CTA, "__thread_scope_block_tag"},
|
||||
std::pair{Scope::Cluster, "__thread_scope_cluster_tag"},
|
||||
std::pair{Scope::GPU, "__thread_scope_device_tag"},
|
||||
std::pair{Scope::System, "__thread_scope_system_tag"},
|
||||
};
|
||||
return sco_map[sco];
|
||||
}
|
||||
|
||||
#endif // DEFINITIONS_H
|
||||
@@ -1,197 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef EXCHANGE_H
|
||||
#define EXCHANGE_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline void FormatExchange(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange_memory_order_dispatch(_Fn& __cuda_exch, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __cuda_exch(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_exch(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_exch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_exch(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_exch(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Type Constraint
|
||||
// 3 - Memory Order
|
||||
// 4 - Memory Order function tag
|
||||
// 5 - Scope Constraint
|
||||
// 6 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
|
||||
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b exchange is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_90, (),
|
||||
NV_ANY_TARGET, (__atomic_exchange_128b_unsupported_before_SM_90();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
.reg .b128 _v;
|
||||
mov.b128 _v, {{%3, %4}};
|
||||
atom.exch{3}{5}.b128 _d,[%2],_v;
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__old.__x),"=l"(__old.__y) : "l"(__ptr), "l"(__new.__x),"l"(__new.__y) : "memory");
|
||||
}})XXX";
|
||||
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_exchange(
|
||||
_Type* __ptr, _Type& __old, _Type __new, {4}, __atomic_cuda_operand_{0}{1}, {6})
|
||||
{{ asm volatile("atom.exch{3}{5}.{0}{1} %0,[%1],%2;" : "={2}"(__old) : "l"(__ptr), "{2}"(__new) : "memory"); }})XXX";
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
if (size == 2 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
operand(type),
|
||||
size,
|
||||
constraints(type, size),
|
||||
semantic(sem),
|
||||
semantic_tag(sem),
|
||||
scope(sco),
|
||||
scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_exchange {
|
||||
_Type* __ptr;
|
||||
_Type* __old;
|
||||
_Type* __new;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_exchange(__ptr, *__old, *__new, _Atomic_Memorder{}, _Tag{}, _Sco{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
|
||||
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
|
||||
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
|
||||
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
|
||||
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_exchange_cuda(_Type volatile* __ptr, _Type& __old, _Type __new, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __old_proxy = reinterpret_cast<__proxy_t*>(&__old);
|
||||
__proxy_t* __new_proxy = reinterpret_cast<__proxy_t*>(&__new);
|
||||
if(__cuda_exchange_weak_if_local(__ptr_proxy, __new_proxy, __old_proxy)) {{return;}}
|
||||
__cuda_atomic_bind_exchange<__proxy_t, __proxy_tag, _Sco> __bound_swap{__ptr_proxy, __old_proxy, __new_proxy};
|
||||
__cuda_atomic_exchange_memory_order_dispatch(__bound_swap, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // EXCHANGE_H
|
||||
@@ -1,110 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef FENCE_H
|
||||
#define FENCE_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string membar_scope(Scope sco)
|
||||
{
|
||||
static std::map scope_map{
|
||||
std::pair{Scope::GPU, ".gl"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
};
|
||||
|
||||
return scope_map[sco];
|
||||
}
|
||||
|
||||
inline void FormatFence(std::ostream& out)
|
||||
{
|
||||
// Argument ID Reference
|
||||
// 0 - Membar scope tag
|
||||
// 1 - Membar scope
|
||||
constexpr auto intrinsic_membar = R"XXX(
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_membar({0})
|
||||
{{ asm volatile("membar{1};" ::: "memory"); }})XXX";
|
||||
|
||||
const std::map membar_scopes{
|
||||
std::pair{Scope::GPU, ".gl"},
|
||||
std::pair{Scope::System, ".sys"},
|
||||
std::pair{Scope::CTA, ".cta"},
|
||||
};
|
||||
|
||||
for (const auto& sco : membar_scopes)
|
||||
{
|
||||
out << std::format(intrinsic_membar, scope_tag(sco.first), sco.second);
|
||||
}
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Fence scope tag
|
||||
// 1 - Fence scope
|
||||
// 2 - Fence order tag
|
||||
// 3 - Fence order
|
||||
constexpr auto intrinsic_fence = R"XXX(
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fence({0}, {2})
|
||||
{{ asm volatile("fence{1}{3};" ::: "memory"); }})XXX";
|
||||
|
||||
const Scope fence_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
const Semantic fence_semantics[] = {
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Seq_Cst,
|
||||
};
|
||||
|
||||
for (const auto& sco : fence_scopes)
|
||||
{
|
||||
for (const auto& sem : fence_semantics)
|
||||
{
|
||||
out << std::format(intrinsic_fence, scope_tag(sco), semantic(sem), semantic_tag(sem), scope(sco));
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_thread_fence_cuda(int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); break;
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: [[fallthrough]];
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_fence(_Sco{}, __atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELAXED: break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: [[fallthrough]];
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELAXED: break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // FENCE_H
|
||||
@@ -1,219 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef FETCH_OPS_H
|
||||
#define FETCH_OPS_H
|
||||
|
||||
#include <array>
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string fetch_op_skip_v(std::string fetch_op)
|
||||
{
|
||||
if (fetch_op == "add")
|
||||
{
|
||||
return "constexpr auto __skip_v = __atomic_ptr_skip_t<_Type>::__skip;";
|
||||
}
|
||||
return "constexpr auto __skip_v = 1;";
|
||||
}
|
||||
|
||||
inline void FormatFetchOps(std::ostream& out)
|
||||
{
|
||||
const std::vector arithmetic_types = {
|
||||
Operand::Floating,
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
const std::vector minmax_types = {
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
const std::vector bitwise_types = {Operand::Bit};
|
||||
|
||||
const std::map op_support_map{
|
||||
std::pair{std::string{"add"}, std::pair{arithmetic_types, std::string{"arithmetic"}}},
|
||||
std::pair{std::string{"min"}, std::pair{minmax_types, std::string{"minmax"}}},
|
||||
std::pair{std::string{"max"}, std::pair{minmax_types, std::string{"minmax"}}},
|
||||
std::pair{std::string{"or"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
std::pair{std::string{"xor"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
std::pair{std::string{"and"}, std::pair{bitwise_types, std::string{"bitwise"}}},
|
||||
};
|
||||
|
||||
// Memory order dispatcher
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fetch_memory_order_dispatch(_Fn& __cuda_fetch, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_ACQ_REL: __cuda_fetch(__atomic_cuda_acq_rel{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_fetch(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: [[fallthrough]];
|
||||
case __ATOMIC_ACQ_REL: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_fetch(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); __cuda_fetch(__atomic_cuda_volatile{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_fetch(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Atomic Operation
|
||||
// 1 - Operand Type
|
||||
// 2 - Operand Size
|
||||
// 3 - Type Constraint
|
||||
// 4 - Memory Order
|
||||
// 5 - Memory Order function tag
|
||||
// 6 - Scope Constraint
|
||||
// 7 - Scope function tag
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_fetch_{0}(
|
||||
_Type* __ptr, _Type& __dst, _Type __op, {5}, __atomic_cuda_operand_{1}{2}, {7})
|
||||
{{ asm volatile("atom.{0}{4}{6}.{1}{2} %0,[%1],%2;" : "={3}"(__dst) : "l"(__ptr), "{3}"(__op) : "memory"); }})XXX";
|
||||
|
||||
// 0 - Atomic Operation
|
||||
// 1 - Operand type constraint
|
||||
// 2 - Pointer op skip_v
|
||||
constexpr auto fetch_bind_invoke = R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco>
|
||||
struct __cuda_atomic_bind_fetch_{0} {{
|
||||
_Type* __ptr;
|
||||
_Type* __dst;
|
||||
_Type* __op;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {{
|
||||
__cuda_atomic_fetch_{0}(__ptr, *__dst, *__op, _Atomic_Memorder{{}}, _Tag{{}}, _Sco{{}});
|
||||
}}
|
||||
}};
|
||||
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{{
|
||||
{2}
|
||||
__op = __op * __skip_v;
|
||||
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
|
||||
_Type __dst{{}};
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
|
||||
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
|
||||
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
|
||||
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
|
||||
return __dst;
|
||||
}}
|
||||
template <class _Type, class _Up, class _Sco, __atomic_enable_if_native_{1}<_Type> = 0>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_{0}_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{{
|
||||
{2}
|
||||
__op = __op * __skip_v;
|
||||
using __proxy_t = typename __atomic_cuda_deduce_{1}<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_{1}<_Type>::__tag;
|
||||
_Type __dst{{}};
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
__proxy_t* __op_proxy = reinterpret_cast<__proxy_t*>(&__op);
|
||||
if (__cuda_fetch_{0}_weak_if_local(__ptr_proxy, *__op_proxy, __dst_proxy)) {{return __dst;}}
|
||||
__cuda_atomic_bind_fetch_{0}<__proxy_t, __proxy_tag, _Sco> __bound_{0}{{__ptr_proxy, __dst_proxy, __op_proxy}};
|
||||
__cuda_atomic_fetch_memory_order_dispatch(__bound_{0}, __memorder, _Sco{{}});
|
||||
return __dst;
|
||||
}}
|
||||
)XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
32,
|
||||
64,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Release,
|
||||
Semantic::Acq_Rel,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
for (auto& op_kp : op_support_map)
|
||||
{
|
||||
const auto& op_name = op_kp.first;
|
||||
const auto& op_type_kp = op_kp.second;
|
||||
const auto& type_list = op_type_kp.first;
|
||||
const auto& deduction = op_type_kp.second;
|
||||
for (auto type : type_list)
|
||||
{
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
const std::string proxy_type = operand_proxy_type(type, size);
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
// There is no atom.add.s64
|
||||
if (op_name == "add" && type == Operand::Signed && size == 64)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ op_name,
|
||||
/* 1 */ operand(type),
|
||||
/* 2 */ size,
|
||||
/* 3 */ constraints(type, size),
|
||||
/* 4 */ semantic(sem),
|
||||
/* 5 */ semantic_tag(sem),
|
||||
/* 6 */ scope(sco),
|
||||
/* 7 */ scope_tag(sco));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n" << std::format(fetch_bind_invoke, op_name, deduction, fetch_op_skip_v(op_name));
|
||||
}
|
||||
|
||||
out << R"XXX(
|
||||
template <class _Type, class _Up, class _Sco>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{
|
||||
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Up, class _Sco>
|
||||
[[nodiscard]] static inline _CCCL_DEVICE _Type __atomic_fetch_sub_cuda(_Type volatile* __ptr, _Up __op, int __memorder, _Sco)
|
||||
{
|
||||
return __atomic_fetch_add_cuda(__ptr, -__op, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // FETCH_OPS_H
|
||||
@@ -1,88 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef HEADER_H
|
||||
#define HEADER_H
|
||||
|
||||
#include <string>
|
||||
|
||||
inline void FormatHeader(std::ostream& out)
|
||||
{
|
||||
constexpr auto header = R"XXX(//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
// This is an autogenerated file, we want to ensure that it contains exactly the contents we want to generate
|
||||
// clang-format off
|
||||
|
||||
#ifndef _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
#define _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cuda/std/cassert>
|
||||
#include <cuda/std/cstdint>
|
||||
|
||||
#include <cuda/std/__type_traits/enable_if.h>
|
||||
#include <cuda/std/__type_traits/is_signed.h>
|
||||
#include <cuda/std/__type_traits/is_unsigned.h>
|
||||
|
||||
#include <cuda/std/__atomic/scopes.h>
|
||||
#include <cuda/std/__atomic/order.h>
|
||||
#include <cuda/std/__atomic/functions/common.h>
|
||||
#include <cuda/std/__atomic/functions/cuda_ptx_generated_helper.h>
|
||||
#include <cuda/std/__atomic/functions/cuda_local.h>
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_STD
|
||||
|
||||
#if _CCCL_CUDA_COMPILATION()
|
||||
|
||||
extern "C" _CCCL_DEVICE void __atomic_cas_128b_unsupported_before_SM_90();
|
||||
extern "C" _CCCL_DEVICE void __atomic_exchange_128b_unsupported_before_SM_90();
|
||||
extern "C" _CCCL_DEVICE void __atomic_ldst_128b_unsupported_before_SM_70();
|
||||
)XXX";
|
||||
|
||||
out << header;
|
||||
}
|
||||
|
||||
inline void FormatTail(std::ostream& out)
|
||||
{
|
||||
constexpr auto tail = R"XXX(
|
||||
#endif // _CCCL_CUDA_COMPILATION()
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_STD
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA_STD___ATOMIC_FUNCTIONS_CUDA_PTX_GENERATED_H
|
||||
|
||||
// clang-format on
|
||||
)XXX";
|
||||
|
||||
out << tail;
|
||||
}
|
||||
|
||||
#endif // HEADER_H
|
||||
@@ -1,407 +0,0 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef LD_ST_H
|
||||
#define LD_ST_H
|
||||
|
||||
#include <format>
|
||||
#include <string>
|
||||
|
||||
#include "definitions.h"
|
||||
|
||||
inline std::string semantic_ld_st(Semantic sem)
|
||||
{
|
||||
static std::map sem_map = {
|
||||
std::pair{Semantic::Relaxed, ".relaxed"},
|
||||
std::pair{Semantic::Release, ".release"},
|
||||
std::pair{Semantic::Acquire, ".acquire"},
|
||||
std::pair{Semantic::Volatile, ".volatile"},
|
||||
};
|
||||
return sem_map[sem];
|
||||
}
|
||||
|
||||
inline std::string scope_ld_st(Semantic sem, Scope sco)
|
||||
{
|
||||
if (sem == Semantic::Volatile)
|
||||
{
|
||||
return "";
|
||||
}
|
||||
return scope(sco);
|
||||
}
|
||||
|
||||
inline void FormatLoad(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load_memory_order_dispatch(_Fn &__cuda_load, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_acquire{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_CONSUME: [[fallthrough]];
|
||||
case __ATOMIC_ACQUIRE: __cuda_load(__atomic_cuda_volatile{}); __cuda_atomic_membar(_Sco{}); break;
|
||||
case __ATOMIC_RELAXED: __cuda_load(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Constraint
|
||||
// 3 - Memory order
|
||||
// 4 - Memory order semantic
|
||||
// 5 - Scope tag
|
||||
// 6 - Scope semantic
|
||||
// 7 - Mmio tag
|
||||
// 8 - Mmio semantic
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load(
|
||||
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (),
|
||||
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _d;
|
||||
ld{8}{4}{6}.b128 _d,[%2];
|
||||
mov.b128 {{%0, %1}}, _d;
|
||||
}}
|
||||
)YYY" : "=l"(__dst.__x),"=l"(__dst.__y) : "l"(__ptr) : "memory");
|
||||
}})XXX";
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_load(
|
||||
const _Type* __ptr, _Type& __dst, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{ asm volatile("ld{8}{4}{6}.{0}{1} %0,[%1];" : "={2}"(__dst) : "l"(__ptr) : "memory"); }})XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
16,
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
Operand::Floating,
|
||||
Operand::Unsigned,
|
||||
Operand::Signed,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Acquire,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
constexpr Mmio mmio_states[] = {
|
||||
Mmio::Disabled,
|
||||
Mmio::Enabled,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto mm : mmio_states)
|
||||
{
|
||||
if (size == 16 && type == Operand::Floating)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
|
||||
struct __cuda_atomic_bind_load {
|
||||
const _Type* __ptr;
|
||||
_Type* __dst;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_load(__ptr, *__dst, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type* __ptr, _Type& __dst, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(__ptr);
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
|
||||
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_load_cuda(const _Type volatile* __ptr, _Type& __dst, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
const __proxy_t* __ptr_proxy = reinterpret_cast<const __proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __dst_proxy = reinterpret_cast<__proxy_t*>(&__dst);
|
||||
if (__cuda_load_weak_if_local(__ptr_proxy, __dst_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_load<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_load{__ptr_proxy, __dst_proxy};
|
||||
__cuda_atomic_load_memory_order_dispatch(__bound_load, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
inline void FormatStore(std::ostream& out)
|
||||
{
|
||||
out << R"XXX(
|
||||
template <class _Fn, class _Sco>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store_memory_order_dispatch(_Fn &__cuda_store, int __memorder, _Sco) {
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_RELEASE: __cuda_store(__atomic_cuda_release{}); break;
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); [[fallthrough]];
|
||||
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_relaxed{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
),
|
||||
NV_IS_DEVICE, (
|
||||
switch (__memorder) {
|
||||
case __ATOMIC_RELEASE: [[fallthrough]];
|
||||
case __ATOMIC_SEQ_CST: __cuda_atomic_membar(_Sco{}); [[fallthrough]];
|
||||
case __ATOMIC_RELAXED: __cuda_store(__atomic_cuda_volatile{}); break;
|
||||
default: _CCCL_ASSERT(false, "invalid memory order");
|
||||
}
|
||||
)
|
||||
)
|
||||
}
|
||||
)XXX";
|
||||
// Argument ID Reference
|
||||
// 0 - Operand Type
|
||||
// 1 - Operand Size
|
||||
// 2 - Constraint
|
||||
// 3 - Memory order
|
||||
// 4 - Memory order semantic
|
||||
// 5 - Scope tag
|
||||
// 6 - Scope semantic
|
||||
// 7 - Mmio tag
|
||||
// 8 - Mmio semantic
|
||||
constexpr auto asm_intrinsic_format_128 = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store(
|
||||
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{
|
||||
static_assert(__cccl_ptx_isa >= 840 && (sizeof(_Type) == 16), "128b ld/st is not supported until PTX ISA version 840");
|
||||
NV_DISPATCH_TARGET(
|
||||
NV_PROVIDES_SM_70, (),
|
||||
NV_ANY_TARGET, (__atomic_ldst_128b_unsupported_before_SM_70();)
|
||||
)
|
||||
asm volatile(R"YYY(
|
||||
{{
|
||||
.reg .b128 _v;
|
||||
mov.b128 _v, {{%1, %2}};
|
||||
st{8}{4}{6}.b128 [%0],_v;
|
||||
}}
|
||||
)YYY" :: "l"(__ptr), "l"(__val.__x),"l"(__val.__y) : "memory");
|
||||
}})XXX";
|
||||
constexpr auto asm_intrinsic_format = R"XXX(
|
||||
template <class _Type>
|
||||
static inline _CCCL_DEVICE void __cuda_atomic_store(
|
||||
_Type* __ptr, _Type& __val, {3}, __atomic_cuda_operand_{0}{1}, {5}, {7})
|
||||
{{ asm volatile("st{8}{4}{6}.{0}{1} [%0],%1;" :: "l"(__ptr), "{2}"(__val) : "memory"); }})XXX";
|
||||
|
||||
constexpr size_t supported_sizes[] = {
|
||||
16,
|
||||
32,
|
||||
64,
|
||||
128,
|
||||
};
|
||||
|
||||
constexpr Operand supported_types[] = {
|
||||
Operand::Bit,
|
||||
};
|
||||
|
||||
constexpr Semantic supported_semantics[] = {
|
||||
Semantic::Release,
|
||||
Semantic::Relaxed,
|
||||
Semantic::Volatile,
|
||||
};
|
||||
|
||||
constexpr Scope supported_scopes[] = {
|
||||
Scope::CTA,
|
||||
Scope::Cluster,
|
||||
Scope::GPU,
|
||||
Scope::System,
|
||||
};
|
||||
|
||||
constexpr Mmio mmio_states[] = {
|
||||
Mmio::Disabled,
|
||||
Mmio::Enabled,
|
||||
};
|
||||
|
||||
for (auto size : supported_sizes)
|
||||
{
|
||||
for (auto type : supported_types)
|
||||
{
|
||||
for (auto sem : supported_semantics)
|
||||
{
|
||||
for (auto sco : supported_scopes)
|
||||
{
|
||||
for (auto mm : mmio_states)
|
||||
{
|
||||
if (size == 16 && type == Operand::Floating)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if (size == 128 && type != Operand::Bit)
|
||||
{
|
||||
continue;
|
||||
}
|
||||
if ((mm == Mmio::Enabled) && ((sco != Scope::System) || (sem != Semantic::Relaxed)))
|
||||
{
|
||||
continue;
|
||||
}
|
||||
|
||||
if (size == 128)
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format_128,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
else
|
||||
{
|
||||
out << std::format(
|
||||
asm_intrinsic_format,
|
||||
/* 0 */ operand(type),
|
||||
/* 1 */ size,
|
||||
/* 2 */ constraints(type, size),
|
||||
/* 3 */ semantic_tag(sem),
|
||||
/* 4 */ semantic_ld_st(sem),
|
||||
/* 5 */ scope_tag(sco),
|
||||
/* 6 */ scope_ld_st(sem, sco),
|
||||
/* 7 */ mmio_tag(mm),
|
||||
/* 8 */ mmio(mm));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
out << "\n"
|
||||
<< R"XXX(
|
||||
template <typename _Type, typename _Tag, typename _Sco, typename _Mmio>
|
||||
struct __cuda_atomic_bind_store {
|
||||
_Type* __ptr;
|
||||
_Type* __val;
|
||||
|
||||
template <typename _Atomic_Memorder>
|
||||
inline _CCCL_DEVICE void operator()(_Atomic_Memorder) {
|
||||
__cuda_atomic_store(__ptr, *__val, _Atomic_Memorder{}, _Tag{}, _Sco{}, _Mmio{});
|
||||
}
|
||||
};
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_store_cuda(_Type* __ptr, _Type& __val, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(__ptr);
|
||||
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
|
||||
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
|
||||
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
|
||||
}
|
||||
template <class _Type, class _Sco>
|
||||
static inline _CCCL_DEVICE void __atomic_store_cuda(volatile _Type* __ptr, _Type& __val, int __memorder, _Sco)
|
||||
{
|
||||
using __proxy_t = typename __atomic_cuda_deduce_bitwise<_Type>::__type;
|
||||
using __proxy_tag = typename __atomic_cuda_deduce_bitwise<_Type>::__tag;
|
||||
__proxy_t* __ptr_proxy = reinterpret_cast<__proxy_t*>(const_cast<_Type*>(__ptr));
|
||||
__proxy_t* __val_proxy = reinterpret_cast<__proxy_t*>(&__val);
|
||||
if (__cuda_store_weak_if_local(__ptr_proxy, __val_proxy, sizeof(__proxy_t))) {{return;}}
|
||||
__cuda_atomic_bind_store<__proxy_t, __proxy_tag, _Sco, __atomic_cuda_mmio_disable> __bound_store{__ptr_proxy, __val_proxy};
|
||||
__cuda_atomic_store_memory_order_dispatch(__bound_store, __memorder, _Sco{});
|
||||
}
|
||||
)XXX";
|
||||
}
|
||||
|
||||
#endif // LD_ST_H
|
||||
@@ -1,34 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""GDB entry point for CUDA C++ Core Libraries pretty printers.
|
||||
|
||||
Requires Python 3.12 or newer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from pathlib import Path
|
||||
|
||||
import gdb
|
||||
|
||||
_SCRIPT_DIRECTORY = str(Path(__file__).resolve().parent)
|
||||
if _SCRIPT_DIRECTORY not in sys.path:
|
||||
sys.path.insert(0, _SCRIPT_DIRECTORY)
|
||||
|
||||
import buffer # noqa: E402
|
||||
import memory_resource # noqa: E402
|
||||
import std_array # noqa: E402
|
||||
|
||||
_PRINTERS = (memory_resource, buffer, std_array)
|
||||
|
||||
|
||||
def register() -> None:
|
||||
"""Register every CCCL GDB pretty printer."""
|
||||
for printer in _PRINTERS:
|
||||
printer.register(gdb)
|
||||
|
||||
|
||||
register()
|
||||
@@ -1,127 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""GDB pretty printer for cuda::buffer."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Iterator
|
||||
from types import ModuleType
|
||||
|
||||
import memory_resource
|
||||
|
||||
import gdb
|
||||
import gdb.printing
|
||||
|
||||
# GDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
|
||||
# parser does not necessarily import the enum constants. This is the value of
|
||||
# cudaMemcpyDefault from the CUDA Runtime API.
|
||||
_CUDA_MEMCPY_DEFAULT = 4
|
||||
|
||||
|
||||
def _template_name(value_type: gdb.Type) -> str:
|
||||
return str(value_type).split("<", 1)[0]
|
||||
|
||||
|
||||
def _is_cuda_buffer(value_type: gdb.Type) -> bool:
|
||||
# strip_typedefs resolves aliases that can hide accessibility properties.
|
||||
value_type = value_type.strip_typedefs().unqualified()
|
||||
template_name = _template_name(value_type)
|
||||
return (
|
||||
template_name.startswith("cuda::")
|
||||
and template_name.rsplit("::", 1)[-1] == "buffer"
|
||||
)
|
||||
|
||||
|
||||
class BufferPrinter:
|
||||
"""Expose cuda::buffer metadata and elements to GDB."""
|
||||
|
||||
def __init__(self, value: gdb.Value) -> None:
|
||||
self.value = value
|
||||
self.type = value.type.strip_typedefs().unqualified()
|
||||
self.type_name = memory_resource.public_type_name(self.type)
|
||||
self.value_type = self.type.template_argument(0)
|
||||
|
||||
storage = value["__buf_"]
|
||||
self.memory_resource = storage["__mr_"]
|
||||
self.stream = int(storage["__stream_"]["__stream"])
|
||||
self.size = int(storage["__count_"])
|
||||
self.alignment = int(storage["__alignment_"])
|
||||
raw_address = int(storage["__buf_"])
|
||||
self.data_address = (raw_address + self.alignment - 1) & ~(self.alignment - 1)
|
||||
|
||||
host_accessible = "host_accessible" in self.type_name
|
||||
device_accessible = "device_accessible" in self.type_name
|
||||
if host_accessible and device_accessible:
|
||||
self.accessibility = "host/device"
|
||||
elif device_accessible:
|
||||
self.accessibility = "device"
|
||||
elif host_accessible:
|
||||
self.accessibility = "host"
|
||||
else:
|
||||
self.accessibility = "unknown"
|
||||
|
||||
self.host_copy: gdb.Value | None = None
|
||||
self._copy_to_host()
|
||||
|
||||
def __del__(self) -> None:
|
||||
self.clear()
|
||||
|
||||
def clear(self) -> None:
|
||||
"""Release state staged in the inferior for synthetic children."""
|
||||
if self.host_copy is None:
|
||||
return
|
||||
try:
|
||||
gdb.parse_and_eval(f"(void)free((void*){int(self.host_copy):#x})")
|
||||
except gdb.error:
|
||||
pass
|
||||
self.host_copy = None
|
||||
|
||||
def _copy_to_host(self) -> None:
|
||||
if self.size == 0:
|
||||
return
|
||||
|
||||
byte_count = self.size * self.value_type.sizeof
|
||||
self.host_copy = gdb.parse_and_eval(f"(void*)malloc({byte_count})")
|
||||
host_address = int(self.host_copy)
|
||||
status = gdb.parse_and_eval(
|
||||
"(int)cudaMemcpy((void*)"
|
||||
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
|
||||
f"{_CUDA_MEMCPY_DEFAULT})"
|
||||
)
|
||||
if int(status) != 0:
|
||||
self.clear()
|
||||
|
||||
def children(self) -> Iterator[tuple[str, gdb.Value]]:
|
||||
if self.host_copy is None:
|
||||
return
|
||||
|
||||
pointer = self.host_copy.cast(self.value_type.pointer())
|
||||
for index in range(self.size):
|
||||
yield f"[{index}]", (pointer + index).dereference()
|
||||
|
||||
def to_string(self) -> str:
|
||||
resource = memory_resource.memory_resource_description(self.memory_resource)
|
||||
return (
|
||||
f"{self.type_name} mr={resource}, stream={self.stream:#x}, "
|
||||
f"size={self.size}, align={self.alignment}, "
|
||||
f"data={self.data_address:#x} ({self.accessibility})"
|
||||
)
|
||||
|
||||
|
||||
class BufferPrinterLookup(gdb.printing.PrettyPrinter):
|
||||
"""Select the cuda::buffer printer by its public class name."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__("cuda::buffer")
|
||||
|
||||
def __call__(self, value: gdb.Value) -> BufferPrinter | None:
|
||||
if _is_cuda_buffer(value.type):
|
||||
return BufferPrinter(value)
|
||||
return None
|
||||
|
||||
|
||||
def register(objfile: ModuleType) -> None:
|
||||
"""Register the cuda::buffer printer with GDB."""
|
||||
gdb.printing.register_pretty_printer(objfile, BufferPrinterLookup(), replace=True)
|
||||
@@ -1,76 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""GDB pretty printer for CUDA type-erased memory resources."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from types import ModuleType
|
||||
|
||||
import gdb
|
||||
import gdb.printing
|
||||
|
||||
_ABI_NAMESPACE_PATTERN = re.compile(r"::__(?:\d+|version_bump_ver\d+_)(?=::)")
|
||||
_RESOURCE_NAMES = frozenset(
|
||||
{"any_resource", "any_synchronous_resource", "basic_any_resource"}
|
||||
)
|
||||
|
||||
|
||||
def public_type_name(value_type: gdb.Type) -> str:
|
||||
"""Return a type name without CUDA ABI inline namespaces."""
|
||||
return _ABI_NAMESPACE_PATTERN.sub("", str(value_type))
|
||||
|
||||
|
||||
def _template_name(value_type: gdb.Type) -> str:
|
||||
return str(value_type).split("<", 1)[0]
|
||||
|
||||
|
||||
def _is_memory_resource(value_type: gdb.Type) -> bool:
|
||||
value_type = value_type.strip_typedefs().unqualified()
|
||||
type_name = public_type_name(value_type)
|
||||
template_name = _template_name(value_type)
|
||||
return (
|
||||
type_name.startswith("cuda::mr::")
|
||||
and template_name.rsplit("::", 1)[-1] in _RESOURCE_NAMES
|
||||
)
|
||||
|
||||
|
||||
def memory_resource_description(value: gdb.Value) -> str:
|
||||
value_type = value.type.strip_typedefs().unqualified()
|
||||
type_name = public_type_name(value_type)
|
||||
try:
|
||||
address = int(value.address)
|
||||
except (gdb.error, TypeError):
|
||||
return type_name
|
||||
return f"{type_name} @ {address:#x}"
|
||||
|
||||
|
||||
class MemoryResourcePrinter:
|
||||
"""Summarize a CUDA type-erased memory resource."""
|
||||
|
||||
def __init__(self, value: gdb.Value) -> None:
|
||||
self.value = value
|
||||
|
||||
def to_string(self) -> str:
|
||||
return memory_resource_description(self.value)
|
||||
|
||||
|
||||
class MemoryResourcePrinterLookup(gdb.printing.PrettyPrinter):
|
||||
"""Select printers for public CUDA type-erased resource types."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__("cuda::mr::any_resource")
|
||||
|
||||
def __call__(self, value: gdb.Value) -> MemoryResourcePrinter | None:
|
||||
if _is_memory_resource(value.type):
|
||||
return MemoryResourcePrinter(value)
|
||||
return None
|
||||
|
||||
|
||||
def register(objfile: ModuleType) -> None:
|
||||
"""Register CUDA memory-resource formatters with GDB."""
|
||||
gdb.printing.register_pretty_printer(
|
||||
objfile, MemoryResourcePrinterLookup(), replace=True
|
||||
)
|
||||
@@ -1,63 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""GDB pretty printer for cuda::std::array."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from collections.abc import Iterator
|
||||
from types import ModuleType
|
||||
|
||||
import memory_resource
|
||||
|
||||
import gdb
|
||||
import gdb.printing
|
||||
|
||||
|
||||
def _template_name(value_type: gdb.Type) -> str:
|
||||
return str(value_type).split("<", 1)[0]
|
||||
|
||||
|
||||
def _is_cuda_array(value_type: gdb.Type) -> bool:
|
||||
value_type = value_type.strip_typedefs().unqualified()
|
||||
template_name = _template_name(value_type)
|
||||
return (
|
||||
template_name.startswith("cuda::std::")
|
||||
and template_name.rsplit("::", 1)[-1] == "array"
|
||||
)
|
||||
|
||||
|
||||
class ArrayPrinter:
|
||||
"""Expose cuda::std::array metadata and elements to GDB."""
|
||||
|
||||
def __init__(self, value: gdb.Value) -> None:
|
||||
self.value = value
|
||||
self.type = value.type.strip_typedefs().unqualified()
|
||||
self.type_name = memory_resource.public_type_name(self.type)
|
||||
self.size = int(self.type.template_argument(1))
|
||||
|
||||
def children(self) -> Iterator[tuple[str, gdb.Value]]:
|
||||
elems = self.value["__elems_"]
|
||||
for index in range(self.size):
|
||||
yield f"[{index}]", elems[index]
|
||||
|
||||
def to_string(self) -> str:
|
||||
return self.type_name
|
||||
|
||||
|
||||
class ArrayPrinterLookup(gdb.printing.PrettyPrinter):
|
||||
"""Select the cuda::std::array printer by its public class name."""
|
||||
|
||||
def __init__(self) -> None:
|
||||
super().__init__("cuda::std::array")
|
||||
|
||||
def __call__(self, value: gdb.Value) -> ArrayPrinter | None:
|
||||
if _is_cuda_array(value.type):
|
||||
return ArrayPrinter(value)
|
||||
return None
|
||||
|
||||
|
||||
def register(objfile: ModuleType) -> None:
|
||||
"""Register the cuda::std::array printer with GDB."""
|
||||
gdb.printing.register_pretty_printer(objfile, ArrayPrinterLookup(), replace=True)
|
||||
@@ -1,28 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""LLDB entry point for CUDA C++ Core Libraries pretty printers.
|
||||
|
||||
Requires Python 3.12 or newer.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import buffer
|
||||
import memory_resource
|
||||
import std_array
|
||||
|
||||
import lldb
|
||||
|
||||
_CATEGORY = "cccl"
|
||||
_FORMATTERS = (memory_resource, buffer, std_array)
|
||||
InternalDict = dict[str, object]
|
||||
|
||||
|
||||
def __lldb_init_module(debugger: lldb.SBDebugger, _internal_dict: InternalDict) -> None:
|
||||
debugger.HandleCommand(f"type category define {_CATEGORY}")
|
||||
for formatter in _FORMATTERS:
|
||||
module = f"{__name__}.{formatter.__name__.rsplit('.', 1)[-1]}"
|
||||
formatter.register(debugger, _CATEGORY, module)
|
||||
debugger.HandleCommand(f"type category enable {_CATEGORY}")
|
||||
@@ -1,219 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""LLDB pretty printer for cuda::buffer."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import NamedTuple
|
||||
|
||||
import memory_resource
|
||||
|
||||
import lldb
|
||||
|
||||
_BUFFER_PATTERN = re.compile(r"^cuda::buffer<.+>$")
|
||||
# LLDB sees cudaMemcpyKind through cudaMemcpy's declaration, but its expression
|
||||
# parser does not necessarily import the enum constants. This is the value of
|
||||
# cudaMemcpyDefault from the CUDA Runtime API.
|
||||
_CUDA_MEMCPY_DEFAULT = 4
|
||||
InternalDict = dict[str, object]
|
||||
|
||||
|
||||
class BufferInfo(NamedTuple):
|
||||
size: int
|
||||
data_address: int
|
||||
value_type: lldb.SBType
|
||||
accessibility: str
|
||||
memory_resource: lldb.SBValue
|
||||
stream: lldb.SBValue
|
||||
alignment: lldb.SBValue
|
||||
|
||||
|
||||
def is_cuda_buffer(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
|
||||
type_name = (
|
||||
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
|
||||
)
|
||||
return _BUFFER_PATTERN.fullmatch(type_name) is not None
|
||||
|
||||
|
||||
def _buffer_info(value: lldb.SBValue) -> BufferInfo | None:
|
||||
value = value.GetNonSyntheticValue()
|
||||
storage = value.GetChildMemberWithName("__buf_")
|
||||
if not storage.IsValid():
|
||||
return None
|
||||
|
||||
count = storage.GetChildMemberWithName("__count_")
|
||||
memory_resource = storage.GetChildMemberWithName("__mr_")
|
||||
stream_ref = storage.GetChildMemberWithName("__stream_")
|
||||
stream = stream_ref.GetChildMemberWithName("__stream")
|
||||
alignment = storage.GetChildMemberWithName("__alignment_")
|
||||
allocation = storage.GetChildMemberWithName("__buf_")
|
||||
if not all(
|
||||
child.IsValid()
|
||||
for child in (count, memory_resource, stream, alignment, allocation)
|
||||
):
|
||||
return None
|
||||
|
||||
# A source-level alias can hide the accessibility properties from
|
||||
# GetTypeName(), so use the canonical public type for all property checks.
|
||||
buffer_type = value.GetType().GetCanonicalType().GetUnqualifiedType()
|
||||
value_type = buffer_type.GetTemplateArgumentType(0)
|
||||
if not value_type.IsValid():
|
||||
return None
|
||||
|
||||
type_name = buffer_type.GetDisplayTypeName() or ""
|
||||
host_accessible = "host_accessible" in type_name
|
||||
device_accessible = "device_accessible" in type_name
|
||||
if host_accessible and device_accessible:
|
||||
accessibility = "host/device"
|
||||
elif device_accessible:
|
||||
accessibility = "device"
|
||||
elif host_accessible:
|
||||
accessibility = "host"
|
||||
else:
|
||||
accessibility = "unknown"
|
||||
size = count.GetValueAsUnsigned(0)
|
||||
align = alignment.GetValueAsUnsigned(1)
|
||||
raw_address = allocation.GetValueAsUnsigned(0)
|
||||
data_address = (raw_address + align - 1) & ~(align - 1)
|
||||
return BufferInfo(
|
||||
size,
|
||||
data_address,
|
||||
value_type,
|
||||
accessibility,
|
||||
memory_resource,
|
||||
stream,
|
||||
alignment,
|
||||
)
|
||||
|
||||
|
||||
def buffer_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str | None:
|
||||
info = _buffer_info(value)
|
||||
if info is None:
|
||||
return None
|
||||
resource = memory_resource.memory_resource_description(info.memory_resource)
|
||||
stream = info.stream.GetValueAsUnsigned(0)
|
||||
alignment = info.alignment.GetValueAsUnsigned(0)
|
||||
return (
|
||||
f"mr={resource}, stream={stream:#x}, size={info.size}, align={alignment}, "
|
||||
f"data={info.data_address:#x} ({info.accessibility})"
|
||||
)
|
||||
|
||||
|
||||
class BufferSyntheticProvider:
|
||||
"""Expose cuda::buffer elements as LLDB synthetic children."""
|
||||
|
||||
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
|
||||
self.value = value.GetNonSyntheticValue()
|
||||
self.host_copy = lldb.SBValue()
|
||||
self.clear()
|
||||
self.update()
|
||||
|
||||
def __del__(self) -> None:
|
||||
self.clear()
|
||||
|
||||
def _evaluate(self, expression: str) -> lldb.SBValue:
|
||||
frame = self.value.GetFrame()
|
||||
if not frame.IsValid():
|
||||
return lldb.SBValue()
|
||||
options = lldb.SBExpressionOptions()
|
||||
options.SetIgnoreBreakpoints(True)
|
||||
options.SetUnwindOnError(True)
|
||||
return frame.EvaluateExpression(expression, options)
|
||||
|
||||
def clear(self) -> None:
|
||||
"""Release the staged copy and reset all cached buffer information."""
|
||||
if self.host_copy.IsValid():
|
||||
address = self.host_copy.GetValueAsUnsigned(0)
|
||||
if address:
|
||||
self._evaluate(f"(void)free((void*){address:#x})")
|
||||
self.host_copy = lldb.SBValue()
|
||||
self.size = 0
|
||||
self.data_address = 0
|
||||
self.value_type = lldb.SBType()
|
||||
self.value_size = 0
|
||||
|
||||
def _copy_to_host(self) -> bool:
|
||||
if self.size == 0:
|
||||
return True
|
||||
|
||||
byte_count = self.size * self.value_size
|
||||
self.host_copy = self._evaluate(f"(void*)malloc({byte_count})")
|
||||
if not self.host_copy.IsValid() or self.host_copy.GetError().Fail():
|
||||
return False
|
||||
|
||||
host_address = self.host_copy.GetValueAsUnsigned(0)
|
||||
result = self._evaluate(
|
||||
"(int)cudaMemcpy((void*)"
|
||||
f"{host_address:#x}, (const void*){self.data_address:#x}, {byte_count}, "
|
||||
f"{_CUDA_MEMCPY_DEFAULT})"
|
||||
)
|
||||
if (
|
||||
not result.IsValid()
|
||||
or result.GetError().Fail()
|
||||
or result.GetValueAsSigned(-1) != 0
|
||||
):
|
||||
self.clear()
|
||||
return False
|
||||
return True
|
||||
|
||||
def update(self) -> bool:
|
||||
self.clear()
|
||||
|
||||
info = _buffer_info(self.value)
|
||||
if info is None:
|
||||
return False
|
||||
|
||||
self.size = info.size
|
||||
self.data_address = info.data_address
|
||||
self.value_type = info.value_type
|
||||
self.value_size = self.value_type.GetByteSize()
|
||||
self._copy_to_host()
|
||||
return True
|
||||
|
||||
def num_children(self) -> int:
|
||||
return self.size
|
||||
|
||||
def has_children(self) -> bool:
|
||||
return self.size != 0
|
||||
|
||||
def get_type_name(self) -> str:
|
||||
# STL element access can preserve an alloc_traits::value_type typedef.
|
||||
# Report the canonical display name so LLDB shows cuda::buffer instead.
|
||||
return (
|
||||
self.value.GetType()
|
||||
.GetCanonicalType()
|
||||
.GetUnqualifiedType()
|
||||
.GetDisplayTypeName()
|
||||
or ""
|
||||
)
|
||||
|
||||
def get_child_index(self, name: str) -> int:
|
||||
if name.startswith("[") and name.endswith("]"):
|
||||
try:
|
||||
return int(name[1:-1])
|
||||
except ValueError:
|
||||
pass
|
||||
return -1
|
||||
|
||||
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
|
||||
if index < 0:
|
||||
return None
|
||||
if index >= self.size:
|
||||
return None
|
||||
offset = index * self.value_size
|
||||
return self.host_copy.CreateChildAtOffset(f"[{index}]", offset, self.value_type)
|
||||
|
||||
|
||||
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
|
||||
"""Register the cuda::buffer formatter in an LLDB category."""
|
||||
debugger.HandleCommand(
|
||||
f"type summary add --category {category} --expand --python-function {module}.buffer_summary "
|
||||
f"--recognizer-function {module}.is_cuda_buffer"
|
||||
)
|
||||
debugger.HandleCommand(
|
||||
f"type synthetic add --category {category} --python-class {module}.BufferSyntheticProvider "
|
||||
f"--recognizer-function {module}.is_cuda_buffer"
|
||||
)
|
||||
@@ -1,50 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""LLDB pretty printer for CUDA type-erased memory resources."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import lldb
|
||||
|
||||
_RESOURCE_PATTERN = re.compile(
|
||||
r"^cuda::mr::(?:basic_any_resource|any_resource|any_synchronous_resource)<.+>$"
|
||||
)
|
||||
InternalDict = dict[str, object]
|
||||
|
||||
|
||||
def is_memory_resource(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
|
||||
type_name = (
|
||||
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
|
||||
)
|
||||
return _RESOURCE_PATTERN.fullmatch(type_name) is not None
|
||||
|
||||
|
||||
def memory_resource_description(value: lldb.SBValue) -> str:
|
||||
"""Describe a type-erased resource using only public type information."""
|
||||
type_name = (
|
||||
value.GetType().GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName()
|
||||
)
|
||||
if not type_name:
|
||||
type_name = "type-erased resource"
|
||||
|
||||
address = value.GetLoadAddress()
|
||||
if address == lldb.LLDB_INVALID_ADDRESS:
|
||||
return type_name
|
||||
return f"{type_name} @ {address:#x}"
|
||||
|
||||
|
||||
def memory_resource_summary(value: lldb.SBValue, _internal_dict: InternalDict) -> str:
|
||||
return memory_resource_description(value)
|
||||
|
||||
|
||||
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
|
||||
"""Register CUDA memory-resource formatters in an LLDB category."""
|
||||
debugger.HandleCommand(
|
||||
f"type summary add --category {category} --python-function "
|
||||
f"{module}.memory_resource_summary --recognizer-function "
|
||||
f"{module}.is_memory_resource"
|
||||
)
|
||||
@@ -1,78 +0,0 @@
|
||||
# Copyright (c) 2026, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
"""LLDB pretty printer for cuda::std::array."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
|
||||
import lldb
|
||||
|
||||
_ARRAY_PATTERN = re.compile(r"^cuda::std::array<.+,\s*(\d+)>$")
|
||||
InternalDict = dict[str, object]
|
||||
|
||||
|
||||
def is_cuda_array(value_type: lldb.SBType, _internal_dict: InternalDict) -> bool:
|
||||
type_name = (
|
||||
value_type.GetCanonicalType().GetUnqualifiedType().GetDisplayTypeName() or ""
|
||||
)
|
||||
return _ARRAY_PATTERN.fullmatch(type_name) is not None
|
||||
|
||||
|
||||
class ArraySyntheticProvider:
|
||||
"""Expose cuda::std::array elements as LLDB synthetic children."""
|
||||
|
||||
def __init__(self, value: lldb.SBValue, _internal_dict: InternalDict) -> None:
|
||||
self.value = value.GetNonSyntheticValue()
|
||||
self.update()
|
||||
|
||||
def update(self) -> bool:
|
||||
type_name = (
|
||||
self.value.GetType()
|
||||
.GetCanonicalType()
|
||||
.GetUnqualifiedType()
|
||||
.GetDisplayTypeName()
|
||||
or ""
|
||||
)
|
||||
self.type_name = type_name
|
||||
match = _ARRAY_PATTERN.fullmatch(type_name)
|
||||
self.elems = self.value.GetChildMemberWithName("__elems_")
|
||||
self.size = 0
|
||||
if not self.elems.IsValid() or not match:
|
||||
return False
|
||||
self.size = int(match.group(1))
|
||||
return True
|
||||
|
||||
def num_children(self) -> int:
|
||||
return self.size
|
||||
|
||||
def has_children(self) -> bool:
|
||||
return self.size != 0
|
||||
|
||||
def get_type_name(self) -> str:
|
||||
return self.type_name
|
||||
|
||||
def get_child_index(self, name: str) -> int:
|
||||
if name.startswith("[") and name.endswith("]"):
|
||||
try:
|
||||
return int(name[1:-1])
|
||||
except ValueError:
|
||||
pass
|
||||
return -1
|
||||
|
||||
def get_child_at_index(self, index: int) -> lldb.SBValue | None:
|
||||
if index < 0:
|
||||
return None
|
||||
if index >= self.size:
|
||||
return None
|
||||
return self.elems.GetChildAtIndex(index)
|
||||
|
||||
|
||||
def register(debugger: lldb.SBDebugger, category: str, module: str) -> None:
|
||||
"""Register the cuda::std::array formatter in an LLDB category."""
|
||||
debugger.HandleCommand(
|
||||
f"type synthetic add --category {category} --python-class {module}.ArraySyntheticProvider "
|
||||
f"--recognizer-function {module}.is_cuda_array"
|
||||
)
|
||||
68
cccl_upstream/libcudacxx/test/.gitignore
vendored
68
cccl_upstream/libcudacxx/test/.gitignore
vendored
@@ -1,68 +0,0 @@
|
||||
# Byte-compiled / optimized / DLL files
|
||||
__pycache__/
|
||||
*.py[cod]
|
||||
|
||||
# C extensions
|
||||
*.so
|
||||
|
||||
# Distribution / packaging
|
||||
.Python
|
||||
env/
|
||||
build/
|
||||
develop-eggs/
|
||||
dist/
|
||||
downloads/
|
||||
eggs/
|
||||
#lib/ # We actually have things checked in to lib/
|
||||
lib64/
|
||||
parts/
|
||||
sdist/
|
||||
var/
|
||||
*.egg-info/
|
||||
.installed.cfg
|
||||
*.egg
|
||||
|
||||
# PyInstaller
|
||||
# Usually these files are written by a python script from a template
|
||||
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
||||
*.manifest
|
||||
*.spec
|
||||
!*.spec/
|
||||
|
||||
# Installer logs
|
||||
pip-log.txt
|
||||
pip-delete-this-directory.txt
|
||||
|
||||
# Unit test / coverage reports
|
||||
htmlcov/
|
||||
.tox/
|
||||
.coverage
|
||||
.cache
|
||||
nosetests.xml
|
||||
coverage.xml
|
||||
|
||||
# Translations
|
||||
*.mo
|
||||
*.pot
|
||||
|
||||
# Django stuff:
|
||||
*.log
|
||||
|
||||
# Sphinx documentation
|
||||
docs/_build/
|
||||
|
||||
# PyBuilder
|
||||
target/
|
||||
|
||||
# MSVC libraries test harness
|
||||
env.lst
|
||||
keep.lst
|
||||
|
||||
# Editor by-products
|
||||
.vscode/
|
||||
|
||||
# Random things
|
||||
build-*
|
||||
|
||||
# Perforce files
|
||||
.p4config
|
||||
@@ -1,102 +0,0 @@
|
||||
find_package(Python COMPONENTS Interpreter)
|
||||
if (NOT Python_Interpreter_FOUND)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Failed to find python interpreter, which is required for running tests and building a libcu++ static library."
|
||||
)
|
||||
endif()
|
||||
|
||||
# Determine the host triple to avoid invoking `${CXX} -dumpmachine`.
|
||||
include(GetHostTriple)
|
||||
get_host_triple(LLVM_INFERRED_HOST_TRIPLE)
|
||||
|
||||
set(
|
||||
LLVM_HOST_TRIPLE
|
||||
"${LLVM_INFERRED_HOST_TRIPLE}"
|
||||
CACHE STRING
|
||||
"Host on which LLVM binaries will run"
|
||||
)
|
||||
|
||||
# By default, we target the host, but this can be overridden at CMake
|
||||
# invocation time.
|
||||
set(
|
||||
LLVM_DEFAULT_TARGET_TRIPLE
|
||||
"${LLVM_HOST_TRIPLE}"
|
||||
CACHE STRING
|
||||
"Default target for which LLVM will generate code."
|
||||
)
|
||||
set(TARGET_TRIPLE "${LLVM_DEFAULT_TARGET_TRIPLE}")
|
||||
message(STATUS "LLVM host triple: ${LLVM_HOST_TRIPLE}")
|
||||
message(STATUS "LLVM default target triple: ${LLVM_DEFAULT_TARGET_TRIPLE}")
|
||||
|
||||
set(LIT_EXTRA_ARGS "" CACHE STRING "Use for additional options (e.g. -j12)")
|
||||
find_program(LLVM_DEFAULT_EXTERNAL_LIT lit)
|
||||
set(LLVM_LIT_ARGS "-sv ${LIT_EXTRA_ARGS}")
|
||||
|
||||
# Libcudacxx's main lit tests
|
||||
add_subdirectory(libcudacxx)
|
||||
|
||||
add_subdirectory(cmake)
|
||||
|
||||
# Set appropriate warning levels for MSVC/sane
|
||||
if ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "NVIDIA")
|
||||
# CUDA 11.5 and down do not support '-use-local-env'
|
||||
if (MSVC)
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Xcompiler=/W4
|
||||
-Xcompiler=/WX
|
||||
-Wno-deprecated-gpu-targets
|
||||
)
|
||||
if ("${CMAKE_CUDA_COMPILER_VERSION}" GREATER_EQUAL "11.6.0")
|
||||
list(APPEND headertest_warning_levels_device --use-local-env)
|
||||
endif()
|
||||
else()
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Wall
|
||||
-Werror
|
||||
all-warnings
|
||||
-Wno-deprecated-gpu-targets
|
||||
)
|
||||
endif()
|
||||
|
||||
if (
|
||||
CCCL_ENABLE_TILE
|
||||
AND "${CMAKE_CUDA_COMPILER_VERSION}" VERSION_LESS_EQUAL "13.2"
|
||||
)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"tile programs require NVCC 13.3 or later; found NVCC ${CMAKE_CUDA_COMPILER_VERSION}"
|
||||
)
|
||||
endif()
|
||||
# Set warnings for Clang as device compiler
|
||||
elseif ("${CMAKE_CUDA_COMPILER_ID}" STREQUAL "Clang")
|
||||
set(
|
||||
headertest_warning_levels_device
|
||||
-Wall
|
||||
-Werror
|
||||
-Wno-unknown-cuda-version
|
||||
-Xclang=-fcuda-allow-variadic-functions
|
||||
)
|
||||
# If the CMAKE_CUDA_COMPILER is unknown, try to use gcc style warnings
|
||||
else()
|
||||
set(headertest_warning_levels_device -Wall -Werror)
|
||||
endif()
|
||||
|
||||
# Set raw host/device warnings
|
||||
if (MSVC)
|
||||
set(headertest_warning_levels_host /W4 /WX)
|
||||
else()
|
||||
set(headertest_warning_levels_host -Wall -Werror)
|
||||
endif()
|
||||
|
||||
# Enable building the nvrtcc project if NVRTC is enabled
|
||||
if (LIBCUDACXX_TEST_WITH_NVRTC)
|
||||
add_subdirectory(utils/nvidia/nvrtc)
|
||||
endif()
|
||||
|
||||
add_subdirectory(nvtarget)
|
||||
add_subdirectory(atomic_codegen)
|
||||
add_subdirectory(simd_codegen)
|
||||
add_subdirectory(debugging)
|
||||
@@ -1,150 +0,0 @@
|
||||
This file is a partial list of people who have contributed to the LLVM/libc++
|
||||
project. If you have contributed a patch or made some other contribution to
|
||||
LLVM/libc++, please submit a patch to this file to add yourself, and it will be
|
||||
done!
|
||||
|
||||
The list is sorted by surname and formatted to allow easy grepping and
|
||||
beautification by scripts. The fields are: name (N), email (E), web-address
|
||||
(W), PGP key ID and fingerprint (P), description (D), and snail-mail address
|
||||
(S).
|
||||
|
||||
N: Saleem Abdulrasool
|
||||
E: compnerd@compnerd.org
|
||||
D: Minor patches and Linux fixes.
|
||||
|
||||
N: Dan Albert
|
||||
E: danalbert@google.com
|
||||
D: Android support and test runner improvements.
|
||||
|
||||
N: Dimitry Andric
|
||||
E: dimitry@andric.com
|
||||
D: Visibility fixes, minor FreeBSD portability patches.
|
||||
|
||||
N: Holger Arnold
|
||||
E: holgerar@gmail.com
|
||||
D: Minor fix.
|
||||
|
||||
N: Ruben Van Boxem
|
||||
E: vanboxem dot ruben at gmail dot com
|
||||
D: Initial Windows patches.
|
||||
|
||||
N: David Chisnall
|
||||
E: theraven at theravensnest dot org
|
||||
D: FreeBSD and Solaris ports, libcxxrt support, some atomics work.
|
||||
|
||||
N: Marshall Clow
|
||||
E: mclow.lists@gmail.com
|
||||
E: marshall@idio.com
|
||||
D: C++14 support, patches and bug fixes.
|
||||
|
||||
N: Jonathan B Coe
|
||||
E: jbcoe@me.com
|
||||
D: Implementation of propagate_const.
|
||||
|
||||
N: Glen Joseph Fernandes
|
||||
E: glenjofe@gmail.com
|
||||
D: Implementation of to_address.
|
||||
|
||||
N: Eric Fiselier
|
||||
E: eric@efcs.ca
|
||||
D: LFTS support, patches and bug fixes.
|
||||
|
||||
N: Bill Fisher
|
||||
E: william.w.fisher@gmail.com
|
||||
D: Regex bug fixes.
|
||||
|
||||
N: Matthew Dempsky
|
||||
E: matthew@dempsky.org
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Google Inc.
|
||||
D: Copyright owner and contributor of the CityHash algorithm
|
||||
|
||||
N: Howard Hinnant
|
||||
E: hhinnant@apple.com
|
||||
D: Architect and primary author of libc++
|
||||
|
||||
N: Hyeon-bin Jeong
|
||||
E: tuhertz@gmail.com
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Argyrios Kyrtzidis
|
||||
E: kyrtzidis@apple.com
|
||||
D: Bug fixes.
|
||||
|
||||
N: Bruce Mitchener, Jr.
|
||||
E: bruce.mitchener@gmail.com
|
||||
D: Emscripten-related changes.
|
||||
|
||||
N: Michel Morin
|
||||
E: mimomorin@gmail.com
|
||||
D: Minor patches to is_convertible.
|
||||
|
||||
N: Andrew Morrow
|
||||
E: andrew.c.morrow@gmail.com
|
||||
D: Minor patches and Linux fixes.
|
||||
|
||||
N: Michael Park
|
||||
E: mcypark@gmail.com
|
||||
D: Implementation of <variant>.
|
||||
|
||||
N: Arvid Picciani
|
||||
E: aep at exys dot org
|
||||
D: Minor patches and musl port.
|
||||
|
||||
N: Bjorn Reese
|
||||
E: breese@users.sourceforge.net
|
||||
D: Initial regex prototype
|
||||
|
||||
N: Nico Rieck
|
||||
E: nico.rieck@gmail.com
|
||||
D: Windows fixes
|
||||
|
||||
N: Jon Roelofs
|
||||
E: jroelofS@jroelofs.com
|
||||
D: Remote testing, Newlib port, baremetal/single-threaded support.
|
||||
|
||||
N: Jonathan Sauer
|
||||
D: Minor patches, mostly related to constexpr
|
||||
|
||||
N: Craig Silverstein
|
||||
E: csilvers@google.com
|
||||
D: Implemented Cityhash as the string hash function on 64-bit machines
|
||||
|
||||
N: Richard Smith
|
||||
D: Minor patches.
|
||||
|
||||
N: Joerg Sonnenberger
|
||||
E: joerg@NetBSD.org
|
||||
D: NetBSD port.
|
||||
|
||||
N: Stephan Tolksdorf
|
||||
E: st@quanttec.com
|
||||
D: Minor <atomic> fix
|
||||
|
||||
N: Michael van der Westhuizen
|
||||
E: r1mikey at gmail dot com
|
||||
|
||||
N: Larisse Voufo
|
||||
D: Minor patches.
|
||||
|
||||
N: Klaas de Vries
|
||||
E: klaas at klaasgaaf dot nl
|
||||
D: Minor bug fix.
|
||||
|
||||
N: Zhang Xiongpang
|
||||
E: zhangxiongpang@gmail.com
|
||||
D: Minor patches and bug fixes.
|
||||
|
||||
N: Xing Xue
|
||||
E: xingxue@ca.ibm.com
|
||||
D: AIX port
|
||||
|
||||
N: Zhihao Yuan
|
||||
E: lichray@gmail.com
|
||||
D: Standard compatibility fixes.
|
||||
|
||||
N: Jeffrey Yasskin
|
||||
E: jyasskin@gmail.com
|
||||
E: jyasskin@google.com
|
||||
D: Linux fixes.
|
||||
@@ -1,311 +0,0 @@
|
||||
==============================================================================
|
||||
The LLVM Project is under the Apache License v2.0 with LLVM Exceptions:
|
||||
==============================================================================
|
||||
|
||||
Apache License
|
||||
Version 2.0, January 2004
|
||||
http://www.apache.org/licenses/
|
||||
|
||||
TERMS AND CONDITIONS FOR USE, REPRODUCTION, AND DISTRIBUTION
|
||||
|
||||
1. Definitions.
|
||||
|
||||
"License" shall mean the terms and conditions for use, reproduction,
|
||||
and distribution as defined by Sections 1 through 9 of this document.
|
||||
|
||||
"Licensor" shall mean the copyright owner or entity authorized by
|
||||
the copyright owner that is granting the License.
|
||||
|
||||
"Legal Entity" shall mean the union of the acting entity and all
|
||||
other entities that control, are controlled by, or are under common
|
||||
control with that entity. For the purposes of this definition,
|
||||
"control" means (i) the power, direct or indirect, to cause the
|
||||
direction or management of such entity, whether by contract or
|
||||
otherwise, or (ii) ownership of fifty percent (50%) or more of the
|
||||
outstanding shares, or (iii) beneficial ownership of such entity.
|
||||
|
||||
"You" (or "Your") shall mean an individual or Legal Entity
|
||||
exercising permissions granted by this License.
|
||||
|
||||
"Source" form shall mean the preferred form for making modifications,
|
||||
including but not limited to software source code, documentation
|
||||
source, and configuration files.
|
||||
|
||||
"Object" form shall mean any form resulting from mechanical
|
||||
transformation or translation of a Source form, including but
|
||||
not limited to compiled object code, generated documentation,
|
||||
and conversions to other media types.
|
||||
|
||||
"Work" shall mean the work of authorship, whether in Source or
|
||||
Object form, made available under the License, as indicated by a
|
||||
copyright notice that is included in or attached to the work
|
||||
(an example is provided in the Appendix below).
|
||||
|
||||
"Derivative Works" shall mean any work, whether in Source or Object
|
||||
form, that is based on (or derived from) the Work and for which the
|
||||
editorial revisions, annotations, elaborations, or other modifications
|
||||
represent, as a whole, an original work of authorship. For the purposes
|
||||
of this License, Derivative Works shall not include works that remain
|
||||
separable from, or merely link (or bind by name) to the interfaces of,
|
||||
the Work and Derivative Works thereof.
|
||||
|
||||
"Contribution" shall mean any work of authorship, including
|
||||
the original version of the Work and any modifications or additions
|
||||
to that Work or Derivative Works thereof, that is intentionally
|
||||
submitted to Licensor for inclusion in the Work by the copyright owner
|
||||
or by an individual or Legal Entity authorized to submit on behalf of
|
||||
the copyright owner. For the purposes of this definition, "submitted"
|
||||
means any form of electronic, verbal, or written communication sent
|
||||
to the Licensor or its representatives, including but not limited to
|
||||
communication on electronic mailing lists, source code control systems,
|
||||
and issue tracking systems that are managed by, or on behalf of, the
|
||||
Licensor for the purpose of discussing and improving the Work, but
|
||||
excluding communication that is conspicuously marked or otherwise
|
||||
designated in writing by the copyright owner as "Not a Contribution."
|
||||
|
||||
"Contributor" shall mean Licensor and any individual or Legal Entity
|
||||
on behalf of whom a Contribution has been received by Licensor and
|
||||
subsequently incorporated within the Work.
|
||||
|
||||
2. Grant of Copyright License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
copyright license to reproduce, prepare Derivative Works of,
|
||||
publicly display, publicly perform, sublicense, and distribute the
|
||||
Work and such Derivative Works in Source or Object form.
|
||||
|
||||
3. Grant of Patent License. Subject to the terms and conditions of
|
||||
this License, each Contributor hereby grants to You a perpetual,
|
||||
worldwide, non-exclusive, no-charge, royalty-free, irrevocable
|
||||
(except as stated in this section) patent license to make, have made,
|
||||
use, offer to sell, sell, import, and otherwise transfer the Work,
|
||||
where such license applies only to those patent claims licensable
|
||||
by such Contributor that are necessarily infringed by their
|
||||
Contribution(s) alone or by combination of their Contribution(s)
|
||||
with the Work to which such Contribution(s) was submitted. If You
|
||||
institute patent litigation against any entity (including a
|
||||
cross-claim or counterclaim in a lawsuit) alleging that the Work
|
||||
or a Contribution incorporated within the Work constitutes direct
|
||||
or contributory patent infringement, then any patent licenses
|
||||
granted to You under this License for that Work shall terminate
|
||||
as of the date such litigation is filed.
|
||||
|
||||
4. Redistribution. You may reproduce and distribute copies of the
|
||||
Work or Derivative Works thereof in any medium, with or without
|
||||
modifications, and in Source or Object form, provided that You
|
||||
meet the following conditions:
|
||||
|
||||
(a) You must give any other recipients of the Work or
|
||||
Derivative Works a copy of this License; and
|
||||
|
||||
(b) You must cause any modified files to carry prominent notices
|
||||
stating that You changed the files; and
|
||||
|
||||
(c) You must retain, in the Source form of any Derivative Works
|
||||
that You distribute, all copyright, patent, trademark, and
|
||||
attribution notices from the Source form of the Work,
|
||||
excluding those notices that do not pertain to any part of
|
||||
the Derivative Works; and
|
||||
|
||||
(d) If the Work includes a "NOTICE" text file as part of its
|
||||
distribution, then any Derivative Works that You distribute must
|
||||
include a readable copy of the attribution notices contained
|
||||
within such NOTICE file, excluding those notices that do not
|
||||
pertain to any part of the Derivative Works, in at least one
|
||||
of the following places: within a NOTICE text file distributed
|
||||
as part of the Derivative Works; within the Source form or
|
||||
documentation, if provided along with the Derivative Works; or,
|
||||
within a display generated by the Derivative Works, if and
|
||||
wherever such third-party notices normally appear. The contents
|
||||
of the NOTICE file are for informational purposes only and
|
||||
do not modify the License. You may add Your own attribution
|
||||
notices within Derivative Works that You distribute, alongside
|
||||
or as an addendum to the NOTICE text from the Work, provided
|
||||
that such additional attribution notices cannot be construed
|
||||
as modifying the License.
|
||||
|
||||
You may add Your own copyright statement to Your modifications and
|
||||
may provide additional or different license terms and conditions
|
||||
for use, reproduction, or distribution of Your modifications, or
|
||||
for any such Derivative Works as a whole, provided Your use,
|
||||
reproduction, and distribution of the Work otherwise complies with
|
||||
the conditions stated in this License.
|
||||
|
||||
5. Submission of Contributions. Unless You explicitly state otherwise,
|
||||
any Contribution intentionally submitted for inclusion in the Work
|
||||
by You to the Licensor shall be under the terms and conditions of
|
||||
this License, without any additional terms or conditions.
|
||||
Notwithstanding the above, nothing herein shall supersede or modify
|
||||
the terms of any separate license agreement you may have executed
|
||||
with Licensor regarding such Contributions.
|
||||
|
||||
6. Trademarks. This License does not grant permission to use the trade
|
||||
names, trademarks, service marks, or product names of the Licensor,
|
||||
except as required for reasonable and customary use in describing the
|
||||
origin of the Work and reproducing the content of the NOTICE file.
|
||||
|
||||
7. Disclaimer of Warranty. Unless required by applicable law or
|
||||
agreed to in writing, Licensor provides the Work (and each
|
||||
Contributor provides its Contributions) on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or
|
||||
implied, including, without limitation, any warranties or conditions
|
||||
of TITLE, NON-INFRINGEMENT, MERCHANTABILITY, or FITNESS FOR A
|
||||
PARTICULAR PURPOSE. You are solely responsible for determining the
|
||||
appropriateness of using or redistributing the Work and assume any
|
||||
risks associated with Your exercise of permissions under this License.
|
||||
|
||||
8. Limitation of Liability. In no event and under no legal theory,
|
||||
whether in tort (including negligence), contract, or otherwise,
|
||||
unless required by applicable law (such as deliberate and grossly
|
||||
negligent acts) or agreed to in writing, shall any Contributor be
|
||||
liable to You for damages, including any direct, indirect, special,
|
||||
incidental, or consequential damages of any character arising as a
|
||||
result of this License or out of the use or inability to use the
|
||||
Work (including but not limited to damages for loss of goodwill,
|
||||
work stoppage, computer failure or malfunction, or any and all
|
||||
other commercial damages or losses), even if such Contributor
|
||||
has been advised of the possibility of such damages.
|
||||
|
||||
9. Accepting Warranty or Additional Liability. While redistributing
|
||||
the Work or Derivative Works thereof, You may choose to offer,
|
||||
and charge a fee for, acceptance of support, warranty, indemnity,
|
||||
or other liability obligations and/or rights consistent with this
|
||||
License. However, in accepting such obligations, You may act only
|
||||
on Your own behalf and on Your sole responsibility, not on behalf
|
||||
of any other Contributor, and only if You agree to indemnify,
|
||||
defend, and hold each Contributor harmless for any liability
|
||||
incurred by, or claims asserted against, such Contributor by reason
|
||||
of your accepting any such warranty or additional liability.
|
||||
|
||||
END OF TERMS AND CONDITIONS
|
||||
|
||||
APPENDIX: How to apply the Apache License to your work.
|
||||
|
||||
To apply the Apache License to your work, attach the following
|
||||
boilerplate notice, with the fields enclosed by brackets "[]"
|
||||
replaced with your own identifying information. (Don't include
|
||||
the brackets!) The text should be enclosed in the appropriate
|
||||
comment syntax for the file format. We also recommend that a
|
||||
file or class name and description of purpose be included on the
|
||||
same "printed page" as the copyright notice for easier
|
||||
identification within third-party archives.
|
||||
|
||||
Copyright [yyyy] [name of copyright owner]
|
||||
|
||||
Licensed under the Apache License, Version 2.0 (the "License");
|
||||
you may not use this file except in compliance with the License.
|
||||
You may obtain a copy of the License at
|
||||
|
||||
http://www.apache.org/licenses/LICENSE-2.0
|
||||
|
||||
Unless required by applicable law or agreed to in writing, software
|
||||
distributed under the License is distributed on an "AS IS" BASIS,
|
||||
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
See the License for the specific language governing permissions and
|
||||
limitations under the License.
|
||||
|
||||
|
||||
---- LLVM Exceptions to the Apache 2.0 License ----
|
||||
|
||||
As an exception, if, as a result of your compiling your source code, portions
|
||||
of this Software are embedded into an Object form of such source code, you
|
||||
may redistribute such embedded portions in such Object form without complying
|
||||
with the conditions of Sections 4(a), 4(b) and 4(d) of the License.
|
||||
|
||||
In addition, if you combine or link compiled forms of this Software with
|
||||
software that is licensed under the GPLv2 ("Combined Software") and if a
|
||||
court of competent jurisdiction determines that the patent provision (Section
|
||||
3), the indemnity provision (Section 9) or other Section of the License
|
||||
conflicts with the conditions of the GPLv2, you may retroactively and
|
||||
prospectively choose to deem waived or otherwise exclude such Section(s) of
|
||||
the License, but only in their entirety and only with respect to the Combined
|
||||
Software.
|
||||
|
||||
==============================================================================
|
||||
Software from third parties included in the LLVM Project:
|
||||
==============================================================================
|
||||
The LLVM Project contains third party software which is under different license
|
||||
terms. All such code will be identified clearly using at least one of two
|
||||
mechanisms:
|
||||
1) It will be in a separate directory tree with its own `LICENSE.txt` or
|
||||
`LICENSE` file at the top containing the specific license and restrictions
|
||||
which apply to that software, or
|
||||
2) It will contain specific license and restriction terms at the top of every
|
||||
file.
|
||||
|
||||
==============================================================================
|
||||
Legacy LLVM License (https://llvm.org/docs/DeveloperPolicy.html#legacy):
|
||||
==============================================================================
|
||||
|
||||
The libc++ library is dual licensed under both the University of Illinois
|
||||
"BSD-Like" license and the MIT license. As a user of this code you may choose
|
||||
to use it under either license. As a contributor, you agree to allow your code
|
||||
to be used under both.
|
||||
|
||||
Full text of the relevant licenses is included below.
|
||||
|
||||
==============================================================================
|
||||
|
||||
University of Illinois/NCSA
|
||||
Open Source License
|
||||
|
||||
Copyright (c) 2009-2019 by the contributors listed in CREDITS.TXT
|
||||
|
||||
All rights reserved.
|
||||
|
||||
Developed by:
|
||||
|
||||
LLVM Team
|
||||
|
||||
University of Illinois at Urbana-Champaign
|
||||
|
||||
http://llvm.org
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy of
|
||||
this software and associated documentation files (the "Software"), to deal with
|
||||
the Software without restriction, including without limitation the rights to
|
||||
use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies
|
||||
of the Software, and to permit persons to whom the Software is furnished to do
|
||||
so, subject to the following conditions:
|
||||
|
||||
* Redistributions of source code must retain the above copyright notice,
|
||||
this list of conditions and the following disclaimers.
|
||||
|
||||
* Redistributions in binary form must reproduce the above copyright notice,
|
||||
this list of conditions and the following disclaimers in the
|
||||
documentation and/or other materials provided with the distribution.
|
||||
|
||||
* Neither the names of the LLVM Team, University of Illinois at
|
||||
Urbana-Champaign, nor the names of its contributors may be used to
|
||||
endorse or promote products derived from this Software without specific
|
||||
prior written permission.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS
|
||||
FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
CONTRIBUTORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS WITH THE
|
||||
SOFTWARE.
|
||||
|
||||
==============================================================================
|
||||
|
||||
Copyright (c) 2009-2014 by the contributors listed in CREDITS.TXT
|
||||
|
||||
Permission is hereby granted, free of charge, to any person obtaining a copy
|
||||
of this software and associated documentation files (the "Software"), to deal
|
||||
in the Software without restriction, including without limitation the rights
|
||||
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
||||
copies of the Software, and to permit persons to whom the Software is
|
||||
furnished to do so, subject to the following conditions:
|
||||
|
||||
The above copyright notice and this permission notice shall be included in
|
||||
all copies or substantial portions of the Software.
|
||||
|
||||
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
||||
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
||||
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
||||
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
||||
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
||||
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN
|
||||
THE SOFTWARE.
|
||||
@@ -1,29 +0,0 @@
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Notes relating to various libc++ tasks
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
This file contains notes about various libc++ tasks and processes.
|
||||
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Post-Release TODO
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
These notes contain a list of things that must be done after branching for
|
||||
an LLVM release.
|
||||
|
||||
1. Update _LIBCUDACXX_VERSION in `__config`
|
||||
2. Update the __cccl_version file.
|
||||
3. Update the version number in `docs/conf.py`
|
||||
4. Create ABI lists for the previous release under `lib/abi`
|
||||
|
||||
//===---------------------------------------------------------------------===//
|
||||
// Adding a new header TODO
|
||||
//===---------------------------------------------------------------------===//
|
||||
|
||||
These notes contain a list of things that must be done upon adding a new header
|
||||
to libc++.
|
||||
|
||||
1. Add a test under `test/libcxx` that the header defines `_LIBCUDACXX_VERSION`.
|
||||
2. Update `test/libcxx/double_include.sh.cpp` to include the new header.
|
||||
3. Create a submodule in `include/module.modulemap` for the new header.
|
||||
4. Update the include/CMakeLists.txt file to include the new header.
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user