[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
92
cccl_upstream/libcudacxx/benchmarks/CMakeLists.txt
Normal file
92
cccl_upstream/libcudacxx/benchmarks/CMakeLists.txt
Normal file
@@ -0,0 +1,92 @@
|
||||
include(${CMAKE_SOURCE_DIR}/benchmarks/cmake/CCCLBenchmarkRegistry.cmake)
|
||||
|
||||
cccl_get_nvbench_helper()
|
||||
|
||||
set(benches_root "${CMAKE_CURRENT_LIST_DIR}")
|
||||
|
||||
if (NOT CMAKE_BUILD_TYPE STREQUAL "Release")
|
||||
set(message_type FATAL_ERROR)
|
||||
if (CCCL_ENABLE_CLANG_TIDY)
|
||||
# We are here because CI has force-enabled clang-tidy. We must use a debug build for
|
||||
# this because certain clang-tidy checks (such as out of bounds or clang static
|
||||
# analyzer) work better when they see assert()'s. In this case we don't actually
|
||||
# intend to run any of the benchmarks, we just need them to be compilable, so a simple
|
||||
# warning is enough.
|
||||
#
|
||||
# We don't ignore this outright (by making it say, DEBUG or VERBOSE), because it's
|
||||
# possible that a user may accidentally stumble into enabling the option.
|
||||
set(message_type WARNING)
|
||||
endif()
|
||||
message(${message_type} "libcu++ benchmarks must be built in release mode.")
|
||||
endif()
|
||||
|
||||
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"CMAKE_CUDA_ARCHITECTURES must be set to build libcu++ benchmarks."
|
||||
)
|
||||
endif()
|
||||
|
||||
set(benches_meta_target libcudacxx.all.benches)
|
||||
add_custom_target(${benches_meta_target})
|
||||
|
||||
function(get_recursive_subdirs subdirs)
|
||||
set(dirs)
|
||||
file(
|
||||
GLOB_RECURSE contents
|
||||
CONFIGURE_DEPENDS
|
||||
LIST_DIRECTORIES ON
|
||||
"${CMAKE_CURRENT_LIST_DIR}/bench/*"
|
||||
)
|
||||
|
||||
foreach (test_dir IN LISTS contents)
|
||||
if (IS_DIRECTORY "${test_dir}")
|
||||
list(APPEND dirs "${test_dir}")
|
||||
endif()
|
||||
endforeach()
|
||||
|
||||
set(${subdirs} "${dirs}" PARENT_SCOPE)
|
||||
endfunction()
|
||||
|
||||
create_benchmark_registry()
|
||||
|
||||
function(add_bench target_name bench_name bench_src)
|
||||
set(bench_target ${bench_name})
|
||||
set(${target_name} ${bench_target} PARENT_SCOPE)
|
||||
|
||||
cccl_add_executable(${bench_target} SOURCES "${bench_src}")
|
||||
target_link_libraries(
|
||||
${bench_target}
|
||||
PRIVATE libcudacxx::libcudacxx cccl.nvbench_helper nvbench::main
|
||||
)
|
||||
endfunction()
|
||||
|
||||
function(add_bench_dir bench_dir)
|
||||
file(GLOB bench_srcs CONFIGURE_DEPENDS "${bench_dir}/*.cu")
|
||||
file(RELATIVE_PATH bench_prefix "${benches_root}" "${bench_dir}")
|
||||
file(TO_CMAKE_PATH "${bench_prefix}" bench_prefix)
|
||||
string(REPLACE "/" "." bench_prefix "${bench_prefix}")
|
||||
|
||||
foreach (bench_src IN LISTS bench_srcs)
|
||||
# base tuning
|
||||
get_filename_component(bench_name "${bench_src}" NAME_WLE)
|
||||
string(PREPEND bench_name "libcudacxx.${bench_prefix}.")
|
||||
|
||||
set(base_bench_name "${bench_name}.base")
|
||||
add_bench(base_bench_target ${base_bench_name} "${bench_src}")
|
||||
add_dependencies(${benches_meta_target} ${base_bench_target})
|
||||
target_compile_definitions(${base_bench_target} PRIVATE TUNE_BASE=1)
|
||||
target_compile_options(
|
||||
${base_bench_target}
|
||||
PRIVATE "$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>"
|
||||
)
|
||||
# benchmarking
|
||||
register_cccl_benchmark("${bench_name}" "")
|
||||
endforeach()
|
||||
endfunction()
|
||||
|
||||
get_recursive_subdirs(subdirs)
|
||||
|
||||
foreach (subdir IN LISTS subdirs)
|
||||
add_bench_dir("${subdir}")
|
||||
endforeach()
|
||||
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/adjacent_difference.h>
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_difference(
|
||||
cuda_policy(alloc, launch), in.cbegin(), in.cend(), out.begin(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,77 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 2);
|
||||
|
||||
thrust::device_vector<T> in(elements, thrust::no_init);
|
||||
thrust::sequence(in.begin(), in.end(), 0);
|
||||
in[mismatch_point] = in[mismatch_point + 1];
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(mismatch_point);
|
||||
state.add_global_memory_writes<T>(0);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::adjacent_find(cuda_policy(alloc, launch), in.cbegin(), in.cend(), ::cuda::std::greater<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/all_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/all_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::all_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/any_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/any_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::any_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
70
cccl_upstream/libcudacxx/benchmarks/bench/copy/basic.cu
Normal file
70
cccl_upstream/libcudacxx/benchmarks/bench/copy/basic.cu
Normal file
@@ -0,0 +1,70 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::counting_iterator<std::size_t>{0},
|
||||
cuda::counting_iterator{elements},
|
||||
out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
51
cccl_upstream/libcudacxx/benchmarks/bench/copy_if/basic.cu
Normal file
51
cccl_upstream/libcudacxx/benchmarks/bench/copy_if/basic.cu
Normal file
@@ -0,0 +1,51 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
67
cccl_upstream/libcudacxx/benchmarks/bench/copy_n/basic.cu
Normal file
67
cccl_upstream/libcudacxx/benchmarks/bench/copy_n/basic.cu
Normal file
@@ -0,0 +1,67 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::copy_n(cuda_policy(alloc, launch), in.begin(), elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("contiguous")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void random_access(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::copy_n(cuda_policy(alloc, launch), cuda::counting_iterator<std::size_t>{0}, elements, out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(random_access, NVBENCH_TYPE_AXES(integral_types))
|
||||
.set_name("random_access")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/count/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/count/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count(cuda_policy(alloc, launch), in.begin(), in.end(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
50
cccl_upstream/libcudacxx/benchmarks/bench/count_if/basic.cu
Normal file
50
cccl_upstream/libcudacxx/benchmarks/bench/count_if/basic.cu
Normal file
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::count_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
82
cccl_upstream/libcudacxx/benchmarks/bench/equal/basic.cu
Normal file
82
cccl_upstream/libcudacxx/benchmarks/bench/equal/basic.cu
Normal file
@@ -0,0 +1,82 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::equal(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::equal(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,68 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_init_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::exclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_init_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_init_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
40
cccl_upstream/libcudacxx/benchmarks/bench/fill/basic.cu
Normal file
40
cccl_upstream/libcudacxx/benchmarks/bench/fill/basic.cu
Normal file
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::fill(cuda_policy(alloc, launch), output.begin(), output.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
40
cccl_upstream/libcudacxx/benchmarks/bench/fill_n/basic.cu
Normal file
40
cccl_upstream/libcudacxx/benchmarks/bench/fill_n/basic.cu
Normal file
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::fill_n(cuda_policy(alloc, launch), output.begin(), elements, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
46
cccl_upstream/libcudacxx/benchmarks/bench/find/basic.cu
Normal file
46
cccl_upstream/libcudacxx/benchmarks/bench/find/basic.cu
Normal file
@@ -0,0 +1,46 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), val));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/find_if/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/find_if/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::find_if(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::find_if_not(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::not_fn(cuda::equal_to_value{val})));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
52
cccl_upstream/libcudacxx/benchmarks/bench/for_each/basic.cu
Normal file
52
cccl_upstream/libcudacxx/benchmarks/bench/for_each/basic.cu
Normal file
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::for_each(cuda_policy(alloc, launch), in.begin(), in.end(), op);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct square_t
|
||||
{
|
||||
__device__ void operator()(T& x) const
|
||||
{
|
||||
x = x * x;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in(elements, T{1});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
square_t<T> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::for_each_n(cuda_policy(alloc, launch), in.begin(), elements, op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
42
cccl_upstream/libcudacxx/benchmarks/bench/generate/basic.cu
Normal file
42
cccl_upstream/libcudacxx/benchmarks/bench/generate/basic.cu
Normal file
@@ -0,0 +1,42 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate(cuda_policy(alloc, launch), output.begin(), output.end(), generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,42 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
struct generator
|
||||
{
|
||||
_CCCL_DEVICE_API _CCCL_FORCEINLINE auto operator()() const -> T
|
||||
{
|
||||
return 42;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> output(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::generate_n(cuda_policy(alloc, launch), output.begin(), elements, generator<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,94 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), ::cuda::std::plus<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,69 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void range_iter_op_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::inclusive_scan(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), max_t{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter_op_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("range_iter_op_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
85
cccl_upstream/libcudacxx/benchmarks/bench/is_heap/basic.cu
Normal file
85
cccl_upstream/libcudacxx/benchmarks/bench/is_heap/basic.cu
Normal file
@@ -0,0 +1,85 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,85 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/fill.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// All-zero is a valid heap; setting one element to 1 forces a violation at
|
||||
// that child index since its parent is still 0.
|
||||
template <typename T>
|
||||
static void prepare_input(thrust::device_vector<T>& d, std::size_t violation_point)
|
||||
{
|
||||
thrust::fill(d.begin(), d.end(), T{0});
|
||||
if (violation_point >= 1 && violation_point < d.size())
|
||||
{
|
||||
d[violation_point] = T{1};
|
||||
}
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto violation_frac = state.get_float64("ViolationAt");
|
||||
const auto violation_point = cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * violation_frac), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
prepare_input(dinput, violation_point);
|
||||
|
||||
state.add_global_memory_reads<T>(2 * violation_point);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_heap_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ViolationAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
#include <thrust/sequence.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_partitioned(
|
||||
cuda_policy(alloc, launch), dinput.begin(), dinput.end(), select_op_t{static_cast<T>(mismatch_point)}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
78
cccl_upstream/libcudacxx/benchmarks/bench/is_sorted/basic.cu
Normal file
78
cccl_upstream/libcudacxx/benchmarks/bench/is_sorted/basic.cu
Normal file
@@ -0,0 +1,78 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::greater<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,78 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sequence.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = ::cuda::std::clamp<std::size_t>(
|
||||
static_cast<std::size_t>(static_cast<double>(elements) * common_prefix), std::size_t{0}, elements - 1);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
thrust::sequence(dinput.begin(), dinput.end(), T{0});
|
||||
dinput[mismatch_point] = T{-1};
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::is_sorted_until(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::std::less<>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
@@ -0,0 +1,66 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::max_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
95
cccl_upstream/libcudacxx/benchmarks/bench/merge/basic.cu
Normal file
95
cccl_upstream/libcudacxx/benchmarks/bench/merge/basic.cu
Normal file
@@ -0,0 +1,95 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/merge.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs);
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end());
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto size_ratio = static_cast<std::size_t>(state.get_int64("InputSizeRatio"));
|
||||
const auto entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
const auto elements_in_lhs = static_cast<std::size_t>(static_cast<double>(size_ratio * elements) / 100.0);
|
||||
|
||||
thrust::device_vector<T> out(elements);
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
thrust::sort(in.begin(), in.begin() + elements_in_lhs, ::cuda::std::greater<T>{});
|
||||
thrust::sort(in.begin() + elements_in_lhs, in.end(), ::cuda::std::greater<T>{});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::merge(
|
||||
cuda_policy(alloc, launch),
|
||||
in.cbegin(),
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cbegin() + elements_in_lhs,
|
||||
in.cend(),
|
||||
out.begin(),
|
||||
::cuda::std::greater<T>{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"})
|
||||
.add_int64_axis("InputSizeRatio", {25, 50, 75});
|
||||
@@ -0,0 +1,66 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<typename thrust::device_vector<T>::iterator::difference_type>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::min_element(cuda_policy(alloc, launch), in.begin(), in.end(), less_t{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
82
cccl_upstream/libcudacxx/benchmarks/bench/mismatch/basic.cu
Normal file
82
cccl_upstream/libcudacxx/benchmarks/bench/mismatch/basic.cu
Normal file
@@ -0,0 +1,82 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void range_iter(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::mismatch(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::constant_iterator<T>{0}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_iter, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_iter")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
|
||||
template <typename T>
|
||||
static void range_range(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::mismatch(
|
||||
cuda_policy(alloc, launch),
|
||||
dinput.begin(),
|
||||
dinput.end(),
|
||||
cuda::constant_iterator<T>{0},
|
||||
cuda::constant_iterator<T>{0, elements}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(range_range, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base_range_range")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/none_of/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/none_of/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
T val = 1;
|
||||
// set up input
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto common_prefix = state.get_float64("MismatchAt");
|
||||
const auto mismatch_point = static_cast<std::size_t>(static_cast<double>(elements) * common_prefix);
|
||||
|
||||
thrust::device_vector<T> dinput(elements, thrust::no_init);
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin(), dinput.begin() + mismatch_point, T{0});
|
||||
cuda::std::fill(cuda::execution::gpu, dinput.begin() + mismatch_point, dinput.end(), val);
|
||||
|
||||
state.add_global_memory_reads<T>(mismatch_point + 1);
|
||||
state.add_global_memory_writes<size_t>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::none_of(cuda_policy(alloc, launch), dinput.begin(), dinput.end(), cuda::equal_to_value{val}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MismatchAt", std::vector{1.0, 0.5, 0.01});
|
||||
48
cccl_upstream/libcudacxx/benchmarks/bench/partition/basic.cu
Normal file
48
cccl_upstream/libcudacxx/benchmarks/bench/partition/basic.cu
Normal file
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -0,0 +1,55 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::partition_copy(
|
||||
cuda_policy(alloc, launch),
|
||||
input.begin(),
|
||||
input.end(),
|
||||
output.begin(),
|
||||
cuda::std::make_reverse_iterator(output.begin() + elements),
|
||||
select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/reduce/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/reduce/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reduce(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
43
cccl_upstream/libcudacxx/benchmarks/bench/remove/basic.cu
Normal file
43
cccl_upstream/libcudacxx/benchmarks/bench/remove/basic.cu
Normal file
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove(cuda_policy(alloc, launch), in.begin(), in.end(), T{42});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,44 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/complex>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
const auto count = cuda::std::count(cuda::execution::gpu, in.begin(), in.end(), T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements - count);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::remove_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::remove_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), is_even{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
50
cccl_upstream/libcudacxx/benchmarks/bench/remove_if/basic.cu
Normal file
50
cccl_upstream/libcudacxx/benchmarks/bench/remove_if/basic.cu
Normal file
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct is_even
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return static_cast<int>(val) % 2 == 0;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::remove_if(cuda_policy(alloc, launch), in.begin(), in.end(), is_even{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
41
cccl_upstream/libcudacxx/benchmarks/bench/replace/basic.cu
Normal file
41
cccl_upstream/libcudacxx/benchmarks/bench/replace/basic.cu
Normal file
@@ -0,0 +1,41 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace(cuda_policy(alloc, launch), in.begin(), in.end(), 42, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::replace_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), 42, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::replace_copy_if(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), equal_to_42{}, 1337));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,50 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
struct equal_to_42
|
||||
{
|
||||
template <class T>
|
||||
__device__ constexpr bool operator()(const T& val) const noexcept
|
||||
{
|
||||
return val == static_cast<T>(42);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::replace_if(cuda_policy(alloc, launch), in.begin(), in.end(), equal_to_42{}, 1337);
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
42
cccl_upstream/libcudacxx/benchmarks/bench/reverse/basic.cu
Normal file
42
cccl_upstream/libcudacxx/benchmarks/bench/reverse/basic.cu
Normal file
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::reverse(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/reverse.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::reverse_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
44
cccl_upstream/libcudacxx/benchmarks/bench/rotate/basic.cu
Normal file
44
cccl_upstream/libcudacxx/benchmarks/bench/rotate/basic.cu
Normal file
@@ -0,0 +1,44 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::rotate(cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,45 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("MidpointAt");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements, bit_entropy::_1_000);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::rotate_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), cuda::std::next(in.begin(), midpoint), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("MidpointAt", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,43 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_left(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.5, 0.01});
|
||||
@@ -0,0 +1,42 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const auto midpoint_float = state.get_float64("ShiftedTo");
|
||||
const auto midpoint = static_cast<std::size_t>(static_cast<double>(elements) * midpoint_float);
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements - midpoint);
|
||||
state.add_global_memory_writes<T>(elements - midpoint);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::shift_right(cuda_policy(alloc, launch), in.begin(), in.end(), midpoint));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_float64_axis("ShiftedTo", std::vector{0.9, 0.6, 0.45, 0.01});
|
||||
81
cccl_upstream/libcudacxx/benchmarks/bench/sort/basic.cu
Normal file
81
cccl_upstream/libcudacxx/benchmarks/bench/sort/basic.cu
Normal file
@@ -0,0 +1,81 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
|
||||
struct fake_less
|
||||
{
|
||||
template <class T, class U>
|
||||
[[nodiscard]] _CCCL_API constexpr bool operator()(const T& t, const U& u) const
|
||||
{
|
||||
// complex is not less than comparable, so just compare the first element
|
||||
if constexpr (cuda::std::__is_cpp17_less_than_comparable_v<T, U>)
|
||||
{
|
||||
return t < u;
|
||||
}
|
||||
else
|
||||
{
|
||||
return cuda::std::get<0>(t) < cuda::std::get<0>(u);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void with_predicate(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
thrust::device_vector<T> in = generate(elements, entropy);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::sort(cuda_policy(alloc, launch), in.begin(), in.end(), fake_less{});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_predicate, NVBENCH_TYPE_AXES(all_types))
|
||||
.set_name("with_predicate")
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
@@ -0,0 +1,48 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDA Experimental in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/partition.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
using select_op_t = less_then_t<T>;
|
||||
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const T val = lerp_min_max<T>(entropy_to_probability(entropy));
|
||||
select_op_t select_op{val};
|
||||
|
||||
thrust::device_vector<T> input = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::stable_partition(cuda_policy(alloc, launch), input.begin(), input.end(), select_op));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.544", "0.000"});
|
||||
@@ -0,0 +1,72 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/swap.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(cuda_policy(alloc, launch), in1.begin(), in1.end(), in2.begin());
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_iter_swap(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in1 = generate(elements);
|
||||
thrust::device_vector<T> in2 = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(2 * elements);
|
||||
state.add_global_memory_writes<T>(2 * elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
cuda::std::swap_ranges(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::std::reverse_iterator{in1.end()},
|
||||
cuda::std::reverse_iterator{in1.begin()},
|
||||
cuda::std::reverse_iterator{in2.end()});
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_iter_swap, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_iter_swap")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,156 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/iterator/zip_iterator.h>
|
||||
|
||||
#include <cuda/functional>
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
// The benchmarks are inspired by the BabelStream thrust version:
|
||||
// https://github.com/UoB-HPC/BabelStream/blob/main/src/thrust/ThrustStream.cu
|
||||
|
||||
// Modified from BabelStream to also work for integers
|
||||
constexpr auto startA = 1; // BabelStream: 0.1
|
||||
constexpr auto startB = 2; // BabelStream: 0.2
|
||||
constexpr auto startC = 3; // BabelStream: 0.1
|
||||
constexpr auto startScalar = 4; // BabelStream: 0.4
|
||||
|
||||
using element_types = nvbench::type_list<std::int8_t, std::int16_t, float, double, __int128>;
|
||||
// Different benchmarks use a different number of buffers. H200/B200 can fit 2^31 elements for all benchmarks and types.
|
||||
// Upstream BabelStream uses 2^25. Allocation failure just skips the benchmark
|
||||
auto array_size_powers = std::vector<std::int64_t>{25, 31};
|
||||
|
||||
template <typename T>
|
||||
static void mul(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), c.begin(), c.end(), b.begin(), [=] _CCCL_HOST_DEVICE(const T& ci) {
|
||||
return ci * scalar;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(mul, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("mul")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void add(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch), a.begin(), a.end(), b.begin(), c.begin(), cuda::std::plus<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(add, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("add")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void triad(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(2 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
b.begin(),
|
||||
b.end(),
|
||||
c.begin(),
|
||||
a.begin(),
|
||||
[=] _CCCL_HOST_DEVICE(const T& bi, const T& ci) {
|
||||
return bi + scalar * ci;
|
||||
}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(triad, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("triad")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
|
||||
template <typename T>
|
||||
static void nstream(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto n = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
thrust::device_vector<T> a(n, startA);
|
||||
thrust::device_vector<T> b(n, startB);
|
||||
thrust::device_vector<T> c(n, startC);
|
||||
|
||||
state.add_element_count(n);
|
||||
state.add_global_memory_reads<T>(3 * n);
|
||||
state.add_global_memory_writes<T>(n);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
const T scalar = startScalar;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform(
|
||||
cuda_policy(alloc, launch),
|
||||
cuda::make_zip_iterator(a.begin(), b.begin(), c.begin()),
|
||||
cuda::make_zip_iterator(a.end(), b.end(), c.end()),
|
||||
a.begin(),
|
||||
cuda::zip_function{[=] _CCCL_HOST_DEVICE(const T& ai, const T& bi, const T& ci) {
|
||||
return ai + bi + scalar * ci;
|
||||
}}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(nstream, NVBENCH_TYPE_AXES(element_types))
|
||||
.set_name("nstream")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", array_size_powers);
|
||||
75
cccl_upstream/libcudacxx/benchmarks/bench/transform/fib.cu
Normal file
75
cccl_upstream/libcudacxx/benchmarks/bench/transform/fib.cu
Normal file
@@ -0,0 +1,75 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
template <class InT, class OutT>
|
||||
struct fib_t
|
||||
{
|
||||
__device__ OutT operator()(InT n)
|
||||
{
|
||||
OutT t1 = 0;
|
||||
OutT t2 = 1;
|
||||
|
||||
if (n <= 1)
|
||||
{
|
||||
return t1;
|
||||
}
|
||||
else if (n == 2)
|
||||
{
|
||||
return t2;
|
||||
}
|
||||
for (InT i = 3; i <= n; ++i)
|
||||
{
|
||||
const auto next = t1 + t2;
|
||||
t1 = t2;
|
||||
t2 = next;
|
||||
}
|
||||
|
||||
return t2;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void fib(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> input = generate(elements, bit_entropy::_1_000, T{0}, T{42});
|
||||
thrust::device_vector<T> output(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<nvbench::uint32_t>(elements);
|
||||
|
||||
fib_t<T, nvbench::uint32_t> op{};
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(
|
||||
cuda::std::transform(cuda_policy(alloc, launch), input.cbegin(), input.cend(), output.begin(), op));
|
||||
});
|
||||
}
|
||||
|
||||
using types = nvbench::type_list<nvbench::uint32_t, nvbench::uint64_t>;
|
||||
|
||||
NVBENCH_BENCH_TYPES(fib, NVBENCH_TYPE_AXES(types))
|
||||
.set_name("fib")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,53 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_exclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), T{42}, cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,79 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/transform_scan.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct times_two
|
||||
{
|
||||
_CCCL_DEVICE constexpr T operator()(const T val) const noexcept
|
||||
{
|
||||
return 2 * val;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("basic")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_init(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(elements);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_inclusive_scan(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::plus<T>{}, times_two<T>{}, T{42}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_init, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_init")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,49 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <typename T>
|
||||
static void binary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch),
|
||||
in.begin(),
|
||||
in.end(),
|
||||
cuda::constant_iterator<int>{42},
|
||||
42,
|
||||
cuda::std::plus<T>{},
|
||||
cuda::std::multiplies<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(binary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,52 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
template <class T>
|
||||
struct plus_one
|
||||
{
|
||||
template <class U>
|
||||
[[nodiscard]] __device__ constexpr T operator()(const U val) const noexcept
|
||||
{
|
||||
return static_cast<T>(val + 1);
|
||||
}
|
||||
};
|
||||
|
||||
template <typename T>
|
||||
static void unary(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in = generate(elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
state.add_global_memory_writes<T>(1);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::transform_reduce(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), 42, cuda::std::plus<T>{}, plus_one<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(unary, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
88
cccl_upstream/libcudacxx/benchmarks/bench/unique/basic.cu
Normal file
88
cccl_upstream/libcudacxx/benchmarks/bench/unique/basic.cu
Normal file
@@ -0,0 +1,88 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
// This seems like a clang-tidy bug. Yes we end up converting to double, but the division
|
||||
// is done entirely in integer land...
|
||||
return static_cast<T>(i / 2ULL); // NOLINT(bugprone-integer-division)
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(
|
||||
nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync, [&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique(cuda_policy(alloc, launch), in.begin(), in.end(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
@@ -0,0 +1,90 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/transform.h>
|
||||
#include <thrust/unique.h>
|
||||
|
||||
#include <cuda/memory_pool>
|
||||
#include <cuda/std/execution>
|
||||
#include <cuda/stream_ref>
|
||||
|
||||
#include "nvbench_helper.cuh"
|
||||
|
||||
// Input with runs of equal elements: 0,0,1,1,2,2,... (segment size 2)
|
||||
template <typename T>
|
||||
static void make_unique_input(thrust::device_vector<T>& in, std::size_t elements)
|
||||
{
|
||||
in.resize(elements);
|
||||
thrust::transform(
|
||||
thrust::counting_iterator<std::size_t>(0),
|
||||
thrust::counting_iterator<std::size_t>(elements),
|
||||
in.begin(),
|
||||
[] __device__(std::size_t i) {
|
||||
const auto run = i / 2;
|
||||
return static_cast<T>(run);
|
||||
});
|
||||
}
|
||||
|
||||
template <typename T>
|
||||
static void basic(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(cuda_policy(alloc, launch), in.begin(), in.end(), out.begin()));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(basic, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
|
||||
template <typename T>
|
||||
static void with_comp(nvbench::state& state, nvbench::type_list<T>)
|
||||
{
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements"));
|
||||
|
||||
thrust::device_vector<T> in;
|
||||
make_unique_input(in, elements);
|
||||
thrust::device_vector<T> out(elements, thrust::no_init);
|
||||
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<T>(elements);
|
||||
// unique_copy writes at most elements
|
||||
state.add_global_memory_writes<T>(elements / 2);
|
||||
|
||||
caching_allocator_t alloc{};
|
||||
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch | nvbench::exec_tag::sync,
|
||||
[&](nvbench::launch& launch) {
|
||||
do_not_optimize(cuda::std::unique_copy(
|
||||
cuda_policy(alloc, launch), in.begin(), in.end(), out.begin(), cuda::std::equal_to<T>{}));
|
||||
});
|
||||
}
|
||||
|
||||
NVBENCH_BENCH_TYPES(with_comp, NVBENCH_TYPE_AXES(fundamental_types))
|
||||
.set_name("with_comp")
|
||||
.set_type_axes_names({"T{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements", nvbench::range(16, 28, 4));
|
||||
Reference in New Issue
Block a user