[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
78
cccl_upstream/cub/benchmarks/bench/merge/keys.cu
Normal file
78
cccl_upstream/cub/benchmarks/bench/merge/keys.cu
Normal file
@@ -0,0 +1,78 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include <cub/device/device_merge.cuh>
|
||||
|
||||
#include <thrust/detail/raw_pointer_cast.h>
|
||||
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include "merge_common.cuh"
|
||||
|
||||
// %RANGE% TUNE_TRANSPOSE trp 0:1:1
|
||||
// %RANGE% TUNE_LOAD ld 0:3:1
|
||||
// %RANGE% TUNE_ITEMS_PER_THREAD ipt 7:24:1
|
||||
// %RANGE% TUNE_THREADS_PER_BLOCK_POW2 tpb 6:10:1
|
||||
|
||||
template <typename KeyT>
|
||||
void keys(nvbench::state& state, nvbench::type_list<KeyT>)
|
||||
{
|
||||
using offset_t = int64_t;
|
||||
using compare_op_t = less_t;
|
||||
|
||||
// Retrieve axis parameters
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements{io}"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const auto num_items_lhs = elements / 2;
|
||||
const auto num_items_rhs = elements - num_items_lhs;
|
||||
auto [keys_lhs, keys_rhs] = generate_lhs_rhs<KeyT>(num_items_lhs, num_items_rhs, entropy);
|
||||
|
||||
thrust::device_vector<KeyT> keys_out(elements, thrust::no_init);
|
||||
KeyT* d_keys_lhs = thrust::raw_pointer_cast(keys_lhs.data());
|
||||
KeyT* d_keys_rhs = thrust::raw_pointer_cast(keys_rhs.data());
|
||||
KeyT* d_keys_out = thrust::raw_pointer_cast(keys_out.data());
|
||||
|
||||
// Enable throughput calculations and add "Size" column to results.
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<KeyT>(elements);
|
||||
state.add_global_memory_writes<KeyT>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch, [&](nvbench::launch& launch) {
|
||||
auto env = cub_bench_env(
|
||||
alloc,
|
||||
launch
|
||||
#if !TUNE_BASE
|
||||
,
|
||||
cuda::execution::tune(bench_policy_selector<key_t>{})
|
||||
#endif // !TUNE_BASE
|
||||
);
|
||||
_CCCL_TRY_CUDA_API(
|
||||
cub::DeviceMerge::MergeKeys,
|
||||
"MergePairs failed",
|
||||
d_keys_lhs,
|
||||
static_cast<offset_t>(num_items_lhs),
|
||||
d_keys_rhs,
|
||||
static_cast<offset_t>(num_items_rhs),
|
||||
d_keys_out,
|
||||
compare_op_t{},
|
||||
env);
|
||||
});
|
||||
}
|
||||
|
||||
#ifdef TUNE_KeyT
|
||||
using key_types = nvbench::type_list<TUNE_KeyT>;
|
||||
#else // !defined(TUNE_KeyT)
|
||||
using key_types = fundamental_types;
|
||||
#endif // TUNE_KeyT
|
||||
|
||||
NVBENCH_BENCH_TYPES(keys, NVBENCH_TYPE_AXES(key_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"KeyT{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements{io}", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
135
cccl_upstream/cub/benchmarks/bench/merge/merge_common.cuh
Normal file
135
cccl_upstream/cub/benchmarks/bench/merge/merge_common.cuh
Normal file
@@ -0,0 +1,135 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2024, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <thrust/copy.h>
|
||||
#include <thrust/count.h>
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/sort.h>
|
||||
|
||||
#include <cuda/iterator>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#if !TUNE_BASE
|
||||
# if TUNE_LOAD == 0
|
||||
# define TUNE_LOAD_MODIFIER cub::LOAD_DEFAULT
|
||||
# define TUNE_USE_BL2SH false
|
||||
# elif TUNE_LOAD == 1
|
||||
# define TUNE_LOAD_MODIFIER cub::LOAD_LDG
|
||||
# define TUNE_USE_BL2SH false
|
||||
# elif TUNE_LOAD == 2
|
||||
# define TUNE_LOAD_MODIFIER cub::LOAD_CA
|
||||
# define TUNE_USE_BL2SH false
|
||||
# else // TUNE_LOAD == 3
|
||||
# define TUNE_LOAD_MODIFIER cub::LOAD_DEFAULT
|
||||
# define TUNE_USE_BL2SH true
|
||||
# endif // TUNE_LOAD
|
||||
|
||||
template <typename KeyT>
|
||||
struct bench_policy_selector
|
||||
{
|
||||
[[nodiscard]] _CCCL_HOST_DEVICE constexpr auto operator()(cuda::compute_capability) const -> cub::MergePolicy
|
||||
{
|
||||
return cub::MergePolicy{
|
||||
(1 << TUNE_THREADS_PER_BLOCK_POW2),
|
||||
cub::Nominal4BItemsToItems<KeyT>(TUNE_ITEMS_PER_THREAD),
|
||||
TUNE_LOAD_MODIFIER,
|
||||
TUNE_TRANSPOSE == 0 ? cub::BLOCK_STORE_DIRECT : cub::BLOCK_STORE_WARP_TRANSPOSE,
|
||||
TUNE_USE_BL2SH};
|
||||
}
|
||||
};
|
||||
#endif // TUNE_BASE
|
||||
|
||||
struct select_if_less_than_t
|
||||
{
|
||||
bool negate;
|
||||
uint8_t threshold;
|
||||
|
||||
__device__ __forceinline__ bool operator()(uint8_t val) const
|
||||
{
|
||||
return negate ? !(val < threshold) : val < threshold;
|
||||
}
|
||||
};
|
||||
|
||||
template <typename OffsetT>
|
||||
struct write_pivot_point_t
|
||||
{
|
||||
OffsetT threshold;
|
||||
OffsetT* pivot_point;
|
||||
|
||||
__device__ void operator()(OffsetT output_index, OffsetT input_index) const
|
||||
{
|
||||
if (output_index == threshold)
|
||||
{
|
||||
*pivot_point = input_index;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
template <typename KeyT>
|
||||
std::pair<thrust::device_vector<KeyT>, thrust::device_vector<KeyT>>
|
||||
generate_lhs_rhs(std::size_t num_items_lhs, std::size_t num_items_rhs, bit_entropy entropy)
|
||||
{
|
||||
using offset_t = std::size_t;
|
||||
|
||||
const auto elements = num_items_lhs + num_items_rhs;
|
||||
|
||||
// We generate data distributions in the range [0, 255], which, with lower entropy, get skewed towards 0.
|
||||
// We use this to generate increasingly large *consecutive* segments of data that are getting selected from the lhs
|
||||
thrust::device_vector<uint8_t> rnd_selector_val = generate(elements, entropy);
|
||||
uint8_t threshold = 128;
|
||||
select_if_less_than_t select_lhs_op{false, threshold};
|
||||
select_if_less_than_t select_rhs_op{true, threshold};
|
||||
|
||||
// The following algorithm only works under the precondition that there's at least 50% of the data in the lhs
|
||||
// If that's not the case, we simply swap the logic for selecting into lhs and rhs
|
||||
const auto num_items_selected_into_lhs =
|
||||
static_cast<offset_t>(thrust::count_if(rnd_selector_val.begin(), rnd_selector_val.end(), select_lhs_op));
|
||||
if (num_items_selected_into_lhs < num_items_lhs)
|
||||
{
|
||||
using ::cuda::std::swap;
|
||||
swap(select_lhs_op, select_rhs_op);
|
||||
}
|
||||
|
||||
// We want lhs and rhs to be of equal size. We also want to have skewed distributions, such that we put different
|
||||
// workloads on the binary search part. For this reason, we identify the index from the input, referred to as pivot
|
||||
// point, after which the lhs is "full". We compose the rhs by selecting all items up to the pivot point that were not
|
||||
// selected for lhs and *all* items after the pivot point.
|
||||
constexpr std::size_t num_pivot_points = 1;
|
||||
thrust::device_vector<offset_t> pivot_point(num_pivot_points);
|
||||
auto counting_it = thrust::make_counting_iterator(offset_t{0});
|
||||
using counting_difference_t = typename decltype(counting_it)::difference_type;
|
||||
thrust::copy_if(
|
||||
counting_it,
|
||||
counting_it + static_cast<counting_difference_t>(elements),
|
||||
rnd_selector_val.begin(),
|
||||
cuda::make_tabulate_output_iterator(write_pivot_point_t<offset_t>{
|
||||
static_cast<offset_t>(num_items_lhs), thrust::raw_pointer_cast(pivot_point.data())}),
|
||||
select_lhs_op);
|
||||
|
||||
thrust::device_vector<KeyT> keys_lhs(num_items_lhs);
|
||||
thrust::device_vector<KeyT> keys_rhs(num_items_rhs);
|
||||
|
||||
thrust::device_vector<KeyT> increasing_input = generate(elements);
|
||||
thrust::sort(increasing_input.begin(), increasing_input.end());
|
||||
|
||||
offset_t pivot_point_val = pivot_point[0];
|
||||
auto const end_lhs = thrust::copy_if(
|
||||
increasing_input.cbegin(),
|
||||
increasing_input.cbegin() + pivot_point_val,
|
||||
rnd_selector_val.cbegin(),
|
||||
keys_lhs.begin(),
|
||||
select_lhs_op);
|
||||
|
||||
auto const end_rhs = thrust::copy_if(
|
||||
increasing_input.cbegin(),
|
||||
increasing_input.cbegin() + pivot_point_val,
|
||||
rnd_selector_val.cbegin(),
|
||||
keys_rhs.begin(),
|
||||
select_rhs_op);
|
||||
thrust::copy(increasing_input.cbegin() + pivot_point_val, increasing_input.cbegin() + elements, end_rhs);
|
||||
|
||||
return {keys_lhs, keys_rhs};
|
||||
}
|
||||
103
cccl_upstream/cub/benchmarks/bench/merge/pairs.cu
Normal file
103
cccl_upstream/cub/benchmarks/bench/merge/pairs.cu
Normal file
@@ -0,0 +1,103 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3-Clause
|
||||
|
||||
#include <cub/device/device_merge.cuh>
|
||||
|
||||
#include <thrust/detail/raw_pointer_cast.h>
|
||||
|
||||
#include <cuda/std/utility>
|
||||
|
||||
#include <cstdint>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include "merge_common.cuh"
|
||||
|
||||
// %RANGE% TUNE_TRANSPOSE trp 0:1:1
|
||||
// %RANGE% TUNE_LOAD ld 0:3:1
|
||||
// %RANGE% TUNE_ITEMS_PER_THREAD ipt 7:24:1
|
||||
// %RANGE% TUNE_THREADS_PER_BLOCK_POW2 tpb 6:10:1
|
||||
|
||||
template <typename KeyT, typename ValueT>
|
||||
void pairs(nvbench::state& state, nvbench::type_list<KeyT, ValueT>)
|
||||
{
|
||||
using offset_t = int64_t;
|
||||
using compare_op_t = less_t;
|
||||
|
||||
// Retrieve axis parameters
|
||||
const auto elements = static_cast<std::size_t>(state.get_int64("Elements{io}"));
|
||||
const bit_entropy entropy = str_to_entropy(state.get_string("Entropy"));
|
||||
|
||||
const auto num_items_lhs = elements / 2;
|
||||
const auto num_items_rhs = elements - num_items_lhs;
|
||||
|
||||
thrust::device_vector<KeyT> keys_out(elements, thrust::no_init);
|
||||
thrust::device_vector<ValueT> values_lhs(num_items_lhs, thrust::no_init);
|
||||
thrust::device_vector<ValueT> values_rhs(num_items_rhs, thrust::no_init);
|
||||
thrust::device_vector<ValueT> values_out(elements, thrust::no_init);
|
||||
|
||||
auto [keys_lhs, keys_rhs] = generate_lhs_rhs<KeyT>(num_items_lhs, num_items_rhs, entropy);
|
||||
|
||||
KeyT* d_keys_lhs = thrust::raw_pointer_cast(keys_lhs.data());
|
||||
KeyT* d_keys_rhs = thrust::raw_pointer_cast(keys_rhs.data());
|
||||
KeyT* d_keys_out = thrust::raw_pointer_cast(keys_out.data());
|
||||
ValueT* d_values_lhs = thrust::raw_pointer_cast(values_lhs.data());
|
||||
ValueT* d_values_rhs = thrust::raw_pointer_cast(values_rhs.data());
|
||||
ValueT* d_values_out = thrust::raw_pointer_cast(values_out.data());
|
||||
|
||||
// Enable throughput calculations and add "Size" column to results.
|
||||
state.add_element_count(elements);
|
||||
state.add_global_memory_reads<KeyT>(elements);
|
||||
state.add_global_memory_reads<ValueT>(elements);
|
||||
state.add_global_memory_writes<KeyT>(elements);
|
||||
state.add_global_memory_writes<ValueT>(elements);
|
||||
|
||||
caching_allocator_t alloc;
|
||||
state.exec(nvbench::exec_tag::gpu | nvbench::exec_tag::no_batch, [&](nvbench::launch& launch) {
|
||||
auto env = cub_bench_env(
|
||||
alloc,
|
||||
launch
|
||||
#if !TUNE_BASE
|
||||
,
|
||||
cuda::execution::tune(bench_policy_selector<key_t>{})
|
||||
#endif // !TUNE_BASE
|
||||
);
|
||||
_CCCL_TRY_CUDA_API(
|
||||
cub::DeviceMerge::MergePairs,
|
||||
"MergePairs failed",
|
||||
d_keys_lhs,
|
||||
d_values_lhs,
|
||||
static_cast<offset_t>(num_items_lhs),
|
||||
d_keys_rhs,
|
||||
d_values_rhs,
|
||||
static_cast<offset_t>(num_items_rhs),
|
||||
d_keys_out,
|
||||
d_values_out,
|
||||
compare_op_t{},
|
||||
env);
|
||||
});
|
||||
}
|
||||
|
||||
#ifdef TUNE_KeyT
|
||||
using key_types = nvbench::type_list<TUNE_KeyT>;
|
||||
#else // !defined(TUNE_KeyT)
|
||||
using key_types = fundamental_types;
|
||||
#endif // TUNE_KeyT
|
||||
|
||||
#ifdef TUNE_ValueT
|
||||
using value_types = nvbench::type_list<TUNE_ValueT>;
|
||||
#else // !defined(TUNE_ValueT)
|
||||
using value_types = nvbench::type_list<int8_t, int16_t, int32_t, int64_t
|
||||
# if _CCCL_HAS_INT128()
|
||||
// nvcc currently hangs for __int128 value type with the fallback policy of {CTA: 64, IPT: 1}. NVBug 4384075
|
||||
// ,
|
||||
// int128_t
|
||||
# endif
|
||||
>;
|
||||
#endif // TUNE_ValueT
|
||||
|
||||
NVBENCH_BENCH_TYPES(pairs, NVBENCH_TYPE_AXES(key_types, value_types))
|
||||
.set_name("base")
|
||||
.set_type_axes_names({"KeyT{ct}", "ValueT{ct}"})
|
||||
.add_int64_power_of_two_axis("Elements{io}", nvbench::range(16, 28, 4))
|
||||
.add_string_axis("Entropy", {"1.000", "0.201"});
|
||||
Reference in New Issue
Block a user