Files
project_6/cccl_upstream/thrust/examples/scan_by_key.cu
EngineX CI 56fd68e7dd [INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
2026-07-30 09:35:51 +00:00

92 lines
2.7 KiB
Plaintext

#include <thrust/copy.h>
#include <thrust/device_vector.h>
#include <thrust/scan.h>
#include <iostream>
// BinaryPredicate for the head flag segment representation
// equivalent to cuda::std::not_fn(thrust::project2nd<int,int>()));
template <typename HeadFlagType>
struct head_flag_predicate
{
__host__ __device__ bool operator()(HeadFlagType, HeadFlagType right) const
{
return !right;
}
};
template <typename Vector>
void print(const Vector& v)
{
for (const auto& e : v)
{
std::cout << e << " ";
}
std::cout << '\n';
}
int main()
{
int keys[] = {0, 0, 0, 1, 1, 2, 2, 2, 2, 3, 4, 4, 5, 5, 5}; // segments represented with keys
int flags[] = {1, 0, 0, 1, 0, 1, 0, 0, 0, 1, 1, 0, 1, 0, 0}; // segments represented with head flags
int values[] = {2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2, 2}; // values corresponding to each key
int N = sizeof(keys) / sizeof(int); // number of elements
// copy input data to device
thrust::device_vector<int> d_keys(keys, keys + N);
thrust::device_vector<int> d_flags(flags, flags + N);
thrust::device_vector<int> d_values(values, values + N);
// allocate storage for output
thrust::device_vector<int> d_output(N);
// inclusive scan using keys
thrust::inclusive_scan_by_key(d_keys.begin(), d_keys.end(), d_values.begin(), d_output.begin());
std::cout << "Inclusive Segmented Scan w/ Key Sequence\n";
std::cout << " keys : ";
print(d_keys);
std::cout << " input values : ";
print(d_values);
std::cout << " output values : ";
print(d_output);
// inclusive scan using head flags
thrust::inclusive_scan_by_key(
d_flags.begin(), d_flags.end(), d_values.begin(), d_output.begin(), head_flag_predicate<int>());
std::cout << "\nInclusive Segmented Scan w/ Head Flag Sequence\n";
std::cout << " head flags : ";
print(d_flags);
std::cout << " input values : ";
print(d_values);
std::cout << " output values : ";
print(d_output);
// exclusive scan using keys
thrust::exclusive_scan_by_key(d_keys.begin(), d_keys.end(), d_values.begin(), d_output.begin());
std::cout << "\nExclusive Segmented Scan w/ Key Sequence\n";
std::cout << " keys : ";
print(d_keys);
std::cout << " input values : ";
print(d_values);
std::cout << " output values : ";
print(d_output);
// exclusive scan using head flags
thrust::exclusive_scan_by_key(
d_flags.begin(), d_flags.end(), d_values.begin(), d_output.begin(), 0, head_flag_predicate<int>());
std::cout << "\nExclusive Segmented Scan w/ Head Flag Sequence\n";
std::cout << " head flags : ";
print(d_flags);
std::cout << " input values : ";
print(d_values);
std::cout << " output values : ";
print(d_output);
return 0;
}