[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
7
cccl_upstream/cub/examples/block/.gitignore
vendored
Normal file
7
cccl_upstream/cub/examples/block/.gitignore
vendored
Normal file
@@ -0,0 +1,7 @@
|
||||
/bin
|
||||
/Debug
|
||||
/Release
|
||||
/cuda55.sdf
|
||||
/cuda55.suo
|
||||
/cuda60.sdf
|
||||
/cuda60.suo
|
||||
18
cccl_upstream/cub/examples/block/CMakeLists.txt
Normal file
18
cccl_upstream/cub/examples/block/CMakeLists.txt
Normal file
@@ -0,0 +1,18 @@
|
||||
file(
|
||||
GLOB_RECURSE example_srcs
|
||||
RELATIVE "${CMAKE_CURRENT_LIST_DIR}"
|
||||
CONFIGURE_DEPENDS
|
||||
example_*.cu
|
||||
)
|
||||
|
||||
foreach (example_src IN LISTS example_srcs)
|
||||
get_filename_component(example_name "${example_src}" NAME_WE)
|
||||
string(
|
||||
REGEX REPLACE
|
||||
"^example_block_"
|
||||
"block."
|
||||
example_name
|
||||
"${example_name}"
|
||||
)
|
||||
cub_add_example(target_name ${example_name} "${example_src}")
|
||||
endforeach()
|
||||
315
cccl_upstream/cub/examples/block/example_block_radix_sort.cu
Normal file
315
cccl_upstream/cub/examples/block/example_block_radix_sort.cu
Normal file
@@ -0,0 +1,315 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011, Duane Merrill. All rights reserved.
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
/******************************************************************************
|
||||
* Simple demonstration of cub::BlockRadixSort
|
||||
*
|
||||
* To compile using the command line:
|
||||
* nvcc -arch=sm_XX example_block_radix_sort.cu -I../.. -lcudart -O3
|
||||
*
|
||||
******************************************************************************/
|
||||
|
||||
// Ensure printing of CUDA runtime errors to console (define before including cub.h)
|
||||
#define CUB_STDERR
|
||||
|
||||
#include <cub/block/block_load.cuh>
|
||||
#include <cub/block/block_radix_sort.cuh>
|
||||
#include <cub/block/block_store.cuh>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <iostream>
|
||||
|
||||
#include "../../test/test_util.h"
|
||||
|
||||
using namespace cub;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Globals, constants and aliases
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/// Verbose output
|
||||
bool g_verbose = false;
|
||||
|
||||
/// Timing iterations
|
||||
int g_timing_iterations = 100;
|
||||
|
||||
/// Default grid size
|
||||
int g_grid_size = 1;
|
||||
|
||||
/// Uniform key samples
|
||||
bool g_uniform_keys;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Kernels
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Simple kernel for performing a block-wide sorting over integers
|
||||
*/
|
||||
template <typename Key,
|
||||
int BLOCK_THREADS,
|
||||
int ITEMS_PER_THREAD>
|
||||
__launch_bounds__(BLOCK_THREADS) __global__
|
||||
void BlockSortKernel(Key* d_in, // Tile of input
|
||||
Key* d_out, // Tile of output
|
||||
clock_t* d_elapsed) // Elapsed cycle count of block scan
|
||||
{
|
||||
static constexpr int TILE_SIZE = BLOCK_THREADS * ITEMS_PER_THREAD;
|
||||
|
||||
// Specialize BlockLoad type for our thread block (uses warp-striped loads for coalescing, then transposes in shared
|
||||
// memory to a blocked arrangement)
|
||||
using BlockLoadT = BlockLoad<Key, BLOCK_THREADS, ITEMS_PER_THREAD, BLOCK_LOAD_WARP_TRANSPOSE>;
|
||||
|
||||
// Specialize BlockRadixSort type for our thread block
|
||||
using BlockRadixSortT = BlockRadixSort<Key, BLOCK_THREADS, ITEMS_PER_THREAD>;
|
||||
|
||||
// Shared memory
|
||||
__shared__ union TempStorage
|
||||
{
|
||||
typename BlockLoadT::TempStorage load;
|
||||
typename BlockRadixSortT::TempStorage sort;
|
||||
} temp_storage;
|
||||
|
||||
// Per-thread tile items
|
||||
Key items[ITEMS_PER_THREAD];
|
||||
|
||||
// Our current block's offset
|
||||
int block_offset = blockIdx.x * TILE_SIZE;
|
||||
|
||||
// Load items into a blocked arrangement
|
||||
BlockLoadT(temp_storage.load).Load(d_in + block_offset, items);
|
||||
|
||||
// Barrier for smem reuse
|
||||
__syncthreads();
|
||||
|
||||
// Start cycle timer
|
||||
clock_t start = clock();
|
||||
|
||||
// Sort keys
|
||||
BlockRadixSortT(temp_storage.sort).SortBlockedToStriped(items);
|
||||
|
||||
// Stop cycle timer
|
||||
clock_t stop = clock();
|
||||
|
||||
// Store output in striped fashion
|
||||
StoreDirectStriped<BLOCK_THREADS>(threadIdx.x, d_out + block_offset, items);
|
||||
|
||||
// Store elapsed clocks
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
d_elapsed[blockIdx.x] = (start > stop) ? start - stop : stop - start;
|
||||
}
|
||||
}
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Host utilities
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Initialize sorting problem (and solution).
|
||||
*/
|
||||
template <typename Key>
|
||||
void Initialize(Key* h_in, Key* h_reference, int num_items, int tile_size)
|
||||
{
|
||||
for (int i = 0; i < num_items; ++i)
|
||||
{
|
||||
if (g_uniform_keys)
|
||||
{
|
||||
h_in[i] = 0;
|
||||
}
|
||||
else
|
||||
{
|
||||
RandomBits(h_in[i]);
|
||||
}
|
||||
h_reference[i] = h_in[i];
|
||||
}
|
||||
|
||||
// Only sort the first tile
|
||||
std::sort(h_reference, h_reference + tile_size);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test BlockScan
|
||||
*/
|
||||
template <typename Key, int BLOCK_THREADS, int ITEMS_PER_THREAD>
|
||||
void Test()
|
||||
{
|
||||
constexpr int TILE_SIZE = BLOCK_THREADS * ITEMS_PER_THREAD;
|
||||
|
||||
// Allocate host arrays
|
||||
Key* h_in = new Key[TILE_SIZE * g_grid_size];
|
||||
Key* h_reference = new Key[TILE_SIZE * g_grid_size];
|
||||
clock_t* h_elapsed = new clock_t[g_grid_size];
|
||||
|
||||
// Initialize problem and reference output on host
|
||||
Initialize(h_in, h_reference, TILE_SIZE * g_grid_size, TILE_SIZE);
|
||||
|
||||
// Initialize device arrays
|
||||
Key* d_in = nullptr;
|
||||
Key* d_out = nullptr;
|
||||
clock_t* d_elapsed = nullptr;
|
||||
CubDebugExit(cudaMalloc((void**) &d_in, sizeof(Key) * TILE_SIZE * g_grid_size));
|
||||
CubDebugExit(cudaMalloc((void**) &d_out, sizeof(Key) * TILE_SIZE * g_grid_size));
|
||||
CubDebugExit(cudaMalloc((void**) &d_elapsed, sizeof(clock_t) * g_grid_size));
|
||||
|
||||
// Display input problem data
|
||||
if (g_verbose)
|
||||
{
|
||||
printf("Input data: ");
|
||||
for (int i = 0; i < TILE_SIZE; i++)
|
||||
{
|
||||
std::cout << h_in[i] << ", "; // NOLINT(bugprone-unintended-char-ostream-output)
|
||||
}
|
||||
printf("\n\n");
|
||||
}
|
||||
|
||||
// Kernel props
|
||||
int max_sm_occupancy;
|
||||
CubDebugExit(MaxSmOccupancy(max_sm_occupancy, BlockSortKernel<Key, BLOCK_THREADS, ITEMS_PER_THREAD>, BLOCK_THREADS));
|
||||
|
||||
// Copy problem to device
|
||||
CubDebugExit(cudaMemcpy(d_in, h_in, sizeof(Key) * TILE_SIZE * g_grid_size, cudaMemcpyHostToDevice));
|
||||
|
||||
printf(
|
||||
"BlockRadixSort %d items (%d timing iterations, %d blocks, %d threads, %d items per thread, %d SM occupancy):\n",
|
||||
TILE_SIZE * g_grid_size,
|
||||
g_timing_iterations,
|
||||
g_grid_size,
|
||||
BLOCK_THREADS,
|
||||
ITEMS_PER_THREAD,
|
||||
max_sm_occupancy);
|
||||
fflush(stdout);
|
||||
|
||||
// Run kernel once to prime caches and check result
|
||||
BlockSortKernel<Key, BLOCK_THREADS, ITEMS_PER_THREAD><<<g_grid_size, BLOCK_THREADS>>>(d_in, d_out, d_elapsed);
|
||||
|
||||
// Check for kernel errors and STDIO from the kernel, if any
|
||||
CubDebugExit(cudaPeekAtLastError());
|
||||
CubDebugExit(cudaDeviceSynchronize());
|
||||
|
||||
// Check results
|
||||
printf("\tOutput items: ");
|
||||
int compare = CompareDeviceResults(h_reference, d_out, TILE_SIZE, g_verbose, g_verbose);
|
||||
printf("%s\n", compare ? "FAIL" : "PASS");
|
||||
AssertEquals(0, compare);
|
||||
fflush(stdout);
|
||||
|
||||
// Run this several times and average the performance results
|
||||
GpuTimer timer;
|
||||
float elapsed_millis = 0.0;
|
||||
unsigned long long elapsed_clocks = 0;
|
||||
|
||||
for (int i = 0; i < g_timing_iterations; ++i)
|
||||
{
|
||||
timer.Start();
|
||||
|
||||
// Run kernel
|
||||
BlockSortKernel<Key, BLOCK_THREADS, ITEMS_PER_THREAD><<<g_grid_size, BLOCK_THREADS>>>(d_in, d_out, d_elapsed);
|
||||
|
||||
timer.Stop();
|
||||
elapsed_millis += timer.ElapsedMillis();
|
||||
|
||||
// Copy clocks from device
|
||||
CubDebugExit(cudaMemcpy(h_elapsed, d_elapsed, sizeof(clock_t) * g_grid_size, cudaMemcpyDeviceToHost));
|
||||
for (int j = 0; j < g_grid_size; j++)
|
||||
{
|
||||
elapsed_clocks += h_elapsed[j];
|
||||
}
|
||||
}
|
||||
|
||||
// Check for kernel errors and STDIO from the kernel, if any
|
||||
CubDebugExit(cudaDeviceSynchronize());
|
||||
|
||||
// Display timing results
|
||||
float avg_millis = elapsed_millis / static_cast<float>(g_timing_iterations);
|
||||
float avg_items_per_sec = float(TILE_SIZE * g_grid_size) / avg_millis / 1000.0f;
|
||||
double avg_clocks = double(elapsed_clocks) / g_timing_iterations / g_grid_size;
|
||||
double avg_clocks_per_item = avg_clocks / TILE_SIZE;
|
||||
|
||||
printf("\tAverage BlockRadixSort::SortBlocked clocks: %.3f\n", avg_clocks);
|
||||
printf("\tAverage BlockRadixSort::SortBlocked clocks per item: %.3f\n", avg_clocks_per_item);
|
||||
printf("\tAverage kernel millis: %.4f\n", avg_millis);
|
||||
printf("\tAverage million items / sec: %.4f\n", avg_items_per_sec);
|
||||
fflush(stdout);
|
||||
|
||||
// Cleanup
|
||||
if (h_in)
|
||||
{
|
||||
delete[] h_in;
|
||||
}
|
||||
if (h_reference)
|
||||
{
|
||||
delete[] h_reference;
|
||||
}
|
||||
if (h_elapsed)
|
||||
{
|
||||
delete[] h_elapsed;
|
||||
}
|
||||
if (d_in)
|
||||
{
|
||||
CubDebugExit(cudaFree(d_in));
|
||||
}
|
||||
if (d_out)
|
||||
{
|
||||
CubDebugExit(cudaFree(d_out));
|
||||
}
|
||||
if (d_elapsed)
|
||||
{
|
||||
CubDebugExit(cudaFree(d_elapsed));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Main
|
||||
*/
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
// Initialize command line
|
||||
CommandLineArgs args(argc, argv);
|
||||
g_verbose = args.CheckCmdLineFlag("v");
|
||||
g_uniform_keys = args.CheckCmdLineFlag("uniform");
|
||||
args.GetCmdLineArgument("i", g_timing_iterations);
|
||||
args.GetCmdLineArgument("grid-size", g_grid_size);
|
||||
|
||||
// Print usage
|
||||
if (args.CheckCmdLineFlag("help"))
|
||||
{
|
||||
printf("%s "
|
||||
"[--device=<device-id>] "
|
||||
"[--i=<timing iterations (default:%d)>]"
|
||||
"[--grid-size=<grid size (default:%d)>]"
|
||||
"[--v] "
|
||||
"\n",
|
||||
argv[0],
|
||||
g_timing_iterations,
|
||||
g_grid_size);
|
||||
exit(0);
|
||||
}
|
||||
|
||||
// Initialize device
|
||||
CubDebugExit(args.DeviceInit());
|
||||
fflush(stdout);
|
||||
|
||||
// Run tests
|
||||
printf("\nuint32:\n");
|
||||
fflush(stdout);
|
||||
Test<unsigned int, 128, 13>();
|
||||
printf("\n");
|
||||
fflush(stdout);
|
||||
|
||||
printf("\nfp32:\n");
|
||||
fflush(stdout);
|
||||
Test<float, 128, 13>();
|
||||
printf("\n");
|
||||
fflush(stdout);
|
||||
|
||||
printf("\nuint8:\n");
|
||||
fflush(stdout);
|
||||
Test<unsigned char, 128, 13>();
|
||||
printf("\n");
|
||||
fflush(stdout);
|
||||
|
||||
return 0;
|
||||
}
|
||||
273
cccl_upstream/cub/examples/block/example_block_reduce.cu
Normal file
273
cccl_upstream/cub/examples/block/example_block_reduce.cu
Normal file
@@ -0,0 +1,273 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011, Duane Merrill. All rights reserved.
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2018, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
/******************************************************************************
|
||||
* Simple demonstration of cub::BlockReduce
|
||||
*
|
||||
* To compile using the command line:
|
||||
* nvcc -arch=sm_XX example_block_reduce.cu -I../.. -lcudart -O3
|
||||
*
|
||||
******************************************************************************/
|
||||
|
||||
// Ensure printing of CUDA runtime errors to console (define before including cub.h)
|
||||
#define CUB_STDERR
|
||||
|
||||
#include <cub/block/block_load.cuh>
|
||||
#include <cub/block/block_reduce.cuh>
|
||||
#include <cub/block/block_store.cuh>
|
||||
|
||||
#include <cstdio>
|
||||
#include <iostream>
|
||||
|
||||
#include "../../test/test_util.h"
|
||||
|
||||
using namespace cub;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Globals, constants and aliases
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/// Verbose output
|
||||
bool g_verbose = false;
|
||||
|
||||
/// Timing iterations
|
||||
int g_timing_iterations = 100;
|
||||
|
||||
/// Default grid size
|
||||
int g_grid_size = 1;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Kernels
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Simple kernel for performing a block-wide reduction.
|
||||
*/
|
||||
template <int BLOCK_THREADS,
|
||||
int ITEMS_PER_THREAD,
|
||||
BlockReduceAlgorithm ALGORITHM>
|
||||
__global__ void BlockReduceKernel(int* d_in, // Tile of input
|
||||
int* d_out, // Tile aggregate
|
||||
clock_t* d_elapsed) // Elapsed cycle count of block reduction
|
||||
{
|
||||
// Specialize BlockReduce type for our thread block
|
||||
using BlockReduceT = BlockReduce<int, BLOCK_THREADS, ALGORITHM>;
|
||||
|
||||
// Shared memory
|
||||
__shared__ typename BlockReduceT::TempStorage temp_storage;
|
||||
|
||||
// Per-thread tile data
|
||||
int data[ITEMS_PER_THREAD];
|
||||
LoadDirectStriped<BLOCK_THREADS>(threadIdx.x, d_in, data);
|
||||
|
||||
// Start cycle timer
|
||||
clock_t start = clock();
|
||||
|
||||
// Compute sum
|
||||
int aggregate = BlockReduceT(temp_storage).Sum(data);
|
||||
|
||||
// Stop cycle timer
|
||||
clock_t stop = clock();
|
||||
|
||||
// Store aggregate and elapsed clocks
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
*d_elapsed = (start > stop) ? start - stop : stop - start;
|
||||
*d_out = aggregate;
|
||||
}
|
||||
}
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Host utilities
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Initialize reduction problem (and solution).
|
||||
* Returns the aggregate
|
||||
*/
|
||||
int Initialize(int* h_in, int num_items)
|
||||
{
|
||||
int inclusive = 0;
|
||||
|
||||
for (int i = 0; i < num_items; ++i)
|
||||
{
|
||||
h_in[i] = i % 17;
|
||||
inclusive += h_in[i];
|
||||
}
|
||||
|
||||
return inclusive;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test thread block reduction
|
||||
*/
|
||||
template <int BLOCK_THREADS, int ITEMS_PER_THREAD, BlockReduceAlgorithm ALGORITHM>
|
||||
void Test()
|
||||
{
|
||||
constexpr int TILE_SIZE = BLOCK_THREADS * ITEMS_PER_THREAD;
|
||||
|
||||
// Allocate host arrays
|
||||
int* h_in = new int[TILE_SIZE];
|
||||
int* h_gpu = new int[TILE_SIZE + 1];
|
||||
|
||||
// Initialize problem and reference output on host
|
||||
int h_aggregate = Initialize(h_in, TILE_SIZE);
|
||||
|
||||
// Initialize device arrays
|
||||
int* d_in = nullptr;
|
||||
int* d_out = nullptr;
|
||||
clock_t* d_elapsed = nullptr;
|
||||
cudaMalloc((void**) &d_in, sizeof(int) * TILE_SIZE);
|
||||
cudaMalloc((void**) &d_out, sizeof(int) * 1);
|
||||
cudaMalloc((void**) &d_elapsed, sizeof(clock_t));
|
||||
|
||||
// Display input problem data
|
||||
if (g_verbose)
|
||||
{
|
||||
printf("Input data: ");
|
||||
for (int i = 0; i < TILE_SIZE; i++)
|
||||
{
|
||||
printf("%d, ", h_in[i]);
|
||||
}
|
||||
printf("\n\n");
|
||||
}
|
||||
|
||||
// Kernel props
|
||||
int max_sm_occupancy;
|
||||
CubDebugExit(
|
||||
MaxSmOccupancy(max_sm_occupancy, BlockReduceKernel<BLOCK_THREADS, ITEMS_PER_THREAD, ALGORITHM>, BLOCK_THREADS));
|
||||
|
||||
// Copy problem to device
|
||||
cudaMemcpy(d_in, h_in, sizeof(int) * TILE_SIZE, cudaMemcpyHostToDevice);
|
||||
|
||||
printf("BlockReduce algorithm %s on %d items (%d timing iterations, %d blocks, %d threads, %d items per thread, %d "
|
||||
"SM occupancy):\n",
|
||||
(ALGORITHM == BLOCK_REDUCE_RAKING) ? "BLOCK_REDUCE_RAKING" : "BLOCK_REDUCE_WARP_REDUCTIONS",
|
||||
TILE_SIZE,
|
||||
g_timing_iterations,
|
||||
g_grid_size,
|
||||
BLOCK_THREADS,
|
||||
ITEMS_PER_THREAD,
|
||||
max_sm_occupancy);
|
||||
|
||||
// Run kernel
|
||||
BlockReduceKernel<BLOCK_THREADS, ITEMS_PER_THREAD, ALGORITHM><<<g_grid_size, BLOCK_THREADS>>>(d_in, d_out, d_elapsed);
|
||||
|
||||
// Check total aggregate
|
||||
printf("\tAggregate: ");
|
||||
int compare = CompareDeviceResults(&h_aggregate, d_out, 1, g_verbose, g_verbose);
|
||||
printf("%s\n", compare ? "FAIL" : "PASS");
|
||||
AssertEquals(0, compare);
|
||||
|
||||
// Run this several times and average the performance results
|
||||
GpuTimer timer;
|
||||
float elapsed_millis = 0.0;
|
||||
clock_t elapsed_clocks = 0;
|
||||
|
||||
for (int i = 0; i < g_timing_iterations; ++i)
|
||||
{
|
||||
// Copy problem to device
|
||||
cudaMemcpy(d_in, h_in, sizeof(int) * TILE_SIZE, cudaMemcpyHostToDevice);
|
||||
|
||||
timer.Start();
|
||||
|
||||
// Run kernel
|
||||
BlockReduceKernel<BLOCK_THREADS, ITEMS_PER_THREAD, ALGORITHM>
|
||||
<<<g_grid_size, BLOCK_THREADS>>>(d_in, d_out, d_elapsed);
|
||||
|
||||
timer.Stop();
|
||||
elapsed_millis += timer.ElapsedMillis();
|
||||
|
||||
// Copy clocks from device
|
||||
clock_t clocks;
|
||||
CubDebugExit(cudaMemcpy(&clocks, d_elapsed, sizeof(clock_t), cudaMemcpyDeviceToHost));
|
||||
elapsed_clocks += clocks;
|
||||
}
|
||||
|
||||
// Check for kernel errors and STDIO from the kernel, if any
|
||||
CubDebugExit(cudaPeekAtLastError());
|
||||
CubDebugExit(cudaDeviceSynchronize());
|
||||
|
||||
// Display timing results
|
||||
float avg_millis = elapsed_millis / static_cast<float>(g_timing_iterations);
|
||||
float avg_items_per_sec = float(TILE_SIZE * g_grid_size) / avg_millis / 1000.0f;
|
||||
float avg_clocks = float(elapsed_clocks) / static_cast<float>(g_timing_iterations);
|
||||
float avg_clocks_per_item = avg_clocks / TILE_SIZE;
|
||||
|
||||
printf("\tAverage BlockReduce::Sum clocks: %.3f\n", avg_clocks);
|
||||
printf("\tAverage BlockReduce::Sum clocks per item: %.3f\n", avg_clocks_per_item);
|
||||
printf("\tAverage kernel millis: %.4f\n", avg_millis);
|
||||
printf("\tAverage million items / sec: %.4f\n", avg_items_per_sec);
|
||||
|
||||
// Cleanup
|
||||
if (h_in)
|
||||
{
|
||||
delete[] h_in;
|
||||
}
|
||||
if (h_gpu)
|
||||
{
|
||||
delete[] h_gpu;
|
||||
}
|
||||
if (d_in)
|
||||
{
|
||||
cudaFree(d_in);
|
||||
}
|
||||
if (d_out)
|
||||
{
|
||||
cudaFree(d_out);
|
||||
}
|
||||
if (d_elapsed)
|
||||
{
|
||||
cudaFree(d_elapsed);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Main
|
||||
*/
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
// Initialize command line
|
||||
CommandLineArgs args(argc, argv);
|
||||
g_verbose = args.CheckCmdLineFlag("v");
|
||||
args.GetCmdLineArgument("i", g_timing_iterations);
|
||||
args.GetCmdLineArgument("grid-size", g_grid_size);
|
||||
|
||||
// Print usage
|
||||
if (args.CheckCmdLineFlag("help"))
|
||||
{
|
||||
printf("%s "
|
||||
"[--device=<device-id>] "
|
||||
"[--i=<timing iterations>] "
|
||||
"[--grid-size=<grid size>] "
|
||||
"[--v] "
|
||||
"\n",
|
||||
argv[0]);
|
||||
exit(0);
|
||||
}
|
||||
|
||||
// Initialize device
|
||||
CubDebugExit(args.DeviceInit());
|
||||
|
||||
// Run tests
|
||||
Test<1024, 1, BLOCK_REDUCE_RAKING>();
|
||||
Test<512, 2, BLOCK_REDUCE_RAKING>();
|
||||
Test<256, 4, BLOCK_REDUCE_RAKING>();
|
||||
Test<128, 8, BLOCK_REDUCE_RAKING>();
|
||||
Test<64, 16, BLOCK_REDUCE_RAKING>();
|
||||
Test<32, 32, BLOCK_REDUCE_RAKING>();
|
||||
Test<16, 64, BLOCK_REDUCE_RAKING>();
|
||||
|
||||
printf("-------------\n");
|
||||
|
||||
Test<1024, 1, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<512, 2, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<256, 4, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<128, 8, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<64, 16, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<32, 32, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
Test<16, 64, BLOCK_REDUCE_WARP_REDUCTIONS>();
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,215 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2021, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
/******************************************************************************
|
||||
* Simple demonstration of cub::BlockReduce with dynamic shared memory
|
||||
*
|
||||
* To compile using the command line:
|
||||
* nvcc -arch=sm_XX example_block_reduce_dyn_smem.cu -I../.. -lcudart -O3 -std=c++17
|
||||
*
|
||||
******************************************************************************/
|
||||
|
||||
// Ensure printing of CUDA runtime errors to console (define before including cub.h)
|
||||
#define CUB_STDERR
|
||||
|
||||
#include <cub/block/block_load.cuh>
|
||||
#include <cub/block/block_reduce.cuh>
|
||||
#include <cub/block/block_store.cuh>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstdio>
|
||||
#include <iostream>
|
||||
|
||||
#include "../../test/test_util.h"
|
||||
|
||||
using namespace cub;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Globals, constants and aliases
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/// Verbose output
|
||||
bool g_verbose = false;
|
||||
|
||||
/// Default grid size
|
||||
int g_grid_size = 1;
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Kernels
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Simple kernel for performing a block-wide reduction.
|
||||
*/
|
||||
template <int BLOCK_THREADS>
|
||||
__global__ void BlockReduceKernel(int* d_in, // Tile of input
|
||||
int* d_out // Tile aggregate
|
||||
)
|
||||
{
|
||||
// Specialize BlockReduce type for our thread block
|
||||
using BlockReduceT = cub::BlockReduce<int, BLOCK_THREADS>;
|
||||
using TempStorageT = typename BlockReduceT::TempStorage;
|
||||
|
||||
union ShmemLayout
|
||||
{
|
||||
TempStorageT reduce;
|
||||
int aggregate;
|
||||
};
|
||||
|
||||
// shared memory byte-array
|
||||
extern __shared__ __align__(alignof(ShmemLayout)) char smem[];
|
||||
|
||||
// cast to lvalue reference of expected type
|
||||
auto& temp_storage = reinterpret_cast<TempStorageT&>(smem);
|
||||
|
||||
int data = d_in[threadIdx.x];
|
||||
|
||||
// Compute sum
|
||||
int aggregate = BlockReduceT(temp_storage).Sum(data);
|
||||
|
||||
// block-wide sync barrier necessary to re-use shared mem safely
|
||||
__syncthreads();
|
||||
int* smem_integers = reinterpret_cast<int*>(smem);
|
||||
if (threadIdx.x == 0)
|
||||
{
|
||||
smem_integers[0] = aggregate;
|
||||
}
|
||||
|
||||
// sync to make new shared value available to all threads
|
||||
__syncthreads();
|
||||
aggregate = smem_integers[0];
|
||||
|
||||
// all threads write the aggregate to output
|
||||
d_out[threadIdx.x] = aggregate;
|
||||
}
|
||||
|
||||
//---------------------------------------------------------------------
|
||||
// Host utilities
|
||||
//---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Initialize reduction problem (and solution).
|
||||
* Returns the aggregate
|
||||
*/
|
||||
int Initialize(int* h_in, int num_items)
|
||||
{
|
||||
int inclusive = 0;
|
||||
|
||||
for (int i = 0; i < num_items; ++i)
|
||||
{
|
||||
h_in[i] = i % 17;
|
||||
inclusive += h_in[i];
|
||||
}
|
||||
|
||||
return inclusive;
|
||||
}
|
||||
|
||||
/**
|
||||
* Test thread block reduction
|
||||
*/
|
||||
template <int BLOCK_THREADS>
|
||||
void Test()
|
||||
{
|
||||
// Allocate host arrays
|
||||
int* h_in = new int[BLOCK_THREADS];
|
||||
|
||||
// Initialize problem and reference output on host
|
||||
int h_aggregate = Initialize(h_in, BLOCK_THREADS);
|
||||
|
||||
// Initialize device arrays
|
||||
int* d_in = nullptr;
|
||||
int* d_out = nullptr;
|
||||
cudaMalloc((void**) &d_in, sizeof(int) * BLOCK_THREADS);
|
||||
cudaMalloc((void**) &d_out, sizeof(int) * BLOCK_THREADS);
|
||||
|
||||
// Display input problem data
|
||||
if (g_verbose)
|
||||
{
|
||||
printf("Input data: ");
|
||||
for (int i = 0; i < BLOCK_THREADS; i++)
|
||||
{
|
||||
printf("%d, ", h_in[i]);
|
||||
}
|
||||
printf("\n\n");
|
||||
}
|
||||
|
||||
// Copy problem to device
|
||||
cudaMemcpy(d_in, h_in, sizeof(int) * BLOCK_THREADS, cudaMemcpyHostToDevice);
|
||||
|
||||
// determine necessary storage size:
|
||||
auto block_reduce_temp_bytes = sizeof(typename cub::BlockReduce<int, BLOCK_THREADS>::TempStorage);
|
||||
// finally, we need to make sure that we can hold at least one integer
|
||||
// needed in the kernel to exchange data after reduction
|
||||
auto smem_size = (std::max) (1 * sizeof(int), block_reduce_temp_bytes);
|
||||
|
||||
// use default stream
|
||||
cudaStream_t stream = nullptr;
|
||||
|
||||
// Run reduction kernel
|
||||
BlockReduceKernel<BLOCK_THREADS><<<g_grid_size, BLOCK_THREADS, smem_size, stream>>>(d_in, d_out);
|
||||
|
||||
// Check total aggregate
|
||||
printf("\tAggregate: ");
|
||||
int compare = 0;
|
||||
for (int i = 0; i < BLOCK_THREADS; i++)
|
||||
{
|
||||
compare = compare || CompareDeviceResults(&h_aggregate, d_out + i, 1, g_verbose, g_verbose);
|
||||
}
|
||||
printf("%s\n", compare ? "FAIL" : "PASS");
|
||||
AssertEquals(0, compare);
|
||||
|
||||
// Check for kernel errors and STDIO from the kernel, if any
|
||||
CubDebugExit(cudaPeekAtLastError());
|
||||
CubDebugExit(cudaDeviceSynchronize());
|
||||
|
||||
// Cleanup
|
||||
if (h_in)
|
||||
{
|
||||
delete[] h_in;
|
||||
}
|
||||
if (d_in)
|
||||
{
|
||||
cudaFree(d_in);
|
||||
}
|
||||
if (d_out)
|
||||
{
|
||||
cudaFree(d_out);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Main
|
||||
*/
|
||||
int main(int argc, char** argv)
|
||||
{
|
||||
// Initialize command line
|
||||
CommandLineArgs args(argc, argv);
|
||||
g_verbose = args.CheckCmdLineFlag("v");
|
||||
args.GetCmdLineArgument("grid-size", g_grid_size);
|
||||
|
||||
// Print usage
|
||||
if (args.CheckCmdLineFlag("help"))
|
||||
{
|
||||
printf("%s "
|
||||
"[--device=<device-id>] "
|
||||
"[--grid-size=<grid size>] "
|
||||
"[--v] "
|
||||
"\n",
|
||||
argv[0]);
|
||||
exit(0);
|
||||
}
|
||||
|
||||
// Initialize device
|
||||
CubDebugExit(args.DeviceInit());
|
||||
|
||||
// Run tests
|
||||
Test<1024>();
|
||||
Test<512>();
|
||||
Test<256>();
|
||||
Test<128>();
|
||||
Test<64>();
|
||||
Test<32>();
|
||||
Test<16>();
|
||||
|
||||
return 0;
|
||||
}
|
||||
1265
cccl_upstream/cub/examples/block/example_block_scan.cu
Normal file
1265
cccl_upstream/cub/examples/block/example_block_scan.cu
Normal file
File diff suppressed because it is too large
Load Diff
Reference in New Issue
Block a user