[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
76
cccl_upstream/thrust/examples/monte_carlo.cu
Normal file
76
cccl_upstream/thrust/examples/monte_carlo.cu
Normal file
@@ -0,0 +1,76 @@
|
||||
#include <thrust/functional.h>
|
||||
#include <thrust/iterator/counting_iterator.h>
|
||||
#include <thrust/random.h>
|
||||
#include <thrust/transform_reduce.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <iomanip>
|
||||
#include <iostream>
|
||||
|
||||
// we could vary M & N to find the perf sweet spot
|
||||
|
||||
__host__ __device__ unsigned int hash(unsigned int a)
|
||||
{
|
||||
a = (a + 0x7ed55d16) + (a << 12);
|
||||
a = (a ^ 0xc761c23c) ^ (a >> 19);
|
||||
a = (a + 0x165667b1) + (a << 5);
|
||||
a = (a + 0xd3a2646c) ^ (a << 9);
|
||||
a = (a + 0xfd7046c5) + (a << 3);
|
||||
a = (a ^ 0xb55a4f09) ^ (a >> 16);
|
||||
return a;
|
||||
}
|
||||
|
||||
struct estimate_pi
|
||||
{
|
||||
__host__ __device__ float operator()(unsigned int thread_id)
|
||||
{
|
||||
float sum = 0;
|
||||
unsigned int N = 10000; // samples per thread
|
||||
|
||||
unsigned int seed = hash(thread_id);
|
||||
|
||||
// seed a random number generator
|
||||
thrust::default_random_engine rng(seed);
|
||||
|
||||
// create a mapping from random numbers to [0,1)
|
||||
thrust::uniform_real_distribution<float> u01(0, 1);
|
||||
|
||||
// take N samples in a quarter circle
|
||||
for (unsigned int i = 0; i < N; ++i)
|
||||
{
|
||||
// draw a sample from the unit square
|
||||
float x = u01(rng);
|
||||
float y = u01(rng);
|
||||
|
||||
// measure distance from the origin
|
||||
float dist = sqrtf(x * x + y * y);
|
||||
|
||||
// add 1.0f if (u0,u1) is inside the quarter circle
|
||||
if (dist <= 1.0f)
|
||||
{
|
||||
sum += 1.0f;
|
||||
}
|
||||
}
|
||||
|
||||
// multiply by 4 to get the area of the whole circle
|
||||
sum *= 4.0f;
|
||||
|
||||
// divide by N
|
||||
return sum / static_cast<float>(N);
|
||||
}
|
||||
};
|
||||
|
||||
int main()
|
||||
{
|
||||
// use 30K independent seeds
|
||||
int M = 30000;
|
||||
|
||||
float estimate = thrust::transform_reduce(
|
||||
thrust::counting_iterator<int>(0), thrust::counting_iterator<int>(M), estimate_pi(), 0.0f, cuda::std::plus<float>());
|
||||
estimate /= static_cast<float>(M);
|
||||
|
||||
std::cout << std::setprecision(3);
|
||||
std::cout << "pi is approximately " << estimate << '\n';
|
||||
|
||||
return 0;
|
||||
}
|
||||
Reference in New Issue
Block a user