[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,138 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef LIBCUDACXX_TEST_SUPPORT_STATS_FUNCTIONS_H
#define LIBCUDACXX_TEST_SUPPORT_STATS_FUNCTIONS_H
#include <cuda/std/cmath>
#include "test_macros.h"
// Regularized incomplete gamma function P(a,x)
// Adapted from numerical recipes
TEST_FUNC inline double incomplete_gamma(double a, double x)
{
if (x <= 0.0)
{
return 0.0;
}
const int max_iter = 100;
double sum = 1.0 / a;
double term = 1.0 / a;
for (int n = 1; n < max_iter; ++n)
{
term *= x / (a + n);
sum += term;
if (cuda::std::abs(term) < 1e-12 * cuda::std::abs(sum))
{
break;
}
}
return cuda::std::exp(-x + a * cuda::std::log(x) - cuda::std::lgamma(a)) * sum;
}
// Regularized incomplete beta function I_x(a,b)
// Adapted from numerical recipes
TEST_FUNC inline double incomplete_beta(double a, double b, double x)
{
if (x <= 0.0)
{
return 0.0;
}
if (x >= 1.0)
{
return 1.0;
}
double log_beta = cuda::std::lgamma(a) + cuda::std::lgamma(b) - cuda::std::lgamma(a + b);
const int max_iter = 200;
const double eps = 1e-12;
bool use_complement = false;
double xx = x;
double aa = a;
double bb = b;
if (x > (a + 1.0) / (a + b + 2.0))
{
// Use symmetry relation: I_x(a,b) = 1 - I_{1-x}(b,a)
use_complement = true;
xx = 1.0 - x;
aa = b;
bb = a;
}
// Continued fraction expansion
double qab = aa + bb;
double qap = aa + 1.0;
double qam = aa - 1.0;
double c = 1.0;
double d = 1.0 - qab * xx / qap;
if (cuda::std::abs(d) < 1e-30)
{
d = 1e-30;
}
d = 1.0 / d;
double h = d;
for (int m = 1; m <= max_iter; ++m)
{
int m2 = 2 * m;
double aa1 = m * (bb - m) * xx / ((qam + m2) * (aa + m2));
d = 1.0 + aa1 * d;
if (cuda::std::abs(d) < 1e-30)
{
d = 1e-30;
}
c = 1.0 + aa1 / c;
if (cuda::std::abs(c) < 1e-30)
{
c = 1e-30;
}
d = 1.0 / d;
h *= d * c;
double aa2 = -(aa + m) * (qab + m) * xx / ((aa + m2) * (qap + m2));
d = 1.0 + aa2 * d;
if (cuda::std::abs(d) < 1e-30)
{
d = 1e-30;
}
c = 1.0 + aa2 / c;
if (cuda::std::abs(c) < 1e-30)
{
c = 1e-30;
}
d = 1.0 / d;
double del = d * c;
h *= del;
if (cuda::std::abs(del - 1.0) < eps)
{
break;
}
}
double log_prefix = aa * cuda::std::log(xx) + bb * cuda::std::log(1.0 - xx) - log_beta;
double result = cuda::std::exp(log_prefix) * h / aa;
if (use_complement)
{
return 1.0 - result;
}
return result;
}
#endif // LIBCUDACXX_TEST_SUPPORT_STATS_FUNCTIONS_H

View File

@@ -0,0 +1,277 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef LIBCUDACXX_TEST_SUPPORT_RANDOM_UTILITIES_TEST_DISTRIBUTION_H
#define LIBCUDACXX_TEST_SUPPORT_RANDOM_UTILITIES_TEST_DISTRIBUTION_H
#include <cuda/std/__algorithm/partial_sort.h>
#include <cuda/std/__memory_>
#include <cuda/std/array>
#include <cuda/std/cstddef>
#if _CCCL_HOSTED()
# include <sstream>
#endif // _CCCL_HOSTED()
#include "test_macros.h"
namespace detail
{
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_ctor_assign(Param param)
{
D d1(param);
D d2;
d2 = d1;
assert(d1 == d2);
assert(d1.param() == param);
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_copy(Param param)
{
D d1(param);
D d2(d1);
assert(d1 == d2);
static_assert(noexcept(D(d1)));
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_eq(Param param)
{
D d1(param);
D d2(param);
assert(d1 == d2);
assert(!(d1 != d2));
static_assert(noexcept(d1 == d2));
static_assert(noexcept(d1 != d2));
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_get_param(Param param)
{
D d1(param);
assert(d1.param() == param);
static_assert(noexcept(d1.param()));
return true;
}
#if _CCCL_HOSTED()
template <class D, class URNG, class Param>
bool test_io(Param param)
{
D d1(param);
std::stringstream ss;
ss << d1;
D d2;
ss >> d2;
assert(d1 == d2);
return true;
}
#endif
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_min_max(Param param)
{
D d1(param);
static_assert(noexcept(d1.min()));
static_assert(noexcept(d1.max()));
assert(d1.min() <= d1.max());
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_set_param(Param param)
{
D d1;
d1.param(param);
assert(d1.param() == param);
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_types(Param param)
{
D d1(param);
[[maybe_unused]] URNG g{};
using result_type = typename D::result_type;
static_assert(cuda::std::is_same_v<result_type, decltype(d1.min())>);
static_assert(cuda::std::is_same_v<result_type, decltype(d1.max())>);
static_assert(cuda::std::is_same_v<result_type, decltype(d1(g))>);
static_assert(cuda::std::is_same_v<result_type, decltype(d1(g, param))>);
return true;
}
template <class D, class URNG, class Param>
TEST_FUNC constexpr bool test_param(Param param)
{
static_assert(cuda::std::is_same_v<typename D::param_type, Param>);
static_assert(cuda::std::is_same_v<typename Param::distribution_type, D>);
Param p2(param);
assert(p2 == param);
assert(!(p2 != param));
Param p3 = param;
assert(p3 == param);
static_assert(noexcept(p3 = p2));
static_assert(noexcept(p2 == p3));
static_assert(noexcept(p2 != p3));
return true;
}
// Compute KS test statistic for continuous distributions
template <class D, class CDF>
TEST_FUNC double ks_test_statistic_continuous(
const typename D::result_type* samples, cuda::std::size_t num_samples, const typename D::param_type& param, CDF cdf)
{
double d_max = 0.0;
for (cuda::std::size_t i = 0; i < num_samples; ++i)
{
double f_x = static_cast<double>(i + 1) / static_cast<double>(num_samples);
double f_x_lower = static_cast<double>(i) / static_cast<double>(num_samples);
double g_x = cdf(samples[i], param);
double diff1 = cuda::std::abs(f_x - g_x);
double diff2 = cuda::std::abs(g_x - f_x_lower);
d_max = cuda::std::max(d_max, cuda::std::max(diff1, diff2));
}
return d_max;
}
// Compute KS test statistic for discrete distributions
template <class D, class CDF>
TEST_FUNC double ks_test_statistic_discrete(
const typename D::result_type* samples, cuda::std::size_t num_samples, const typename D::param_type& param, CDF cdf)
{
// Compute empirical CDF
// Find unique values and their frequencies
auto unique_values = cuda::std::make_unique<typename D::result_type[]>(num_samples);
auto empirical_cdf = cuda::std::make_unique<double[]>(num_samples);
cuda::std::size_t unique_count = 0;
for (cuda::std::size_t i = 0; i < num_samples; ++i)
{
if (samples[i] != samples[i + 1] || i == num_samples - 1)
{
unique_values[unique_count] = samples[i];
empirical_cdf[unique_count] = (i + 1) / static_cast<double>(num_samples);
unique_count++;
}
}
// Compute KS statistic
double d_max = 0.0;
for (cuda::std::size_t j = 0; j < unique_count; ++j)
{
double f_x = empirical_cdf[j];
double f_x_lower = j == 0 ? 0.0 : empirical_cdf[j - 1];
double g_x = cdf(unique_values[j], param);
double g_x_lower = j == 0 ? 0.0 : cdf(unique_values[j] - 1, param);
double diff1 = cuda::std::abs(f_x - g_x);
double diff2 = cuda::std::abs(g_x_lower - f_x_lower);
d_max = cuda::std::max(d_max, cuda::std::max(diff1, diff2));
}
return d_max;
}
// Perform a kolmogorov-Smirnov test, comparing the observed and expected cumulative
// distribution function from a continuous distribution.
// Generates a fixed size of 10000 samples
template <class D, bool continuous, class URNG, bool test_constexpr, class CDF>
TEST_FUNC bool test_eval(const typename D::param_type param, CDF cdf)
{
// First check the operator with param is equivalent to the constructor param
{
D d1(param);
D d2(param);
URNG g_1{};
URNG g_2{};
for (cuda::std::size_t i = 0; i < 100; ++i)
{
auto dist_val = d1(g_1, param);
auto dist2_val = d2(g_2);
assert((dist_val == dist2_val) || (cuda::std::isnan(dist_val) && cuda::std::isnan(dist2_val)));
}
}
D dist(param);
URNG g{};
const cuda::std::size_t num_samples = 10000;
auto samples = cuda::std::make_unique<typename D::result_type[]>(num_samples);
for (cuda::std::size_t i = 0; i < num_samples; ++i)
{
samples[i] = dist(g, param);
}
// Use sort when available
cuda::std::partial_sort(samples.get(), samples.get() + num_samples, samples.get() + num_samples);
// Compute the KS statistic - specially handle discrete case
// Arnold, Taylor B., and John W. Emerson. "Nonparametric goodness-of-fit tests for discrete null distributions."
// (2011).
double d_max = 0.0;
if constexpr (continuous)
{
d_max = ks_test_statistic_continuous<D>(samples.get(), num_samples, param, cdf);
}
else
{
d_max = ks_test_statistic_discrete<D>(samples.get(), num_samples, param, cdf);
}
// Note that this critical value from the KS distribution is only valid for discrete distributions when num_samples is
// large
const double critical_value = 0.016259280113043572; // for alpha = 0.01 and n = 10000
assert(d_max < critical_value);
return true;
}
template <class D, class URNG>
TEST_FUNC constexpr bool test_eval_constexpr()
{
typename D::param_type param;
D dist(param);
URNG g{};
unused(dist(g, param));
unused(dist(g));
return true;
}
} // namespace detail
template <class D, bool continuous, class URNG, bool test_constexpr, class CDF, cuda::std::size_t N>
TEST_FUNC void constexpr test_distribution(cuda::std::array<typename D::param_type, N> params, CDF cdf)
{
for (cuda::std::size_t i = 0; i < N; ++i)
{
detail::test_eval<D, continuous, URNG, test_constexpr>(params[i], cdf);
detail::test_ctor_assign<D, URNG>(params[i]);
detail::test_copy<D, URNG>(params[i]);
detail::test_eq<D, URNG>(params[i]);
detail::test_get_param<D, URNG>(params[i]);
detail::test_min_max<D, URNG>(params[i]);
detail::test_set_param<D, URNG>(params[i]);
detail::test_types<D, URNG>(params[i]);
detail::test_param<D, URNG, typename D::param_type>(params[i]);
NV_IF_TARGET(NV_IS_HOST, ({ detail::test_io<D, URNG>(params[i]); }));
}
if constexpr (test_constexpr)
{
constexpr typename D::param_type param{};
static_assert(detail::test_eval_constexpr<D, URNG>());
static_assert(detail::test_ctor_assign<D, URNG>(param));
static_assert(detail::test_eq<D, URNG>(param));
static_assert(detail::test_get_param<D, URNG>(param));
static_assert(detail::test_min_max<D, URNG>(param));
static_assert(detail::test_set_param<D, URNG>(param));
static_assert(detail::test_types<D, URNG>(param));
static_assert(detail::test_param<D, URNG, typename D::param_type>(param));
}
}
#endif // LIBCUDACXX_TEST_SUPPORT_RANDOM_UTILITIES_TEST_DISTRIBUTION_H

View File

@@ -0,0 +1,190 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#include <cuda/std/random>
#if _CCCL_HOSTED()
# include <sstream>
#endif // _CCCL_HOSTED()
#include "test_macros.h"
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_ctor()
{
Engine e1;
Engine e2(Engine::default_seed);
assert(e1 == e2);
Engine e3(42);
assert(e3 != e2);
auto seq = cuda::std::seed_seq{};
Engine e4(seq);
Engine e5 = e4;
assert(e4 == e5);
static_assert(noexcept(Engine()));
static_assert(noexcept(Engine(42)));
return true;
}
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_copy()
{
Engine e1;
Engine e2 = e1;
assert(e1 == e2);
e1();
assert(e1 != e2);
e2 = e1;
assert(e1 == e2);
static_assert(noexcept(Engine(e1)));
static_assert(noexcept(e2 = e1));
return true;
}
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_seed()
{
Engine e1(23);
Engine e2;
e2.seed(Engine::default_seed);
assert(e1 != e2);
e1.seed(Engine::default_seed);
assert(e1 == e2);
auto seq = cuda::std::seed_seq{};
static_assert(cuda::std::is_void_v<decltype(e1.seed(seq))>);
static_assert(cuda::std::is_void_v<decltype(e1.seed())>);
static_assert(cuda::std::is_void_v<decltype(e1.seed(23))>);
static_assert(noexcept(e1.seed()));
static_assert(noexcept(e1.seed(23)));
return true;
}
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_operator()
{
Engine e1;
static_assert(cuda::std::is_same_v<decltype(e1()), typename Engine::result_type>);
e1();
Engine e2;
assert(e1 != e2);
e2();
assert(e1 == e2);
return true;
}
template <typename Engine, typename Engine::result_type value_10000>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_discard()
{
Engine e;
for (int i = 0; i < 100; ++i)
{
Engine e2;
e2.discard(i);
assert(e == e2);
e();
}
e = Engine();
e.discard(9999);
assert(e() == value_10000);
static_assert(cuda::std::is_void_v<decltype(e.discard(10))>);
static_assert(noexcept(e.discard(10)));
return true;
}
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_equality()
{
Engine e;
assert(e == e);
Engine e2;
assert(e == e2);
e();
assert(e != e2);
e = Engine(3);
e2 = Engine(3);
assert(e == e2);
e2 = Engine(4);
assert(e != e2);
static_assert(noexcept(e == e2));
static_assert(noexcept(e != e2));
return true;
}
template <typename Engine>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_min_max()
{
const auto seeds = {0, 29332, 9000};
for (auto seed : seeds)
{
Engine e(seed);
for (int i = 0; i < 100; ++i)
{
auto val = e();
assert(val <= Engine::max());
// Avoid pointless comparison of unsigned values with 0 warning
if constexpr (Engine::min() > 0)
{
assert(val >= Engine::min());
}
}
}
static_assert(Engine::min() <= Engine::max());
static_assert(noexcept(Engine::min()));
static_assert(noexcept(Engine::max()));
static_assert(cuda::std::is_same_v<decltype(Engine::min()), typename Engine::result_type>);
static_assert(cuda::std::is_same_v<decltype(Engine::max()), typename Engine::result_type>);
return true;
}
#if _CCCL_HOSTED()
template <typename Engine>
void test_save_restore()
{
Engine e0;
e0.discard(10000);
std::stringstream ss;
ss << e0;
e0.discard(10000);
Engine e1;
ss >> e1;
e1.discard(10000);
assert(e0() == e1());
}
#endif // _CCCL_HOSTED()
template <typename Engine, typename Engine::result_type value_10000>
TEST_FUNC TEST_CONSTEXPR_CXX20 bool test_engine()
{
test_ctor<Engine>();
test_seed<Engine>();
test_copy<Engine>();
test_operator<Engine>();
test_discard<Engine, value_10000>();
test_equality<Engine>();
test_min_max<Engine>();
NV_IF_TARGET(NV_IS_HOST, ({ test_save_restore<Engine>(); }));
#if TEST_STD_VER >= 2020
static_assert(test_ctor<Engine>());
static_assert(test_seed<Engine>());
static_assert(test_copy<Engine>());
static_assert(test_operator<Engine>());
static_assert(test_discard<Engine, value_10000>());
static_assert(test_equality<Engine>());
static_assert(test_min_max<Engine>());
#endif
return true;
}