[CCCL] Add missing CCCL components: c2h, nvbench_helper, cmake, cudax, AGENTS.md
Added 863 files from NVIDIA/cccl sparse checkout: - c2h/ (27 files): Catch2 test helpers — generators, validators, runner - nvbench_helper/ (10 files): Benchmark harness utilities - cmake/ (29 files): CMake presets and build helpers - cudax/ (794 files): Experimental CUDA extensions - AGENTS.md: NVIDIA's official AI agent instructions for CCCL - CMakePresets.json: Standardized build configurations - cccl-version.json: Version tracking Also added CCCL_ASSET_MAP.md mapping all 4295 CCCL files to competition value and PRD items. cccl_upstream now covers 100% of competition-critical assets: - 27 tuning headers (SM80/90/100 benchmark data) - 32 dispatch headers (algorithm implementations) - 60 Thrust examples (correctness verification) - 217 CUB Catch2 tests (regression matrix) - 153 CUB benchmarks (parameter space search) - 18 CUB examples (API verification) - 27 test helpers + benchmark harness - 794 cudax experimental extensions
This commit is contained in:
157
cccl_upstream/nvbench_helper/test/gen_entropy.cu
Normal file
157
cccl_upstream/nvbench_helper/test/gen_entropy.cu
Normal file
@@ -0,0 +1,157 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <cub/device/device_run_length_encode.cuh>
|
||||
|
||||
#include <thrust/count.h>
|
||||
#include <thrust/execution_policy.h>
|
||||
#include <thrust/host_vector.h>
|
||||
#include <thrust/sort.h>
|
||||
#include <thrust/transform.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include <catch2/catch_template_test_macros.hpp>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
template <class T>
|
||||
double get_expected_entropy(bit_entropy in_entropy)
|
||||
{
|
||||
if (in_entropy == bit_entropy::_0_000)
|
||||
{
|
||||
return 0.0;
|
||||
}
|
||||
|
||||
if (in_entropy == bit_entropy::_1_000)
|
||||
{
|
||||
return sizeof(T) * 8;
|
||||
}
|
||||
|
||||
const int samples = static_cast<int>(in_entropy) + 1;
|
||||
const double p1 = std::pow(0.5, samples);
|
||||
const double p2 = 1 - p1;
|
||||
const double entropy = (-p1 * std::log2(p1)) + (-p2 * std::log2(p2));
|
||||
return sizeof(T) * 8 * entropy;
|
||||
}
|
||||
|
||||
template <class T>
|
||||
double compute_actual_entropy(thrust::device_vector<T> in)
|
||||
{
|
||||
const int n = static_cast<int>(in.size());
|
||||
|
||||
#if THRUST_DEVICE_SYSTEM == THRUST_DEVICE_SYSTEM_CUDA
|
||||
thrust::device_vector<T> unique(n);
|
||||
thrust::device_vector<int> counts(n);
|
||||
thrust::device_vector<int> num_runs(1);
|
||||
thrust::sort(in.begin(), in.end(), less_t{});
|
||||
|
||||
// RLE
|
||||
void* d_temp_storage = nullptr;
|
||||
std::size_t temp_storage_bytes = 0;
|
||||
|
||||
T* d_in = thrust::raw_pointer_cast(in.data());
|
||||
T* d_unique_out = thrust::raw_pointer_cast(unique.data());
|
||||
int* d_counts_out = thrust::raw_pointer_cast(counts.data());
|
||||
int* d_num_runs_out = thrust::raw_pointer_cast(num_runs.data());
|
||||
|
||||
cub::DeviceRunLengthEncode::Encode(
|
||||
d_temp_storage, temp_storage_bytes, d_in, d_unique_out, d_counts_out, d_num_runs_out, n);
|
||||
|
||||
thrust::device_vector<std::uint8_t> temp_storage(temp_storage_bytes);
|
||||
d_temp_storage = thrust::raw_pointer_cast(temp_storage.data());
|
||||
|
||||
cub::DeviceRunLengthEncode::Encode(
|
||||
d_temp_storage, temp_storage_bytes, d_in, d_unique_out, d_counts_out, d_num_runs_out, n);
|
||||
|
||||
thrust::host_vector<int> h_counts = counts;
|
||||
thrust::host_vector<int> h_num_runs = num_runs;
|
||||
#else
|
||||
std::vector<T> h_in(in.begin(), in.end());
|
||||
std::sort(h_in.begin(), h_in.end(), less_t{});
|
||||
thrust::host_vector<int> h_counts;
|
||||
T prev = h_in[0];
|
||||
int length = 1;
|
||||
|
||||
for (std::size_t i = 1; i < h_in.size(); i++)
|
||||
{
|
||||
const T next = h_in[i];
|
||||
if (next == prev)
|
||||
{
|
||||
length++;
|
||||
}
|
||||
else
|
||||
{
|
||||
h_counts.push_back(length);
|
||||
prev = next;
|
||||
length = 1;
|
||||
}
|
||||
}
|
||||
h_counts.push_back(length);
|
||||
|
||||
thrust::host_vector<int> h_num_runs(1, h_counts.size());
|
||||
#endif
|
||||
|
||||
// normalize counts
|
||||
thrust::host_vector<double> ps(h_num_runs[0]);
|
||||
for (std::size_t i = 0; i < ps.size(); i++)
|
||||
{
|
||||
ps[i] = static_cast<double>(h_counts[i]) / n;
|
||||
}
|
||||
|
||||
double entropy = 0.0;
|
||||
|
||||
if (ps.size())
|
||||
{
|
||||
for (double p : ps)
|
||||
{
|
||||
entropy -= p * std::log2(p);
|
||||
}
|
||||
}
|
||||
|
||||
return entropy;
|
||||
}
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce data with given entropy", "[gen]", fundamental_types)
|
||||
{
|
||||
constexpr int num_entropy_levels = 6;
|
||||
std::array<bit_entropy, num_entropy_levels> entropy_levels{
|
||||
bit_entropy::_0_000,
|
||||
bit_entropy::_0_201,
|
||||
bit_entropy::_0_337,
|
||||
bit_entropy::_0_544,
|
||||
bit_entropy::_0_811,
|
||||
bit_entropy::_1_000};
|
||||
|
||||
std::vector<double> entropy(num_entropy_levels);
|
||||
std::transform(entropy_levels.cbegin(), entropy_levels.cend(), entropy.begin(), [](bit_entropy entropy) {
|
||||
const thrust::device_vector<TestType> data = generate(1 << 24, entropy);
|
||||
return compute_actual_entropy(data);
|
||||
});
|
||||
|
||||
REQUIRE(std::is_sorted(entropy.begin(), entropy.end(), less_t{}));
|
||||
REQUIRE(std::unique(entropy.begin(), entropy.end()) == entropy.end());
|
||||
}
|
||||
|
||||
TEST_CASE("Generators support bool", "[gen]")
|
||||
{
|
||||
constexpr int num_entropy_levels = 6;
|
||||
std::array<bit_entropy, num_entropy_levels> entropy_levels{
|
||||
bit_entropy::_0_000,
|
||||
bit_entropy::_0_201,
|
||||
bit_entropy::_0_337,
|
||||
bit_entropy::_0_544,
|
||||
bit_entropy::_0_811,
|
||||
bit_entropy::_1_000};
|
||||
|
||||
std::vector<std::size_t> number_of_set(num_entropy_levels);
|
||||
std::transform(entropy_levels.cbegin(), entropy_levels.cend(), number_of_set.begin(), [](bit_entropy entropy) {
|
||||
const thrust::device_vector<bool> data = generate(1 << 24, entropy);
|
||||
return thrust::count(data.begin(), data.end(), true);
|
||||
});
|
||||
|
||||
REQUIRE(std::is_sorted(number_of_set.begin(), number_of_set.end()));
|
||||
REQUIRE(std::unique(number_of_set.begin(), number_of_set.end()) == number_of_set.end());
|
||||
}
|
||||
@@ -0,0 +1,49 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/host_vector.h>
|
||||
|
||||
#include <algorithm>
|
||||
#include <cmath>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include <boost/math/statistics/anderson_darling.hpp>
|
||||
#include <boost/math/statistics/univariate_statistics.hpp>
|
||||
#include <catch2/catch_template_test_macros.hpp>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
|
||||
bool is_normal(thrust::host_vector<double> data)
|
||||
{
|
||||
std::sort(data.begin(), data.end());
|
||||
const double A2 = boost::math::statistics::anderson_darling_normality_statistic(data);
|
||||
return A2 / data.size() < 0.05;
|
||||
}
|
||||
|
||||
using types = nvbench::type_list<uint32_t, uint64_t>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce power law distributed data", "[gen][power-law]", types)
|
||||
{
|
||||
const std::size_t elements = 1 << 28;
|
||||
const std::size_t segments = 4 * 1024;
|
||||
const thrust::device_vector<TestType> d_segment_offsets = generate.power_law.segment_offsets(elements, segments);
|
||||
REQUIRE(d_segment_offsets.size() == segments + 1);
|
||||
|
||||
std::size_t actual_elements = 0;
|
||||
thrust::host_vector<double> log_sizes(segments);
|
||||
const thrust::host_vector<TestType> h_segment_offsets = d_segment_offsets;
|
||||
for (std::size_t i = 0; i < segments; ++i)
|
||||
{
|
||||
const TestType begin = h_segment_offsets[i];
|
||||
const TestType end = h_segment_offsets[i + 1];
|
||||
REQUIRE(begin <= end);
|
||||
|
||||
const std::size_t size = end - begin;
|
||||
actual_elements += size;
|
||||
log_sizes[i] = std::log(size);
|
||||
}
|
||||
|
||||
REQUIRE(actual_elements == elements);
|
||||
REQUIRE(is_normal(std::move(log_sizes)));
|
||||
}
|
||||
37
cccl_upstream/nvbench_helper/test/gen_range.cu
Normal file
37
cccl_upstream/nvbench_helper/test/gen_range.cu
Normal file
@@ -0,0 +1,37 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/extrema.h>
|
||||
|
||||
#include <limits>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include <catch2/catch_template_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_all.hpp>
|
||||
|
||||
using types =
|
||||
nvbench::type_list<int8_t,
|
||||
int16_t,
|
||||
int32_t,
|
||||
int64_t,
|
||||
#if _CCCL_HAS_INT128()
|
||||
int128_t,
|
||||
#endif
|
||||
float,
|
||||
double>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce data within specified range", "[gen]", types)
|
||||
{
|
||||
const auto min = static_cast<TestType>(GENERATE_COPY(take(3, random(-124, 0))));
|
||||
const auto max = static_cast<TestType>(GENERATE_COPY(take(3, random(0, 124))));
|
||||
|
||||
const thrust::device_vector<TestType> data = generate(1 << 16, bit_entropy::_1_000, min, max);
|
||||
|
||||
const TestType min_element = *thrust::min_element(data.begin(), data.end());
|
||||
const TestType max_element = *thrust::max_element(data.begin(), data.end());
|
||||
|
||||
REQUIRE(min_element >= min);
|
||||
REQUIRE(max_element <= max);
|
||||
}
|
||||
34
cccl_upstream/nvbench_helper/test/gen_seed.cu
Normal file
34
cccl_upstream/nvbench_helper/test/gen_seed.cu
Normal file
@@ -0,0 +1,34 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/equal.h>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include <catch2/catch_template_test_macros.hpp>
|
||||
|
||||
using types = nvbench::type_list<
|
||||
bool,
|
||||
int8_t,
|
||||
int16_t,
|
||||
int32_t,
|
||||
int64_t,
|
||||
#if _CCCL_HAS_INT128()
|
||||
int128_t,
|
||||
#endif
|
||||
float,
|
||||
double,
|
||||
complex32,
|
||||
complex64>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generator seeds the data", "[gen]", types)
|
||||
{
|
||||
auto generator = generate(1 << 24, bit_entropy::_0_811);
|
||||
|
||||
const thrust::device_vector<TestType> vec_1 = generator;
|
||||
const thrust::device_vector<TestType> vec_2 = generator;
|
||||
|
||||
REQUIRE(vec_1.size() == vec_2.size());
|
||||
REQUIRE_FALSE(thrust::equal(vec_1.begin(), vec_1.end(), vec_2.begin()));
|
||||
}
|
||||
204
cccl_upstream/nvbench_helper/test/gen_uniform_distribution.cu
Normal file
204
cccl_upstream/nvbench_helper/test/gen_uniform_distribution.cu
Normal file
@@ -0,0 +1,204 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2011-2023, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
#include <thrust/count.h>
|
||||
#include <thrust/device_vector.h>
|
||||
#include <thrust/host_vector.h>
|
||||
|
||||
#include <cmath>
|
||||
#include <limits>
|
||||
#include <map>
|
||||
|
||||
#include <nvbench_helper.cuh>
|
||||
|
||||
#include <boost/math/distributions/chi_squared.hpp>
|
||||
#include <catch2/catch_template_test_macros.hpp>
|
||||
#include <catch2/catch_test_macros.hpp>
|
||||
#include <catch2/generators/catch_generators_all.hpp>
|
||||
|
||||
template <typename T>
|
||||
bool is_uniform(thrust::host_vector<T> data, T min, T max)
|
||||
{
|
||||
const double value_range = static_cast<double>(max) - min;
|
||||
const bool exact_binning = value_range < (1 << 20);
|
||||
const int number_of_bins = exact_binning ? static_cast<int>(max - min + 1) : static_cast<int>(std::sqrt(data.size()));
|
||||
thrust::host_vector<int> bins(number_of_bins, 0);
|
||||
|
||||
const double interval = value_range / static_cast<double>(number_of_bins);
|
||||
const double expected_count = static_cast<double>(data.size()) / number_of_bins;
|
||||
|
||||
for (T val : data)
|
||||
{
|
||||
int bin_index = exact_binning ? val - min : (val - static_cast<double>(min)) / interval;
|
||||
|
||||
if (bin_index >= 0 && bin_index < number_of_bins)
|
||||
{
|
||||
bins[bin_index]++;
|
||||
}
|
||||
}
|
||||
|
||||
double chi_square = 0.0;
|
||||
for (const auto& count : bins)
|
||||
{
|
||||
chi_square += std::pow(count - expected_count, 2) / expected_count;
|
||||
}
|
||||
|
||||
boost::math::chi_squared_distribution<double> chi_squared_dist(number_of_bins - 1);
|
||||
|
||||
const double confidence = 0.95;
|
||||
const double critical_value = boost::math::quantile(chi_squared_dist, confidence);
|
||||
|
||||
return chi_square <= critical_value;
|
||||
}
|
||||
|
||||
using types =
|
||||
nvbench::type_list<int8_t,
|
||||
int16_t,
|
||||
int32_t,
|
||||
int64_t,
|
||||
#if _CCCL_HAS_INT128()
|
||||
int128_t,
|
||||
#endif
|
||||
float,
|
||||
double>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce uniformly distributed data", "[gen][uniform]", types)
|
||||
{
|
||||
const std::size_t elements = 1 << GENERATE_COPY(16, 20, 24, 28);
|
||||
const TestType min = ::cuda::std::numeric_limits<TestType>::min();
|
||||
const TestType max = ::cuda::std::numeric_limits<TestType>::max();
|
||||
|
||||
const thrust::device_vector<TestType> data = generate(elements, bit_entropy::_1_000, min, max);
|
||||
|
||||
REQUIRE(is_uniform<TestType>(data, min, max));
|
||||
}
|
||||
|
||||
struct complex_to_real_t
|
||||
{
|
||||
template <typename T>
|
||||
__host__ __device__ T operator()(const cuda::std::complex<T>& c) const
|
||||
{
|
||||
return c.real();
|
||||
}
|
||||
};
|
||||
|
||||
struct complex_to_imag_t
|
||||
{
|
||||
template <typename T>
|
||||
__host__ __device__ T operator()(const cuda::std::complex<T>& c) const
|
||||
{
|
||||
return c.imag();
|
||||
}
|
||||
};
|
||||
|
||||
using complex_value_types = nvbench::type_list<float, double>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce uniformly distributed complex", "[gen]", complex_value_types)
|
||||
{
|
||||
using value_type = TestType;
|
||||
const auto min = ::cuda::std::numeric_limits<value_type>::min();
|
||||
const auto max = ::cuda::std::numeric_limits<value_type>::max();
|
||||
|
||||
const thrust::device_vector<cuda::std::complex<value_type>> data = generate(1 << 16);
|
||||
|
||||
thrust::device_vector<value_type> component(data.size());
|
||||
thrust::transform(data.begin(), data.end(), component.begin(), complex_to_real_t());
|
||||
REQUIRE(is_uniform<value_type>(component, min, max));
|
||||
|
||||
thrust::transform(data.begin(), data.end(), component.begin(), complex_to_imag_t());
|
||||
REQUIRE(is_uniform<value_type>(component, min, max));
|
||||
}
|
||||
|
||||
TEST_CASE("Generators produce uniformly distributed bools", "[gen]")
|
||||
{
|
||||
const thrust::device_vector<bool> data = generate(1 << 24, bit_entropy::_0_544);
|
||||
|
||||
const std::size_t falses = thrust::count(data.begin(), data.end(), false);
|
||||
const std::size_t trues = thrust::count(data.begin(), data.end(), true);
|
||||
|
||||
REQUIRE(falses > 0);
|
||||
REQUIRE(trues > 0);
|
||||
REQUIRE(falses + trues == data.size());
|
||||
|
||||
const double ratio = static_cast<double>(falses) / trues;
|
||||
REQUIRE(ratio > 0.7);
|
||||
}
|
||||
|
||||
using offsets = nvbench::type_list<uint32_t, uint64_t>;
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce uniformly distributed offsets", "[gen]", offsets)
|
||||
{
|
||||
const std::size_t min_segment_size = 1;
|
||||
const std::size_t max_segment_size = 256;
|
||||
const std::size_t elements = 1 << GENERATE_COPY(16, 20, 24, 28);
|
||||
const thrust::device_vector<TestType> d_segments =
|
||||
generate.uniform.segment_offsets(elements, min_segment_size, max_segment_size);
|
||||
const thrust::host_vector<TestType> h_segments = d_segments;
|
||||
const std::size_t num_segments = h_segments.size() - 1;
|
||||
|
||||
std::size_t actual_elements = 0;
|
||||
thrust::host_vector<int> segment_sizes(num_segments);
|
||||
for (std::size_t sid = 0; sid < num_segments; sid++)
|
||||
{
|
||||
const TestType begin = h_segments[sid];
|
||||
const TestType end = h_segments[sid + 1];
|
||||
REQUIRE(begin <= end);
|
||||
|
||||
const TestType size = end - begin;
|
||||
REQUIRE(size >= min_segment_size);
|
||||
REQUIRE(size <= max_segment_size);
|
||||
|
||||
segment_sizes[sid] = size;
|
||||
actual_elements += size;
|
||||
}
|
||||
|
||||
REQUIRE(actual_elements == elements);
|
||||
REQUIRE(is_uniform<int>(std::move(segment_sizes), min_segment_size, max_segment_size));
|
||||
}
|
||||
|
||||
TEMPLATE_LIST_TEST_CASE("Generators produce uniformly distributed key segments", "[gen]", types)
|
||||
try
|
||||
{
|
||||
const std::size_t min_segment_size = 1;
|
||||
const std::size_t max_segment_size = 128;
|
||||
const std::size_t elements = 1 << GENERATE_COPY(16, 20, 24, 28);
|
||||
const thrust::device_vector<TestType> d_keys =
|
||||
generate.uniform.key_segments(elements, min_segment_size, max_segment_size);
|
||||
REQUIRE(d_keys.size() == elements);
|
||||
|
||||
const thrust::host_vector<TestType> h_keys = d_keys;
|
||||
|
||||
thrust::host_vector<std::size_t> segment_sizes;
|
||||
|
||||
TestType prev = h_keys[0];
|
||||
std::size_t length = 1;
|
||||
|
||||
for (std::size_t kid = 1; kid < elements; kid++)
|
||||
{
|
||||
TestType next = h_keys[kid];
|
||||
|
||||
if (next == prev)
|
||||
{
|
||||
length++;
|
||||
}
|
||||
else
|
||||
{
|
||||
REQUIRE(length >= min_segment_size);
|
||||
REQUIRE(length <= max_segment_size);
|
||||
|
||||
segment_sizes.push_back(length);
|
||||
|
||||
prev = next;
|
||||
length = 1;
|
||||
}
|
||||
}
|
||||
REQUIRE(length >= min_segment_size);
|
||||
REQUIRE(length <= max_segment_size);
|
||||
segment_sizes.push_back(length);
|
||||
|
||||
REQUIRE(is_uniform(std::move(segment_sizes), min_segment_size, max_segment_size));
|
||||
}
|
||||
catch (std::bad_alloc&)
|
||||
{
|
||||
// Skip test on OOM.
|
||||
}
|
||||
Reference in New Issue
Block a user