// SPDX-FileCopyrightText: Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. All rights reserved. // SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception #include "insert_nested_NVTX_range_guard.h" #include #include #include #include #include #include "catch2_test_device_reduce.cuh" #include "catch2_test_launch_helper.h" #include // %PARAM% TEST_LAUNCH lid 0:1:2 DECLARE_LAUNCH_WRAPPER(cub::DeviceFind::FindIf, find_if); DECLARE_LAUNCH_WRAPPER(cub::DeviceFind::LowerBound, lower_bound); DECLARE_LAUNCH_WRAPPER(cub::DeviceFind::UpperBound, upper_bound); // List of types to test using custom_t = c2h::custom_type_t; using value_types = c2h::type_list; using offset_types = c2h::type_list; enum class gen_data_t { GEN_TYPE_RANDOM, /// Uniform random data generation GEN_TYPE_CONST /// Constant value as input data }; template auto compute_find_if_reference(InputIt first, InputIt last, Predicate predicate) -> OffsetT { const auto it = std::find_if(first, last, predicate); // not thrust::find_if because it will rely on cub::FindIf return static_cast(std::distance(first, it)); } C2H_TEST("Device find_if works", "[device][find_if]", value_types, offset_types) { using input_t = c2h::get<0, TestType>; using offset_t = c2h::get<1, TestType>; constexpr offset_t min_items = 1; constexpr offset_t max_items = 10'000'000; // 10M items for reasonable test time // TODO(bgruber): test a value larger than UINT32 // Generate the input sizes to test for const offset_t num_items = GENERATE_COPY(offset_t{1}, offset_t{100}, offset_t{5324}, max_items, take(5, random(min_items, max_items))); const gen_data_t data_gen_mode = GENERATE(gen_data_t::GEN_TYPE_RANDOM, gen_data_t::GEN_TYPE_CONST); const bool value_exists = GENERATE(false, true); CAPTURE(c2h::type_name(), c2h::type_name(), num_items, data_gen_mode, value_exists); constexpr bool is_custom_t = cuda::std::is_same_v; if constexpr (is_custom_t) { if (data_gen_mode == gen_data_t::GEN_TYPE_RANDOM && !value_exists) { // min/max handling is not implemented for c2h::gen and custom_t, so we cannot pick a value that does not exist // in the input sequence return; } } // Generate input data c2h::device_vector in_items(num_items, thrust::default_init); if (data_gen_mode == gen_data_t::GEN_TYPE_RANDOM) { if constexpr (is_custom_t) { c2h::gen(C2H_SEED(1), in_items); } else { // omit the largest value from the random values so we have a value to that does not occur c2h::gen(C2H_SEED(1), in_items, input_t{0}, static_cast(::cuda::std::numeric_limits::max() - 1)); } } else { // fill with 1s input_t default_constant; init_default_constant(default_constant, 1); thrust::fill(c2h::device_policy, in_items.begin(), in_items.end(), default_constant); } auto d_in_it = thrust::raw_pointer_cast(in_items.data()); using predice_t = cuda::equal_to_value; input_t val_to_find{}; // Generate test cases for both "found" and "not found" scenarios if (value_exists) { // take a random value from the input sequence val_to_find = in_items[GENERATE_COPY(take(1, random(offset_t{0}, num_items - 1)))]; } else { // max value is neither in the random input and nor in the constant val_to_find = ::cuda::std::numeric_limits::max(); } auto predicate = predice_t{val_to_find}; SECTION("Generic find if case") { // Prepare verification data c2h::host_vector host_items(in_items); const auto expected_result = compute_find_if_reference(host_items.begin(), host_items.end(), predicate); // Run test c2h::device_vector out_result(1, thrust::no_init); find_if(d_in_it, thrust::raw_pointer_cast(out_result.data()), predicate, num_items); REQUIRE(expected_result == out_result[0]); } SECTION("find_if works with thrust contiguous iterator") { // Prepare verification data c2h::host_vector host_items(in_items); const auto expected = compute_find_if_reference(host_items.begin(), host_items.end(), predicate); // Run test c2h::device_vector out_result(1); find_if(in_items.begin(), out_result.begin(), predicate, num_items); REQUIRE(expected == out_result[0]); } SECTION("find_if works for unaligned input") { for (int offset = 1; offset < 4; ++offset) { if (num_items > offset) { // Prepare verification data c2h::host_vector host_items(in_items); const auto expected = compute_find_if_reference(host_items.begin() + offset, host_items.end(), predicate); // Run test c2h::device_vector out_result(1, thrust::no_init); find_if(d_in_it + offset, thrust::raw_pointer_cast(out_result.data()), predicate, num_items - offset); REQUIRE(expected == out_result[0]); } } } } C2H_TEST("Device find_if works with non primitive iterator", "[device][find_if]") { using input_t = int32_t; using offset_t = int32_t; constexpr offset_t min_items = 1; constexpr offset_t max_items = 10000000; // 10M items for reasonable test time input_t val_to_find = static_cast(GENERATE_COPY(take(1, random(min_items, max_items)))); // Generate the input sizes to test for const offset_t num_items = GENERATE_COPY( take(1, random(min_items, max_items)), values({ min_items, max_items, })); CAPTURE(num_items, val_to_find); const auto expected_if_found = cuda::std::min(static_cast(val_to_find), num_items); // counting_iterator input auto c_it = cuda::make_counting_iterator(input_t{0}); { c2h::device_vector out_result(1, thrust::no_init); auto predicate = cuda::equal_to_value{val_to_find}; find_if(c_it, thrust::raw_pointer_cast(out_result.data()), predicate, num_items); REQUIRE(expected_if_found == out_result[0]); } { // transform_iterator of counting_iterator input and thrust device_ptr output auto t_it = cuda::make_transform_iterator(c_it, ::cuda::std::negate{}); c2h::device_vector out_result(1, thrust::no_init); auto predicate = cuda::equal_to_value{-val_to_find}; find_if(t_it, out_result.data(), predicate, num_items); REQUIRE(expected_if_found == out_result[0]); } { // counting_iterator input and transform_output_iterator output c2h::device_vector out_result(1, thrust::no_init); auto predicate = cuda::equal_to_value{val_to_find}; auto out_it = cuda::make_transform_output_iterator(out_result.begin(), ::cuda::std::negate{}); find_if(c_it, out_it, predicate, num_items); REQUIRE(-expected_if_found == out_result[0]); } } struct NotDefaultConstructible { int value_; __host__ __device__ constexpr explicit NotDefaultConstructible(int value) : value_(value) {} __host__ __device__ friend constexpr bool operator==(const NotDefaultConstructible& lhs, const NotDefaultConstructible& rhs) { return lhs.value_ == rhs.value_; } __host__ __device__ friend constexpr bool operator!=(const NotDefaultConstructible& lhs, const NotDefaultConstructible& rhs) { return lhs.value_ != rhs.value_; } __host__ __device__ constexpr operator int() const noexcept { return value_; } }; struct index_to_value { __host__ __device__ NotDefaultConstructible operator()(int i) { return NotDefaultConstructible{static_cast(i)}; } }; static_assert(!cuda::std::is_default_constructible_v, "NotDefaultConstructible should not be default constructible"); C2H_TEST("Device find_if works with non default constructible types", "[device][find_if]") { using input_t = NotDefaultConstructible; using offset_t = int; constexpr offset_t min_items = 1; constexpr offset_t max_items = 10'000; // 10k items for reasonable test time // Generate the input sizes to test for const offset_t num_items = GENERATE_COPY( take(1, random(min_items, max_items)), values({ min_items, max_items, })); const auto val_to_find = static_cast(num_items - 1); CAPTURE(num_items, val_to_find); // raw device iterator to some device vector so that vectorized path is taken c2h::device_vector d_vec(num_items, NotDefaultConstructible(0)); // fill with arbitrary values dont use c2h gen because NotDefaultConstructible is not default constructible thrust::tabulate(c2h::device_policy, d_vec.begin(), d_vec.end(), index_to_value{}); auto it = thrust::raw_pointer_cast(d_vec.data()); c2h::device_vector out_result(1, thrust::no_init); auto predicate = cuda::equal_to_value{NotDefaultConstructible{val_to_find}}; find_if(it, thrust::raw_pointer_cast(out_result.data()), predicate, num_items); REQUIRE(val_to_find == out_result[0]); } // LowerBound / UpperBound tests (formerly in catch2_test_device_binary_search.cu) using binary_search_types = c2h::type_list; struct std_lower_bound_t { template RangeIteratorT operator()(RangeIteratorT first, RangeIteratorT last, const T& value, CompareOpT comp) const { return std::lower_bound(first, last, value, comp); } } std_lower_bound; struct std_upper_bound_t { template RangeIteratorT operator()(RangeIteratorT first, RangeIteratorT last, const T& value, CompareOpT comp) const { return std::upper_bound(first, last, value, comp); } } std_upper_bound; template > void test_vectorized(Variant variant, HostVariant host_variant, std::size_t num_items = 7492, CompareOp compare_op = {}) { c2h::device_vector target_values_d(num_items / 100, thrust::default_init); c2h::gen(C2H_SEED(1), target_values_d); c2h::device_vector values_d(num_items + target_values_d.size(), thrust::default_init); c2h::gen(C2H_SEED(1), values_d); thrust::copy(c2h::device_policy, target_values_d.begin(), target_values_d.end(), values_d.begin()); thrust::sort(c2h::device_policy, values_d.begin(), values_d.end(), compare_op); using Result = std::ptrdiff_t; c2h::device_vector offsets_d(target_values_d.size(), thrust::default_init); variant(thrust::raw_pointer_cast(values_d.data()), num_items, thrust::raw_pointer_cast(target_values_d.data()), target_values_d.size(), thrust::raw_pointer_cast(offsets_d.data()), compare_op); c2h::host_vector target_values_h = target_values_d; c2h::host_vector values_h = values_d; c2h::host_vector offsets_h = offsets_d; c2h::host_vector offsets_ref(offsets_h.size(), thrust::default_init); for (auto i = 0u; i < target_values_h.size(); ++i) { offsets_ref[i] = host_variant(values_h.data(), values_h.data() + num_items, target_values_h[i], compare_op) - values_h.data(); } CHECK(offsets_ref == offsets_h); } C2H_TEST("DeviceFind::LowerBound works", "[find][device][binary-search]", binary_search_types) { using value_type = c2h::get<0, TestType>; test_vectorized(lower_bound, std_lower_bound); } C2H_TEST("DeviceFind::UpperBound works", "[find][device][binary-search]", binary_search_types) { using value_type = c2h::get<0, TestType>; test_vectorized(upper_bound, std_upper_bound); } // this test exceeds 4GiB of memory and the range of 32-bit integers C2H_TEST("DeviceFind::LowerBound really large input", "[find][device][binary-search][skip-cs-rangecheck][skip-cs-initcheck][skip-cs-synccheck]") { try { using value_type = char; const auto size = std::int64_t{1} << GENERATE(30, 31, 32, 33); test_vectorized(lower_bound, std_lower_bound, size); } catch (const std::bad_alloc&) { // allocation failure is not a test failure, so we can run tests on smaller GPUs SUCCEED("allocation failure is not a test failure"); } } // this test exceeds 4GiB of memory and the range of 32-bit integers C2H_TEST("DeviceFind::UpperBound really large input", "[find][device][binary-search][skip-cs-rangecheck][skip-cs-initcheck][skip-cs-synccheck]") { try { using value_type = char; const auto size = std::int64_t{1} << GENERATE(30, 31, 32, 33); test_vectorized(upper_bound, std_upper_bound, size); } catch (const std::bad_alloc&) { // allocation failure is not a test failure, so we can run tests on smaller GPUs SUCCEED("allocation failure is not a test failure"); } }