#include #include #include #include // std::optional #include #include #include #include "algorithm_execution.h" #include "build_result_caching.h" #include "test_util.h" #include #include #include using BuildResultT = cccl_device_segmented_reduce_build_result_t; struct segmented_reduce_cleanup { CUresult operator()(BuildResultT* build_data) const noexcept { return cccl_device_segmented_reduce_cleanup(build_data); } }; using segmented_reduce_deleter = BuildResultDeleter; using segmented_reduce_build_cache_t = build_cache_t>; template auto& get_cache() { return fixture::get_or_create().get_value(); } struct segmented_reduce_build { CUresult operator()( BuildResultT* build_ptr, cccl_iterator_t input, cccl_iterator_t output, uint64_t, cccl_iterator_t start_offsets, cccl_iterator_t end_offsets, cccl_op_t op, cccl_value_t init, int cc_major, int cc_minor, const char* cub_path, const char* thrust_path, const char* libcudacxx_path, const char* ctk_path) const noexcept { return cccl_device_segmented_reduce_build( build_ptr, input, output, start_offsets, end_offsets, op, init, cc_major, cc_minor, cub_path, thrust_path, libcudacxx_path, ctk_path); } static constexpr bool should_check_sass(int) { return false; } }; struct segmented_reduce_run { template CUresult operator()( cccl_device_segmented_reduce_build_result_t build, void* d_temp_storage, size_t* temp_storage_bytes, cccl_iterator_t d_in, cccl_iterator_t d_out, uint64_t num_segments, cccl_iterator_t start_offset, cccl_iterator_t end_offset, cccl_op_t op, cccl_value_t init, CUstream stream) const noexcept { return cccl_device_segmented_reduce( build, d_temp_storage, temp_storage_bytes, d_in, d_out, num_segments, start_offset, end_offset, op, init, #ifndef CCCL_C_PARALLEL_V2 0, // v1 only: guaranteed_max_segment_size #endif stream); } }; #ifndef CCCL_C_PARALLEL_V2 // v1-only: variant that passes a compile-time guaranteed_max_segment_size to // exercise different dispatch policies. v2's cccl_device_segmented_reduce // signature doesn't accept guaranteed_max_segment_size. template struct segmented_reduce_run_guaranteed { CUresult operator()( cccl_device_segmented_reduce_build_result_t build, void* d_temp_storage, size_t* temp_storage_bytes, cccl_iterator_t d_in, cccl_iterator_t d_out, uint64_t num_segments, cccl_iterator_t start_offset, cccl_iterator_t end_offset, cccl_op_t op, cccl_value_t init, CUstream stream) const noexcept { return cccl_device_segmented_reduce( build, d_temp_storage, temp_storage_bytes, d_in, d_out, num_segments, start_offset, end_offset, op, init, GuaranteedMaxSegmentSize, stream); } }; #endif // CCCL_C_PARALLEL_V2 template void segmented_reduce( cccl_iterator_t input, cccl_iterator_t output, uint64_t num_segments, cccl_iterator_t start_offsets, cccl_iterator_t end_offsets, cccl_op_t op, cccl_value_t init, std::optional& cache, const std::optional& lookup_key) { AlgorithmExecute( cache, lookup_key, input, output, num_segments, start_offsets, end_offsets, op, init); } #ifndef CCCL_C_PARALLEL_V2 template void segmented_reduce_guaranteed( cccl_iterator_t input, cccl_iterator_t output, uint64_t num_segments, cccl_iterator_t start_offsets, cccl_iterator_t end_offsets, cccl_op_t op, cccl_value_t init, std::optional& cache, const std::optional& lookup_key) { AlgorithmExecute, BuildCache, KeyT>(cache, lookup_key, input, output, num_segments, start_offsets, end_offsets, op, init); } #endif // CCCL_C_PARALLEL_V2 // ============== // Test section // ============== struct SegmentedReduce_SumOverRows_Fixture_Tag; C2H_TEST_LIST("segmented_reduce can sum over rows of matrix with integral type", "[segmented_reduce]", std::int32_t, std::int64_t, std::uint32_t, std::uint64_t) { // generate 4 choices for n_rows: 0, 13 and 2 random samples from [1024, 4096) const std::size_t n_rows = GENERATE(0, 13, take(2, random(1 << 10, 1 << 12))); // generate 4 choices for number of columns const std::size_t n_cols = GENERATE(0, 12, take(2, random(1 << 10, 1 << 12))); const std::size_t n_elems = n_rows * n_cols; const std::size_t segment_size = n_cols; const std::vector host_input = generate(n_elems); std::vector host_output(n_rows, 0); REQUIRE(host_input.size() == n_cols * n_rows); REQUIRE(host_output.size() == n_rows); pointer_t input_ptr(host_input); // copy from host to device pointer_t output_ptr(host_output); // copy from host to device using SizeT = unsigned long long; static constexpr std::string_view index_ty_name = "unsigned long long"; struct row_offset_iterator_state_t { SizeT linear_id; SizeT segment_size; }; static constexpr std::string_view offset_iterator_state_name = "row_offset_iterator_state_t"; static constexpr std::string_view advance_offset_method_name = "advance_offset_it"; static constexpr std::string_view deref_offset_method_name = "dereference_offset_it"; const auto& [offset_iterator_state_src, offset_iterator_advance_src, offset_iterator_deref_src] = make_step_counting_iterator_sources( index_ty_name, offset_iterator_state_name, advance_offset_method_name, deref_offset_method_name); iterator_t start_offset_it = make_iterator( {offset_iterator_state_name, offset_iterator_state_src}, {advance_offset_method_name, offset_iterator_advance_src}, {deref_offset_method_name, offset_iterator_deref_src}); start_offset_it.state.linear_id = 0; start_offset_it.state.segment_size = segment_size; // a copy of offset iterator, so no need to define advance/dereference bodies, // just reused those defined above iterator_t end_offset_it = make_iterator( {offset_iterator_state_name, ""}, {advance_offset_method_name, ""}, {deref_offset_method_name, ""}); end_offset_it.state.linear_id = 1; end_offset_it.state.segment_size = segment_size; operation_t op = make_operation("op", get_reduce_op(get_type_info().type)); value_t init{0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce(input_ptr, output_ptr, n_rows, start_offset_it, end_offset_it, op, init, build_cache, test_key); auto host_input_it = host_input.begin(); auto host_output_it = host_output.begin(); for (std::size_t i = 0; i < n_rows; ++i) { std::size_t row_offset = i * segment_size; host_output_it[i] = std::reduce(host_input_it + row_offset, host_input_it + (row_offset + n_cols)); } REQUIRE(host_output == std::vector(output_ptr)); } struct SegmentedReduce_SumOverRows_WellKnown_Fixture_Tag; C2H_TEST_LIST("segmented_reduce can sum over rows of matrix with integral type " "with well-known operations", "[segmented_reduce][well_known]", std::int32_t, std::int64_t, std::uint32_t, std::uint64_t) { // generate 4 choices for n_rows: 0, 13 and 2 random samples from [1024, 4096) const std::size_t n_rows = GENERATE(0, 13, take(2, random(1 << 10, 1 << 12))); // generate 4 choices for number of columns const std::size_t n_cols = GENERATE(0, 12, take(2, random(1 << 10, 1 << 12))); const std::size_t n_elems = n_rows * n_cols; const std::size_t segment_size = n_cols; const std::vector host_input = generate(n_elems); std::vector host_output(n_rows, 0); REQUIRE(host_input.size() == n_cols * n_rows); REQUIRE(host_output.size() == n_rows); pointer_t input_ptr(host_input); // copy from host to device pointer_t output_ptr(host_output); // copy from host to device using SizeT = unsigned long long; static constexpr std::string_view index_ty_name = "unsigned long long"; struct row_offset_iterator_state_t { SizeT linear_id; SizeT segment_size; }; static constexpr std::string_view offset_iterator_state_name = "row_offset_iterator_state_t"; static constexpr std::string_view advance_offset_method_name = "advance_offset_it"; static constexpr std::string_view deref_offset_method_name = "dereference_offset_it"; const auto& [offset_iterator_state_src, offset_iterator_advance_src, offset_iterator_deref_src] = make_step_counting_iterator_sources( index_ty_name, offset_iterator_state_name, advance_offset_method_name, deref_offset_method_name); iterator_t start_offset_it = make_iterator( {offset_iterator_state_name, offset_iterator_state_src}, {advance_offset_method_name, offset_iterator_advance_src}, {deref_offset_method_name, offset_iterator_deref_src}); start_offset_it.state.linear_id = 0; start_offset_it.state.segment_size = segment_size; // a copy of offset iterator, so no need to define advance/dereference bodies, // just reused those defined above iterator_t end_offset_it = make_iterator( {offset_iterator_state_name, ""}, {advance_offset_method_name, ""}, {deref_offset_method_name, ""}); end_offset_it.state.linear_id = 1; end_offset_it.state.segment_size = segment_size; cccl_op_t op = make_well_known_binary_operation(); value_t init{0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce(input_ptr, output_ptr, n_rows, start_offset_it, end_offset_it, op, init, build_cache, test_key); auto host_input_it = host_input.begin(); auto host_output_it = host_output.begin(); for (std::size_t i = 0; i < n_rows; ++i) { std::size_t row_offset = i * segment_size; host_output_it[i] = std::reduce(host_input_it + row_offset, host_input_it + (row_offset + n_cols)); } REQUIRE(host_output == std::vector(output_ptr)); } struct pair { short a; size_t b; bool operator==(const pair& other) const { return a == other.a && b == other.b; } }; struct SegmentedReduce_CustomTypes_Fixture_Tag; C2H_TEST("SegmentedReduce works with custom types", "[segmented_reduce]") { using SizeT = ::cuda::std::size_t; const std::size_t n_segments = 50; auto increments = generate(n_segments); std::vector segments(n_segments + 1, 0); auto binary_op = std::plus<>{}; auto shift_op = [](auto i) { return i + 32; }; std::transform_inclusive_scan(increments.begin(), increments.end(), segments.begin() + 1, binary_op, shift_op); const std::vector a = generate(segments.back()); const std::vector b = generate(segments.back()); std::vector host_input(segments.back()); for (size_t i = 0; i < segments.back(); ++i) { host_input[i] = pair{.a = a[i], .b = b[i]}; } std::vector host_output(n_segments, pair{0, 0}); pointer_t input_ptr(host_input); // copy from host to device pointer_t output_ptr(host_output); // copy from host to device pointer_t offset_ptr(segments); // copy from host to device auto start_offset_it = static_cast(offset_ptr); auto end_offset_it = start_offset_it; end_offset_it.state = offset_ptr.ptr + 1; static constexpr std::string_view device_op_name = "plus_pair"; static constexpr std::string_view plus_pair_op_template = R"XXX( struct pair {{ short a; size_t b; }}; extern "C" __device__ void {0}(void* lhs_ptr, void* rhs_ptr, void* out_ptr) {{ pair* lhs = static_cast(lhs_ptr); pair* rhs = static_cast(rhs_ptr); pair* out = static_cast(out_ptr); *out = pair{{ lhs->a + rhs->a, lhs->b + rhs->b }}; }} )XXX"; std::string plus_pair_op_src = std::format(plus_pair_op_template, device_op_name); operation_t op = make_operation(device_op_name, plus_pair_op_src); pair v0 = pair{4, 2}; value_t init{v0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce(input_ptr, output_ptr, n_segments, start_offset_it, end_offset_it, op, init, build_cache, test_key); for (std::size_t i = 0; i < n_segments; ++i) { auto segment_begin_it = host_input.begin() + static_cast(segments[i]); auto segment_end_it = host_input.begin() + static_cast(segments[i + 1]); host_output[i] = std::reduce(segment_begin_it, segment_end_it, v0, [](pair lhs, pair rhs) { return pair{static_cast(lhs.a + rhs.a), lhs.b + rhs.b}; }); } auto host_actual = std::vector(output_ptr); REQUIRE(host_output == host_actual); } struct SegmentedReduce_CustomTypes_WellKnown_Fixture_Tag; C2H_TEST("SegmentedReduce works with custom types with well-known operations", "[segmented_reduce][well_known]") { using SizeT = ::cuda::std::size_t; const std::size_t n_segments = 50; auto increments = generate(n_segments); std::vector segments(n_segments + 1, 0); auto binary_op = std::plus<>{}; auto shift_op = [](auto i) { return i + 32; }; std::transform_inclusive_scan(increments.begin(), increments.end(), segments.begin() + 1, binary_op, shift_op); const std::vector a = generate(segments.back()); const std::vector b = generate(segments.back()); std::vector host_input(segments.back()); for (size_t i = 0; i < segments.back(); ++i) { host_input[i] = pair{.a = a[i], .b = b[i]}; } std::vector host_output(n_segments, pair{0, 0}); pointer_t input_ptr(host_input); // copy from host to device pointer_t output_ptr(host_output); // copy from host to device pointer_t offset_ptr(segments); // copy from host to device auto start_offset_it = static_cast(offset_ptr); auto end_offset_it = start_offset_it; end_offset_it.state = offset_ptr.ptr + 1; static constexpr std::string_view device_op_name = "plus_pair"; static constexpr std::string_view plus_pair_op_template = R"XXX( struct pair {{ short a; size_t b; }}; extern "C" __device__ void {0}(void* lhs_ptr, void* rhs_ptr, void* out_ptr) {{ pair* lhs = static_cast(lhs_ptr); pair* rhs = static_cast(rhs_ptr); pair* out = static_cast(out_ptr); *out = pair{{ lhs->a + rhs->a, lhs->b + rhs->b }}; }} )XXX"; std::string plus_pair_op_src = std::format(plus_pair_op_template, device_op_name); operation_t op_state = make_operation(device_op_name, plus_pair_op_src); cccl_op_t op = op_state; op.type = cccl_op_kind_t::CCCL_PLUS; pair v0 = pair{4, 2}; value_t init{v0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce(input_ptr, output_ptr, n_segments, start_offset_it, end_offset_it, op, init, build_cache, test_key); for (std::size_t i = 0; i < n_segments; ++i) { auto segment_begin_it = host_input.begin() + static_cast(segments[i]); auto segment_end_it = host_input.begin() + static_cast(segments[i + 1]); host_output[i] = std::reduce(segment_begin_it, segment_end_it, v0, [](pair lhs, pair rhs) { return pair{static_cast(lhs.a + rhs.a), lhs.b + rhs.b}; }); } auto host_actual = std::vector(output_ptr); REQUIRE(host_output == host_actual); } using SizeT = unsigned long long; struct strided_offset_iterator_state_t { SizeT linear_id; SizeT step; }; struct input_transposed_iterator_state_t { float* ptr; SizeT linear_id; SizeT n_rows; SizeT n_cols; }; static std::tuple make_input_transposed_iterator_sources( std::string_view value_type_name, std::string_view index_type_name, std::string_view state_name, std::string_view advance_fn_name, std::string_view dereference_fn_name) { static constexpr std::string_view it_state_src_tmpl = R"XXX( struct {0} {{ {1} *ptr; {2} linear_id; {2} n_rows; {2} n_cols; }}; )XXX"; const std::string it_state_def_src = std::format( it_state_src_tmpl, /* 0 */ state_name, /* 1 */ value_type_name, /* 2 */ index_type_name); static constexpr std::string_view it_advance_fn_def_src_tmpl = R"XXX( extern "C" __device__ void {0}(void* state, const void* offset) {{ auto* typed_state = static_cast<{1}*>(state); auto offset_val = *static_cast(offset); typed_state->linear_id += offset_val; }} )XXX"; const std::string it_advance_fn_def_src = std::format(it_advance_fn_def_src_tmpl, /*0*/ advance_fn_name, state_name, index_type_name); static constexpr std::string_view it_dereference_fn_src_tmpl = R"XXX( extern "C" __device__ void {0}(const void* state, {1}* result) {{ auto* typed_state = static_cast(state); unsigned long long col_id = (typed_state->linear_id) / (typed_state->n_rows); unsigned long long row_id = (typed_state->linear_id) - col_id * (typed_state->n_rows); *result = *(typed_state->ptr + row_id * (typed_state->n_cols) + col_id); }} )XXX"; const std::string it_dereference_fn_def_src = std::format( it_dereference_fn_src_tmpl, /* 0 */ dereference_fn_name, /*1*/ value_type_name, /*2*/ state_name); return std::make_tuple(it_state_def_src, it_advance_fn_def_src, it_dereference_fn_def_src); } struct SegmentedReduce_InputIterators_Fixture_Tag; C2H_TEST("SegmentedReduce works with input iterators", "[segmented_reduce]") { // Sum over columns of matrix const std::size_t n_rows = 2048; const std::size_t n_cols = 128; const std::size_t n_elems = n_rows * n_cols; const std::size_t col_size = n_rows; using ValueT = float; std::vector host_input; host_input.reserve(n_elems); { auto inp_ = generate(n_elems); for (auto&& el : inp_) { host_input.push_back(static_cast(el)); } } std::vector host_output(n_cols, 0); pointer_t input_ptr(host_input); // copy from host to device pointer_t output_ptr(host_output); // copy from host to device static constexpr std::string_view index_ty_name = "unsigned long long"; static constexpr std::string_view offset_it_state_name = "strided_offset_iterator_state_t"; static constexpr std::string_view offset_advance_fn_name = "advance_offset_it"; static constexpr std::string_view offset_deref_fn_name = "dereference_offset_it"; const auto& [offset_iterator_state_src, offset_iterator_advance_src, offset_iterator_deref_src] = make_step_counting_iterator_sources( index_ty_name, offset_it_state_name, offset_advance_fn_name, offset_deref_fn_name); iterator_t start_offset_it = make_iterator( {offset_it_state_name, offset_iterator_state_src}, {offset_advance_fn_name, offset_iterator_advance_src}, {offset_deref_fn_name, offset_iterator_deref_src}); start_offset_it.state.linear_id = 0; start_offset_it.state.step = col_size; // a copy of offset iterator, so no need to define advance/dereference bodies, // just reused those defined above iterator_t end_offset_it = make_iterator( {offset_it_state_name, ""}, {offset_advance_fn_name, ""}, {offset_deref_fn_name, ""}); end_offset_it.state.linear_id = 1; end_offset_it.state.step = col_size; static constexpr std::string_view value_type_name = "float"; static constexpr std::string_view input_it_state_name = "input_transposed_iterator_state_t"; static constexpr std::string_view transpose_it_advance_fn_name = "advance_transposed_it"; static constexpr std::string_view transpose_it_deref_fn_name = "dereference_transposed_it"; const auto& [transpose_it_state_src, transpose_it_advance_fn_src, transpose_it_deref_fn_src] = make_input_transposed_iterator_sources( value_type_name, index_ty_name, input_it_state_name, transpose_it_advance_fn_name, transpose_it_deref_fn_name); iterator_t input_transposed_iterator_it = make_iterator( {input_it_state_name, transpose_it_state_src}, {transpose_it_advance_fn_name, transpose_it_advance_fn_src}, {transpose_it_deref_fn_name, transpose_it_deref_fn_src}); input_transposed_iterator_it.state.ptr = input_ptr.ptr; input_transposed_iterator_it.state.linear_id = 0; input_transposed_iterator_it.state.n_rows = n_rows; input_transposed_iterator_it.state.n_cols = n_cols; operation_t op = make_operation("op", get_reduce_op(get_type_info().type)); value_t init{0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce( input_transposed_iterator_it, output_ptr, n_cols, start_offset_it, end_offset_it, op, init, build_cache, test_key); for (size_t col_id = 0; col_id < n_cols; ++col_id) { ValueT col_sum = 0; for (size_t row_id = 0; row_id < n_rows; ++row_id) { col_sum += host_input[row_id * n_cols + col_id]; } host_output[col_id] = col_sum; } auto host_actual = std::vector(output_ptr); REQUIRE(host_actual == host_output); } using fp_test_types = c2h::type_list< #if _CCCL_HAS_NVFP16() __half, #endif float, double>; struct SegmentedReduce_SumOverRows_FloatingPointTypes_Fixture_Tag; C2H_TEST("segmented_reduce can work with floating point types", "[segmented_reduce]", fp_test_types) { using T = c2h::get<0, TestType>; constexpr std::size_t n_rows = 13; constexpr std::size_t n_cols = 12; constexpr std::size_t n_elems = n_rows * n_cols; constexpr std::size_t row_size = n_cols; const std::vector int_input = generate(n_elems); // Suppress harmless conversion warnings on MSVC _CCCL_DIAG_PUSH _CCCL_DIAG_SUPPRESS_MSVC(4244) const std::vector input(int_input.begin(), int_input.end()); _CCCL_DIAG_POP std::vector output(n_rows, 0); pointer_t input_ptr(input); // copy from host to device pointer_t output_ptr(output); // copy from host to device using SizeT = unsigned long long; static constexpr std::string_view index_ty_name = "unsigned long long"; struct row_offset_iterator_state_t { SizeT linear_id; SizeT row_size; }; static constexpr std::string_view offset_iterator_state_name = "row_offset_iterator_state_t"; static constexpr std::string_view advance_offset_method_name = "advance_offset_it"; static constexpr std::string_view deref_offset_method_name = "dereference_offset_it"; const auto& [offset_iterator_state_src, offset_iterator_advance_src, offset_iterator_deref_src] = make_step_counting_iterator_sources( index_ty_name, offset_iterator_state_name, advance_offset_method_name, deref_offset_method_name); iterator_t start_offset_it = make_iterator( {offset_iterator_state_name, offset_iterator_state_src}, {advance_offset_method_name, offset_iterator_advance_src}, {deref_offset_method_name, offset_iterator_deref_src}); start_offset_it.state.linear_id = 0; start_offset_it.state.row_size = row_size; // a copy of offset iterator, so no need to define advance/dereference bodies, // just reused those defined above iterator_t end_offset_it = make_iterator( {offset_iterator_state_name, ""}, {advance_offset_method_name, ""}, {deref_offset_method_name, ""}); end_offset_it.state.linear_id = 1; end_offset_it.state.row_size = row_size; operation_t op = make_operation("op", get_reduce_op(get_type_info().type)); value_t init{0}; auto& build_cache = get_cache(); const auto& test_key = make_key(); segmented_reduce(input_ptr, output_ptr, n_rows, start_offset_it, end_offset_it, op, init, build_cache, test_key); auto host_input_it = input.begin(); auto host_output_it = output.begin(); for (std::size_t i = 0; i < n_rows; ++i) { std::size_t row_offset = i * row_size; host_output_it[i] = std::reduce(host_input_it + row_offset, host_input_it + (row_offset + n_cols)); } REQUIRE(output == std::vector(output_ptr)); } template struct host_offset_functor_state { ValueT m_p; ValueT m_min; }; template struct host_check_functor_state { ValueT m_p; ValueT m_min; DataT* m_ptr; }; namespace validate { using BuildResultT = cccl_device_reduce_build_result_t; struct reduce_cleanup { CUresult operator()(BuildResultT* build_data) const noexcept { return cccl_device_reduce_cleanup(build_data); } }; struct reduce_build { template CUresult operator()( BuildResultT* build_ptr, cccl_determinism_t determinism, cccl_iterator_t input, cccl_iterator_t output, uint64_t, cccl_op_t op, cccl_value_t init, Ts... args) const noexcept { return cccl_device_reduce_build(build_ptr, input, output, op, init, determinism, args...); } }; struct reduce_run { template CUresult operator()(cccl_device_reduce_build_result_t build, void* d_temp_storage, size_t* temp_storage_bytes, cccl_determinism_t /*determinism*/, Ts... args) const noexcept { return cccl_device_reduce(build, d_temp_storage, temp_storage_bytes, args...); } }; using reduce_deleter = BuildResultDeleter; using reduce_build_cache_t = build_cache_t>; template auto& get_cache() { return fixture::get_or_create().get_value(); } struct Reduce_Pointer_Fixture_Tag; template void reduce_for_pointer_inputs( cccl_iterator_t input, cccl_iterator_t output, uint64_t num_items, cccl_op_t op, cccl_value_t init) { auto& build_cache = get_cache(); const auto& test_key = make_key(); AlgorithmExecute( build_cache, test_key, CCCL_RUN_TO_RUN, input, output, num_items, op, init); } } // namespace validate struct SegmentedReduce_LargeNumSegments_Fixture_Tag; C2H_TEST("SegmentedReduce works with large num_segments", "[segmented_reduce]") { using DataT = signed short; using IndexT = signed long long; static constexpr std::string_view data_ty_name = "signed short"; static constexpr std::string_view index_ty_name = "signed long long"; // Segment sizes vary in range [min, min + p) in a linear progression // and restart periodically. Size of segment with 0-based index k is // min + (k % p) const IndexT min = 265; const IndexT p = 163; static constexpr IndexT n_segments_base = (IndexT(1) << 15) + (IndexT(1) << 3); static constexpr IndexT n_segments_under_int_max = n_segments_base << 10; static_assert(n_segments_under_int_max < INT_MAX); static constexpr IndexT n_segments_over_int_max = n_segments_base << 16; static_assert(n_segments_over_int_max > INT_MAX); const IndexT n_segments = GENERATE(n_segments_under_int_max, n_segments_over_int_max); // first define constant iterator: // iterators.ConstantIterator(np.int8(1)) auto input_const_it = make_constant_iterator(std::string{data_ty_name}); input_const_it.state.value = DataT(1); // Build counting iterator: iterators.CountingIterator(np.int64(-1)) // N.B.: Even though make_counting_iterator helper function exists, we need // source code for advance and dereference functions associated with counting // iterator to build transformed_iterator needed by this example static constexpr std::string_view counting_it_state_name = "counting_iterator_state_t"; static constexpr std::string_view counting_it_advance_fn_name = "advance_counting_it"; static constexpr std::string_view counting_it_deref_fn_name = "dereference_counting_it"; const auto [counting_it_state_src, counting_it_advance_fn_src, counting_it_deref_fn_src] = make_counting_iterator_sources( index_ty_name, counting_it_state_name, counting_it_advance_fn_name, counting_it_deref_fn_name); // Build transformation operation: offset_functor static constexpr std::string_view offset_functor_name = "offset_functor"; static constexpr std::string_view offset_functor_state_name = "offset_functor_state"; static constexpr std::string_view offset_functor_state_src_tmpl = R"XXX( struct {0} {{ {1} m_p; {1} m_min; }}; )XXX"; const std::string offset_functor_state_src = std::format(offset_functor_state_src_tmpl, offset_functor_state_name, index_ty_name); static constexpr std::string_view offset_functor_src_tmpl = R"XXX( extern "C" __device__ {2} {0}({1} *functor_state, {2} n) {{ /* def transform_fn(n): q = n // p r = n - q * p p2 = (p * (p - 1)) // 2 r2 = (r * (r + 1)) // 2 return min*(n + 1) + q * p2 + r2 */ {2} m0 = functor_state->m_min; {2} t = (n + 1) * m0; {2} p = functor_state->m_p; {2} q = n / p; {2} r = n - (q * p); {2} p2 = (p * (p - 1)) / 2; {2} qp2 = q * p2; {2} r2 = (r * (r + 1)) / 2; {2} t2 = t + r2; return (t2 + qp2); }} )XXX"; const std::string offset_functor_src = std::format(offset_functor_src_tmpl, offset_functor_name, offset_functor_state_name, index_ty_name); // Building transform_iterator /* offset_it = iterators.TransformIterator( iterators.CountingIterator(np.int64(0)), make_offset_transform(min, p) ) */ auto start_offsets_it = make_stateful_transform_input_iterator, host_offset_functor_state>( index_ty_name, index_ty_name, {counting_it_state_name, counting_it_state_src}, {counting_it_advance_fn_name, counting_it_advance_fn_src}, {counting_it_deref_fn_name, counting_it_deref_fn_src}, {offset_functor_state_name, offset_functor_state_src}, {offset_functor_name, offset_functor_src}); // Initialize the state of start_offset_it start_offsets_it.state.base_it_state.value = IndexT(-1); start_offsets_it.state.functor_state.m_p = IndexT(p); start_offsets_it.state.functor_state.m_min = IndexT(min); using HostTransformStateT = decltype(start_offsets_it.state); // end_offsets_it reuses advance/dereference definitions provided by // start_offsets_it constexpr std::string_view reuse_prior_definitions = ""; auto end_offsets_it = make_iterator( {start_offsets_it.state_name, reuse_prior_definitions}, {start_offsets_it.advance.name, reuse_prior_definitions}, {start_offsets_it.dereference.name, reuse_prior_definitions}); // Initialize the state of end_offset_it end_offsets_it.state.base_it_state.value = IndexT(0); end_offsets_it.state.functor_state = start_offsets_it.state.functor_state; static constexpr std::string_view binary_op_name = "_plus"; static constexpr std::string_view binary_op_src_tmpl = R"XXX( extern "C" __device__ void {0}(const void *x1_p, const void *x2_p, void *out_p) {{ const {1} *x1_tp = static_cast(x1_p); const {1} *x2_tp = static_cast(x2_p); {1} *out_tp = static_cast<{1}*>(out_p); *out_tp = (*x1_tp) + (*x2_tp); }} )XXX"; const std::string binary_op_src = std::format(binary_op_src_tmpl, binary_op_name, data_ty_name); auto binary_op = make_operation(binary_op_name, binary_op_src); // allocate memory for the result pointer_t res(n_segments); auto cccl_start_offsets_it = static_cast(start_offsets_it); auto cccl_end_offsets_it = static_cast(end_offsets_it); // set host_advance functions cccl_start_offsets_it.host_advance = &host_advance_base_value; cccl_end_offsets_it.host_advance = &host_advance_base_value; value_t h_init{DataT{0}}; auto& build_cache = get_cache(); const auto& test_key = make_key(); // launch segmented reduce segmented_reduce( input_const_it, res, n_segments, cccl_start_offsets_it, cccl_end_offsets_it, binary_op, h_init, build_cache, test_key); // Build validation call using device_reduce using CmpT = int; constexpr std::string_view cmp_ty_name = "int"; // check functor transforms computed values to comparison value against the // expected result static constexpr std::string_view check_functor_name = "check_functor"; static constexpr std::string_view check_functor_state_name = "check_functor_state"; static constexpr std::string_view check_functor_state_src_tmpl = R"XXX( struct {0} {{ {1} m_p; {1} m_min; {2} *m_ptr; }}; )XXX"; const std::string check_functor_state_src = std::format(check_functor_state_src_tmpl, check_functor_state_name, index_ty_name, data_ty_name); static constexpr std::string_view check_functor_src_tmpl = R"XXX( extern "C" __device__ {4} {0}({1} *functor_state, {2} n) {{ /* def expected_fn(n, ptr): q = n % p return (min + q) == ptr[n] */ {2} m0 = functor_state->m_min; {2} p = functor_state->m_p; {2} r = n % p; {3} actual = ({3})((functor_state->m_ptr)[n]); {3} expected = ({3})(m0 + r); return (expected == actual); }} )XXX"; static constexpr std::string_view common_ty_name = index_ty_name; const std::string check_functor_src = std::format( check_functor_src_tmpl, check_functor_name, check_functor_state_name, index_ty_name, common_ty_name, cmp_ty_name); // Building transform_iterator auto check_it = make_stateful_transform_input_iterator, host_check_functor_state>( cmp_ty_name, index_ty_name, {counting_it_state_name, counting_it_state_src}, {counting_it_advance_fn_name, counting_it_advance_fn_src}, {counting_it_deref_fn_name, counting_it_deref_fn_src}, {check_functor_state_name, check_functor_state_src}, {check_functor_name, check_functor_src}); // Initialize the state of check_it check_it.state.base_it_state.value = IndexT(0); check_it.state.functor_state.m_p = IndexT(p); check_it.state.functor_state.m_min = IndexT(min); check_it.state.functor_state.m_ptr = res.ptr; pointer_t as_expected(1); CmpT expected_value{1}; value_t _true{expected_value}; static constexpr std::string_view cmp_combine_op_name = "_logical_and"; static constexpr std::string_view cmp_combine_op_src_tmpl = R"XXX( extern "C" __device__ void {0}(const void *x1_p, const void *x2_p, void *out_p) {{ const {1} one = 1; const {1} zero = 0; {1} b1 = (*static_cast(x1_p)) ? one : zero; {1} b2 = (*static_cast(x2_p)) ? one : zero; *static_cast<{1}*>(out_p) = b1 * b2; }} )XXX"; const std::string cmp_combine_op_src = std::format(cmp_combine_op_src_tmpl, cmp_combine_op_name, cmp_ty_name); auto cmp_combine_op = make_operation(cmp_combine_op_name, cmp_combine_op_src); validate::reduce_for_pointer_inputs(check_it, as_expected, n_segments, cmp_combine_op, _true); REQUIRE(expected_value == std::vector(as_expected)[0]); } #ifndef CCCL_C_PARALLEL_V2 // ============== // guaranteed_max_segment_size tests (v1-only) // These exercise the small / medium / large dispatch policies in CUB segmented reduce. // Segment sizes are fixed so the guarantee exactly matches, verifying correctness across policies. // v2's cccl_device_segmented_reduce doesn't accept a guaranteed_max_segment_size parameter. // ============== // Helper shared by all three tests: builds offset iterators for uniform-size segments // and calls segmented_reduce_guaranteed, then verifies against std::reduce. template void run_guaranteed_max_seg_size_test( std::size_t n_rows, std::size_t n_cols, std::optional& build_cache, const std::optional& test_key) { const std::size_t n_elems = n_rows * n_cols; const std::size_t segment_size = n_cols; const std::vector host_input = generate(n_elems); std::vector host_output(n_rows, 0); pointer_t input_ptr(host_input); pointer_t output_ptr(host_output); using SizeT = unsigned long long; static constexpr std::string_view index_ty_name = "unsigned long long"; struct row_offset_iterator_state_t { SizeT linear_id; SizeT segment_size; }; static constexpr std::string_view offset_iterator_state_name = "row_offset_iterator_state_t"; static constexpr std::string_view advance_offset_method_name = "advance_offset_it"; static constexpr std::string_view deref_offset_method_name = "dereference_offset_it"; const auto& [offset_iterator_state_src, offset_iterator_advance_src, offset_iterator_deref_src] = make_step_counting_iterator_sources( index_ty_name, offset_iterator_state_name, advance_offset_method_name, deref_offset_method_name); iterator_t start_offset_it = make_iterator( {offset_iterator_state_name, offset_iterator_state_src}, {advance_offset_method_name, offset_iterator_advance_src}, {deref_offset_method_name, offset_iterator_deref_src}); start_offset_it.state.linear_id = 0; start_offset_it.state.segment_size = segment_size; iterator_t end_offset_it = make_iterator( {offset_iterator_state_name, ""}, {advance_offset_method_name, ""}, {deref_offset_method_name, ""}); end_offset_it.state.linear_id = 1; end_offset_it.state.segment_size = segment_size; operation_t op = make_operation("op", get_reduce_op(get_type_info().type)); value_t init{0}; segmented_reduce_guaranteed( input_ptr, output_ptr, n_rows, start_offset_it, end_offset_it, op, init, build_cache, test_key); for (std::size_t i = 0; i < n_rows; ++i) { std::size_t row_offset = i * segment_size; host_output[i] = std::reduce(host_input.begin() + row_offset, host_input.begin() + row_offset + n_cols); } REQUIRE(host_output == std::vector(output_ptr)); } // Small segments (≤16 elements): exercises the small warp-level dispatch policy struct SegmentedReduce_GuaranteedMaxSegSize_Small_Fixture_Tag; C2H_TEST_LIST("segmented_reduce respects guaranteed_max_segment_size for small segments", "[segmented_reduce][guaranteed_max_segment_size]", std::int32_t, std::int64_t, std::uint32_t, std::uint64_t) { static constexpr std::size_t segment_size = 8; const std::size_t n_rows = GENERATE(0, 13, take(2, random(100, 200))); auto& build_cache = get_cache(); const auto& test_key = make_key(); run_guaranteed_max_seg_size_test(n_rows, segment_size, build_cache, test_key); } // Medium segments (≤256 elements): exercises the medium warp-level dispatch policy struct SegmentedReduce_GuaranteedMaxSegSize_Medium_Fixture_Tag; C2H_TEST_LIST("segmented_reduce respects guaranteed_max_segment_size for medium segments", "[segmented_reduce][guaranteed_max_segment_size]", std::int32_t, std::int64_t, std::uint32_t, std::uint64_t) { static constexpr std::size_t segment_size = 64; const std::size_t n_rows = GENERATE(0, 13, take(2, random(50, 100))); auto& build_cache = get_cache(); const auto& test_key = make_key(); run_guaranteed_max_seg_size_test(n_rows, segment_size, build_cache, test_key); } // Large segments (≥512 elements): exercises the large block-level dispatch policy struct SegmentedReduce_GuaranteedMaxSegSize_Large_Fixture_Tag; C2H_TEST_LIST("segmented_reduce respects guaranteed_max_segment_size for large segments", "[segmented_reduce][guaranteed_max_segment_size]", std::int32_t, std::int64_t, std::uint32_t, std::uint64_t) { static constexpr std::size_t segment_size = 1024; const std::size_t n_rows = GENERATE(0, 5, take(2, random(10, 20))); auto& build_cache = get_cache(); const auto& test_key = make_key(); run_guaranteed_max_seg_size_test(n_rows, segment_size, build_cache, test_key); } C2H_TEST("SegmentedReduce build result has serialization metadata populated", "[segmented_reduce][serialization]") { using T = int32_t; constexpr int device_id = 0; const auto& build_info = BuildInformation::init(); cccl_op_t op = make_well_known_binary_operation(); pointer_t in(1); pointer_t out(1); pointer_t begin_offsets(1); pointer_t end_offsets(1); value_t init{T{0}}; cccl_device_segmented_reduce_build_result_t build{}; REQUIRE( CUDA_SUCCESS == cccl_device_segmented_reduce_build( &build, in, out, begin_offsets, end_offsets, op, init, build_info.get_cc_major(), build_info.get_cc_minor(), build_info.get_cub_path(), build_info.get_thrust_path(), build_info.get_libcudacxx_path(), build_info.get_ctk_path())); CHECK(build.cc == build_info.get_cc_major() * 10 + build_info.get_cc_minor()); CHECK((build.payload != nullptr && build.payload_kind == CCCL_PAYLOAD_CUBIN)); CHECK(build.payload_size > 0); CHECK(build.runtime_policy != nullptr); CHECK(build.runtime_policy_size > 0); REQUIRE(build.segmented_reduce_kernel_lowered_name != nullptr); CHECK(build.segmented_reduce_kernel_lowered_name[0] != '\0'); REQUIRE(CUDA_SUCCESS == cccl_device_segmented_reduce_cleanup(&build)); } C2H_TEST("SegmentedReduce compile/load round-trip", "[segmented_reduce][serialization]") { using T = int32_t; constexpr int device_id = 0; const auto& build_info = BuildInformation::init(); cccl_op_t op = make_well_known_binary_operation(); pointer_t dummy_in(1); pointer_t dummy_out(1); pointer_t dummy_begin_offsets(1); pointer_t dummy_end_offsets(1); value_t init{T{0}}; cccl_device_segmented_reduce_build_result_t build{}; REQUIRE( CUDA_SUCCESS == cccl_device_segmented_reduce_compile( &build, dummy_in, dummy_out, dummy_begin_offsets, dummy_end_offsets, op, init, build_info.get_cc_major(), build_info.get_cc_minor(), build_info.get_cub_path(), build_info.get_thrust_path(), build_info.get_libcudacxx_path(), build_info.get_ctk_path(), nullptr)); REQUIRE((build.payload != nullptr && build.payload_kind == CCCL_PAYLOAD_CUBIN)); REQUIRE(build.payload_size > 0); REQUIRE(build.segmented_reduce_kernel_lowered_name != nullptr); CHECK(build.library == nullptr); CHECK(build.segmented_reduce_kernel == nullptr); REQUIRE(CUDA_SUCCESS == cccl_device_segmented_reduce_load(&build)); REQUIRE(build.library != nullptr); CHECK(build.segmented_reduce_kernel != nullptr); constexpr std::size_t n = 16; constexpr std::size_t n_segments = 2; const std::vector input = generate(n); pointer_t input_ptr(input); pointer_t output_ptr(n_segments); const std::vector begin_offsets_host = {0, static_cast(n / 2)}; const std::vector end_offsets_host = {static_cast(n / 2), static_cast(n)}; pointer_t begin_offsets_ptr(begin_offsets_host); pointer_t end_offsets_ptr(end_offsets_host); CUstream null_stream = nullptr; size_t temp_storage_bytes = 0; REQUIRE( CUDA_SUCCESS == cccl_device_segmented_reduce( build, nullptr, &temp_storage_bytes, input_ptr, output_ptr, n_segments, begin_offsets_ptr, end_offsets_ptr, op, init, /*max_segment_size=*/0, null_stream)); pointer_t temp_storage(temp_storage_bytes); REQUIRE( CUDA_SUCCESS == cccl_device_segmented_reduce( build, temp_storage.ptr, &temp_storage_bytes, input_ptr, output_ptr, n_segments, begin_offsets_ptr, end_offsets_ptr, op, init, /*max_segment_size=*/0, null_stream)); const T expected0 = std::accumulate(input.begin(), input.begin() + n / 2, T{0}); const T expected1 = std::accumulate(input.begin() + n / 2, input.end(), T{0}); REQUIRE(output_ptr[0] == expected0); REQUIRE(output_ptr[1] == expected1); REQUIRE(CUDA_SUCCESS == cccl_device_segmented_reduce_cleanup(&build)); } #endif // CCCL_C_PARALLEL_V2 (guaranteed_max_segment_size tests)