Add complete CCCL CUB header tree (1394 files) to cccl_preload/include/: - cub/device/ — DeviceRadixSort, DeviceScan, DeviceHistogram, DeviceReduce, DeviceSelect - cub/agent/ — all agent implementations (sort, scan, reduce, histogram, etc) - cub/block/ — BlockScan, BlockReduce, BlockExchange, BlockLoad, BlockStore, etc - cub/warp/ — WarpScan, WarpReduce, WarpExchange, WarpMergeSort - cub/thread/ — thread-level operators - thrust/ — sort_by_key, iterator utilities - cuda/ — execution, stream, memory_resource, functional New kernel: cccl_moe_sort_scatter.cu - Uses CUB DeviceRadixSort::SortPairs to sort (expert_id, token_idx) pairs - O(n) radix sort replaces O(n log n) torch.argsort in MoE prefill path - Boundary detection + fill for expert offsets/sizes - Compiled against CCCL upstream headers (not corex CUB) to avoid BI-V100 bugs Previously only 288 CCCL headers (CachingDeviceAllocator only). Now 1394 headers — full CUB device-level algorithm stack available for all future kernels.
219 lines
8.2 KiB
C++
219 lines
8.2 KiB
C++
// SPDX-FileCopyrightText: Copyright (c) 2008-2013, NVIDIA Corporation. All rights reserved.
|
|
// SPDX-License-Identifier: Apache-2.0
|
|
|
|
/*! \file reverse.h
|
|
* \brief Reverses the order of a range
|
|
*/
|
|
|
|
#pragma once
|
|
|
|
#include <thrust/detail/config.h>
|
|
|
|
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
|
# pragma GCC system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
|
# pragma clang system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
|
# pragma system_header
|
|
#endif // no system header
|
|
#include <thrust/detail/execution_policy.h>
|
|
|
|
THRUST_NAMESPACE_BEGIN
|
|
|
|
/*! \addtogroup reordering
|
|
* \ingroup algorithms
|
|
* \{
|
|
*/
|
|
|
|
/*! \p reverse reverses a range. That is: for every <tt>i</tt> such that
|
|
* <tt>0 <= i <= (last - first) / 2</tt>, it exchanges <tt>*(first + i)</tt>
|
|
* and <tt>*(last - (i + 1))</tt>.
|
|
*
|
|
* The algorithm's execution is parallelized as determined by \p exec.
|
|
*
|
|
* \param exec The execution policy to use for parallelization.
|
|
* \param first The beginning of the range to reverse.
|
|
* \param last The end of the range to reverse.
|
|
*
|
|
* \tparam DerivedPolicy The name of the derived execution policy.
|
|
* \tparam BidirectionalIterator is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a> and \p
|
|
* BidirectionalIterator is mutable.
|
|
*
|
|
* The following code snippet demonstrates how to use \p reverse to reverse a
|
|
* \p device_vector of integers using the \p thrust::device execution policy for
|
|
* parallelization:
|
|
*
|
|
* \code
|
|
* #include <thrust/reverse.h>
|
|
* #include <thrust/execution_policy.h>
|
|
* ...
|
|
* const int N = 6;
|
|
* int data[N] = {0, 1, 2, 3, 4, 5};
|
|
* thrust::device_vector<int> v(data, data + N);
|
|
* thrust::reverse(thrust::device, v.begin(), v.end());
|
|
* // v is now {5, 4, 3, 2, 1, 0}
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/reverse
|
|
* \see \p reverse_copy
|
|
* \see \p reverse_iterator
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename DerivedPolicy, typename BidirectionalIterator>
|
|
_CCCL_HOST_DEVICE void reverse(const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
|
|
BidirectionalIterator first,
|
|
BidirectionalIterator last);
|
|
|
|
/*! \p reverse reverses a range. That is: for every <tt>i</tt> such that
|
|
* <tt>0 <= i <= (last - first) / 2</tt>, it exchanges <tt>*(first + i)</tt>
|
|
* and <tt>*(last - (i + 1))</tt>.
|
|
*
|
|
* \param first The beginning of the range to reverse.
|
|
* \param last The end of the range to reverse.
|
|
*
|
|
* \tparam BidirectionalIterator is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a> and \p
|
|
* BidirectionalIterator is mutable.
|
|
*
|
|
* The following code snippet demonstrates how to use \p reverse to reverse a
|
|
* \p device_vector of integers.
|
|
*
|
|
* \code
|
|
* #include <thrust/reverse.h>
|
|
* ...
|
|
* const int N = 6;
|
|
* int data[N] = {0, 1, 2, 3, 4, 5};
|
|
* thrust::device_vector<int> v(data, data + N);
|
|
* thrust::reverse(v.begin(), v.end());
|
|
* // v is now {5, 4, 3, 2, 1, 0}
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/reverse
|
|
* \see \p reverse_copy
|
|
* \see \p reverse_iterator
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename BidirectionalIterator>
|
|
void reverse(BidirectionalIterator first, BidirectionalIterator last);
|
|
|
|
/*! \p reverse_copy differs from \p reverse only in that the reversed range
|
|
* is written to a different output range, rather than inplace.
|
|
*
|
|
* \p reverse_copy copies elements from the range <tt>[first, last)</tt> to the
|
|
* range <tt>[result, result + (last - first))</tt> such that the copy is a
|
|
* reverse of the original range. Specifically: for every <tt>i</tt> such that
|
|
* <tt>0 <= i < (last - first)</tt>, \p reverse_copy performs the assignment
|
|
* <tt>*(result + (last - first) - i) = *(first + i)</tt>.
|
|
*
|
|
* The return value is <tt>result + (last - first))</tt>.
|
|
*
|
|
* The algorithm's execution is parallelized as determined by \p exec.
|
|
*
|
|
* \param exec The execution policy to use for parallelization.
|
|
* \param first The beginning of the range to reverse.
|
|
* \param last The end of the range to reverse.
|
|
* \param result The beginning of the output range.
|
|
*
|
|
* \tparam DerivedPolicy The name of the derived execution policy.
|
|
* \tparam BidirectionalIterator is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a>, and \p
|
|
* BidirectionalIterator's \p value_type is convertible to \p OutputIterator's \p value_type. \tparam OutputIterator is
|
|
* a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output Iterator</a>.
|
|
*
|
|
* \pre The range <tt>[first, last)</tt> and the range <tt>[result, result + (last - first))</tt> shall not overlap.
|
|
*
|
|
* The following code snippet demonstrates how to use \p reverse_copy to reverse
|
|
* an input \p device_vector of integers to an output \p device_vector using the \p thrust::device
|
|
* execution policy for parallelization:
|
|
*
|
|
* \code
|
|
* #include <thrust/reverse.h>
|
|
* #include <thrust/execution_policy.h>
|
|
* ...
|
|
* const int N = 6;
|
|
* int data[N] = {0, 1, 2, 3, 4, 5};
|
|
* thrust::device_vector<int> input(data, data + N);
|
|
* thrust::device_vector<int> output(N);
|
|
* thrust::reverse_copy(thrust::device, v.begin(), v.end(), output.begin());
|
|
* // input is still {0, 1, 2, 3, 4, 5}
|
|
* // output is now {5, 4, 3, 2, 1, 0}
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/reverse_copy
|
|
* \see \p reverse
|
|
* \see \p reverse_iterator
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename DerivedPolicy, typename BidirectionalIterator, typename OutputIterator>
|
|
_CCCL_HOST_DEVICE OutputIterator reverse_copy(
|
|
const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
|
|
BidirectionalIterator first,
|
|
BidirectionalIterator last,
|
|
OutputIterator result);
|
|
|
|
/*! \p reverse_copy differs from \p reverse only in that the reversed range
|
|
* is written to a different output range, rather than inplace.
|
|
*
|
|
* \p reverse_copy copies elements from the range <tt>[first, last)</tt> to the
|
|
* range <tt>[result, result + (last - first))</tt> such that the copy is a
|
|
* reverse of the original range. Specifically: for every <tt>i</tt> such that
|
|
* <tt>0 <= i < (last - first)</tt>, \p reverse_copy performs the assignment
|
|
* <tt>*(result + (last - first) - i) = *(first + i)</tt>.
|
|
*
|
|
* The return value is <tt>result + (last - first))</tt>.
|
|
*
|
|
* \param first The beginning of the range to reverse.
|
|
* \param last The end of the range to reverse.
|
|
* \param result The beginning of the output range.
|
|
*
|
|
* \tparam BidirectionalIterator is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a>, and \p
|
|
* BidirectionalIterator's \p value_type is convertible to \p OutputIterator's \p value_type. \tparam OutputIterator is
|
|
* a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output Iterator</a>.
|
|
*
|
|
* \pre The range <tt>[first, last)</tt> and the range <tt>[result, result + (last - first))</tt> shall not overlap.
|
|
*
|
|
* The following code snippet demonstrates how to use \p reverse_copy to reverse
|
|
* an input \p device_vector of integers to an output \p device_vector.
|
|
*
|
|
* \code
|
|
* #include <thrust/reverse.h>
|
|
* ...
|
|
* const int N = 6;
|
|
* int data[N] = {0, 1, 2, 3, 4, 5};
|
|
* thrust::device_vector<int> input(data, data + N);
|
|
* thrust::device_vector<int> output(N);
|
|
* thrust::reverse_copy(v.begin(), v.end(), output.begin());
|
|
* // input is still {0, 1, 2, 3, 4, 5}
|
|
* // output is now {5, 4, 3, 2, 1, 0}
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/reverse_copy
|
|
* \see \p reverse
|
|
* \see \p reverse_iterator
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename BidirectionalIterator, typename OutputIterator>
|
|
OutputIterator reverse_copy(BidirectionalIterator first, BidirectionalIterator last, OutputIterator result);
|
|
|
|
/*!
|
|
* \} end group reordering
|
|
*/
|
|
|
|
THRUST_NAMESPACE_END
|
|
|
|
#include <thrust/detail/reverse.inl>
|