Files
project_6/qwen3_6_scripts/cccl_preload/include/thrust/reverse.h
project6-dev 4c365b8c03 feat(CCCL): device-level CUB algorithms for MoE dispatch
Add complete CCCL CUB header tree (1394 files) to cccl_preload/include/:
- cub/device/ — DeviceRadixSort, DeviceScan, DeviceHistogram, DeviceReduce, DeviceSelect
- cub/agent/ — all agent implementations (sort, scan, reduce, histogram, etc)
- cub/block/ — BlockScan, BlockReduce, BlockExchange, BlockLoad, BlockStore, etc
- cub/warp/ — WarpScan, WarpReduce, WarpExchange, WarpMergeSort
- cub/thread/ — thread-level operators
- thrust/ — sort_by_key, iterator utilities
- cuda/ — execution, stream, memory_resource, functional

New kernel: cccl_moe_sort_scatter.cu
- Uses CUB DeviceRadixSort::SortPairs to sort (expert_id, token_idx) pairs
- O(n) radix sort replaces O(n log n) torch.argsort in MoE prefill path
- Boundary detection + fill for expert offsets/sizes
- Compiled against CCCL upstream headers (not corex CUB) to avoid BI-V100 bugs

Previously only 288 CCCL headers (CachingDeviceAllocator only).
Now 1394 headers — full CUB device-level algorithm stack available for
all future kernels.
2026-08-13 11:18:52 +00:00

219 lines
8.2 KiB
C++

// SPDX-FileCopyrightText: Copyright (c) 2008-2013, NVIDIA Corporation. All rights reserved.
// SPDX-License-Identifier: Apache-2.0
/*! \file reverse.h
* \brief Reverses the order of a range
*/
#pragma once
#include <thrust/detail/config.h>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <thrust/detail/execution_policy.h>
THRUST_NAMESPACE_BEGIN
/*! \addtogroup reordering
* \ingroup algorithms
* \{
*/
/*! \p reverse reverses a range. That is: for every <tt>i</tt> such that
* <tt>0 <= i <= (last - first) / 2</tt>, it exchanges <tt>*(first + i)</tt>
* and <tt>*(last - (i + 1))</tt>.
*
* The algorithm's execution is parallelized as determined by \p exec.
*
* \param exec The execution policy to use for parallelization.
* \param first The beginning of the range to reverse.
* \param last The end of the range to reverse.
*
* \tparam DerivedPolicy The name of the derived execution policy.
* \tparam BidirectionalIterator is a model of <a
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a> and \p
* BidirectionalIterator is mutable.
*
* The following code snippet demonstrates how to use \p reverse to reverse a
* \p device_vector of integers using the \p thrust::device execution policy for
* parallelization:
*
* \code
* #include <thrust/reverse.h>
* #include <thrust/execution_policy.h>
* ...
* const int N = 6;
* int data[N] = {0, 1, 2, 3, 4, 5};
* thrust::device_vector<int> v(data, data + N);
* thrust::reverse(thrust::device, v.begin(), v.end());
* // v is now {5, 4, 3, 2, 1, 0}
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/reverse
* \see \p reverse_copy
* \see \p reverse_iterator
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename DerivedPolicy, typename BidirectionalIterator>
_CCCL_HOST_DEVICE void reverse(const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
BidirectionalIterator first,
BidirectionalIterator last);
/*! \p reverse reverses a range. That is: for every <tt>i</tt> such that
* <tt>0 <= i <= (last - first) / 2</tt>, it exchanges <tt>*(first + i)</tt>
* and <tt>*(last - (i + 1))</tt>.
*
* \param first The beginning of the range to reverse.
* \param last The end of the range to reverse.
*
* \tparam BidirectionalIterator is a model of <a
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a> and \p
* BidirectionalIterator is mutable.
*
* The following code snippet demonstrates how to use \p reverse to reverse a
* \p device_vector of integers.
*
* \code
* #include <thrust/reverse.h>
* ...
* const int N = 6;
* int data[N] = {0, 1, 2, 3, 4, 5};
* thrust::device_vector<int> v(data, data + N);
* thrust::reverse(v.begin(), v.end());
* // v is now {5, 4, 3, 2, 1, 0}
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/reverse
* \see \p reverse_copy
* \see \p reverse_iterator
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename BidirectionalIterator>
void reverse(BidirectionalIterator first, BidirectionalIterator last);
/*! \p reverse_copy differs from \p reverse only in that the reversed range
* is written to a different output range, rather than inplace.
*
* \p reverse_copy copies elements from the range <tt>[first, last)</tt> to the
* range <tt>[result, result + (last - first))</tt> such that the copy is a
* reverse of the original range. Specifically: for every <tt>i</tt> such that
* <tt>0 <= i < (last - first)</tt>, \p reverse_copy performs the assignment
* <tt>*(result + (last - first) - i) = *(first + i)</tt>.
*
* The return value is <tt>result + (last - first))</tt>.
*
* The algorithm's execution is parallelized as determined by \p exec.
*
* \param exec The execution policy to use for parallelization.
* \param first The beginning of the range to reverse.
* \param last The end of the range to reverse.
* \param result The beginning of the output range.
*
* \tparam DerivedPolicy The name of the derived execution policy.
* \tparam BidirectionalIterator is a model of <a
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a>, and \p
* BidirectionalIterator's \p value_type is convertible to \p OutputIterator's \p value_type. \tparam OutputIterator is
* a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output Iterator</a>.
*
* \pre The range <tt>[first, last)</tt> and the range <tt>[result, result + (last - first))</tt> shall not overlap.
*
* The following code snippet demonstrates how to use \p reverse_copy to reverse
* an input \p device_vector of integers to an output \p device_vector using the \p thrust::device
* execution policy for parallelization:
*
* \code
* #include <thrust/reverse.h>
* #include <thrust/execution_policy.h>
* ...
* const int N = 6;
* int data[N] = {0, 1, 2, 3, 4, 5};
* thrust::device_vector<int> input(data, data + N);
* thrust::device_vector<int> output(N);
* thrust::reverse_copy(thrust::device, v.begin(), v.end(), output.begin());
* // input is still {0, 1, 2, 3, 4, 5}
* // output is now {5, 4, 3, 2, 1, 0}
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/reverse_copy
* \see \p reverse
* \see \p reverse_iterator
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename DerivedPolicy, typename BidirectionalIterator, typename OutputIterator>
_CCCL_HOST_DEVICE OutputIterator reverse_copy(
const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
BidirectionalIterator first,
BidirectionalIterator last,
OutputIterator result);
/*! \p reverse_copy differs from \p reverse only in that the reversed range
* is written to a different output range, rather than inplace.
*
* \p reverse_copy copies elements from the range <tt>[first, last)</tt> to the
* range <tt>[result, result + (last - first))</tt> such that the copy is a
* reverse of the original range. Specifically: for every <tt>i</tt> such that
* <tt>0 <= i < (last - first)</tt>, \p reverse_copy performs the assignment
* <tt>*(result + (last - first) - i) = *(first + i)</tt>.
*
* The return value is <tt>result + (last - first))</tt>.
*
* \param first The beginning of the range to reverse.
* \param last The end of the range to reverse.
* \param result The beginning of the output range.
*
* \tparam BidirectionalIterator is a model of <a
* href="https://en.cppreference.com/w/cpp/iterator/bidirectional_iterator">Bidirectional Iterator</a>, and \p
* BidirectionalIterator's \p value_type is convertible to \p OutputIterator's \p value_type. \tparam OutputIterator is
* a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output Iterator</a>.
*
* \pre The range <tt>[first, last)</tt> and the range <tt>[result, result + (last - first))</tt> shall not overlap.
*
* The following code snippet demonstrates how to use \p reverse_copy to reverse
* an input \p device_vector of integers to an output \p device_vector.
*
* \code
* #include <thrust/reverse.h>
* ...
* const int N = 6;
* int data[N] = {0, 1, 2, 3, 4, 5};
* thrust::device_vector<int> input(data, data + N);
* thrust::device_vector<int> output(N);
* thrust::reverse_copy(v.begin(), v.end(), output.begin());
* // input is still {0, 1, 2, 3, 4, 5}
* // output is now {5, 4, 3, 2, 1, 0}
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/reverse_copy
* \see \p reverse
* \see \p reverse_iterator
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename BidirectionalIterator, typename OutputIterator>
OutputIterator reverse_copy(BidirectionalIterator first, BidirectionalIterator last, OutputIterator result);
/*!
* \} end group reordering
*/
THRUST_NAMESPACE_END
#include <thrust/detail/reverse.inl>