feat(CCCL): device-level CUB algorithms for MoE dispatch
Add complete CCCL CUB header tree (1394 files) to cccl_preload/include/: - cub/device/ — DeviceRadixSort, DeviceScan, DeviceHistogram, DeviceReduce, DeviceSelect - cub/agent/ — all agent implementations (sort, scan, reduce, histogram, etc) - cub/block/ — BlockScan, BlockReduce, BlockExchange, BlockLoad, BlockStore, etc - cub/warp/ — WarpScan, WarpReduce, WarpExchange, WarpMergeSort - cub/thread/ — thread-level operators - thrust/ — sort_by_key, iterator utilities - cuda/ — execution, stream, memory_resource, functional New kernel: cccl_moe_sort_scatter.cu - Uses CUB DeviceRadixSort::SortPairs to sort (expert_id, token_idx) pairs - O(n) radix sort replaces O(n log n) torch.argsort in MoE prefill path - Boundary detection + fill for expert offsets/sizes - Compiled against CCCL upstream headers (not corex CUB) to avoid BI-V100 bugs Previously only 288 CCCL headers (CachingDeviceAllocator only). Now 1394 headers — full CUB device-level algorithm stack available for all future kernels.
This commit is contained in:
142
qwen3_6_scripts/cccl_preload/include/thrust/swap.h
Normal file
142
qwen3_6_scripts/cccl_preload/include/thrust/swap.h
Normal file
@@ -0,0 +1,142 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2008-2013, NVIDIA Corporation. All rights reserved.
|
||||
// SPDX-License-Identifier: Apache-2.0
|
||||
|
||||
/*! \file swap.h
|
||||
* \brief Functions for swapping the value of elements
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <thrust/detail/config.h>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
#include <thrust/detail/execution_policy.h>
|
||||
|
||||
#include <cuda/std/__utility/swap.h>
|
||||
|
||||
THRUST_NAMESPACE_BEGIN
|
||||
|
||||
using ::cuda::std::swap;
|
||||
|
||||
/*! \addtogroup copying
|
||||
* \{
|
||||
*/
|
||||
|
||||
/*! \p swap_ranges swaps each of the elements in the range <tt>[first1, last1)</tt>
|
||||
* with the corresponding element in the range <tt>[first2, first2 + (last1 - first1))</tt>.
|
||||
* That is, for each integer \c n such that <tt>0 <= n < (last1 - first1)</tt>, it swaps
|
||||
* <tt>*(first1 + n)</tt> and <tt>*(first2 + n)</tt>. The return value is
|
||||
* <tt>first2 + (last1 - first1)</tt>.
|
||||
*
|
||||
* The algorithm's execution is parallelized as determined by \p exec.
|
||||
*
|
||||
* \param exec The execution policy to use for parallelization.
|
||||
* \param first1 The beginning of the first sequence to swap.
|
||||
* \param last1 One position past the last element of the first sequence to swap.
|
||||
* \param first2 The beginning of the second sequence to swap.
|
||||
* \return An iterator pointing to one position past the last element of the second
|
||||
* sequence to swap.
|
||||
*
|
||||
* \tparam DerivedPolicy The name of the derived execution policy.
|
||||
* \tparam ForwardIterator1 is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
||||
* Iterator</a>, and \p ForwardIterator1's \c value_type must be convertible to \p ForwardIterator2's \c value_type.
|
||||
* \tparam ForwardIterator2 is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
||||
* Iterator</a>, and \p ForwardIterator2's \c value_type must be convertible to \p ForwardIterator1's \c value_type.
|
||||
*
|
||||
* \pre \p first1 may equal \p first2, but the range <tt>[first1, last1)</tt> shall not overlap the range <tt>[first2,
|
||||
* first2 + (last1 - first1))</tt> otherwise.
|
||||
*
|
||||
* The following code snippet demonstrates how to use \p swap_ranges to
|
||||
* swap the contents of two \c thrust::device_vectors using the \p thrust::device execution
|
||||
* policy for parallelization:
|
||||
*
|
||||
* \code
|
||||
* #include <thrust/swap.h>
|
||||
* #include <thrust/device_vector.h>
|
||||
* #include <thrust/execution_policy.h>
|
||||
* ...
|
||||
* thrust::device_vector<int> v1(2), v2(2);
|
||||
* v1[0] = 1;
|
||||
* v1[1] = 2;
|
||||
* v2[0] = 3;
|
||||
* v2[1] = 4;
|
||||
*
|
||||
* thrust::swap_ranges(thrust::device, v1.begin(), v1.end(), v2.begin());
|
||||
*
|
||||
* // v1[0] == 3, v1[1] == 4, v2[0] == 1, v2[1] == 2
|
||||
* \endcode
|
||||
*
|
||||
* \see https://en.cppreference.com/w/cpp/algorithm/swap_ranges
|
||||
* \see \c swap
|
||||
*
|
||||
* \verbatim embed:rst:leading-asterisk
|
||||
* .. versionadded:: 2.2.0
|
||||
* \endverbatim
|
||||
*/
|
||||
template <typename DerivedPolicy, typename ForwardIterator1, typename ForwardIterator2>
|
||||
_CCCL_HOST_DEVICE ForwardIterator2 swap_ranges(
|
||||
const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
|
||||
ForwardIterator1 first1,
|
||||
ForwardIterator1 last1,
|
||||
ForwardIterator2 first2);
|
||||
|
||||
/*! \p swap_ranges swaps each of the elements in the range <tt>[first1, last1)</tt>
|
||||
* with the corresponding element in the range <tt>[first2, first2 + (last1 - first1))</tt>.
|
||||
* That is, for each integer \c n such that <tt>0 <= n < (last1 - first1)</tt>, it swaps
|
||||
* <tt>*(first1 + n)</tt> and <tt>*(first2 + n)</tt>. The return value is
|
||||
* <tt>first2 + (last1 - first1)</tt>.
|
||||
*
|
||||
* \param first1 The beginning of the first sequence to swap.
|
||||
* \param last1 One position past the last element of the first sequence to swap.
|
||||
* \param first2 The beginning of the second sequence to swap.
|
||||
* \return An iterator pointing to one position past the last element of the second
|
||||
* sequence to swap.
|
||||
*
|
||||
* \tparam ForwardIterator1 is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
||||
* Iterator</a>, and \p ForwardIterator1's \c value_type must be convertible to \p ForwardIterator2's \c value_type.
|
||||
* \tparam ForwardIterator2 is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
||||
* Iterator</a>, and \p ForwardIterator2's \c value_type must be convertible to \p ForwardIterator1's \c value_type.
|
||||
*
|
||||
* \pre \p first1 may equal \p first2, but the range <tt>[first1, last1)</tt> shall not overlap the range <tt>[first2,
|
||||
* first2 + (last1 - first1))</tt> otherwise.
|
||||
*
|
||||
* The following code snippet demonstrates how to use \p swap_ranges to
|
||||
* swap the contents of two \c thrust::device_vectors.
|
||||
*
|
||||
* \code
|
||||
* #include <thrust/swap.h>
|
||||
* #include <thrust/device_vector.h>
|
||||
* ...
|
||||
* thrust::device_vector<int> v1(2), v2(2);
|
||||
* v1[0] = 1;
|
||||
* v1[1] = 2;
|
||||
* v2[0] = 3;
|
||||
* v2[1] = 4;
|
||||
*
|
||||
* thrust::swap_ranges(v1.begin(), v1.end(), v2.begin());
|
||||
*
|
||||
* // v1[0] == 3, v1[1] == 4, v2[0] == 1, v2[1] == 2
|
||||
* \endcode
|
||||
*
|
||||
* \see https://en.cppreference.com/w/cpp/algorithm/swap_ranges
|
||||
* \see \c swap
|
||||
*
|
||||
* \verbatim embed:rst:leading-asterisk
|
||||
* .. versionadded:: 2.2.0
|
||||
* \endverbatim
|
||||
*/
|
||||
template <typename ForwardIterator1, typename ForwardIterator2>
|
||||
ForwardIterator2 swap_ranges(ForwardIterator1 first1, ForwardIterator1 last1, ForwardIterator2 first2);
|
||||
|
||||
/*! \} // copying
|
||||
*/
|
||||
|
||||
THRUST_NAMESPACE_END
|
||||
|
||||
#include <thrust/detail/swap_ranges.inl>
|
||||
Reference in New Issue
Block a user