Files
project_6/qwen3_6_scripts/cccl_preload/include/thrust/fill.h
project6-dev 4c365b8c03 feat(CCCL): device-level CUB algorithms for MoE dispatch
Add complete CCCL CUB header tree (1394 files) to cccl_preload/include/:
- cub/device/ — DeviceRadixSort, DeviceScan, DeviceHistogram, DeviceReduce, DeviceSelect
- cub/agent/ — all agent implementations (sort, scan, reduce, histogram, etc)
- cub/block/ — BlockScan, BlockReduce, BlockExchange, BlockLoad, BlockStore, etc
- cub/warp/ — WarpScan, WarpReduce, WarpExchange, WarpMergeSort
- cub/thread/ — thread-level operators
- thrust/ — sort_by_key, iterator utilities
- cuda/ — execution, stream, memory_resource, functional

New kernel: cccl_moe_sort_scatter.cu
- Uses CUB DeviceRadixSort::SortPairs to sort (expert_id, token_idx) pairs
- O(n) radix sort replaces O(n log n) torch.argsort in MoE prefill path
- Boundary detection + fill for expert offsets/sizes
- Compiled against CCCL upstream headers (not corex CUB) to avoid BI-V100 bugs

Previously only 288 CCCL headers (CachingDeviceAllocator only).
Now 1394 headers — full CUB device-level algorithm stack available for
all future kernels.
2026-08-13 11:18:52 +00:00

204 lines
7.2 KiB
C++

// SPDX-FileCopyrightText: Copyright (c) 2008-2013, NVIDIA Corporation. All rights reserved.
// SPDX-License-Identifier: Apache-2.0
/*! \file fill.h
* \brief Fills a range with a constant value
*/
#pragma once
#include <thrust/detail/config.h>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <thrust/detail/execution_policy.h>
THRUST_NAMESPACE_BEGIN
/*! \addtogroup filling
* \ingroup transformations
* \{
*/
/*! \p fill assigns the value \p value to every element in
* the range <tt>[first, last)</tt>. That is, for every
* iterator \c i in <tt>[first, last)</tt>, it performs
* the assignment <tt>*i = value</tt>.
*
* The algorithm's execution is parallelized as determined by \p exec.
*
* \param exec The execution policy to use for parallelization.
* \param first The beginning of the sequence.
* \param last The end of the sequence.
* \param value The value to be copied.
*
* \tparam DerivedPolicy The name of the derived execution policy.
* \tparam ForwardIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
* Iterator</a>, and \p ForwardIterator is mutable. \tparam T is a model of <a
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
* convertible to \p ForwardIterator's \c value_type.
*
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
* elements to a given value using the \p thrust::device execution policy for parallelization:
*
* \code
* #include <thrust/fill.h>
* #include <thrust/device_vector.h>
* #include <thrust/execution_policy.h>
* ...
* thrust::device_vector<int> v(4);
* thrust::fill(thrust::device, v.begin(), v.end(), 137);
*
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/fill
* \see \c fill_n
* \see \c uninitialized_fill
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename DerivedPolicy, typename ForwardIterator, typename T>
_CCCL_HOST_DEVICE void
fill(const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
ForwardIterator first,
ForwardIterator last,
const T& value);
/*! \p fill assigns the value \p value to every element in
* the range <tt>[first, last)</tt>. That is, for every
* iterator \c i in <tt>[first, last)</tt>, it performs
* the assignment <tt>*i = value</tt>.
*
* \param first The beginning of the sequence.
* \param last The end of the sequence.
* \param value The value to be copied.
*
* \tparam ForwardIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
* Iterator</a>, and \p ForwardIterator is mutable. \tparam T is a model of <a
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
* convertible to \p ForwardIterator's \c value_type.
*
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
* elements to a given value.
*
* \code
* #include <thrust/fill.h>
* #include <thrust/device_vector.h>
* ...
* thrust::device_vector<int> v(4);
* thrust::fill(v.begin(), v.end(), 137);
*
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/fill
* \see \c fill_n
* \see \c uninitialized_fill
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename ForwardIterator, typename T>
_CCCL_HOST_DEVICE void fill(ForwardIterator first, ForwardIterator last, const T& value);
/*! \p fill_n assigns the value \p value to every element in
* the range <tt>[first, first+n)</tt>. That is, for every
* iterator \c i in <tt>[first, first+n)</tt>, it performs
* the assignment <tt>*i = value</tt>.
*
* The algorithm's execution is parallelized as determined by \p exec.
*
* \param exec The execution policy to use for parallelization.
* \param first The beginning of the sequence.
* \param n The size of the sequence.
* \param value The value to be copied.
* \return <tt>first + n</tt>
*
* \tparam DerivedPolicy The name of the derived execution policy.
* \tparam OutputIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output
* Iterator</a>. \tparam T is a model of <a
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
* convertible to a type in \p OutputIterator's set of \c value_type.
*
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
* elements to a given value using the \p thrust::device execution policy for parallelization:
*
* \code
* #include <thrust/fill.h>
* #include <thrust/device_vector.h>
* #include <thrust/execution_policy.h>
* ...
* thrust::device_vector<int> v(4);
* thrust::fill_n(thrust::device, v.begin(), v.size(), 137);
*
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/fill_n
* \see \c fill
* \see \c uninitialized_fill_n
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename DerivedPolicy, typename OutputIterator, typename Size, typename T>
_CCCL_HOST_DEVICE OutputIterator
fill_n(const thrust::detail::execution_policy_base<DerivedPolicy>& exec, OutputIterator first, Size n, const T& value);
/*! \p fill_n assigns the value \p value to every element in
* the range <tt>[first, first+n)</tt>. That is, for every
* iterator \c i in <tt>[first, first+n)</tt>, it performs
* the assignment <tt>*i = value</tt>.
*
* \param first The beginning of the sequence.
* \param n The size of the sequence.
* \param value The value to be copied.
* \return <tt>first + n</tt>
*
* \tparam OutputIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output
* Iterator</a>. \tparam T is a model of <a
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
* convertible to a type in \p OutputIterator's set of \c value_type.
*
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
* elements to a given value.
*
* \code
* #include <thrust/fill.h>
* #include <thrust/device_vector.h>
* ...
* thrust::device_vector<int> v(4);
* thrust::fill_n(v.begin(), v.size(), 137);
*
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
* \endcode
*
* \see https://en.cppreference.com/w/cpp/algorithm/fill_n
* \see \c fill
* \see \c uninitialized_fill_n
*
* \verbatim embed:rst:leading-asterisk
* .. versionadded:: 2.2.0
* \endverbatim
*/
template <typename OutputIterator, typename Size, typename T>
_CCCL_HOST_DEVICE OutputIterator fill_n(OutputIterator first, Size n, const T& value);
/*!
* \} end group filling
*/
THRUST_NAMESPACE_END
#include <thrust/detail/fill.inl>