CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
204 lines
7.2 KiB
C++
204 lines
7.2 KiB
C++
// SPDX-FileCopyrightText: Copyright (c) 2008-2013, NVIDIA Corporation. All rights reserved.
|
|
// SPDX-License-Identifier: Apache-2.0
|
|
|
|
/*! \file fill.h
|
|
* \brief Fills a range with a constant value
|
|
*/
|
|
|
|
#pragma once
|
|
|
|
#include <thrust/detail/config.h>
|
|
|
|
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
|
# pragma GCC system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
|
# pragma clang system_header
|
|
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
|
# pragma system_header
|
|
#endif // no system header
|
|
#include <thrust/detail/execution_policy.h>
|
|
|
|
THRUST_NAMESPACE_BEGIN
|
|
|
|
/*! \addtogroup filling
|
|
* \ingroup transformations
|
|
* \{
|
|
*/
|
|
|
|
/*! \p fill assigns the value \p value to every element in
|
|
* the range <tt>[first, last)</tt>. That is, for every
|
|
* iterator \c i in <tt>[first, last)</tt>, it performs
|
|
* the assignment <tt>*i = value</tt>.
|
|
*
|
|
* The algorithm's execution is parallelized as determined by \p exec.
|
|
*
|
|
* \param exec The execution policy to use for parallelization.
|
|
* \param first The beginning of the sequence.
|
|
* \param last The end of the sequence.
|
|
* \param value The value to be copied.
|
|
*
|
|
* \tparam DerivedPolicy The name of the derived execution policy.
|
|
* \tparam ForwardIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
|
* Iterator</a>, and \p ForwardIterator is mutable. \tparam T is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
|
|
* convertible to \p ForwardIterator's \c value_type.
|
|
*
|
|
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
|
|
* elements to a given value using the \p thrust::device execution policy for parallelization:
|
|
*
|
|
* \code
|
|
* #include <thrust/fill.h>
|
|
* #include <thrust/device_vector.h>
|
|
* #include <thrust/execution_policy.h>
|
|
* ...
|
|
* thrust::device_vector<int> v(4);
|
|
* thrust::fill(thrust::device, v.begin(), v.end(), 137);
|
|
*
|
|
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/fill
|
|
* \see \c fill_n
|
|
* \see \c uninitialized_fill
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename DerivedPolicy, typename ForwardIterator, typename T>
|
|
_CCCL_HOST_DEVICE void
|
|
fill(const thrust::detail::execution_policy_base<DerivedPolicy>& exec,
|
|
ForwardIterator first,
|
|
ForwardIterator last,
|
|
const T& value);
|
|
|
|
/*! \p fill assigns the value \p value to every element in
|
|
* the range <tt>[first, last)</tt>. That is, for every
|
|
* iterator \c i in <tt>[first, last)</tt>, it performs
|
|
* the assignment <tt>*i = value</tt>.
|
|
*
|
|
* \param first The beginning of the sequence.
|
|
* \param last The end of the sequence.
|
|
* \param value The value to be copied.
|
|
*
|
|
* \tparam ForwardIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/forward_iterator">Forward
|
|
* Iterator</a>, and \p ForwardIterator is mutable. \tparam T is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
|
|
* convertible to \p ForwardIterator's \c value_type.
|
|
*
|
|
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
|
|
* elements to a given value.
|
|
*
|
|
* \code
|
|
* #include <thrust/fill.h>
|
|
* #include <thrust/device_vector.h>
|
|
* ...
|
|
* thrust::device_vector<int> v(4);
|
|
* thrust::fill(v.begin(), v.end(), 137);
|
|
*
|
|
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/fill
|
|
* \see \c fill_n
|
|
* \see \c uninitialized_fill
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename ForwardIterator, typename T>
|
|
_CCCL_HOST_DEVICE void fill(ForwardIterator first, ForwardIterator last, const T& value);
|
|
|
|
/*! \p fill_n assigns the value \p value to every element in
|
|
* the range <tt>[first, first+n)</tt>. That is, for every
|
|
* iterator \c i in <tt>[first, first+n)</tt>, it performs
|
|
* the assignment <tt>*i = value</tt>.
|
|
*
|
|
* The algorithm's execution is parallelized as determined by \p exec.
|
|
*
|
|
* \param exec The execution policy to use for parallelization.
|
|
* \param first The beginning of the sequence.
|
|
* \param n The size of the sequence.
|
|
* \param value The value to be copied.
|
|
* \return <tt>first + n</tt>
|
|
*
|
|
* \tparam DerivedPolicy The name of the derived execution policy.
|
|
* \tparam OutputIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output
|
|
* Iterator</a>. \tparam T is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
|
|
* convertible to a type in \p OutputIterator's set of \c value_type.
|
|
*
|
|
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
|
|
* elements to a given value using the \p thrust::device execution policy for parallelization:
|
|
*
|
|
* \code
|
|
* #include <thrust/fill.h>
|
|
* #include <thrust/device_vector.h>
|
|
* #include <thrust/execution_policy.h>
|
|
* ...
|
|
* thrust::device_vector<int> v(4);
|
|
* thrust::fill_n(thrust::device, v.begin(), v.size(), 137);
|
|
*
|
|
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/fill_n
|
|
* \see \c fill
|
|
* \see \c uninitialized_fill_n
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename DerivedPolicy, typename OutputIterator, typename Size, typename T>
|
|
_CCCL_HOST_DEVICE OutputIterator
|
|
fill_n(const thrust::detail::execution_policy_base<DerivedPolicy>& exec, OutputIterator first, Size n, const T& value);
|
|
|
|
/*! \p fill_n assigns the value \p value to every element in
|
|
* the range <tt>[first, first+n)</tt>. That is, for every
|
|
* iterator \c i in <tt>[first, first+n)</tt>, it performs
|
|
* the assignment <tt>*i = value</tt>.
|
|
*
|
|
* \param first The beginning of the sequence.
|
|
* \param n The size of the sequence.
|
|
* \param value The value to be copied.
|
|
* \return <tt>first + n</tt>
|
|
*
|
|
* \tparam OutputIterator is a model of <a href="https://en.cppreference.com/w/cpp/iterator/output_iterator">Output
|
|
* Iterator</a>. \tparam T is a model of <a
|
|
* href="https://en.cppreference.com/w/cpp/named_req/CopyAssignable">Assignable</a>, and \p T's \c value_type is
|
|
* convertible to a type in \p OutputIterator's set of \c value_type.
|
|
*
|
|
* The following code snippet demonstrates how to use \p fill to set a thrust::device_vector's
|
|
* elements to a given value.
|
|
*
|
|
* \code
|
|
* #include <thrust/fill.h>
|
|
* #include <thrust/device_vector.h>
|
|
* ...
|
|
* thrust::device_vector<int> v(4);
|
|
* thrust::fill_n(v.begin(), v.size(), 137);
|
|
*
|
|
* // v[0] == 137, v[1] == 137, v[2] == 137, v[3] == 137
|
|
* \endcode
|
|
*
|
|
* \see https://en.cppreference.com/w/cpp/algorithm/fill_n
|
|
* \see \c fill
|
|
* \see \c uninitialized_fill_n
|
|
*
|
|
* \verbatim embed:rst:leading-asterisk
|
|
* .. versionadded:: 2.2.0
|
|
* \endverbatim
|
|
*/
|
|
template <typename OutputIterator, typename Size, typename T>
|
|
_CCCL_HOST_DEVICE OutputIterator fill_n(OutputIterator first, Size n, const T& value);
|
|
|
|
/*!
|
|
* \} end group filling
|
|
*/
|
|
|
|
THRUST_NAMESPACE_END
|
|
|
|
#include <thrust/detail/fill.inl>
|