[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
114
cccl_upstream/libcudacxx/include/cuda/__stream/get_stream.h
Normal file
114
cccl_upstream/libcudacxx/include/cuda/__stream/get_stream.h
Normal file
@@ -0,0 +1,114 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___STREAM_GET_STREAM_H
|
||||
#define _CUDA___STREAM_GET_STREAM_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__fwd/get_stream.h>
|
||||
# include <cuda/__stream/stream_ref.h>
|
||||
# include <cuda/std/__concepts/concept_macros.h>
|
||||
# include <cuda/std/__concepts/convertible_to.h>
|
||||
# include <cuda/std/__execution/env.h>
|
||||
# include <cuda/std/__type_traits/is_convertible.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
class stream_ref;
|
||||
|
||||
template <class _Tp>
|
||||
_CCCL_CONCEPT __convertible_to_stream_ref = ::cuda::std::convertible_to<_Tp, ::cuda::stream_ref>;
|
||||
|
||||
template <class _Tp>
|
||||
_CCCL_CONCEPT __has_member_stream = _CCCL_REQUIRES_EXPR((_Tp), const _Tp& __t)(
|
||||
requires(!__convertible_to_stream_ref<_Tp>), //
|
||||
requires(__convertible_to_stream_ref<decltype(__t.stream())>));
|
||||
|
||||
template <class _Tp>
|
||||
_CCCL_CONCEPT __has_member_get_stream = _CCCL_REQUIRES_EXPR((_Tp), const _Tp& __t)(
|
||||
requires(!__convertible_to_stream_ref<_Tp>), //
|
||||
requires(__convertible_to_stream_ref<decltype(__t.get_stream())>));
|
||||
|
||||
template <class _Env>
|
||||
_CCCL_CONCEPT __has_query_get_stream = _CCCL_REQUIRES_EXPR((_Env), const _Env& __env, const get_stream_t& __cpo)(
|
||||
requires(!__convertible_to_stream_ref<_Env>),
|
||||
requires(!__has_member_stream<_Env>),
|
||||
requires(__convertible_to_stream_ref<decltype(__env.query(__cpo))>));
|
||||
|
||||
//! @brief `get_stream` is a customization point object that queries a type `T` for an associated stream
|
||||
struct get_stream_t
|
||||
{
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(::cudaStream_t __stream) const noexcept
|
||||
{
|
||||
return ::cuda::stream_ref{__stream};
|
||||
}
|
||||
|
||||
_CCCL_EXEC_CHECK_DISABLE
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(__convertible_to_stream_ref<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const
|
||||
noexcept(noexcept(static_cast<::cuda::stream_ref>(__t)))
|
||||
{
|
||||
return static_cast<::cuda::stream_ref>(__t);
|
||||
}
|
||||
|
||||
_CCCL_EXEC_CHECK_DISABLE
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(__has_member_stream<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const noexcept(noexcept(__t.stream()))
|
||||
{
|
||||
return __t.stream();
|
||||
}
|
||||
|
||||
_CCCL_EXEC_CHECK_DISABLE
|
||||
_CCCL_TEMPLATE(class _Tp)
|
||||
_CCCL_REQUIRES(__has_member_get_stream<_Tp>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const
|
||||
noexcept(noexcept(__t.get_stream()))
|
||||
{
|
||||
return __t.get_stream();
|
||||
}
|
||||
|
||||
_CCCL_EXEC_CHECK_DISABLE
|
||||
_CCCL_TEMPLATE(class _Env)
|
||||
_CCCL_REQUIRES(__has_query_get_stream<_Env>)
|
||||
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Env& __env) const noexcept
|
||||
{
|
||||
static_assert(noexcept(__env.query(*this)));
|
||||
return __env.query(*this);
|
||||
}
|
||||
|
||||
[[nodiscard]] _CCCL_API static constexpr auto query(::cuda::std::execution::forwarding_query_t) noexcept -> bool
|
||||
{
|
||||
return true;
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_GLOBAL_CONSTANT auto get_stream = get_stream_t{};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___STREAM_GET_STREAM_H
|
||||
@@ -0,0 +1,59 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___STREAM_INTERNAL_STREAMS_H
|
||||
#define _CUDA___STREAM_INTERNAL_STREAMS_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__stream/stream.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
// We make __cccl_allocation_stream() noexcept because the only way it could potentially fail
|
||||
// is e.g. bad driver state or some other deeper corruption so we are pretty much in an
|
||||
// unusable state anyways.
|
||||
|
||||
// NOLINTBEGIN(bugprone-exception-escape)
|
||||
|
||||
//! @brief internal stream used for memory allocations, no real blocking work
|
||||
//! should ever be pushed into it
|
||||
inline ::cuda::stream_ref __cccl_allocation_stream() noexcept
|
||||
{
|
||||
// Intentionally leak the stream here to avoid stream destruction when the program exits, which is not guaraneed to
|
||||
// work.
|
||||
static ::cuda::stream_ref __stream = []() {
|
||||
::cuda::stream __str{::cuda::device_ref{0}};
|
||||
return __str.release();
|
||||
}();
|
||||
return __stream;
|
||||
}
|
||||
|
||||
// NOLINTEND(bugprone-exception-escape)
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA___STREAM_INTERNAL_STREAMS_H
|
||||
@@ -0,0 +1,47 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___STREAM_INVALID_STREAM_H
|
||||
#define _CUDA___STREAM_INVALID_STREAM_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
enum class invalid_stream_t : unsigned char
|
||||
{
|
||||
};
|
||||
|
||||
_CCCL_GLOBAL_CONSTANT invalid_stream_t invalid_stream{};
|
||||
|
||||
[[nodiscard]] _CCCL_API _CCCL_FORCEINLINE ::cudaStream_t __invalid_stream() noexcept
|
||||
{
|
||||
return reinterpret_cast<::cudaStream_t>(~0ull); // NOLINT(performance-no-int-to-ptr)
|
||||
}
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif //_CUDA___STREAM_INVALID_STREAM_H
|
||||
@@ -0,0 +1,203 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA__STREAM_LAUNCH_TRANSFORM_H
|
||||
#define _CUDA__STREAM_LAUNCH_TRANSFORM_H
|
||||
|
||||
#include <cuda/__cccl_config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK()
|
||||
|
||||
# include <cuda/__stream/stream_ref.h>
|
||||
# include <cuda/__type_traits/is_instantiable_with.h>
|
||||
# include <cuda/std/__memory/addressof.h>
|
||||
# include <cuda/std/__memory/construct_at.h>
|
||||
# include <cuda/std/__new/launder.h>
|
||||
# include <cuda/std/__optional/optional.h>
|
||||
# include <cuda/std/__tuple_dir/ignore.h>
|
||||
# include <cuda/std/__type_traits/decay.h>
|
||||
# include <cuda/std/__type_traits/is_callable.h>
|
||||
# include <cuda/std/__type_traits/is_reference.h>
|
||||
# include <cuda/std/__utility/declval.h>
|
||||
# include <cuda/std/__utility/forward.h>
|
||||
# include <cuda/std/__utility/move.h>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
namespace __detail
|
||||
{
|
||||
// This function turns rvalues into prvalues and leaves lvalues as is.
|
||||
template <typename _Tp>
|
||||
_CCCL_API constexpr auto __ixnay_xvalue(_Tp&& __value) noexcept(::cuda::std::is_nothrow_move_constructible_v<_Tp>)
|
||||
-> _Tp
|
||||
{
|
||||
return ::cuda::std::forward<_Tp>(__value);
|
||||
}
|
||||
} // namespace __detail
|
||||
|
||||
template <typename _Tp>
|
||||
using __remove_rvalue_reference_t =
|
||||
decltype(__detail::__ixnay_xvalue(::cuda::std::declval<_Tp>()) // NOLINT(modernize-type-traits)
|
||||
);
|
||||
|
||||
namespace __tfx
|
||||
{
|
||||
// Launch transform:
|
||||
//
|
||||
// The launch transform is a mechanism to transform arguments passed to the
|
||||
// algorithms prior to actually enqueueing work on a stream. This is useful for
|
||||
// example, to automatically convert contiguous ranges into spans. It is also
|
||||
// useful for executing per-argument actions before and after the kernel launch.
|
||||
// A host_vector might want a pre-launch action to copy data from host to device
|
||||
// and a post-launch action to copy data back from device to host.
|
||||
//
|
||||
// The expression `launch_transform(stream, arg)` is expression-equivalent to
|
||||
// the first of the following expressions that is valid:
|
||||
//
|
||||
// 1. `transform_launch_argument(stream, arg).transformed_argument()`
|
||||
// 2. `transform_launch_argument(stream, arg)`
|
||||
// 3. `arg.transformed_argument()`
|
||||
// 4. `arg`
|
||||
_CCCL_HOST_API void transform_launch_argument();
|
||||
|
||||
struct _CCCL_TYPE_VISIBILITY_DEFAULT __launch_transform_t
|
||||
{
|
||||
// Types that want to customize `launch_transform` should define overloads of
|
||||
// transform_launch_argument that are find-able by ADL.
|
||||
template <typename _Arg>
|
||||
using __transform_result_t = __remove_rvalue_reference_t<decltype(transform_launch_argument(
|
||||
::cuda::stream_ref{::cudaStream_t{}}, ::cuda::std::declval<_Arg>()))>;
|
||||
|
||||
template <typename _Arg>
|
||||
using __transformed_argument_t =
|
||||
__remove_rvalue_reference_t<decltype(::cuda::std::declval<_Arg>().transformed_argument())>;
|
||||
|
||||
// The use of `optional` here is to move the destruction of the object returned from
|
||||
// transform_launch_argument into the caller's stack frame. Objects created for default arguments
|
||||
// are located in the caller's stack frame. This is so that a use of `launch_transform`
|
||||
// such as:
|
||||
//
|
||||
// kernel<<<grid, block, 0, stream>>>(launch_transform(stream, arg));
|
||||
//
|
||||
// is equivalent to:
|
||||
//
|
||||
// kernel<<<grid, block, 0, stream>>>(transform_launch_argument(stream, arg).transformed_argument());
|
||||
//
|
||||
// where the object returned from `transform_launch_argument` is destroyed *after* the kernel
|
||||
// launch.
|
||||
//
|
||||
// What I really wanted to do was:
|
||||
//
|
||||
// template <typename Arg>
|
||||
// auto operator()(::cuda::stream_ref stream, Arg&& arg, auto&& action = transform_launch_argument(arg))
|
||||
//
|
||||
// but sadly that is not valid C++.
|
||||
// TODO move to use __variant type once cuda/experimental/execution/__variant is moved to libcudacxx
|
||||
// NOTE: The above seems to only apply if the type is not trivially destructible. To use the optional here I had to
|
||||
// add a destructor.
|
||||
template <typename _Tp>
|
||||
struct __optional_with_a_destructor : ::cuda::std::optional<_Tp>
|
||||
{
|
||||
using ::cuda::std::optional<_Tp>::optional;
|
||||
// Use of explicit destructor is intentional. Without it, the argument may have a trivial
|
||||
// destructor and hence would be performed by the callee.
|
||||
~__optional_with_a_destructor() {} // NOLINT(modernize-use-equals-default)
|
||||
|
||||
template <class _Fn>
|
||||
_CCCL_API inline _CCCL_CONSTEXPR_CXX20 _Tp& __emplace_from_fn(_Fn&& __fn)
|
||||
{
|
||||
_CCCL_ASSERT(!this->has_value(), "__construct called for engaged __optional_storage");
|
||||
new (::cuda::std::addressof(this->__get())) _Tp(::cuda::std::invoke(::cuda::std::forward<_Fn>(__fn)));
|
||||
this->__set_engaged(true);
|
||||
return this->__get();
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_TEMPLATE(typename _Stream, typename _Arg)
|
||||
_CCCL_REQUIRES(::cuda::std::convertible_to<_Stream, ::cuda::stream_ref> _CCCL_AND(
|
||||
!::cuda::std::is_reference_v<__transform_result_t<_Arg>>))
|
||||
[[nodiscard]] _CCCL_HOST_API auto operator()(
|
||||
_Stream&& __stream,
|
||||
_Arg&& __arg,
|
||||
__optional_with_a_destructor<__transform_result_t<_Arg>> __storage = cuda::std::nullopt) const -> decltype(auto)
|
||||
{
|
||||
// Calls to transform_launch_argument are intentionally unqualified so as to use ADL.
|
||||
if constexpr (__is_instantiable_with<__transformed_argument_t, __transform_result_t<_Arg>>)
|
||||
{
|
||||
return _CCCL_MOVE(__storage.__emplace_from_fn([&]() {
|
||||
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
|
||||
}))
|
||||
.transformed_argument();
|
||||
}
|
||||
else
|
||||
{
|
||||
return _CCCL_MOVE(__storage.__emplace_from_fn([&]() {
|
||||
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
|
||||
}));
|
||||
}
|
||||
}
|
||||
|
||||
// If transform_launch_argument returns a reference type, then there are no pre- and
|
||||
// post-launch actions. (References types don't have ctors/dtors.) There is no need to
|
||||
// store the result of transform_launch_argument.
|
||||
_CCCL_TEMPLATE(typename _Stream, typename _Arg)
|
||||
_CCCL_REQUIRES(::cuda::std::convertible_to<_Stream, ::cuda::stream_ref>
|
||||
_CCCL_AND ::cuda::std::is_reference_v<__transform_result_t<_Arg>>)
|
||||
[[nodiscard]] _CCCL_HOST_API auto operator()(_Stream&& __stream, _Arg&& __arg) const -> decltype(auto)
|
||||
{
|
||||
// Calls to transform_launch_argument are intentionally unqualified so as to use ADL.
|
||||
if constexpr (__is_instantiable_with<__transformed_argument_t, __transform_result_t<_Arg>>)
|
||||
{
|
||||
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg)).transformed_argument();
|
||||
}
|
||||
else
|
||||
{
|
||||
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
|
||||
}
|
||||
}
|
||||
|
||||
template <typename _Arg>
|
||||
[[nodiscard]] _CCCL_HOST_API auto operator()(::cuda::std::__ignore_t, _Arg&& __arg) const -> decltype(auto)
|
||||
{
|
||||
if constexpr (__is_instantiable_with<__transformed_argument_t, _Arg>)
|
||||
{
|
||||
return ::cuda::std::forward<_Arg>(__arg).transformed_argument();
|
||||
}
|
||||
else
|
||||
{
|
||||
return static_cast<_Arg>(::cuda::std::forward<_Arg>(__arg));
|
||||
}
|
||||
}
|
||||
};
|
||||
} // namespace __tfx
|
||||
|
||||
_CCCL_GLOBAL_CONSTANT auto launch_transform = __tfx::__launch_transform_t{};
|
||||
|
||||
# ifndef _CCCL_DOXYGEN_INVOKED // Doxygen chokes here
|
||||
template <typename _Arg>
|
||||
using transformed_device_argument_t _CCCL_NODEBUG_ALIAS =
|
||||
__remove_rvalue_reference_t<::cuda::std::__call_result_t<__tfx::__launch_transform_t, ::cuda::stream_ref, _Arg>>;
|
||||
# endif // ^^^ _CCCL_DOXYGEN_INVOKED ^^^
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK()
|
||||
|
||||
#endif // _CUDA__STREAM_LAUNCH_TRANSFORM_H
|
||||
145
cccl_upstream/libcudacxx/include/cuda/__stream/stream.h
Normal file
145
cccl_upstream/libcudacxx/include/cuda/__stream/stream.h
Normal file
@@ -0,0 +1,145 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___STREAM_STREAM_H
|
||||
#define _CUDA___STREAM_STREAM_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
# include <cuda/__device/device_ref.h>
|
||||
# include <cuda/__driver/driver_api.h>
|
||||
# include <cuda/__runtime/ensure_current_context.h>
|
||||
# include <cuda/__stream/invalid_stream.h>
|
||||
# include <cuda/__stream/stream_ref.h> // IWYU pragma: export
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief An owning wrapper for cudaStream_t.
|
||||
struct stream : stream_ref
|
||||
{
|
||||
// 0 is documented as default priority
|
||||
static constexpr int default_priority = 0;
|
||||
|
||||
//! @brief Constructs a stream on a specified device and with specified priority
|
||||
//!
|
||||
//! Priority is defaulted to stream::default_priority
|
||||
//!
|
||||
//! @throws cuda_error if stream creation fails
|
||||
_CCCL_HOST_API explicit stream(device_ref __dev, int __priority = default_priority)
|
||||
: stream_ref(::cuda::__invalid_stream())
|
||||
{
|
||||
[[maybe_unused]] __ensure_current_context __ctx_setter(__dev);
|
||||
__stream = ::cuda::__driver::__streamCreateWithPriority(cudaStreamNonBlocking, __priority);
|
||||
}
|
||||
|
||||
//! @brief Construct a new `stream` object into the moved-from state.
|
||||
//!
|
||||
//! @post `stream()` returns an invalid stream handle
|
||||
// Can't be constexpr because __invalid_stream isn't
|
||||
_CCCL_HOST_API explicit stream(no_init_t) noexcept
|
||||
: stream_ref(::cuda::__invalid_stream())
|
||||
{}
|
||||
|
||||
//! @brief Move-construct a new `stream` object
|
||||
//!
|
||||
//! @param __other
|
||||
//!
|
||||
//! @post `__other` is in moved-from state.
|
||||
_CCCL_HOST_API stream(stream&& __other) noexcept
|
||||
: stream(::cuda::std::exchange(__other.__stream, ::cuda::__invalid_stream()))
|
||||
{}
|
||||
|
||||
stream(const stream&) = delete;
|
||||
|
||||
//! Destroy the `stream` object
|
||||
//!
|
||||
//! @note If the stream fails to be destroyed, the error is silently ignored.
|
||||
_CCCL_HOST_API ~stream()
|
||||
{
|
||||
if (__stream != ::cuda::__invalid_stream())
|
||||
{
|
||||
// Needs to call driver API in case current device is not set, runtime version would set dev 0 current
|
||||
// Alternative would be to store the device and push/pop here
|
||||
[[maybe_unused]] auto status = ::cuda::__driver::__streamDestroyNoThrow(__stream);
|
||||
}
|
||||
}
|
||||
|
||||
//! @brief Move-assign a `stream` object
|
||||
//!
|
||||
//! @param __other
|
||||
//!
|
||||
//! @post `__other` is in a moved-from state.
|
||||
_CCCL_HOST_API stream& operator=(stream&& __other) noexcept
|
||||
{
|
||||
stream __tmp(::cuda::std::move(__other));
|
||||
::cuda::std::swap(__stream, __tmp.__stream);
|
||||
return *this;
|
||||
}
|
||||
|
||||
stream& operator=(const stream&) = delete;
|
||||
|
||||
//! @brief Construct an `stream` object from a native `cudaStream_t` handle.
|
||||
//!
|
||||
//! @param __handle The native handle
|
||||
//!
|
||||
//! @return stream The constructed `stream` object
|
||||
//!
|
||||
//! @note The constructed `stream` object takes ownership of the native handle.
|
||||
[[nodiscard]] static _CCCL_HOST_API stream from_native_handle(::cudaStream_t __handle)
|
||||
{
|
||||
return stream(__handle);
|
||||
}
|
||||
|
||||
// Disallow construction from an `int`, e.g., `0`.
|
||||
static stream from_native_handle(int) = delete;
|
||||
|
||||
// Disallow construction from `nullptr`.
|
||||
static stream from_native_handle(::cuda::std::nullptr_t) = delete;
|
||||
|
||||
// Disallow construction from `invalid_stream_t`.
|
||||
static stream from_native_handle(invalid_stream_t) = delete;
|
||||
|
||||
//! @brief Retrieve the native `cudaStream_t` handle and give up ownership.
|
||||
//!
|
||||
//! @return cudaStream_t The native handle being held by the `stream` object.
|
||||
//!
|
||||
//! @post The stream object is in a moved-from state.
|
||||
[[nodiscard]] _CCCL_HOST_API ::cudaStream_t release()
|
||||
{
|
||||
return ::cuda::std::exchange(__stream, ::cuda::__invalid_stream());
|
||||
}
|
||||
|
||||
private:
|
||||
// Use `stream::from_native_handle(s)` to construct an owning `stream`
|
||||
// object from a `cudaStream_t` handle.
|
||||
_CCCL_HOST_API explicit stream(::cudaStream_t __handle)
|
||||
: stream_ref(__handle)
|
||||
{}
|
||||
};
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
#endif // _CUDA___STREAM_STREAM_H
|
||||
356
cccl_upstream/libcudacxx/include/cuda/__stream/stream_ref.h
Normal file
356
cccl_upstream/libcudacxx/include/cuda/__stream/stream_ref.h
Normal file
@@ -0,0 +1,356 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA___STREAM_STREAM_REF_H
|
||||
#define _CUDA___STREAM_STREAM_REF_H
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
# include <cuda/__device/device_ref.h>
|
||||
# include <cuda/__driver/driver_api.h>
|
||||
# include <cuda/__event/timed_event.h>
|
||||
# include <cuda/__fwd/get_stream.h>
|
||||
# include <cuda/__runtime/ensure_current_context.h>
|
||||
# include <cuda/__stream/invalid_stream.h>
|
||||
# include <cuda/__utility/no_init.h>
|
||||
# include <cuda/std/__exception/cuda_error.h>
|
||||
# include <cuda/std/__exception/exception_macros.h>
|
||||
# include <cuda/std/__utility/to_underlying.h>
|
||||
# include <cuda/std/cstddef>
|
||||
|
||||
# include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA
|
||||
|
||||
//! @brief A type representing a stream ID.
|
||||
enum class stream_id : unsigned long long
|
||||
{
|
||||
};
|
||||
|
||||
//! @brief A non-owning wrapper for a `cudaStream_t`.
|
||||
class stream_ref
|
||||
{
|
||||
protected:
|
||||
::cudaStream_t __stream{nullptr};
|
||||
|
||||
public:
|
||||
using value_type = ::cudaStream_t;
|
||||
|
||||
//! @brief Constructs a `stream_ref` of the "default" CUDA stream.
|
||||
//!
|
||||
//! For behavior of the default stream,
|
||||
//! @see //! https://docs.nvidia.com/cuda/cuda-runtime-api/stream-sync-behavior.html
|
||||
CCCL_DEPRECATED_BECAUSE("Using the default/null stream is generally discouraged. If you need to use it, please "
|
||||
"construct a "
|
||||
"stream_ref from cudaStream_t{nullptr}") _CCCL_HIDE_FROM_ABI
|
||||
stream_ref() = default;
|
||||
|
||||
//! @brief Constructs a `stream_ref` from a `cudaStream_t` handle.
|
||||
//!
|
||||
//! This constructor provides implicit conversion from `cudaStream_t`.
|
||||
//!
|
||||
//! @note: It is the callers responsibility to ensure the `stream_ref` does not
|
||||
//! outlive the stream identified by the `cudaStream_t` handle.
|
||||
_CCCL_API constexpr stream_ref(value_type __stream_) noexcept
|
||||
: __stream{__stream_}
|
||||
{}
|
||||
|
||||
//! @brief Constructs a `stream_ref` from the `cuda::invalid_stream_t`.
|
||||
//!
|
||||
//! @note Any CUDA APIs called on the created object will result in an CUDA error.
|
||||
_CCCL_API explicit stream_ref(invalid_stream_t) noexcept
|
||||
: __stream{::cuda::__invalid_stream()}
|
||||
{}
|
||||
|
||||
//! Disallow construction from an `int`, e.g., `0`.
|
||||
stream_ref(int) = delete;
|
||||
|
||||
//! Disallow construction from `nullptr`.
|
||||
stream_ref(::cuda::std::nullptr_t) = delete;
|
||||
|
||||
//! @brief Compares two `stream_ref`s for equality
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __lhs The first `stream_ref` to compare
|
||||
//! @param __rhs The second `stream_ref` to compare
|
||||
//! @return true if equal, false if unequal
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool operator==(const stream_ref& __lhs, const stream_ref& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__stream == __rhs.__stream;
|
||||
}
|
||||
|
||||
//! @brief Compares `stream_ref` with `invalid_stream_t` for equality.
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __lhs The `stream_ref` to compare
|
||||
//! @return true if equal, false if unequal
|
||||
[[nodiscard]] _CCCL_API friend bool operator==(const stream_ref& __lhs, const invalid_stream_t&) noexcept
|
||||
{
|
||||
return __lhs.__stream == ::cuda::__invalid_stream();
|
||||
}
|
||||
|
||||
//! @brief Compares `invalid_stream_t` with `stream_ref` for equality.
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __rhs The `stream_ref` to compare
|
||||
//! @return true if equal, false if unequal
|
||||
[[nodiscard]] _CCCL_API friend bool operator==(const invalid_stream_t&, const stream_ref& __rhs) noexcept
|
||||
{
|
||||
return ::cuda::__invalid_stream() == __rhs.__stream;
|
||||
}
|
||||
|
||||
//! @brief Compares two `stream_ref`s for inequality
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __lhs The first `stream_ref` to compare
|
||||
//! @param __rhs The second `stream_ref` to compare
|
||||
//! @return true if unequal, false if equal
|
||||
[[nodiscard]] _CCCL_API friend constexpr bool operator!=(const stream_ref& __lhs, const stream_ref& __rhs) noexcept
|
||||
{
|
||||
return __lhs.__stream != __rhs.__stream;
|
||||
}
|
||||
|
||||
//! @brief Compares `stream_ref` with `invalid_stream_t` for inequality.
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __lhs The `stream_ref` to compare
|
||||
//! @return false if equal, true if unequal
|
||||
[[nodiscard]] _CCCL_API friend bool operator!=(const stream_ref& __lhs, const invalid_stream_t&) noexcept
|
||||
{
|
||||
return __lhs.__stream != ::cuda::__invalid_stream();
|
||||
}
|
||||
|
||||
//! @brief Compares `invalid_stream_t` with `stream_ref` for inequality.
|
||||
//!
|
||||
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
|
||||
//! `stream_ref`.
|
||||
//!
|
||||
//! @param __rhs The `stream_ref` to compare
|
||||
//! @return false if equal, true if unequal
|
||||
[[nodiscard]] _CCCL_API friend bool operator!=(const invalid_stream_t&, const stream_ref& __rhs) noexcept
|
||||
{
|
||||
return ::cuda::__invalid_stream() != __rhs.__stream;
|
||||
}
|
||||
|
||||
//! Returns the wrapped `cudaStream_t` handle.
|
||||
[[nodiscard]] _CCCL_API constexpr value_type get() const noexcept
|
||||
{
|
||||
return __stream;
|
||||
}
|
||||
|
||||
//! @brief Synchronizes the wrapped stream.
|
||||
//!
|
||||
//! @throws cuda::cuda_error if synchronization fails.
|
||||
_CCCL_HOST_API void sync() const
|
||||
{
|
||||
::cuda::__driver::__streamSynchronize(__stream);
|
||||
}
|
||||
|
||||
//! @brief Deprecated. Use sync() instead.
|
||||
//!
|
||||
//! @deprecated Use sync() instead.
|
||||
CCCL_DEPRECATED_BECAUSE("Use sync() instead.") _CCCL_HOST_API void wait() const
|
||||
{
|
||||
sync();
|
||||
}
|
||||
|
||||
//! @brief Make all future work submitted into this stream depend on completion of the specified event
|
||||
//!
|
||||
//! @param __ev Event that this stream should wait for
|
||||
//!
|
||||
//! @throws cuda_error if inserting the dependency fails
|
||||
_CCCL_HOST_API void wait(event_ref __ev) const
|
||||
{
|
||||
_CCCL_ASSERT(__ev.get() != nullptr, "cuda::stream_ref::wait invalid event passed");
|
||||
// Need to use driver API, cudaStreamWaitEvent would push dev 0 if stack was empty
|
||||
::cuda::__driver::__streamWaitEvent(get(), __ev.get());
|
||||
}
|
||||
|
||||
//! @brief Make all future work submitted into this stream depend on completion of all work from the specified
|
||||
//! stream
|
||||
//!
|
||||
//! @param __other Stream that this stream should wait for
|
||||
//!
|
||||
//! @throws cuda_error if inserting the dependency fails
|
||||
_CCCL_HOST_API void wait(stream_ref __other) const
|
||||
{
|
||||
// TODO consider an optimization to not create an event every time and instead have one persistent event or one
|
||||
// per stream
|
||||
_CCCL_ASSERT(__stream != ::cuda::__invalid_stream(), "cuda::stream_ref::wait invalid stream passed");
|
||||
if (*this != __other)
|
||||
{
|
||||
event __tmp(__other);
|
||||
wait(__tmp);
|
||||
}
|
||||
}
|
||||
|
||||
//! \brief Queries if all operations on the stream have completed.
|
||||
//!
|
||||
//! \throws cuda::cuda_error if the query fails.
|
||||
//!
|
||||
//! \return `true` if all operations have completed, or `false` if not.
|
||||
[[nodiscard]] _CCCL_HOST_API bool is_done() const
|
||||
{
|
||||
const auto __result = ::cuda::__driver::__streamQueryNoThrow(__stream);
|
||||
switch (__result)
|
||||
{
|
||||
case ::cudaErrorNotReady:
|
||||
return false;
|
||||
case ::cudaSuccess:
|
||||
return true;
|
||||
default:
|
||||
_CCCL_THROW(::cuda::cuda_error, __result, "Failed to query stream.");
|
||||
}
|
||||
}
|
||||
|
||||
//! @brief Queries if all operations on the wrapped stream have completed.
|
||||
//!
|
||||
//! @throws cuda::cuda_error if the query fails.
|
||||
//!
|
||||
//! @return `true` if all operations have completed, or `false` if not.
|
||||
[[nodiscard]] CCCL_DEPRECATED_BECAUSE("Use is_done() instead.") _CCCL_HOST_API bool ready() const
|
||||
{
|
||||
return is_done();
|
||||
}
|
||||
|
||||
//! @brief Queries the priority of the wrapped stream.
|
||||
//!
|
||||
//! @throws cuda::cuda_error if the query fails.
|
||||
//!
|
||||
//! @return value representing the priority of the wrapped stream.
|
||||
[[nodiscard]] _CCCL_HOST_API int priority() const
|
||||
{
|
||||
return ::cuda::__driver::__streamGetPriority(__stream);
|
||||
}
|
||||
|
||||
//! @brief Get the unique ID of the stream
|
||||
//!
|
||||
//! Stream handles are sometimes reused, but ID is guaranteed to be unique.
|
||||
//!
|
||||
//! @return The unique ID of the stream
|
||||
//!
|
||||
//! @throws cuda_error if the ID query fails
|
||||
[[nodiscard]] _CCCL_HOST_API stream_id id() const
|
||||
{
|
||||
return stream_id{::cuda::__driver::__streamGetId(__stream)};
|
||||
}
|
||||
|
||||
//! @brief Create a new event and record it into this stream
|
||||
//!
|
||||
//! @return A new event that was recorded into this stream
|
||||
//!
|
||||
//! @throws cuda_error if event creation or record failed
|
||||
[[nodiscard]] _CCCL_HOST_API event record_event(event_flags __flags = event_flags::none) const
|
||||
{
|
||||
return event(*this, __flags);
|
||||
}
|
||||
|
||||
//! @brief Create a new timed event and record it into this stream
|
||||
//!
|
||||
//! @return A new timed event that was recorded into this stream
|
||||
//!
|
||||
//! @throws cuda_error if event creation or record failed
|
||||
[[nodiscard]] _CCCL_HOST_API timed_event record_timed_event(event_flags __flags = event_flags::none) const
|
||||
{
|
||||
return timed_event(*this, __flags);
|
||||
}
|
||||
|
||||
//! @brief Get device under which this stream was created.
|
||||
//!
|
||||
//! Note: In case of a stream created under a `green_context` the device on which that `green_context` was created is
|
||||
//! returned
|
||||
//!
|
||||
//! @throws cuda_error if device check fails
|
||||
[[nodiscard]] _CCCL_HOST_API device_ref device() const
|
||||
{
|
||||
::CUdevice __device{};
|
||||
# if _CCCL_CTK_AT_LEAST(13, 0)
|
||||
__device = ::cuda::__driver::__streamGetDevice(__stream);
|
||||
# else // ^^^ _CCCL_CTK_AT_LEAST(13, 0) ^^^ / vvv _CCCL_CTK_BELOW(13, 0) vvv
|
||||
{
|
||||
::CUcontext __stream_ctx = ::cuda::__driver::__streamGetCtx(__stream);
|
||||
__ensure_current_context __setter(__stream_ctx);
|
||||
__device = ::cuda::__driver::__ctxGetDevice();
|
||||
}
|
||||
# endif // ^^^ _CCCL_CTK_BELOW(13, 0) ^^^
|
||||
return device_ref{::cuda::__driver::__cudevice_to_ordinal(__device)};
|
||||
}
|
||||
|
||||
//! @brief Queries the \c stream_ref for itself. This makes \c stream_ref usable in places where we expect an
|
||||
//! environment with a \c get_stream_t query
|
||||
[[nodiscard]] _CCCL_API constexpr stream_ref query(const ::cuda::get_stream_t&) const noexcept
|
||||
{
|
||||
return *this;
|
||||
}
|
||||
};
|
||||
|
||||
_CCCL_HOST_API inline void event_ref::record(stream_ref __stream) const
|
||||
{
|
||||
_CCCL_ASSERT(__event_ != nullptr, "cuda::event_ref::record no event set");
|
||||
_CCCL_ASSERT(__stream.get() != nullptr, "cuda::event_ref::record invalid stream passed");
|
||||
// Need to use driver API, cudaEventRecord will push dev 0 if stack is empty
|
||||
::cuda::__driver::__eventRecord(__event_, __stream.get());
|
||||
}
|
||||
|
||||
_CCCL_HOST_API inline event::event(stream_ref __stream, event_flags __flags)
|
||||
: event(__stream, ::cuda::std::to_underlying(__flags) | cudaEventDisableTiming)
|
||||
{
|
||||
record(__stream);
|
||||
}
|
||||
|
||||
_CCCL_HOST_API inline event::event(stream_ref __stream, unsigned __flags)
|
||||
: event_ref(::cudaEvent_t{})
|
||||
{
|
||||
[[maybe_unused]] __ensure_current_context __ctx_setter(__stream);
|
||||
__event_ = ::cuda::__driver::__eventCreate(static_cast<unsigned>(__flags));
|
||||
}
|
||||
|
||||
_CCCL_HOST_API inline timed_event::timed_event(stream_ref __stream, event_flags __flags)
|
||||
: event(__stream, ::cuda::std::to_underlying(__flags))
|
||||
{
|
||||
record(__stream);
|
||||
}
|
||||
|
||||
// Hide from Doxygen — __ensure_current_context is an internal symbol excluded by EXCLUDE_SYMBOLS.
|
||||
# ifndef _CCCL_DOXYGEN_INVOKED
|
||||
_CCCL_HOST_API inline __ensure_current_context::__ensure_current_context(stream_ref __stream)
|
||||
{
|
||||
auto __ctx = __driver::__streamGetCtx(__stream.get());
|
||||
::cuda::__driver::__ctxPush(__ctx);
|
||||
}
|
||||
# endif // !_CCCL_DOXYGEN_INVOKED
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA
|
||||
|
||||
# include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
|
||||
|
||||
#endif //_CUDA___STREAM_STREAM_REF_H
|
||||
Reference in New Issue
Block a user