[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,114 @@
//===----------------------------------------------------------------------===//
//
// Part of the libcu++ Project, under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___STREAM_GET_STREAM_H
#define _CUDA___STREAM_GET_STREAM_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__fwd/get_stream.h>
# include <cuda/__stream/stream_ref.h>
# include <cuda/std/__concepts/concept_macros.h>
# include <cuda/std/__concepts/convertible_to.h>
# include <cuda/std/__execution/env.h>
# include <cuda/std/__type_traits/is_convertible.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
class stream_ref;
template <class _Tp>
_CCCL_CONCEPT __convertible_to_stream_ref = ::cuda::std::convertible_to<_Tp, ::cuda::stream_ref>;
template <class _Tp>
_CCCL_CONCEPT __has_member_stream = _CCCL_REQUIRES_EXPR((_Tp), const _Tp& __t)(
requires(!__convertible_to_stream_ref<_Tp>), //
requires(__convertible_to_stream_ref<decltype(__t.stream())>));
template <class _Tp>
_CCCL_CONCEPT __has_member_get_stream = _CCCL_REQUIRES_EXPR((_Tp), const _Tp& __t)(
requires(!__convertible_to_stream_ref<_Tp>), //
requires(__convertible_to_stream_ref<decltype(__t.get_stream())>));
template <class _Env>
_CCCL_CONCEPT __has_query_get_stream = _CCCL_REQUIRES_EXPR((_Env), const _Env& __env, const get_stream_t& __cpo)(
requires(!__convertible_to_stream_ref<_Env>),
requires(!__has_member_stream<_Env>),
requires(__convertible_to_stream_ref<decltype(__env.query(__cpo))>));
//! @brief `get_stream` is a customization point object that queries a type `T` for an associated stream
struct get_stream_t
{
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(::cudaStream_t __stream) const noexcept
{
return ::cuda::stream_ref{__stream};
}
_CCCL_EXEC_CHECK_DISABLE
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(__convertible_to_stream_ref<_Tp>)
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const
noexcept(noexcept(static_cast<::cuda::stream_ref>(__t)))
{
return static_cast<::cuda::stream_ref>(__t);
}
_CCCL_EXEC_CHECK_DISABLE
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(__has_member_stream<_Tp>)
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const noexcept(noexcept(__t.stream()))
{
return __t.stream();
}
_CCCL_EXEC_CHECK_DISABLE
_CCCL_TEMPLATE(class _Tp)
_CCCL_REQUIRES(__has_member_get_stream<_Tp>)
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Tp& __t) const
noexcept(noexcept(__t.get_stream()))
{
return __t.get_stream();
}
_CCCL_EXEC_CHECK_DISABLE
_CCCL_TEMPLATE(class _Env)
_CCCL_REQUIRES(__has_query_get_stream<_Env>)
[[nodiscard]] _CCCL_API constexpr ::cuda::stream_ref operator()(const _Env& __env) const noexcept
{
static_assert(noexcept(__env.query(*this)));
return __env.query(*this);
}
[[nodiscard]] _CCCL_API static constexpr auto query(::cuda::std::execution::forwarding_query_t) noexcept -> bool
{
return true;
}
};
_CCCL_GLOBAL_CONSTANT auto get_stream = get_stream_t{};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___STREAM_GET_STREAM_H

View File

@@ -0,0 +1,59 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___STREAM_INTERNAL_STREAMS_H
#define _CUDA___STREAM_INTERNAL_STREAMS_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__stream/stream.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
// We make __cccl_allocation_stream() noexcept because the only way it could potentially fail
// is e.g. bad driver state or some other deeper corruption so we are pretty much in an
// unusable state anyways.
// NOLINTBEGIN(bugprone-exception-escape)
//! @brief internal stream used for memory allocations, no real blocking work
//! should ever be pushed into it
inline ::cuda::stream_ref __cccl_allocation_stream() noexcept
{
// Intentionally leak the stream here to avoid stream destruction when the program exits, which is not guaraneed to
// work.
static ::cuda::stream_ref __stream = []() {
::cuda::stream __str{::cuda::device_ref{0}};
return __str.release();
}();
return __stream;
}
// NOLINTEND(bugprone-exception-escape)
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___STREAM_INTERNAL_STREAMS_H

View File

@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___STREAM_INVALID_STREAM_H
#define _CUDA___STREAM_INVALID_STREAM_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
enum class invalid_stream_t : unsigned char
{
};
_CCCL_GLOBAL_CONSTANT invalid_stream_t invalid_stream{};
[[nodiscard]] _CCCL_API _CCCL_FORCEINLINE ::cudaStream_t __invalid_stream() noexcept
{
return reinterpret_cast<::cudaStream_t>(~0ull); // NOLINT(performance-no-int-to-ptr)
}
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif //_CUDA___STREAM_INVALID_STREAM_H

View File

@@ -0,0 +1,203 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA__STREAM_LAUNCH_TRANSFORM_H
#define _CUDA__STREAM_LAUNCH_TRANSFORM_H
#include <cuda/__cccl_config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/__stream/stream_ref.h>
# include <cuda/__type_traits/is_instantiable_with.h>
# include <cuda/std/__memory/addressof.h>
# include <cuda/std/__memory/construct_at.h>
# include <cuda/std/__new/launder.h>
# include <cuda/std/__optional/optional.h>
# include <cuda/std/__tuple_dir/ignore.h>
# include <cuda/std/__type_traits/decay.h>
# include <cuda/std/__type_traits/is_callable.h>
# include <cuda/std/__type_traits/is_reference.h>
# include <cuda/std/__utility/declval.h>
# include <cuda/std/__utility/forward.h>
# include <cuda/std/__utility/move.h>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
namespace __detail
{
// This function turns rvalues into prvalues and leaves lvalues as is.
template <typename _Tp>
_CCCL_API constexpr auto __ixnay_xvalue(_Tp&& __value) noexcept(::cuda::std::is_nothrow_move_constructible_v<_Tp>)
-> _Tp
{
return ::cuda::std::forward<_Tp>(__value);
}
} // namespace __detail
template <typename _Tp>
using __remove_rvalue_reference_t =
decltype(__detail::__ixnay_xvalue(::cuda::std::declval<_Tp>()) // NOLINT(modernize-type-traits)
);
namespace __tfx
{
// Launch transform:
//
// The launch transform is a mechanism to transform arguments passed to the
// algorithms prior to actually enqueueing work on a stream. This is useful for
// example, to automatically convert contiguous ranges into spans. It is also
// useful for executing per-argument actions before and after the kernel launch.
// A host_vector might want a pre-launch action to copy data from host to device
// and a post-launch action to copy data back from device to host.
//
// The expression `launch_transform(stream, arg)` is expression-equivalent to
// the first of the following expressions that is valid:
//
// 1. `transform_launch_argument(stream, arg).transformed_argument()`
// 2. `transform_launch_argument(stream, arg)`
// 3. `arg.transformed_argument()`
// 4. `arg`
_CCCL_HOST_API void transform_launch_argument();
struct _CCCL_TYPE_VISIBILITY_DEFAULT __launch_transform_t
{
// Types that want to customize `launch_transform` should define overloads of
// transform_launch_argument that are find-able by ADL.
template <typename _Arg>
using __transform_result_t = __remove_rvalue_reference_t<decltype(transform_launch_argument(
::cuda::stream_ref{::cudaStream_t{}}, ::cuda::std::declval<_Arg>()))>;
template <typename _Arg>
using __transformed_argument_t =
__remove_rvalue_reference_t<decltype(::cuda::std::declval<_Arg>().transformed_argument())>;
// The use of `optional` here is to move the destruction of the object returned from
// transform_launch_argument into the caller's stack frame. Objects created for default arguments
// are located in the caller's stack frame. This is so that a use of `launch_transform`
// such as:
//
// kernel<<<grid, block, 0, stream>>>(launch_transform(stream, arg));
//
// is equivalent to:
//
// kernel<<<grid, block, 0, stream>>>(transform_launch_argument(stream, arg).transformed_argument());
//
// where the object returned from `transform_launch_argument` is destroyed *after* the kernel
// launch.
//
// What I really wanted to do was:
//
// template <typename Arg>
// auto operator()(::cuda::stream_ref stream, Arg&& arg, auto&& action = transform_launch_argument(arg))
//
// but sadly that is not valid C++.
// TODO move to use __variant type once cuda/experimental/execution/__variant is moved to libcudacxx
// NOTE: The above seems to only apply if the type is not trivially destructible. To use the optional here I had to
// add a destructor.
template <typename _Tp>
struct __optional_with_a_destructor : ::cuda::std::optional<_Tp>
{
using ::cuda::std::optional<_Tp>::optional;
// Use of explicit destructor is intentional. Without it, the argument may have a trivial
// destructor and hence would be performed by the callee.
~__optional_with_a_destructor() {} // NOLINT(modernize-use-equals-default)
template <class _Fn>
_CCCL_API inline _CCCL_CONSTEXPR_CXX20 _Tp& __emplace_from_fn(_Fn&& __fn)
{
_CCCL_ASSERT(!this->has_value(), "__construct called for engaged __optional_storage");
new (::cuda::std::addressof(this->__get())) _Tp(::cuda::std::invoke(::cuda::std::forward<_Fn>(__fn)));
this->__set_engaged(true);
return this->__get();
}
};
_CCCL_TEMPLATE(typename _Stream, typename _Arg)
_CCCL_REQUIRES(::cuda::std::convertible_to<_Stream, ::cuda::stream_ref> _CCCL_AND(
!::cuda::std::is_reference_v<__transform_result_t<_Arg>>))
[[nodiscard]] _CCCL_HOST_API auto operator()(
_Stream&& __stream,
_Arg&& __arg,
__optional_with_a_destructor<__transform_result_t<_Arg>> __storage = cuda::std::nullopt) const -> decltype(auto)
{
// Calls to transform_launch_argument are intentionally unqualified so as to use ADL.
if constexpr (__is_instantiable_with<__transformed_argument_t, __transform_result_t<_Arg>>)
{
return _CCCL_MOVE(__storage.__emplace_from_fn([&]() {
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
}))
.transformed_argument();
}
else
{
return _CCCL_MOVE(__storage.__emplace_from_fn([&]() {
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
}));
}
}
// If transform_launch_argument returns a reference type, then there are no pre- and
// post-launch actions. (References types don't have ctors/dtors.) There is no need to
// store the result of transform_launch_argument.
_CCCL_TEMPLATE(typename _Stream, typename _Arg)
_CCCL_REQUIRES(::cuda::std::convertible_to<_Stream, ::cuda::stream_ref>
_CCCL_AND ::cuda::std::is_reference_v<__transform_result_t<_Arg>>)
[[nodiscard]] _CCCL_HOST_API auto operator()(_Stream&& __stream, _Arg&& __arg) const -> decltype(auto)
{
// Calls to transform_launch_argument are intentionally unqualified so as to use ADL.
if constexpr (__is_instantiable_with<__transformed_argument_t, __transform_result_t<_Arg>>)
{
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg)).transformed_argument();
}
else
{
return transform_launch_argument(__stream, ::cuda::std::forward<_Arg>(__arg));
}
}
template <typename _Arg>
[[nodiscard]] _CCCL_HOST_API auto operator()(::cuda::std::__ignore_t, _Arg&& __arg) const -> decltype(auto)
{
if constexpr (__is_instantiable_with<__transformed_argument_t, _Arg>)
{
return ::cuda::std::forward<_Arg>(__arg).transformed_argument();
}
else
{
return static_cast<_Arg>(::cuda::std::forward<_Arg>(__arg));
}
}
};
} // namespace __tfx
_CCCL_GLOBAL_CONSTANT auto launch_transform = __tfx::__launch_transform_t{};
# ifndef _CCCL_DOXYGEN_INVOKED // Doxygen chokes here
template <typename _Arg>
using transformed_device_argument_t _CCCL_NODEBUG_ALIAS =
__remove_rvalue_reference_t<::cuda::std::__call_result_t<__tfx::__launch_transform_t, ::cuda::stream_ref, _Arg>>;
# endif // ^^^ _CCCL_DOXYGEN_INVOKED ^^^
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA__STREAM_LAUNCH_TRANSFORM_H

View File

@@ -0,0 +1,145 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___STREAM_STREAM_H
#define _CUDA___STREAM_STREAM_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
# include <cuda/__device/device_ref.h>
# include <cuda/__driver/driver_api.h>
# include <cuda/__runtime/ensure_current_context.h>
# include <cuda/__stream/invalid_stream.h>
# include <cuda/__stream/stream_ref.h> // IWYU pragma: export
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief An owning wrapper for cudaStream_t.
struct stream : stream_ref
{
// 0 is documented as default priority
static constexpr int default_priority = 0;
//! @brief Constructs a stream on a specified device and with specified priority
//!
//! Priority is defaulted to stream::default_priority
//!
//! @throws cuda_error if stream creation fails
_CCCL_HOST_API explicit stream(device_ref __dev, int __priority = default_priority)
: stream_ref(::cuda::__invalid_stream())
{
[[maybe_unused]] __ensure_current_context __ctx_setter(__dev);
__stream = ::cuda::__driver::__streamCreateWithPriority(cudaStreamNonBlocking, __priority);
}
//! @brief Construct a new `stream` object into the moved-from state.
//!
//! @post `stream()` returns an invalid stream handle
// Can't be constexpr because __invalid_stream isn't
_CCCL_HOST_API explicit stream(no_init_t) noexcept
: stream_ref(::cuda::__invalid_stream())
{}
//! @brief Move-construct a new `stream` object
//!
//! @param __other
//!
//! @post `__other` is in moved-from state.
_CCCL_HOST_API stream(stream&& __other) noexcept
: stream(::cuda::std::exchange(__other.__stream, ::cuda::__invalid_stream()))
{}
stream(const stream&) = delete;
//! Destroy the `stream` object
//!
//! @note If the stream fails to be destroyed, the error is silently ignored.
_CCCL_HOST_API ~stream()
{
if (__stream != ::cuda::__invalid_stream())
{
// Needs to call driver API in case current device is not set, runtime version would set dev 0 current
// Alternative would be to store the device and push/pop here
[[maybe_unused]] auto status = ::cuda::__driver::__streamDestroyNoThrow(__stream);
}
}
//! @brief Move-assign a `stream` object
//!
//! @param __other
//!
//! @post `__other` is in a moved-from state.
_CCCL_HOST_API stream& operator=(stream&& __other) noexcept
{
stream __tmp(::cuda::std::move(__other));
::cuda::std::swap(__stream, __tmp.__stream);
return *this;
}
stream& operator=(const stream&) = delete;
//! @brief Construct an `stream` object from a native `cudaStream_t` handle.
//!
//! @param __handle The native handle
//!
//! @return stream The constructed `stream` object
//!
//! @note The constructed `stream` object takes ownership of the native handle.
[[nodiscard]] static _CCCL_HOST_API stream from_native_handle(::cudaStream_t __handle)
{
return stream(__handle);
}
// Disallow construction from an `int`, e.g., `0`.
static stream from_native_handle(int) = delete;
// Disallow construction from `nullptr`.
static stream from_native_handle(::cuda::std::nullptr_t) = delete;
// Disallow construction from `invalid_stream_t`.
static stream from_native_handle(invalid_stream_t) = delete;
//! @brief Retrieve the native `cudaStream_t` handle and give up ownership.
//!
//! @return cudaStream_t The native handle being held by the `stream` object.
//!
//! @post The stream object is in a moved-from state.
[[nodiscard]] _CCCL_HOST_API ::cudaStream_t release()
{
return ::cuda::std::exchange(__stream, ::cuda::__invalid_stream());
}
private:
// Use `stream::from_native_handle(s)` to construct an owning `stream`
// object from a `cudaStream_t` handle.
_CCCL_HOST_API explicit stream(::cudaStream_t __handle)
: stream_ref(__handle)
{}
};
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
#endif // _CUDA___STREAM_STREAM_H

View File

@@ -0,0 +1,356 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___STREAM_STREAM_REF_H
#define _CUDA___STREAM_STREAM_REF_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
# include <cuda/__device/device_ref.h>
# include <cuda/__driver/driver_api.h>
# include <cuda/__event/timed_event.h>
# include <cuda/__fwd/get_stream.h>
# include <cuda/__runtime/ensure_current_context.h>
# include <cuda/__stream/invalid_stream.h>
# include <cuda/__utility/no_init.h>
# include <cuda/std/__exception/cuda_error.h>
# include <cuda/std/__exception/exception_macros.h>
# include <cuda/std/__utility/to_underlying.h>
# include <cuda/std/cstddef>
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief A type representing a stream ID.
enum class stream_id : unsigned long long
{
};
//! @brief A non-owning wrapper for a `cudaStream_t`.
class stream_ref
{
protected:
::cudaStream_t __stream{nullptr};
public:
using value_type = ::cudaStream_t;
//! @brief Constructs a `stream_ref` of the "default" CUDA stream.
//!
//! For behavior of the default stream,
//! @see //! https://docs.nvidia.com/cuda/cuda-runtime-api/stream-sync-behavior.html
CCCL_DEPRECATED_BECAUSE("Using the default/null stream is generally discouraged. If you need to use it, please "
"construct a "
"stream_ref from cudaStream_t{nullptr}") _CCCL_HIDE_FROM_ABI
stream_ref() = default;
//! @brief Constructs a `stream_ref` from a `cudaStream_t` handle.
//!
//! This constructor provides implicit conversion from `cudaStream_t`.
//!
//! @note: It is the callers responsibility to ensure the `stream_ref` does not
//! outlive the stream identified by the `cudaStream_t` handle.
_CCCL_API constexpr stream_ref(value_type __stream_) noexcept
: __stream{__stream_}
{}
//! @brief Constructs a `stream_ref` from the `cuda::invalid_stream_t`.
//!
//! @note Any CUDA APIs called on the created object will result in an CUDA error.
_CCCL_API explicit stream_ref(invalid_stream_t) noexcept
: __stream{::cuda::__invalid_stream()}
{}
//! Disallow construction from an `int`, e.g., `0`.
stream_ref(int) = delete;
//! Disallow construction from `nullptr`.
stream_ref(::cuda::std::nullptr_t) = delete;
//! @brief Compares two `stream_ref`s for equality
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __lhs The first `stream_ref` to compare
//! @param __rhs The second `stream_ref` to compare
//! @return true if equal, false if unequal
[[nodiscard]] _CCCL_API friend constexpr bool operator==(const stream_ref& __lhs, const stream_ref& __rhs) noexcept
{
return __lhs.__stream == __rhs.__stream;
}
//! @brief Compares `stream_ref` with `invalid_stream_t` for equality.
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __lhs The `stream_ref` to compare
//! @return true if equal, false if unequal
[[nodiscard]] _CCCL_API friend bool operator==(const stream_ref& __lhs, const invalid_stream_t&) noexcept
{
return __lhs.__stream == ::cuda::__invalid_stream();
}
//! @brief Compares `invalid_stream_t` with `stream_ref` for equality.
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __rhs The `stream_ref` to compare
//! @return true if equal, false if unequal
[[nodiscard]] _CCCL_API friend bool operator==(const invalid_stream_t&, const stream_ref& __rhs) noexcept
{
return ::cuda::__invalid_stream() == __rhs.__stream;
}
//! @brief Compares two `stream_ref`s for inequality
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __lhs The first `stream_ref` to compare
//! @param __rhs The second `stream_ref` to compare
//! @return true if unequal, false if equal
[[nodiscard]] _CCCL_API friend constexpr bool operator!=(const stream_ref& __lhs, const stream_ref& __rhs) noexcept
{
return __lhs.__stream != __rhs.__stream;
}
//! @brief Compares `stream_ref` with `invalid_stream_t` for inequality.
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __lhs The `stream_ref` to compare
//! @return false if equal, true if unequal
[[nodiscard]] _CCCL_API friend bool operator!=(const stream_ref& __lhs, const invalid_stream_t&) noexcept
{
return __lhs.__stream != ::cuda::__invalid_stream();
}
//! @brief Compares `invalid_stream_t` with `stream_ref` for inequality.
//!
//! @note Allows comparison with `cudaStream_t` due to implicit conversion to
//! `stream_ref`.
//!
//! @param __rhs The `stream_ref` to compare
//! @return false if equal, true if unequal
[[nodiscard]] _CCCL_API friend bool operator!=(const invalid_stream_t&, const stream_ref& __rhs) noexcept
{
return ::cuda::__invalid_stream() != __rhs.__stream;
}
//! Returns the wrapped `cudaStream_t` handle.
[[nodiscard]] _CCCL_API constexpr value_type get() const noexcept
{
return __stream;
}
//! @brief Synchronizes the wrapped stream.
//!
//! @throws cuda::cuda_error if synchronization fails.
_CCCL_HOST_API void sync() const
{
::cuda::__driver::__streamSynchronize(__stream);
}
//! @brief Deprecated. Use sync() instead.
//!
//! @deprecated Use sync() instead.
CCCL_DEPRECATED_BECAUSE("Use sync() instead.") _CCCL_HOST_API void wait() const
{
sync();
}
//! @brief Make all future work submitted into this stream depend on completion of the specified event
//!
//! @param __ev Event that this stream should wait for
//!
//! @throws cuda_error if inserting the dependency fails
_CCCL_HOST_API void wait(event_ref __ev) const
{
_CCCL_ASSERT(__ev.get() != nullptr, "cuda::stream_ref::wait invalid event passed");
// Need to use driver API, cudaStreamWaitEvent would push dev 0 if stack was empty
::cuda::__driver::__streamWaitEvent(get(), __ev.get());
}
//! @brief Make all future work submitted into this stream depend on completion of all work from the specified
//! stream
//!
//! @param __other Stream that this stream should wait for
//!
//! @throws cuda_error if inserting the dependency fails
_CCCL_HOST_API void wait(stream_ref __other) const
{
// TODO consider an optimization to not create an event every time and instead have one persistent event or one
// per stream
_CCCL_ASSERT(__stream != ::cuda::__invalid_stream(), "cuda::stream_ref::wait invalid stream passed");
if (*this != __other)
{
event __tmp(__other);
wait(__tmp);
}
}
//! \brief Queries if all operations on the stream have completed.
//!
//! \throws cuda::cuda_error if the query fails.
//!
//! \return `true` if all operations have completed, or `false` if not.
[[nodiscard]] _CCCL_HOST_API bool is_done() const
{
const auto __result = ::cuda::__driver::__streamQueryNoThrow(__stream);
switch (__result)
{
case ::cudaErrorNotReady:
return false;
case ::cudaSuccess:
return true;
default:
_CCCL_THROW(::cuda::cuda_error, __result, "Failed to query stream.");
}
}
//! @brief Queries if all operations on the wrapped stream have completed.
//!
//! @throws cuda::cuda_error if the query fails.
//!
//! @return `true` if all operations have completed, or `false` if not.
[[nodiscard]] CCCL_DEPRECATED_BECAUSE("Use is_done() instead.") _CCCL_HOST_API bool ready() const
{
return is_done();
}
//! @brief Queries the priority of the wrapped stream.
//!
//! @throws cuda::cuda_error if the query fails.
//!
//! @return value representing the priority of the wrapped stream.
[[nodiscard]] _CCCL_HOST_API int priority() const
{
return ::cuda::__driver::__streamGetPriority(__stream);
}
//! @brief Get the unique ID of the stream
//!
//! Stream handles are sometimes reused, but ID is guaranteed to be unique.
//!
//! @return The unique ID of the stream
//!
//! @throws cuda_error if the ID query fails
[[nodiscard]] _CCCL_HOST_API stream_id id() const
{
return stream_id{::cuda::__driver::__streamGetId(__stream)};
}
//! @brief Create a new event and record it into this stream
//!
//! @return A new event that was recorded into this stream
//!
//! @throws cuda_error if event creation or record failed
[[nodiscard]] _CCCL_HOST_API event record_event(event_flags __flags = event_flags::none) const
{
return event(*this, __flags);
}
//! @brief Create a new timed event and record it into this stream
//!
//! @return A new timed event that was recorded into this stream
//!
//! @throws cuda_error if event creation or record failed
[[nodiscard]] _CCCL_HOST_API timed_event record_timed_event(event_flags __flags = event_flags::none) const
{
return timed_event(*this, __flags);
}
//! @brief Get device under which this stream was created.
//!
//! Note: In case of a stream created under a `green_context` the device on which that `green_context` was created is
//! returned
//!
//! @throws cuda_error if device check fails
[[nodiscard]] _CCCL_HOST_API device_ref device() const
{
::CUdevice __device{};
# if _CCCL_CTK_AT_LEAST(13, 0)
__device = ::cuda::__driver::__streamGetDevice(__stream);
# else // ^^^ _CCCL_CTK_AT_LEAST(13, 0) ^^^ / vvv _CCCL_CTK_BELOW(13, 0) vvv
{
::CUcontext __stream_ctx = ::cuda::__driver::__streamGetCtx(__stream);
__ensure_current_context __setter(__stream_ctx);
__device = ::cuda::__driver::__ctxGetDevice();
}
# endif // ^^^ _CCCL_CTK_BELOW(13, 0) ^^^
return device_ref{::cuda::__driver::__cudevice_to_ordinal(__device)};
}
//! @brief Queries the \c stream_ref for itself. This makes \c stream_ref usable in places where we expect an
//! environment with a \c get_stream_t query
[[nodiscard]] _CCCL_API constexpr stream_ref query(const ::cuda::get_stream_t&) const noexcept
{
return *this;
}
};
_CCCL_HOST_API inline void event_ref::record(stream_ref __stream) const
{
_CCCL_ASSERT(__event_ != nullptr, "cuda::event_ref::record no event set");
_CCCL_ASSERT(__stream.get() != nullptr, "cuda::event_ref::record invalid stream passed");
// Need to use driver API, cudaEventRecord will push dev 0 if stack is empty
::cuda::__driver::__eventRecord(__event_, __stream.get());
}
_CCCL_HOST_API inline event::event(stream_ref __stream, event_flags __flags)
: event(__stream, ::cuda::std::to_underlying(__flags) | cudaEventDisableTiming)
{
record(__stream);
}
_CCCL_HOST_API inline event::event(stream_ref __stream, unsigned __flags)
: event_ref(::cudaEvent_t{})
{
[[maybe_unused]] __ensure_current_context __ctx_setter(__stream);
__event_ = ::cuda::__driver::__eventCreate(static_cast<unsigned>(__flags));
}
_CCCL_HOST_API inline timed_event::timed_event(stream_ref __stream, event_flags __flags)
: event(__stream, ::cuda::std::to_underlying(__flags))
{
record(__stream);
}
// Hide from Doxygen — __ensure_current_context is an internal symbol excluded by EXCLUDE_SYMBOLS.
# ifndef _CCCL_DOXYGEN_INVOKED
_CCCL_HOST_API inline __ensure_current_context::__ensure_current_context(stream_ref __stream)
{
auto __ctx = __driver::__streamGetCtx(__stream.get());
::cuda::__driver::__ctxPush(__ctx);
}
# endif // !_CCCL_DOXYGEN_INVOKED
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK() && !_CCCL_COMPILER(NVRTC)
#endif //_CUDA___STREAM_STREAM_REF_H