[INFRA] Import NVIDIA/CCCL upstream as optimization reference library

CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
This commit is contained in:
EngineX CI
2026-07-30 09:35:51 +00:00
parent b4d01f481e
commit 56fd68e7dd
8871 changed files with 1454674 additions and 0 deletions

View File

@@ -0,0 +1,43 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_BARRIER_H
#define _CUDA___FWD_BARRIER_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__atomic/scopes.h>
#include <cuda/std/__barrier/empty_completion.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <thread_scope _Sco, class _CompletionF = ::cuda::std::__empty_completion>
class barrier;
template <class _Tp>
inline constexpr bool __is_cuda_barrier_v = false;
template <thread_scope _Sco, class _ComplFn>
inline constexpr bool __is_cuda_barrier_v<barrier<_Sco, _ComplFn>> = true;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_BARRIER_H

View File

@@ -0,0 +1,48 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_COMPLEX_H
#define _CUDA___FWD_COMPLEX_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <class _Tp>
class _CCCL_TYPE_VISIBILITY_DEFAULT complex;
// __is_cuda_complex_v
template <class _Tp>
inline constexpr bool __is_cuda_complex_v = false;
template <class _Tp>
inline constexpr bool __is_cuda_complex_v<const _Tp> = __is_cuda_complex_v<_Tp>;
template <class _Tp>
inline constexpr bool __is_cuda_complex_v<volatile _Tp> = __is_cuda_complex_v<_Tp>;
template <class _Tp>
inline constexpr bool __is_cuda_complex_v<const volatile _Tp> = __is_cuda_complex_v<_Tp>;
template <class _Tp>
inline constexpr bool __is_cuda_complex_v<complex<_Tp>> = true;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_COMPLEX_H

View File

@@ -0,0 +1,47 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_DEVICES_H
#define _CUDA___FWD_DEVICES_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__fwd/span.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
#if _CCCL_HAS_CTK()
class __physical_device;
class device_ref;
template <::cudaDeviceAttr _Attr>
struct __dev_attr;
#endif // _CCCL_HAS_CTK()
struct arch_traits_t;
class compute_capability;
enum class arch_id : int;
inline constexpr int __arch_specific_id_multiplier = 100000;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_DEVICES_H

View File

@@ -0,0 +1,38 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_GET_MEMORY_RESOURCE_H
#define _CUDA___FWD_GET_MEMORY_RESOURCE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA_MR
struct __get_memory_resource_t;
_CCCL_END_NAMESPACE_CUDA_MR
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___FWD_GET_MEMORY_RESOURCE_H

View File

@@ -0,0 +1,38 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2025 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_GET_STREAM_H
#define _CUDA___FWD_GET_STREAM_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#if _CCCL_HAS_CTK()
# include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
struct get_stream_t;
_CCCL_END_NAMESPACE_CUDA
# include <cuda/std/__cccl/epilogue.h>
#endif // _CCCL_HAS_CTK()
#endif // _CUDA___FWD_GET_STREAM_H

View File

@@ -0,0 +1,103 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_HIERARCHY_H
#define _CUDA___FWD_HIERARCHY_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__type_traits/is_base_of.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
using dimensions_index_type = unsigned;
// hierarchy level
template <class _Level>
struct hierarchy_level_base;
template <class _Level>
struct __native_hierarchy_level_base;
struct grid_level;
struct cluster_level;
struct block_level;
struct warp_level;
struct thread_level;
template <class _Tp>
inline constexpr bool __is_hierarchy_level_v = ::cuda::std::is_base_of_v<hierarchy_level_base<_Tp>, _Tp>;
template <class _Tp>
inline constexpr bool __is_native_hierarchy_level_v =
::cuda::std::is_base_of_v<__native_hierarchy_level_base<_Tp>, _Tp>;
// hierarchy
template <class _BottomUnit, class... _Levels>
class hierarchy;
template <class _Tp>
inline constexpr bool __is_hierarchy_v = false;
template <class _BottomUnit, class... _Levels>
inline constexpr bool __is_hierarchy_v<hierarchy<_BottomUnit, _Levels...>> = true;
template <typename... _Levels>
struct __allowed_levels;
struct __hierarchy_level_desc_base
{};
template <class _Level, class _Exts>
class hierarchy_level_desc;
template <class _Tp>
inline constexpr bool __is_hierarchy_level_desc_v = ::cuda::std::is_base_of_v<__hierarchy_level_desc_base, _Tp>;
template <class _Unit, class _Level>
struct __extents_query_native;
template <class _Unit, class _Level>
struct __extents_query;
template <class _Unit, class _Level>
struct __count_query_native;
template <class _Unit, class _Level>
struct __count_query;
template <class _Unit, class _Level>
struct __index_query_native;
template <class _Unit, class _Level>
struct __index_query;
template <class _Unit, class _Level>
struct __rank_query_native;
template <class _Unit, class _Level>
struct __rank_query;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_HIERARCHY_H

View File

@@ -0,0 +1,141 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_ITERATOR_H
#define _CUDA___FWD_ITERATOR_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/__fwd/random.h>
#include <cuda/std/__concepts/arithmetic.h>
#include <cuda/std/__concepts/copyable.h>
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__iterator/concepts.h>
#include <cuda/std/__iterator/incrementable_traits.h>
#include <cuda/std/__type_traits/enable_if.h>
#include <cuda/std/__type_traits/type_identity.h>
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <class _Tp, class _Index = ::cuda::std::ptrdiff_t>
class constant_iterator;
template <class _Tp>
[[nodiscard]] _CCCL_API _CCCL_CONSTEVAL auto __get_wider_signed() noexcept
{
if constexpr (sizeof(_Tp) < sizeof(int))
{
return ::cuda::std::type_identity<int>{};
}
else if constexpr (sizeof(_Tp) < sizeof(long))
{
return ::cuda::std::type_identity<long>{};
}
#if _CCCL_HAS_INT128()
else if constexpr (sizeof(_Tp) < sizeof(long long))
{
return ::cuda::std::type_identity<long long>{};
}
else // if constexpr (sizeof(_Start) < sizeof(__int128_t))
{
return ::cuda::std::type_identity<__int128_t>{};
}
#else // ^^^ _CCCL_HAS_INT128() ^^^ / vvv !_CCCL_HAS_INT128() vvv
else // if constexpr (sizeof(_Start) < sizeof(long long))
{
return ::cuda::std::type_identity<long long>{};
}
#endif // _CCCL_HAS_INT128()
}
template <class _Start>
using _IotaDiffT = typename ::cuda::std::conditional_t<
(!::cuda::std::integral<_Start> || sizeof(::cuda::std::iter_difference_t<_Start>) > sizeof(_Start)),
::cuda::std::type_identity<::cuda::std::iter_difference_t<_Start>>,
decltype(::cuda::__get_wider_signed<_Start>())>::type;
#if _CCCL_HAS_CONCEPTS()
template <::cuda::std::weakly_incrementable _Start, ::cuda::std::signed_integral _DiffT = _IotaDiffT<_Start>>
requires ::cuda::std::copyable<_Start>
#else // ^^^ _CCCL_HAS_CONCEPTS() ^^^ / vvv !_CCCL_HAS_CONCEPTS() vvv
template <class _Start,
class _DiffT = _IotaDiffT<_Start>,
::cuda::std::enable_if_t<::cuda::std::weakly_incrementable<_Start>, int> = 0,
::cuda::std::enable_if_t<::cuda::std::copyable<_Start>, int> = 0,
::cuda::std::enable_if_t<::cuda::std::signed_integral<_DiffT>, int> = 0>
#endif // ^^^ !_CCCL_HAS_CONCEPTS() ^^^
class counting_iterator;
class discard_iterator;
template <class _Iter, class _Index = _Iter>
class permutation_iterator;
template <class _IndexType = ::cuda::std::size_t, class _Bijection = random_bijection<_IndexType>>
class shuffle_iterator;
template <class _Iter, class _Stride = ::cuda::std::iter_difference_t<_Iter>>
class strided_iterator;
template <class _Fn, class _Index = ::cuda::std::ptrdiff_t>
class tabulate_output_iterator;
template <class _InputFn, class _OutputFn, class _Iter>
class transform_input_output_iterator;
template <class _Fn, class _Iter>
class transform_iterator;
template <class _Fn, class _Iter>
class transform_output_iterator;
template <class... _Iterators>
class zip_iterator;
template <class>
inline constexpr bool __is_zip_iterator = false;
template <class... _Iterators>
inline constexpr bool __is_zip_iterator<zip_iterator<_Iterators...>> = true;
template <class _Fn>
class zip_function;
template <class>
inline constexpr bool __is_zip_function = false;
template <class _Fn>
inline constexpr bool __is_zip_function<zip_function<_Fn>> = true;
template <class _Fn, class... _Iterators>
class zip_transform_iterator;
template <class>
inline constexpr bool __is_zip_transform_iterator = false;
template <class _Fn, class... _Iterators>
inline constexpr bool __is_zip_transform_iterator<zip_transform_iterator<_Fn, _Iterators...>> = true;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_ITERATOR_H

View File

@@ -0,0 +1,92 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_MDSPAN_H
#define _CUDA___FWD_MDSPAN_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__cstddef/types.h>
#include <cuda/std/__limits/numeric_limits.h>
#include <cuda/std/__type_traits/make_signed.h>
#include <cuda/std/__utility/integer_sequence.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
//! @brief Class to describe the strides of a multi-dimensional array layout.
//!
//! Similar to extents, but for strides. Supports both static (compile-time known)
//! and dynamic (runtime) stride values. Uses dynamic_stride as the tag for dynamic values.
//!
//! @tparam _OffsetType The signed integer type for stride values (supports negative strides)
//! @tparam _Strides... The stride values, where dynamic_stride indicates a runtime value
template <class _OffsetType, ::cuda::std::ptrdiff_t... _Strides>
class strides;
//! @brief Tag value indicating a dynamic stride (similar to dynamic_extent for extents)
inline constexpr ::cuda::std::ptrdiff_t dynamic_stride = (::cuda::std::numeric_limits<::cuda::std::ptrdiff_t>::min)();
namespace __strides_detail
{
template <class _OffsetType, class _Seq>
struct __make_dstrides_impl;
template <class _OffsetType, ::cuda::std::ptrdiff_t... _Idx>
struct __make_dstrides_impl<_OffsetType, ::cuda::std::integer_sequence<::cuda::std::ptrdiff_t, _Idx...>>
{
using type = strides<_OffsetType, ((void) _Idx, dynamic_stride)...>;
};
} // namespace __strides_detail
//! @brief Alias template for strides with all dynamic stride values
template <class _OffsetType, ::cuda::std::size_t _Rank>
using dstrides = typename __strides_detail::
__make_dstrides_impl<_OffsetType, ::cuda::std::make_integer_sequence<::cuda::std::ptrdiff_t, _Rank>>::type;
template <::cuda::std::size_t _Rank, class _OffsetType = ::cuda::std::ptrdiff_t>
using steps = dstrides<_OffsetType, _Rank>;
template <class _Tp>
inline constexpr bool __is_cuda_strides_v = false;
template <class _OffsetType, ::cuda::std::ptrdiff_t... _Strides>
inline constexpr bool __is_cuda_strides_v<strides<_OffsetType, _Strides...>> = true;
//! @brief Layout policy with relaxed stride mapping that supports negative strides and offsets.
//!
//! Unlike `layout_stride`, this layout allows:
//! - Negative strides (for reverse iteration)
//! - Zero strides (for broadcasting)
//! - A base offset (to accommodate negative strides)
//!
//! @note This layout is NOT always unique, exhaustive, or strided in the standard sense.
struct layout_stride_relaxed
{
template <class _Extents,
class _Strides = dstrides<::cuda::std::make_signed_t<typename _Extents::index_type>, _Extents::rank()>,
class _OffsetType = ::cuda::std::ptrdiff_t>
class mapping;
};
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_MDSPAN_H

View File

@@ -0,0 +1,37 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_PIPELINE_H
#define _CUDA___FWD_PIPELINE_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/__atomic/scopes.h>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
template <thread_scope _Scope>
class pipeline;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_PIPELINE_H

View File

@@ -0,0 +1,39 @@
//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2026 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef _CUDA___FWD_RANDOM_H
#define _CUDA___FWD_RANDOM_H
#include <cuda/std/detail/__config>
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
# pragma GCC system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
# pragma clang system_header
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
# pragma system_header
#endif // no system header
#include <cuda/std/cstdint>
#include <cuda/std/__cccl/prologue.h>
_CCCL_BEGIN_NAMESPACE_CUDA
class __feistel_bijection;
template <class _IndexType = ::cuda::std::uint64_t, class _Bijection = __feistel_bijection>
class random_bijection;
_CCCL_END_NAMESPACE_CUDA
#include <cuda/std/__cccl/epilogue.h>
#endif // _CUDA___FWD_RANDOM_H