[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
294
cccl_upstream/libcudacxx/include/cuda/barrier
Normal file
294
cccl_upstream/libcudacxx/include/cuda/barrier
Normal file
@@ -0,0 +1,294 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of libcu++, the C++ Standard Library for your entire system,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#ifndef _CUDA_BARRIER
|
||||
#define _CUDA_BARRIER
|
||||
|
||||
#include <cuda/std/detail/__config>
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#if _CCCL_DEVICE_COMPILATION() && _CCCL_PTX_ARCH() < 700 && !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
# error "CUDA synchronization primitives are only supported for sm_70 and up."
|
||||
#endif // _CCCL_DEVICE_COMPILATION() && _CCCL_PTX_ARCH() < 700 && !_CCCL_CUDA_COMPILER(NVHPC)
|
||||
|
||||
#include <cuda/__barrier/barrier.h>
|
||||
#include <cuda/__barrier/barrier_arrive_tx.h>
|
||||
#include <cuda/__barrier/barrier_block_scope.h>
|
||||
#include <cuda/__barrier/barrier_expect_tx.h>
|
||||
#include <cuda/__barrier/barrier_thread_scope.h>
|
||||
#include <cuda/__memcpy_async/memcpy_async.h>
|
||||
#include <cuda/__memcpy_async/memcpy_async_tx.h>
|
||||
#include <cuda/__memory/address_space.h>
|
||||
#include <cuda/__memory/aligned_size.h>
|
||||
#include <cuda/ptx>
|
||||
#include <cuda/std/barrier>
|
||||
|
||||
// Forward-declare CUtensorMap for use in cp_async_bulk_tensor_* PTX wrapping
|
||||
// functions. These functions take a pointer to CUtensorMap, so do not need to
|
||||
// know its size. This type is defined in cuda.h (driver API) as:
|
||||
//
|
||||
// typedef struct CUtensorMap_st { [ .. snip .. ] } CUtensorMap;
|
||||
//
|
||||
// We need to forward-declare both CUtensorMap_st (the struct) and CUtensorMap
|
||||
// (the typedef):
|
||||
struct CUtensorMap_st;
|
||||
typedef struct CUtensorMap_st CUtensorMap; // NOLINT(modernize-use-using)
|
||||
|
||||
#include <cuda/std/__cccl/prologue.h>
|
||||
|
||||
_CCCL_BEGIN_NAMESPACE_CUDA_DEVICE_EXPERIMENTAL
|
||||
|
||||
// Experimental exposure of TMA PTX:
|
||||
//
|
||||
// - cp_async_bulk_global_to_shared
|
||||
// - cp_async_bulk_shared_to_global
|
||||
// - cp_async_bulk_tensor_{1,2,3,4,5}d_global_to_shared
|
||||
// - cp_async_bulk_tensor_{1,2,3,4,5}d_shared_to_global
|
||||
// - fence_proxy_async_shared_cta
|
||||
// - cp_async_bulk_commit_group
|
||||
// - cp_async_bulk_wait_group_read<0, …, 7>
|
||||
|
||||
// These PTX wrappers are only available when the code is compiled compute
|
||||
// capability 9.0 and above. The check for (!defined(__CUDA_MINIMUM_ARCH__)) is
|
||||
// necessary to prevent cudafe from ripping out the device functions before
|
||||
// device compilation begins.
|
||||
#ifdef __cccl_lib_experimental_ctk12_cp_async_exposure
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_global_to_shared(
|
||||
void* __dest, const void* __src, ::cuda::std::uint32_t __size, ::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
_CCCL_ASSERT(__size % 16 == 0, "Size must be multiple of 16.");
|
||||
_CCCL_ASSERT(::cuda::device::is_address_from(__dest, ::cuda::device::address_space::shared),
|
||||
"Destination must be shared memory address.");
|
||||
_CCCL_ASSERT(::cuda::device::is_address_from(__src, ::cuda::device::address_space::global),
|
||||
"Source must be global memory address.");
|
||||
|
||||
::cuda::ptx::cp_async_bulk(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__src,
|
||||
__size,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_shared_to_global(void* __dest, const void* __src, ::cuda::std::uint32_t __size)
|
||||
{
|
||||
_CCCL_ASSERT(__size % 16 == 0, "Size must be multiple of 16.");
|
||||
_CCCL_ASSERT(::cuda::device::is_address_from(__dest, ::cuda::device::address_space::global),
|
||||
"Destination must be global memory address.");
|
||||
_CCCL_ASSERT(::cuda::device::is_address_from(__src, ::cuda::device::address_space::shared),
|
||||
"Source must be shared memory address.");
|
||||
|
||||
::cuda::ptx::cp_async_bulk(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __dest, __src, __size);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_1d_global_to_shared(
|
||||
void* __dest, const CUtensorMap* __tensor_map, int __c0, ::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__tensor_map,
|
||||
__coords,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_2d_global_to_shared(
|
||||
void* __dest, const CUtensorMap* __tensor_map, int __c0, int __c1, ::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__tensor_map,
|
||||
__coords,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_3d_global_to_shared(
|
||||
void* __dest,
|
||||
const CUtensorMap* __tensor_map,
|
||||
int __c0,
|
||||
int __c1,
|
||||
int __c2,
|
||||
::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__tensor_map,
|
||||
__coords,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_4d_global_to_shared(
|
||||
void* __dest,
|
||||
const CUtensorMap* __tensor_map,
|
||||
int __c0,
|
||||
int __c1,
|
||||
int __c2,
|
||||
int __c3,
|
||||
::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2, __c3};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__tensor_map,
|
||||
__coords,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_5d_global_to_shared(
|
||||
void* __dest,
|
||||
const CUtensorMap* __tensor_map,
|
||||
int __c0,
|
||||
int __c1,
|
||||
int __c2,
|
||||
int __c3,
|
||||
int __c4,
|
||||
::cuda::barrier<::cuda::thread_scope_block>& __bar)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2, __c3, __c4};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(
|
||||
::cuda::ptx::space_cluster,
|
||||
::cuda::ptx::space_global,
|
||||
__dest,
|
||||
__tensor_map,
|
||||
__coords,
|
||||
::cuda::device::barrier_native_handle(__bar));
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_1d_shared_to_global(const CUtensorMap* __tensor_map, int __c0, const void* __src)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __tensor_map, __coords, __src);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_2d_shared_to_global(const CUtensorMap* __tensor_map, int __c0, int __c1, const void* __src)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __tensor_map, __coords, __src);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_3d_shared_to_global(
|
||||
const CUtensorMap* __tensor_map, int __c0, int __c1, int __c2, const void* __src)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __tensor_map, __coords, __src);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_4d_shared_to_global(
|
||||
const CUtensorMap* __tensor_map, int __c0, int __c1, int __c2, int __c3, const void* __src)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2, __c3};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __tensor_map, __coords, __src);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-tensor
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_tensor instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_tensor_5d_shared_to_global(
|
||||
const CUtensorMap* __tensor_map, int __c0, int __c1, int __c2, int __c3, int __c4, const void* __src)
|
||||
{
|
||||
const ::cuda::std::int32_t __coords[]{__c0, __c1, __c2, __c3, __c4};
|
||||
|
||||
::cuda::ptx::cp_async_bulk_tensor(::cuda::ptx::space_global, ::cuda::ptx::space_shared, __tensor_map, __coords, __src);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#parallel-synchronization-and-communication-instructions-membar
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::fence_proxy_async instead") _CCCL_DEVICE_API inline void
|
||||
fence_proxy_async_shared_cta()
|
||||
{
|
||||
::cuda::ptx::fence_proxy_async(::cuda::ptx::space_shared);
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-commit-group
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_commit_group instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_commit_group()
|
||||
{
|
||||
::cuda::ptx::cp_async_bulk_commit_group();
|
||||
}
|
||||
|
||||
//! Deprecated [Since 3.2]
|
||||
// https://docs.nvidia.com/cuda/parallel-thread-execution/index.html#data-movement-and-conversion-instructions-cp-async-bulk-wait-group
|
||||
template <int __n_prior>
|
||||
CCCL_DEPRECATED_BECAUSE("Use cuda::ptx::cp_async_bulk_wait_group_read instead") _CCCL_DEVICE_API inline void
|
||||
cp_async_bulk_wait_group_read()
|
||||
{
|
||||
static_assert(__n_prior <= 63, "cp_async_bulk_wait_group_read: waiting for more than 63 groups is not supported.");
|
||||
::cuda::ptx::cp_async_bulk_wait_group_read(::cuda::ptx::n32_t<__n_prior>{});
|
||||
}
|
||||
|
||||
#endif // __cccl_lib_experimental_ctk12_cp_async_exposure
|
||||
|
||||
_CCCL_END_NAMESPACE_CUDA_DEVICE_EXPERIMENTAL
|
||||
|
||||
#include <cuda/std/__cccl/epilogue.h>
|
||||
|
||||
#endif // _CUDA_BARRIER
|
||||
Reference in New Issue
Block a user