Files
project_6/cccl_upstream/libcudacxx/codegen/generators/fence.h
EngineX CI 56fd68e7dd [INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides:
- CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk)
- Thrust: high-level parallel algorithms (transform_reduce, sort, scan)
- libcudacxx: CUDA C++ standard library (atomics, barriers, memory)
- cudax: experimental features (memory resources, allocators)
- Tuning policies: per-SM hardware-specific algorithm parameters

Competition optimization vectors mapped to CCCL:
- Output TPS (83% weight): warp_reduce, block_reduce, device_topk
- Input TPS (14% weight): device_scan, block_load, prefetch
- Cache TPS (3% weight): prefix caching strategy patterns
- Memory (0.9 util): pooled/cached/buddy allocators

Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only)
License: Apache-2.0
2026-07-30 09:35:51 +00:00

111 lines
3.1 KiB
C++

//===----------------------------------------------------------------------===//
//
// Part of libcu++, the C++ Standard Library for your entire system,
// under the Apache License v2.0 with LLVM Exceptions.
// See https://llvm.org/LICENSE.txt for license information.
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
// SPDX-FileCopyrightText: Copyright (c) 2024 NVIDIA CORPORATION & AFFILIATES.
//
//===----------------------------------------------------------------------===//
#ifndef FENCE_H
#define FENCE_H
#include <format>
#include <string>
#include "definitions.h"
inline std::string membar_scope(Scope sco)
{
static std::map scope_map{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
return scope_map[sco];
}
inline void FormatFence(std::ostream& out)
{
// Argument ID Reference
// 0 - Membar scope tag
// 1 - Membar scope
constexpr auto intrinsic_membar = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_membar({0})
{{ asm volatile("membar{1};" ::: "memory"); }})XXX";
const std::map membar_scopes{
std::pair{Scope::GPU, ".gl"},
std::pair{Scope::System, ".sys"},
std::pair{Scope::CTA, ".cta"},
};
for (const auto& sco : membar_scopes)
{
out << std::format(intrinsic_membar, scope_tag(sco.first), sco.second);
}
// Argument ID Reference
// 0 - Fence scope tag
// 1 - Fence scope
// 2 - Fence order tag
// 3 - Fence order
constexpr auto intrinsic_fence = R"XXX(
static inline _CCCL_DEVICE void __cuda_atomic_fence({0}, {2})
{{ asm volatile("fence{1}{3};" ::: "memory"); }})XXX";
const Scope fence_scopes[] = {
Scope::CTA,
Scope::Cluster,
Scope::GPU,
Scope::System,
};
const Semantic fence_semantics[] = {
Semantic::Acq_Rel,
Semantic::Seq_Cst,
};
for (const auto& sco : fence_scopes)
{
for (const auto& sem : fence_semantics)
{
out << std::format(intrinsic_fence, scope_tag(sco), semantic(sem), semantic_tag(sem), scope(sco));
}
}
out << "\n"
<< R"XXX(
template <typename _Sco>
static inline _CCCL_DEVICE void __atomic_thread_fence_cuda(int __memorder, _Sco) {
NV_DISPATCH_TARGET(
NV_PROVIDES_SM_70, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: __cuda_atomic_fence(_Sco{}, __atomic_cuda_seq_cst{}); break;
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_fence(_Sco{}, __atomic_cuda_acq_rel{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
),
NV_IS_DEVICE, (
switch (__memorder) {
case __ATOMIC_SEQ_CST: [[fallthrough]];
case __ATOMIC_CONSUME: [[fallthrough]];
case __ATOMIC_ACQUIRE: [[fallthrough]];
case __ATOMIC_ACQ_REL: [[fallthrough]];
case __ATOMIC_RELEASE: __cuda_atomic_membar(_Sco{}); break;
case __ATOMIC_RELAXED: break;
default: _CCCL_ASSERT(false, "invalid memory order");
}
)
)
}
)XXX";
}
#endif // FENCE_H