[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
62
cccl_upstream/examples/cudax_stf/CMakeLists.txt
Normal file
62
cccl_upstream/examples/cudax_stf/CMakeLists.txt
Normal file
@@ -0,0 +1,62 @@
|
||||
# SPDX-FileCopyrightText: Copyright (c) 2023 NVIDIA CORPORATION & AFFILIATES. All rights reserved.
|
||||
# SPDX-License-Identifier: Apache-2.0
|
||||
#
|
||||
# Licensed under the Apache License, Version 2.0 (the "License");
|
||||
# you may not use this file except in compliance with the License.
|
||||
# You may obtain a copy of the License at
|
||||
#
|
||||
# http://www.apache.org/licenses/LICENSE-2.0
|
||||
#
|
||||
# Unless required by applicable law or agreed to in writing, software
|
||||
# distributed under the License is distributed on an "AS IS" BASIS,
|
||||
# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
||||
# See the License for the specific language governing permissions and
|
||||
# limitations under the License.
|
||||
|
||||
cmake_minimum_required(VERSION 3.18 FATAL_ERROR)
|
||||
|
||||
project(CUDAX_SAMPLES CUDA CXX)
|
||||
|
||||
# This example uses the CMake Package Manager (CPM) to simplify fetching CCCL from GitHub
|
||||
# For more information, see https://github.com/cpm-cmake/CPM.cmake
|
||||
include(cmake/CPM.cmake)
|
||||
|
||||
# We define these as variables so they can be overridden in CI to pull from a PR instead of CCCL `main`
|
||||
# In your project, these variables are unnecessary and you can just use the values directly
|
||||
set(
|
||||
CCCL_REPOSITORY
|
||||
"https://github.com/NVIDIA/cccl"
|
||||
CACHE STRING
|
||||
"Git repository to fetch CCCL from"
|
||||
)
|
||||
set(CCCL_TAG "main" CACHE STRING "Git tag/branch to fetch from CCCL repository")
|
||||
|
||||
# This will automatically clone CCCL from GitHub and make the exported cmake targets available
|
||||
CPMAddPackage(
|
||||
NAME CCCL
|
||||
GIT_REPOSITORY "${CCCL_REPOSITORY}"
|
||||
GIT_TAG ${CCCL_TAG}
|
||||
# The following is required to make the `CCCL::cudax` target available:
|
||||
OPTIONS "CCCL_ENABLE_UNSTABLE ON"
|
||||
)
|
||||
|
||||
# Default to building for the GPU on the current system
|
||||
if (NOT DEFINED CMAKE_CUDA_ARCHITECTURES)
|
||||
set(CMAKE_CUDA_ARCHITECTURES native)
|
||||
endif()
|
||||
|
||||
# If you're building an executable
|
||||
add_executable(simple_stf simple_stf.cu)
|
||||
|
||||
target_link_libraries(simple_stf PUBLIC cuda)
|
||||
|
||||
if (CMAKE_CUDA_COMPILER)
|
||||
target_compile_options(
|
||||
simple_stf
|
||||
PUBLIC
|
||||
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--expt-relaxed-constexpr>
|
||||
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>
|
||||
)
|
||||
endif()
|
||||
|
||||
target_link_libraries(simple_stf PRIVATE CCCL::CCCL CCCL::cudax)
|
||||
1297
cccl_upstream/examples/cudax_stf/cmake/CPM.cmake
Normal file
1297
cccl_upstream/examples/cudax_stf/cmake/CPM.cmake
Normal file
File diff suppressed because it is too large
Load Diff
40
cccl_upstream/examples/cudax_stf/simple_stf.cu
Normal file
40
cccl_upstream/examples/cudax_stf/simple_stf.cu
Normal file
@@ -0,0 +1,40 @@
|
||||
//===----------------------------------------------------------------------===//
|
||||
//
|
||||
// Part of CUDASTF in CUDA C++ Core Libraries,
|
||||
// under the Apache License v2.0 with LLVM Exceptions.
|
||||
// See https://llvm.org/LICENSE.txt for license information.
|
||||
// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2022-2024 NVIDIA CORPORATION & AFFILIATES.
|
||||
//
|
||||
//===----------------------------------------------------------------------===//
|
||||
|
||||
#include <cuda/experimental/stf.cuh>
|
||||
|
||||
#include <cstdio>
|
||||
|
||||
using namespace cuda::experimental::stf;
|
||||
|
||||
int main()
|
||||
{
|
||||
context ctx;
|
||||
|
||||
int array[128];
|
||||
for (size_t i = 0; i < 128; i++)
|
||||
{
|
||||
array[i] = i;
|
||||
}
|
||||
|
||||
auto A = ctx.logical_data(array);
|
||||
|
||||
ctx.parallel_for(A.shape(), A.rw())->*[] __device__(size_t i, auto a) {
|
||||
a(i) += 4;
|
||||
};
|
||||
|
||||
ctx.finalize();
|
||||
|
||||
for (size_t i = 0; i < 128; i++)
|
||||
{
|
||||
printf("array[%ld] = %d\n", i, array[i]);
|
||||
}
|
||||
return 0;
|
||||
}
|
||||
Reference in New Issue
Block a user