CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
57 lines
1.5 KiB
CMake
57 lines
1.5 KiB
CMake
thrust_create_target(cudax.examples.thrust)
|
|
|
|
function(cudax_add_example target_name_var example_src)
|
|
get_filename_component(example_name ${example_src} NAME_WE)
|
|
|
|
# The actual name of the test's target:
|
|
set(example_target cudax.example.${example_name})
|
|
set(${target_name_var} ${example_target} PARENT_SCOPE)
|
|
|
|
cccl_add_executable(${example_target} SOURCES "${example_src}" ADD_CTEST)
|
|
target_link_libraries(
|
|
${example_target}
|
|
PRIVATE #
|
|
cudax.compiler_interface
|
|
cudax.examples.thrust
|
|
)
|
|
target_compile_options(
|
|
${example_target}
|
|
PRIVATE
|
|
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--expt-relaxed-constexpr>
|
|
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>
|
|
)
|
|
target_include_directories(
|
|
${example_target}
|
|
PRIVATE "${CUB_SOURCE_DIR}/examples"
|
|
)
|
|
endfunction()
|
|
|
|
file(
|
|
GLOB example_srcs
|
|
RELATIVE "${cudax_SOURCE_DIR}/examples"
|
|
CONFIGURE_DEPENDS
|
|
*.cu
|
|
*.cpp
|
|
)
|
|
|
|
cccl_get_cudatoolkit()
|
|
|
|
# Example requires pinned_memory_resource.
|
|
if (CUDAToolkit_VERSION VERSION_LESS 12.9)
|
|
list(REMOVE_ITEM example_srcs async_buffer_add.cu cub_reduce.cu)
|
|
endif()
|
|
|
|
foreach (example_src IN LISTS example_srcs)
|
|
cudax_add_example(example_target "${example_src}")
|
|
endforeach()
|
|
|
|
# FIXME: Enable MSVC
|
|
if (cudax_ENABLE_CUDASTF AND NOT "MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
|
# STF examples are handled separately:
|
|
add_subdirectory(stf)
|
|
endif()
|
|
|
|
if (cudax_ENABLE_PLACES AND NOT "MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
|
add_subdirectory(places)
|
|
endif()
|