[INFRA] Import NVIDIA/CCCL upstream as optimization reference library
CCCL (CUDA C++ Core Libraries) provides: - CUB: device/block/warp-level GPU primitives (reduce, scan, sort, topk) - Thrust: high-level parallel algorithms (transform_reduce, sort, scan) - libcudacxx: CUDA C++ standard library (atomics, barriers, memory) - cudax: experimental features (memory resources, allocators) - Tuning policies: per-SM hardware-specific algorithm parameters Competition optimization vectors mapped to CCCL: - Output TPS (83% weight): warp_reduce, block_reduce, device_topk - Input TPS (14% weight): device_scan, block_load, prefetch - Cache TPS (3% weight): prefix caching strategy patterns - Memory (0.9 util): pooled/cached/buddy allocators Source: https://github.com/NVIDIA/cccl (shallow clone, HEAD only) License: Apache-2.0
This commit is contained in:
56
cccl_upstream/cudax/examples/CMakeLists.txt
Normal file
56
cccl_upstream/cudax/examples/CMakeLists.txt
Normal file
@@ -0,0 +1,56 @@
|
||||
thrust_create_target(cudax.examples.thrust)
|
||||
|
||||
function(cudax_add_example target_name_var example_src)
|
||||
get_filename_component(example_name ${example_src} NAME_WE)
|
||||
|
||||
# The actual name of the test's target:
|
||||
set(example_target cudax.example.${example_name})
|
||||
set(${target_name_var} ${example_target} PARENT_SCOPE)
|
||||
|
||||
cccl_add_executable(${example_target} SOURCES "${example_src}" ADD_CTEST)
|
||||
target_link_libraries(
|
||||
${example_target}
|
||||
PRIVATE #
|
||||
cudax.compiler_interface
|
||||
cudax.examples.thrust
|
||||
)
|
||||
target_compile_options(
|
||||
${example_target}
|
||||
PRIVATE
|
||||
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--expt-relaxed-constexpr>
|
||||
$<$<COMPILE_LANG_AND_ID:CUDA,NVIDIA>:--extended-lambda>
|
||||
)
|
||||
target_include_directories(
|
||||
${example_target}
|
||||
PRIVATE "${CUB_SOURCE_DIR}/examples"
|
||||
)
|
||||
endfunction()
|
||||
|
||||
file(
|
||||
GLOB example_srcs
|
||||
RELATIVE "${cudax_SOURCE_DIR}/examples"
|
||||
CONFIGURE_DEPENDS
|
||||
*.cu
|
||||
*.cpp
|
||||
)
|
||||
|
||||
cccl_get_cudatoolkit()
|
||||
|
||||
# Example requires pinned_memory_resource.
|
||||
if (CUDAToolkit_VERSION VERSION_LESS 12.9)
|
||||
list(REMOVE_ITEM example_srcs async_buffer_add.cu cub_reduce.cu)
|
||||
endif()
|
||||
|
||||
foreach (example_src IN LISTS example_srcs)
|
||||
cudax_add_example(example_target "${example_src}")
|
||||
endforeach()
|
||||
|
||||
# FIXME: Enable MSVC
|
||||
if (cudax_ENABLE_CUDASTF AND NOT "MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
||||
# STF examples are handled separately:
|
||||
add_subdirectory(stf)
|
||||
endif()
|
||||
|
||||
if (cudax_ENABLE_PLACES AND NOT "MSVC" STREQUAL "${CMAKE_CXX_COMPILER_ID}")
|
||||
add_subdirectory(places)
|
||||
endif()
|
||||
Reference in New Issue
Block a user