[CCCL] 瘦身 + 补全: 移除 cudax/python/libcudacxx-tests 冗余文件, 新增 c2h 测试助手 + cmake 构建系统 + 8 个 CUDA thrust examples
变更摘要:
- 删除: cudax/ (783 files, 7.2M) — 实验性组件,竞赛不需要
- 删除: python/ (226 files, 2.0M) — Python 绑定,竞赛不需要
- 删除: libcudacxx/{test,benchmarks,codegen,cmake,share} (4432 files, 31M)
保留: libcudacxx/include/ (1463 headers, cuda::std 编译依赖)
- 新增: c2h/ (27 files) — CUB Catch2 测试辅助头文件,编译 243 个测试必需
- 新增: cmake/ (29 files) — CCCL 原生 CMake 构建系统
- 新增: thrust/examples/cuda/ (7 files) + cpp_integration/ (1 file)
async_reduce, custom_temporary_allocation, explicit_cuda_stream,
global_device_vector, range_view, unwrap_pointer, wrap_pointer, device
结果: cccl_upstream 从 74M→35M (瘦身 53%), 核心内容 100% 保留:
27/27 tuning headers, 78 benchmarks, 243 tests,
60 thrust examples, 18 CUB examples, 全部编译头文件
This commit is contained in:
@@ -1,183 +0,0 @@
|
||||
# Copyright (c) 2025, NVIDIA CORPORATION & AFFILIATES. ALL RIGHTS RESERVED.
|
||||
#
|
||||
# SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
|
||||
|
||||
cmake_minimum_required(VERSION 3.30)
|
||||
|
||||
# Must be set before project() initializes the CUDA language; otherwise CMake
|
||||
# < 3.23 defaults to sm_52, which is below CCCL's minimum supported arch.
|
||||
include(../../cmake/CCCLCheckCudaArchitectures.cmake)
|
||||
set(
|
||||
CMAKE_CUDA_ARCHITECTURES
|
||||
"${minimum_cccl_arch}"
|
||||
CACHE STRING
|
||||
"CUDA architectures for CCCL"
|
||||
)
|
||||
|
||||
project(cuda_cccl DESCRIPTION "Python package cuda_cccl" LANGUAGES CUDA CXX C)
|
||||
|
||||
find_package(CUDAToolkit REQUIRED)
|
||||
|
||||
set(CUDA_VERSION_MAJOR ${CUDAToolkit_VERSION_MAJOR})
|
||||
set(CUDA_VERSION_DIR "cu${CUDA_VERSION_MAJOR}")
|
||||
message(
|
||||
STATUS
|
||||
"Building for CUDA ${CUDA_VERSION_MAJOR}, output directory: ${CUDA_VERSION_DIR}"
|
||||
)
|
||||
|
||||
# Build cuda_cccl against either cccl.c.parallel (v1, NVRTC) by default or
|
||||
# cccl.c.parallel.v2 (HostJIT) when CCCL_PYTHON_USE_V2=ON. v2 is opt-in until
|
||||
# it replaces v1 across the matrix.
|
||||
set(_cccl_root ../..)
|
||||
set(CCCL_TOPLEVEL_PROJECT ON) # Enable the developer builds
|
||||
option(
|
||||
CCCL_PYTHON_USE_V2
|
||||
"Build cuda_cccl against cccl.c.parallel.v2 (HostJIT)."
|
||||
OFF
|
||||
)
|
||||
if (CCCL_PYTHON_USE_V2)
|
||||
set(CCCL_ENABLE_C_PARALLEL_V2 ON)
|
||||
set(CCCL_C_PARALLEL_V2_LIBRARY_OUTPUT_DIRECTORY ${SKBUILD_PROJECT_NAME})
|
||||
set(_cccl_c_parallel_target cccl.c.parallel.v2)
|
||||
set(_using_v2_py "True")
|
||||
else()
|
||||
set(CCCL_ENABLE_C_PARALLEL ON)
|
||||
set(CCCL_C_PARALLEL_LIBRARY_OUTPUT_DIRECTORY ${SKBUILD_PROJECT_NAME})
|
||||
set(_cccl_c_parallel_target cccl.c.parallel)
|
||||
set(_using_v2_py "False")
|
||||
endif()
|
||||
|
||||
# Surface the v1/v2 choice to Python (tests use it to skip v2-only failures,
|
||||
# and __init__.py uses it to wire up wheel-bundled hostjit header paths).
|
||||
# Generated into the build dir and installed via CMake — writing into the
|
||||
# source tree would miss scikit-build-core's package-file snapshot.
|
||||
set(_build_info_py "${CMAKE_CURRENT_BINARY_DIR}/_build_info.py")
|
||||
file(
|
||||
WRITE "${_build_info_py}"
|
||||
"# Auto-generated by CMakeLists.txt; do not edit.\nUSING_V2 = ${_using_v2_py}\n"
|
||||
)
|
||||
install(FILES "${_build_info_py}" DESTINATION cuda/compute)
|
||||
# Just install the rest:
|
||||
set(libcudacxx_ENABLE_INSTALL_RULES ON)
|
||||
set(CUB_ENABLE_INSTALL_RULES ON)
|
||||
set(Thrust_ENABLE_INSTALL_RULES ON)
|
||||
# Install to our output location:
|
||||
include(GNUInstallDirs)
|
||||
set(old_libdir "${CMAKE_INSTALL_LIBDIR}") # push
|
||||
set(old_includedir "${CMAKE_INSTALL_INCLUDEDIR}") # push
|
||||
set(CMAKE_INSTALL_LIBDIR "cuda/cccl/headers/lib")
|
||||
set(CMAKE_INSTALL_INCLUDEDIR "cuda/cccl/headers/include")
|
||||
add_subdirectory(${_cccl_root} _parent_cccl)
|
||||
set(CMAKE_INSTALL_LIBDIR "${old_libdir}") # pop
|
||||
set(CMAKE_INSTALL_INCLUDEDIR "${old_includedir}") # pop
|
||||
|
||||
# Install version-specific binaries
|
||||
install(
|
||||
TARGETS ${_cccl_c_parallel_target}
|
||||
DESTINATION cuda/compute/${CUDA_VERSION_DIR}/cccl
|
||||
)
|
||||
|
||||
# Build and install Cython extension
|
||||
find_package(Python3 COMPONENTS Interpreter Development.Module REQUIRED)
|
||||
|
||||
set(CYTHON_version_command "${Python3_EXECUTABLE}" -m cython --version)
|
||||
execute_process(
|
||||
COMMAND ${CYTHON_version_command}
|
||||
OUTPUT_VARIABLE CYTHON_version_output
|
||||
ERROR_VARIABLE CYTHON_version_output
|
||||
OUTPUT_STRIP_TRAILING_WHITESPACE
|
||||
ERROR_STRIP_TRAILING_WHITESPACE
|
||||
COMMAND_ERROR_IS_FATAL ANY
|
||||
)
|
||||
|
||||
if ("${CYTHON_version_output}" MATCHES "^[Cc]ython version ([^,]+)")
|
||||
set(CYTHON_VERSION "${CMAKE_MATCH_1}")
|
||||
else()
|
||||
message(
|
||||
FATAL_ERROR
|
||||
"Failed to parse Cython version from:\n${CYTHON_version_output}"
|
||||
)
|
||||
endif()
|
||||
|
||||
# -3 generates source for Python 3
|
||||
# -M generates depfile
|
||||
# -t cythonizes if PYX is newer than preexisting output
|
||||
# -w sets working directory
|
||||
set(
|
||||
CYTHON_FLAGS
|
||||
-3
|
||||
-M
|
||||
-t
|
||||
-w
|
||||
"${cuda_cccl_SOURCE_DIR}"
|
||||
)
|
||||
|
||||
message(STATUS "Using Cython ${CYTHON_VERSION}")
|
||||
set(pyx_source_file "${cuda_cccl_SOURCE_DIR}/cuda/compute/_bindings_impl.pyx")
|
||||
|
||||
set(_generated_extension_src "${cuda_cccl_BINARY_DIR}/_bindings_impl.c")
|
||||
set(_depfile "${cuda_cccl_BINARY_DIR}/_bindings_impl.c.dep")
|
||||
|
||||
# Backend-conditional Cython .pxi files. Where v1 and v2 expose different
|
||||
# struct layouts or call signatures, the .pyx `include`s a generated .pxi
|
||||
# whose source is chosen here. The helpers inside present a uniform interface
|
||||
# so the rest of _bindings_impl.pyx stays backend-agnostic.
|
||||
if (CCCL_PYTHON_USE_V2)
|
||||
set(_backend_suffix "v2")
|
||||
else()
|
||||
set(_backend_suffix "v1")
|
||||
endif()
|
||||
foreach (
|
||||
_pxi_stem
|
||||
segmented_reduce_backend
|
||||
binary_search_backend
|
||||
op_code_type
|
||||
serialization
|
||||
)
|
||||
configure_file(
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/cuda/compute/_bindings_${_pxi_stem}_${_backend_suffix}.pxi"
|
||||
"${CMAKE_CURRENT_BINARY_DIR}/_bindings_${_pxi_stem}.pxi"
|
||||
COPYONLY
|
||||
)
|
||||
endforeach()
|
||||
|
||||
# Custom Cython compilation command. `-I ${BINARY_DIR}` lets the .pyx's
|
||||
# `include "_bindings_..._backend.pxi"` resolve to the file we configured
|
||||
# above.
|
||||
add_custom_command(
|
||||
OUTPUT "${_generated_extension_src}"
|
||||
COMMAND
|
||||
"${Python3_EXECUTABLE}" -m cython
|
||||
# gersemi: off
|
||||
${CYTHON_FLAGS}
|
||||
-I "${CMAKE_CURRENT_BINARY_DIR}"
|
||||
"${pyx_source_file}"
|
||||
--output-file "${_generated_extension_src}"
|
||||
# gersemi: on
|
||||
DEPENDS "${pyx_source_file}"
|
||||
DEPFILE "${_depfile}"
|
||||
COMMENT "Cythonizing ${pyx_source_file} for CUDA ${CUDA_VERSION_MAJOR}"
|
||||
)
|
||||
|
||||
add_custom_target(
|
||||
cythonize_bindings_impl
|
||||
ALL
|
||||
DEPENDS "${_generated_extension_src}"
|
||||
)
|
||||
|
||||
python3_add_library(
|
||||
_bindings_impl
|
||||
MODULE
|
||||
WITH_SOABI
|
||||
"${_generated_extension_src}"
|
||||
)
|
||||
add_dependencies(_bindings_impl cythonize_bindings_impl)
|
||||
target_link_libraries(
|
||||
_bindings_impl
|
||||
PRIVATE #
|
||||
${_cccl_c_parallel_target}
|
||||
CUDA::cuda_driver
|
||||
)
|
||||
set_target_properties(_bindings_impl PROPERTIES INSTALL_RPATH "$ORIGIN/cccl")
|
||||
|
||||
install(TARGETS _bindings_impl DESTINATION cuda/compute/${CUDA_VERSION_DIR})
|
||||
Reference in New Issue
Block a user