Files
project_6/cccl_upstream/ci/build_cuda_cccl_wheel.sh
muh-bot 2a7ca101d7 feat(cccl): integrate missing CCCL directories — python/, ci/, .agent/, docs/, test/
Sparse-checkout from NVIDIA/cccl main branch to complete cccl_upstream:

Added:
- python/cuda_cccl/ (226 files) — Python bindings for device-level algorithms
  Critical for muh toolchain: cuda.compute.reduce_into, scan, radix_sort, etc.
  Includes 204 .py files with full test coverage for all 27 algorithms
- ci/ (163 files) — Build/test infrastructure
  build_cub.sh, test_cub.sh, build_and_test_targets.sh, matrix.yaml
  Directly maps to our [INFRA-CI] and [INFRA-BUILD] items
- .agent/skills/ (7 files) — NVIDIA's own agent skills for CCCL
  cccl-style/SKILL.md, cccl-test/SKILL.md, sass-diff/SKILL.md
- docs/ (491 files) — Official CCCL documentation
  CI references, CMake guides, Python compute docs, libcudacxx PTX docs
- test/ (12 files) — Top-level integration tests (cuda_smoke, stdpar)
- Root configs: .clang-format, .clang-tidy, CONTRIBUTING.md, pyproject.toml
- CLAUDE.md symlink → AGENTS.md (NVIDIA's standard)

cccl_upstream now mirrors full NVIDIA/cccl structure:
  Before: 42M (cub + thrust + libcudacxx + cudax + c + examples + benchmarks)
  After:  53M (+python +ci +docs +.agent +test +configs)

This completes the CCCL base needed for:
- [muh-bench] items: ci/util/build_and_test_targets.sh for targeted builds
- [CCCL-verify] items: python/cuda_cccl/tests/ as reference implementations
- [CCCL-test] items: ci/test_cub.sh, ci/test_thrust.sh
- Agent workflow: .agent/skills/ for consistent style and test patterns
2026-08-07 02:34:33 +00:00

131 lines
5.3 KiB
Bash
Executable File

#!/usr/bin/env bash
set -euo pipefail
# Target script for `docker run` command in build_cuda_cccl_python.sh
# The /workspace pathnames are hard-wired here.
# Toolchain version, pinned in one place. Every gcc-toolset package name and the
# /opt/rh enable path below derive from it — the -fsanitize=thread libtsan must
# come from the same toolset as the compiler, so these cannot drift apart.
readonly gcc_toolset_version=13
# Install the GCC toolset (needed for the build) and ccache (shared between
# cu12 and cu13 builds via /root/.ccache bind-mount from the host).
/workspace/ci/util/retry.sh 5 30 dnf -y install \
"gcc-toolset-${gcc_toolset_version}-gcc" "gcc-toolset-${gcc_toolset_version}-gcc-c++" ccache
# ThreadSanitizer builds link libcccl.c.parallel.so with -fsanitize=thread, which
# the linker resolves via the toolset's libtsan.so. That runtime lives in a
# separate package not pulled in by -gcc-c++, so install it for the TSan lane.
if [[ "${CCCL_C_PARALLEL_SANITIZE_THREAD:-}" =~ ^(1|true|TRUE|on|ON)$ ]]; then
/workspace/ci/util/retry.sh 5 30 dnf -y install "gcc-toolset-${gcc_toolset_version}-libtsan-devel"
fi
# When the caller bind-mounts a ccache dir, wire it through to CMake. This
# transparently caches every compile, so the second wheel build (cu13 after
# cu12, or vice versa) reuses the entire LLVM/clang object tree.
if [[ -n "${CCACHE_DIR:-}" ]]; then
export CMAKE_C_COMPILER_LAUNCHER=ccache
export CMAKE_CXX_COMPILER_LAUNCHER=ccache
export CMAKE_CUDA_COMPILER_LAUNCHER=ccache
echo "ccache enabled: CCACHE_DIR=${CCACHE_DIR}"
ccache --version 2>&1 | head -1 || true
ccache --show-stats 2>&1 | head -5 || true
fi
echo -e "#!/usr/bin/env bash\nsource /opt/rh/gcc-toolset-${gcc_toolset_version}/enable" >/etc/profile.d/enable_devtools.sh
# shellcheck disable=SC1091
source /etc/profile.d/enable_devtools.sh
# Check what's available
command -v gcc
gcc --version
command -v nvcc
nvcc --version
# Set up Python environment
# shellcheck source=ci/pyenv_helper.sh
source /workspace/ci/pyenv_helper.sh
# shellcheck disable=SC2154
setup_python_env "${py_version}"
command -v python
python --version
echo "Done setting up python env"
# Figure out the version to use for the package, we need repo history
if "$(git rev-parse --is-shallow-repository)"; then
git fetch --unshallow
fi
export PACKAGE_VERSION_PREFIX="0.1."
package_version=$(/workspace/ci/generate_version.sh)
echo "Using package version ${package_version}"
# Override the version used by setuptools_scm to the custom version
export SETUPTOOLS_SCM_PRETEND_VERSION_FOR_CUDA_CCCL="${package_version}"
cd /workspace/python/cuda_cccl
# Determine CUDA version from nvcc
cuda_version=$(nvcc --version | grep -oP 'release \K[0-9]+\.[0-9]+' | cut -d. -f1)
echo "Detected CUDA version: ${cuda_version}"
# Configure compilers:
CXX="$(command -v g++)"
export CXX
CUDACXX="$(command -v nvcc)"
export CUDACXX
CUDAHOSTCXX="$(command -v g++)"
export CUDAHOSTCXX
# When CCCL_PYTHON_USE_V2 is set (=1/true/on), build the wheel against the
# HostJIT-based cccl.c.parallel.v2 library instead of the default v1.
if [[ "${CCCL_PYTHON_USE_V2:-}" =~ ^(1|true|TRUE|on|ON)$ ]]; then
export CMAKE_ARGS="${CMAKE_ARGS:-} -DCCCL_PYTHON_USE_V2=ON"
echo "Building wheel with CCCL v2 backend: CMAKE_ARGS=${CMAKE_ARGS}"
# v2's hostjit links against libnvJitLink and libnvfatbin, which aren't in
# the base rapidsai/ci-wheel image. Install the matching CTK devel packages
# so CMake's FindCUDAToolkit picks them up. nvcc is on PATH; derive the
# version (e.g. "13-0") from it.
ctk_pkg_ver=$(nvcc --version 2>/dev/null \
| grep -oP 'release \K[0-9]+\.[0-9]+' | tr '.' '-')
if [[ -n "${ctk_pkg_ver}" ]]; then
echo "Installing libnvjitlink-devel-${ctk_pkg_ver} libnvfatbin-devel-${ctk_pkg_ver}..."
/workspace/ci/util/retry.sh 5 30 dnf -y install \
"libnvjitlink-devel-${ctk_pkg_ver}" \
"libnvfatbin-devel-${ctk_pkg_ver}"
else
echo "WARNING: could not derive CTK version from nvcc; skipping nvJitLink/nvfatbin install"
fi
# FindCUDAToolkit learned about CUDA::nvfatbin only in CMake 3.27. The base
# rapidsai/ci-wheel image ships an older CMake; install a newer one into
# the active venv so scikit-build-core picks it up over the system cmake.
echo "Pinning cmake>=3.27 for FindCUDAToolkit nvfatbin support..."
python -m pip install --upgrade 'cmake>=3.27'
fi
# When CCCL_C_PARALLEL_SANITIZE_THREAD is set (=1/true/on), instrument the
# c.parallel (v1) host code with ThreadSanitizer for the free-threaded TSan
# nightly lane. Host-only; the libtsan runtime stays external (the shared
# build_cuda_cccl_python.sh --excludes it from auditwheel).
if [[ "${CCCL_C_PARALLEL_SANITIZE_THREAD:-}" =~ ^(1|true|TRUE|on|ON)$ ]]; then
export CMAKE_ARGS="${CMAKE_ARGS:-} -DCCCL_C_PARALLEL_SANITIZE_THREAD=ON"
echo "Building wheel with ThreadSanitizer-instrumented c.parallel: CMAKE_ARGS=${CMAKE_ARGS}"
fi
# Build the wheel
python -m pip wheel --no-deps --verbose --wheel-dir dist .
# Rename wheel to include CUDA version suffix
for wheel in dist/cuda_cccl-*.whl; do
if [[ -f "$wheel" ]]; then
base_name=$(basename "$wheel" .whl)
new_name="${base_name}.cu${cuda_version}.whl"
mv "$wheel" "dist/${new_name}"
echo "Renamed wheel to: ${new_name}"
fi
done
# Move wheel to output directory
mkdir -p /workspace/wheelhouse
mv dist/cuda_cccl-*.cu*.whl /workspace/wheelhouse/