Sparse-checkout from NVIDIA/cccl main branch to complete cccl_upstream: Added: - python/cuda_cccl/ (226 files) — Python bindings for device-level algorithms Critical for muh toolchain: cuda.compute.reduce_into, scan, radix_sort, etc. Includes 204 .py files with full test coverage for all 27 algorithms - ci/ (163 files) — Build/test infrastructure build_cub.sh, test_cub.sh, build_and_test_targets.sh, matrix.yaml Directly maps to our [INFRA-CI] and [INFRA-BUILD] items - .agent/skills/ (7 files) — NVIDIA's own agent skills for CCCL cccl-style/SKILL.md, cccl-test/SKILL.md, sass-diff/SKILL.md - docs/ (491 files) — Official CCCL documentation CI references, CMake guides, Python compute docs, libcudacxx PTX docs - test/ (12 files) — Top-level integration tests (cuda_smoke, stdpar) - Root configs: .clang-format, .clang-tidy, CONTRIBUTING.md, pyproject.toml - CLAUDE.md symlink → AGENTS.md (NVIDIA's standard) cccl_upstream now mirrors full NVIDIA/cccl structure: Before: 42M (cub + thrust + libcudacxx + cudax + c + examples + benchmarks) After: 53M (+python +ci +docs +.agent +test +configs) This completes the CCCL base needed for: - [muh-bench] items: ci/util/build_and_test_targets.sh for targeted builds - [CCCL-verify] items: python/cuda_cccl/tests/ as reference implementations - [CCCL-test] items: ci/test_cub.sh, ci/test_thrust.sh - Agent workflow: .agent/skills/ for consistent style and test patterns
185 lines
6.9 KiB
Bash
Executable File
185 lines
6.9 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
set -euo pipefail
|
|
|
|
ci_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
|
|
usage="Usage: $0 -py-version <python_version> [additional options...]"
|
|
|
|
# shellcheck source=ci/util/python/common_arg_parser.sh
|
|
source "$ci_dir/util/python/common_arg_parser.sh"
|
|
parse_python_args "$@"
|
|
|
|
# Check if py_version was provided (this script requires it)
|
|
require_py_version "$usage" || exit 1
|
|
|
|
echo "Docker socket: " "$(ls /var/run/docker.sock)"
|
|
|
|
if [[ -n "${GITHUB_ACTIONS:-}" ]]; then
|
|
# Prepare mount points etc for getting artifacts in/out of the container.
|
|
# shellcheck source=ci/util/artifacts/common.sh
|
|
source "$ci_dir/util/artifacts/common.sh"
|
|
# Note that these mounts use the runner (not the devcontainer) filesystem for
|
|
# source directories because of docker-out-of-docker quirks.
|
|
# The workflow-job GH actions make sure that they exist before running any
|
|
# scripts.
|
|
action_mounts=(
|
|
--mount "type=bind,source=${ARTIFACT_ARCHIVES},target=${ARTIFACT_ARCHIVES}"
|
|
--mount "type=bind,source=${ARTIFACT_UPLOAD_STAGE},target=${ARTIFACT_UPLOAD_STAGE}"
|
|
)
|
|
else
|
|
# If not running in GitHub Actions, we don't need to set up artifact mounts.
|
|
action_mounts=()
|
|
fi
|
|
|
|
# cuda_cccl must be built in a container that can produce manylinux wheels,
|
|
# and has the CUDA toolkit installed. We use the rapidsai/ci-wheel image for this.
|
|
# We build separate wheels using separate containers for each CUDA version,
|
|
# then merge them into a single wheel.
|
|
|
|
readonly cuda12_version=12.9.1
|
|
readonly cuda13_version=13.1.1
|
|
readonly devcontainer_version=26.04
|
|
readonly devcontainer_distro=rockylinux8
|
|
# Use a baseline Python tag for the rapidsai ci-wheel image. The requested
|
|
# py_version is installed inside the container by setup_python_env (uv).
|
|
# Pinning the image tag avoids relying on a per-py_version image being
|
|
# published (e.g. py3.14 images may not yet exist).
|
|
readonly devcontainer_python_version=3.10
|
|
|
|
if [[ "$(uname -m)" == "aarch64" ]]; then
|
|
cuda12_image="rapidsai/ci-wheel:${devcontainer_version}-cuda${cuda12_version}-${devcontainer_distro}-py${devcontainer_python_version}-arm64"
|
|
cuda13_image="rapidsai/ci-wheel:${devcontainer_version}-cuda${cuda13_version}-${devcontainer_distro}-py${devcontainer_python_version}-arm64"
|
|
else
|
|
cuda12_image="rapidsai/ci-wheel:${devcontainer_version}-cuda${cuda12_version}-${devcontainer_distro}-py${devcontainer_python_version}"
|
|
cuda13_image="rapidsai/ci-wheel:${devcontainer_version}-cuda${cuda13_version}-${devcontainer_distro}-py${devcontainer_python_version}"
|
|
fi
|
|
# shellcheck disable=SC2034
|
|
readonly cuda12_image
|
|
# shellcheck disable=SC2034
|
|
readonly cuda13_image
|
|
|
|
mkdir -p wheelhouse
|
|
|
|
# Shared caches across the cu12 + cu13 wheel builds. Both jobs compile an
|
|
# identical LLVM/clang tree (LLVM has no CUDA dep), so a shared ccache cuts
|
|
# the second build's LLVM phase from ~10 min to under 2 min; a shared CPM
|
|
# source cache skips the second LLVM git clone entirely.
|
|
#
|
|
# The `mkdir`s run inside the (dev)container where only the container-side
|
|
# paths are visible. The docker bind-mount uses the host-side paths
|
|
# (${HOST_WORKSPACE}) since the inner docker daemon is the host's.
|
|
mkdir -p ./.ccache ./.cpm-cache
|
|
host_ccache_dir="${HOST_WORKSPACE:?}/.ccache"
|
|
host_cpm_cache_dir="${HOST_WORKSPACE:?}/.cpm-cache"
|
|
|
|
for ctk in 12 13; do
|
|
image="cuda${ctk}_image"
|
|
image="${!image}"
|
|
echo "::group::⚒️ Building CUDA $ctk wheel on $image"
|
|
(
|
|
set -x
|
|
docker pull "$image"
|
|
docker run --rm -i \
|
|
--workdir /workspace/python/cuda_cccl \
|
|
--mount "type=bind,source=${HOST_WORKSPACE:?},target=/workspace/" \
|
|
--mount "type=bind,source=${host_ccache_dir},target=/root/.ccache" \
|
|
--mount "type=bind,source=${host_cpm_cache_dir},target=/root/.cpm-cache" \
|
|
"${action_mounts[@]}" \
|
|
--env "py_version=${py_version}" \
|
|
--env "GITHUB_ACTIONS=${GITHUB_ACTIONS:-}" \
|
|
--env "GITHUB_RUN_ID=${GITHUB_RUN_ID:-}" \
|
|
--env "JOB_ID=${JOB_ID:-}" \
|
|
--env "CCCL_PYTHON_USE_V2=${CCCL_PYTHON_USE_V2:-}" \
|
|
--env "CCCL_C_PARALLEL_SANITIZE_THREAD=${CCCL_C_PARALLEL_SANITIZE_THREAD:-}" \
|
|
--env "CCACHE_DIR=/root/.ccache" \
|
|
--env "CPM_SOURCE_CACHE=/root/.cpm-cache" \
|
|
"$image" \
|
|
/workspace/ci/build_cuda_cccl_wheel.sh
|
|
# Prevent GHA runners from exhausting available storage with leftover images:
|
|
if [[ -n "${GITHUB_ACTIONS:-}" ]]; then
|
|
docker rmi -f "$image"
|
|
fi
|
|
)
|
|
echo "::endgroup::"
|
|
done
|
|
|
|
echo "Merging CUDA wheels..."
|
|
|
|
# Set up a Python environment for the merge/repair steps.
|
|
source "$ci_dir/pyenv_helper.sh"
|
|
setup_python_env "${py_version}"
|
|
|
|
# Needed for unpacking and repacking wheels.
|
|
python -m pip install wheel
|
|
|
|
# Find the built wheels
|
|
cu12_wheel=$(find wheelhouse -name "*cu12*.whl" | head -1)
|
|
cu13_wheel=$(find wheelhouse -name "*cu13*.whl" | head -1)
|
|
|
|
if [[ -z "$cu12_wheel" ]]; then
|
|
echo "Error: CUDA 12 wheel not found in wheelhouse/"
|
|
ls -la wheelhouse/
|
|
exit 1
|
|
fi
|
|
|
|
if [[ -z "$cu13_wheel" ]]; then
|
|
echo "Error: CUDA 13 wheel not found in wheelhouse/"
|
|
ls -la wheelhouse/
|
|
exit 1
|
|
fi
|
|
|
|
echo "Found CUDA 12 wheel: $cu12_wheel"
|
|
echo "Found CUDA 13 wheel: $cu13_wheel"
|
|
|
|
# Merge the wheels
|
|
python python/cuda_cccl/merge_cuda_wheels.py "$cu12_wheel" "$cu13_wheel" --output-dir wheelhouse_merged
|
|
|
|
# A ThreadSanitizer wheel links libtsan; keep it external (excluded) so it is
|
|
# NOT bundled -- the TSan test lane LD_PRELOADs the runner's matching libtsan
|
|
# instead. Harmless for normal builds (the .so has no libtsan dependency).
|
|
tsan_exclude=()
|
|
if [[ "${CCCL_C_PARALLEL_SANITIZE_THREAD:-}" =~ ^(1|true|TRUE|on|ON)$ ]]; then
|
|
tsan_exclude=(--exclude 'libtsan.so.2')
|
|
fi
|
|
|
|
# Install auditwheel and repair the merged wheel
|
|
python -m pip install patchelf auditwheel
|
|
for wheel in wheelhouse_merged/cuda_cccl-*.whl; do
|
|
echo "Repairing merged wheel: $wheel"
|
|
python -m auditwheel repair \
|
|
--exclude 'libnvrtc.so.12' \
|
|
--exclude 'libnvrtc.so.13' \
|
|
--exclude 'libnvJitLink.so.12' \
|
|
--exclude 'libnvJitLink.so.13' \
|
|
--exclude 'libcudart.so.12' \
|
|
--exclude 'libcudart.so.13' \
|
|
--exclude 'libcuda.so.1' \
|
|
"${tsan_exclude[@]}" \
|
|
"$wheel" \
|
|
--wheel-dir wheelhouse_final
|
|
done
|
|
|
|
# Clean up intermediate files and move only the final merged wheel to wheelhouse
|
|
rm -rf wheelhouse/* # Clean existing wheelhouse
|
|
mkdir -p wheelhouse
|
|
|
|
# Move only the final repaired merged wheel
|
|
if ls wheelhouse_final/cuda_cccl-*.whl 1> /dev/null 2>&1; then
|
|
mv wheelhouse_final/cuda_cccl-*.whl wheelhouse/
|
|
echo "Final merged wheel moved to wheelhouse"
|
|
else
|
|
echo "No final repaired wheel found, moving unrepaired merged wheel"
|
|
mv wheelhouse_merged/cuda_cccl-*.whl wheelhouse/
|
|
fi
|
|
|
|
# Clean up temporary directories
|
|
rm -rf wheelhouse_merged wheelhouse_final
|
|
|
|
echo "Final wheels in wheelhouse:"
|
|
ls -la wheelhouse/
|
|
|
|
if [[ -n "${GITHUB_ACTIONS:-}" ]]; then
|
|
wheel_artifact_name="$(ci/util/workflow/get_wheel_artifact_name.sh)"
|
|
ci/util/artifacts/upload.sh "$wheel_artifact_name" 'wheelhouse/.*'
|
|
fi
|