Files
project_6/qwen3_6_scripts/cccl_preload/verify_preload.py
dylanyunlon 8dc6462a2b feat: CCCL CachingDeviceAllocator LD_PRELOAD — bypass CoreX expandable_segments ASSERT
从 CCCL upstream cub/cub/util_allocator.cuh 提取 CachingDeviceAllocator
核心算法,去掉所有 CUB/CCCL 宏依赖,编译为独立 .so。

用 LD_PRELOAD 拦截 cudaMalloc/cudaFree,路由到 CUB 的 geometric-bin
缓存分配器。同时在 constructor 中 strip PYTORCH_CUDA_ALLOC_CONF 里的
expandable_segments 配置,避免 CoreX CUDACachingAllocator.cpp:545 ASSERT。

BI-V100 调优参数:
  bin_growth=8, min_bin=3 (512B), max_bin=13 (~550MB)
  max_cached_bytes=4GB per device (32GB卡的合理上限)

真机测试步骤:
  1. bash build_cccl_preload.sh
  2. LD_PRELOAD=./libcccl_allocator.so CCCL_ALLOC_DEBUG=1 \
     PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True \
     python3 verify_preload.py
2026-08-13 09:53:42 +00:00

82 lines
2.7 KiB
Python

#!/usr/bin/env python3
"""
Verify CCCL allocator preload on BI-V100.
Run WITHOUT preload (should crash with expandable_segments:True):
PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True python3 verify_preload.py
Run WITH preload (should succeed):
LD_PRELOAD=./libcccl_allocator.so CCCL_ALLOC_DEBUG=1 \
PYTORCH_CUDA_ALLOC_CONF=expandable_segments:True python3 verify_preload.py
"""
import os
import sys
import time
print(f"PYTORCH_CUDA_ALLOC_CONF = {os.environ.get('PYTORCH_CUDA_ALLOC_CONF', '(not set)')}")
print(f"LD_PRELOAD = {os.environ.get('LD_PRELOAD', '(not set)')}")
print()
import torch
print(f"torch version: {torch.__version__}")
print(f"CUDA available: {torch.cuda.is_available()}")
if not torch.cuda.is_available():
print("CUDA not available, exiting")
sys.exit(1)
device = torch.device("cuda:0")
print(f"Device: {torch.cuda.get_device_name(0)}")
print(f"Memory: {torch.cuda.get_device_properties(0).total_mem / 1024**3:.1f} GB")
print()
# Test 1: Basic allocation
print("=== Test 1: Basic allocation ===")
t1 = torch.zeros(1024, 1024, dtype=torch.float16, device=device)
print(f" Allocated 1024x1024 fp16 tensor: {t1.shape}, {t1.element_size() * t1.nelement() / 1024**2:.1f} MB")
# Test 2: Large allocation (simulates model weight loading)
print("=== Test 2: Large allocation (512MB) ===")
t2 = torch.zeros(256 * 1024 * 1024, dtype=torch.float16, device=device)
print(f" Allocated 512MB tensor: {t2.nelement() * t2.element_size() / 1024**2:.0f} MB")
# Test 3: Alloc-free-realloc cycle (tests caching)
print("=== Test 3: Alloc-free-realloc cycle ===")
t3 = torch.zeros(64 * 1024 * 1024, dtype=torch.float16, device=device)
ptr_first = t3.data_ptr()
del t3
torch.cuda.empty_cache()
t3b = torch.zeros(64 * 1024 * 1024, dtype=torch.float16, device=device)
ptr_second = t3b.data_ptr()
reused = "YES (cached)" if ptr_first == ptr_second else "NO (new alloc)"
print(f" First ptr: 0x{ptr_first:x}")
print(f" Second ptr: 0x{ptr_second:x}")
print(f" Block reused: {reused}")
# Test 4: Multiple sizes (tests bin routing)
print("=== Test 4: Multiple bin sizes ===")
sizes = [512, 4096, 32768, 262144, 2*1024*1024, 32*1024*1024]
tensors = []
for sz in sizes:
t = torch.zeros(sz // 2, dtype=torch.float16, device=device)
tensors.append(t)
print(f" {sz:>12} bytes -> allocated at 0x{t.data_ptr():x}")
del tensors
# Test 5: OOM recovery
print("=== Test 5: Memory pressure ===")
mem_free = torch.cuda.mem_get_info()[0]
print(f" Free memory: {mem_free / 1024**3:.2f} GB")
# Cleanup
del t1, t2, t3b
torch.cuda.empty_cache()
mem_after = torch.cuda.mem_get_info()[0]
print(f" After cleanup: {mem_after / 1024**3:.2f} GB")
print(f" Recovered: {(mem_after - mem_free) / 1024**2:.0f} MB")
print()
print("ALL TESTS PASSED")