feat: CCCL CachingDeviceAllocator preload — 完整依赖链 288 files
从 cccl_upstream 递归追踪 cub/util_allocator.cuh 的全部 include 依赖: cub/ 9 files (config, util_*, version, detect_cuda_runtime) cuda/ libcudacxx type_traits, concepts, algorithm, iterator... nv/ target macros, preprocessor 总计 288 个头文件 (1.4MB),打包到 include/ 目录,编译时 -I include 即可完全脱离 CCCL 原始目录结构。 .cu 文件直接 #include <cub/util_allocator.cuh>, 走原版 CUB CachingDeviceAllocator,零 mock。 BI-V100 参数: growth=2 bins=[8..32] max_cached=8GB/device
This commit is contained in:
29
qwen3_6_scripts/cccl_preload/include/cub/config.cuh
Normal file
29
qwen3_6_scripts/cccl_preload/include/cub/config.cuh
Normal file
@@ -0,0 +1,29 @@
|
||||
// SPDX-FileCopyrightText: Copyright (c) 2020, NVIDIA CORPORATION. All rights reserved.
|
||||
// SPDX-License-Identifier: BSD-3
|
||||
|
||||
/**
|
||||
* \file
|
||||
* Static configuration header for the CUB project.
|
||||
*/
|
||||
|
||||
#pragma once
|
||||
|
||||
// For _CCCL_IMPLICIT_SYSTEM_HEADER
|
||||
#include <cuda/__cccl_config> // IWYU pragma: export
|
||||
|
||||
#if defined(_CCCL_IMPLICIT_SYSTEM_HEADER_GCC)
|
||||
# pragma GCC system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_CLANG)
|
||||
# pragma clang system_header
|
||||
#elif defined(_CCCL_IMPLICIT_SYSTEM_HEADER_MSVC)
|
||||
# pragma system_header
|
||||
#endif // no system header
|
||||
|
||||
#include <cub/util_arch.cuh> // IWYU pragma: export
|
||||
#include <cub/util_cpp_dialect.cuh> // IWYU pragma: export
|
||||
#include <cub/util_macro.cuh> // IWYU pragma: export
|
||||
#include <cub/util_namespace.cuh> // IWYU pragma: export
|
||||
|
||||
#if !_CCCL_COMPILER(NVRTC)
|
||||
# include <cuda/__nvtx/nvtx.h>
|
||||
#endif // !_CCCL_COMPILER(NVRTC)
|
||||
Reference in New Issue
Block a user