diff --git a/computility-run.yaml b/computility-run.yaml index 226cd7f1..568b4ca1 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -49,3 +49,5 @@ env: value: '1' - name: PYTORCH_CUDA_ALLOC_CONF value: max_split_size_mb:512 + - name: BI100_MAX_GPU_BLOCKS + value: '5000' diff --git a/qwen3_6_scripts/patch_block_major_worker_capacity.py b/qwen3_6_scripts/patch_block_major_worker_capacity.py index 599e194e..7ef25aa0 100644 --- a/qwen3_6_scripts/patch_block_major_worker_capacity.py +++ b/qwen3_6_scripts/patch_block_major_worker_capacity.py @@ -20,6 +20,14 @@ CAPACITY_ANCHOR = """\ CAPACITY_REPLACEMENT = """\ num_gpu_blocks = reserve_block_major_gpu_blocks( num_gpu_blocks, cache_block_size) + # BI100: cap GPU blocks — profiling with zero-tensor attention + # underestimates memory, causing runtime OOM if uncapped. + _bi100_max = int(os.environ.get("BI100_MAX_GPU_BLOCKS", "0")) + if _bi100_max > 0 and num_gpu_blocks > _bi100_max: + logger.warning( + "[BI100] capping num_gpu_blocks: %d -> %d (BI100_MAX_GPU_BLOCKS)", + num_gpu_blocks, _bi100_max) + num_gpu_blocks = _bi100_max num_gpu_blocks = max(num_gpu_blocks, 0) num_cpu_blocks = max(num_cpu_blocks, 0) """