Adds ALL files needed for Dockerfile build:
- qwen3_6_scripts/ (baseline patches + our optimizations)
- vllm/ (full vllm package)
- paged_attention_v2_pytorch.py (V2 with single-bmm optimization)
- Dockerfile + computility-run.yaml
Our optimizations vs baseline:
1. paged_attn.py: pre-gathered context KV (eliminates 194 gather calls),
Triton try/fallback, V2 heuristic, threshold 32K→64K
2. paged_attention_v2_pytorch.py: fills NotImplementedError,
single-bmm Phase 1 (195 launches → 3)
3. patch_enable_triton.py: HAS_TRITON=True with safety fallback
4. patch_triton_tuning.py: BLOCK=64, NUM_WARPS=4 for BI-V100
5. computility-run.yaml: gpu-memory-utilization 0.9→0.95,
max-num-batched-tokens 8192→16384
This repo can now be submitted to dev.modelhub.org.cn as-is.
109 lines
2.9 KiB
Python
109 lines
2.9 KiB
Python
import enum
|
|
from typing import NamedTuple, Optional, Tuple, Union
|
|
|
|
import torch
|
|
|
|
|
|
class PlatformEnum(enum.Enum):
|
|
CUDA = enum.auto()
|
|
ROCM = enum.auto()
|
|
TPU = enum.auto()
|
|
XPU = enum.auto()
|
|
CPU = enum.auto()
|
|
UNSPECIFIED = enum.auto()
|
|
|
|
|
|
class DeviceCapability(NamedTuple):
|
|
major: int
|
|
minor: int
|
|
|
|
def as_version_str(self) -> str:
|
|
return f"{self.major}.{self.minor}"
|
|
|
|
def to_int(self) -> int:
|
|
"""
|
|
Express device capability as an integer ``<major><minor>``.
|
|
|
|
It is assumed that the minor version is always a single digit.
|
|
"""
|
|
assert 0 <= self.minor < 10
|
|
return self.major * 10 + self.minor
|
|
|
|
|
|
class Platform:
|
|
_enum: PlatformEnum
|
|
|
|
def is_cuda(self) -> bool:
|
|
return self._enum == PlatformEnum.CUDA
|
|
|
|
def is_rocm(self) -> bool:
|
|
return self._enum == PlatformEnum.ROCM
|
|
|
|
def is_tpu(self) -> bool:
|
|
return self._enum == PlatformEnum.TPU
|
|
|
|
def is_xpu(self) -> bool:
|
|
return self._enum == PlatformEnum.XPU
|
|
|
|
def is_cpu(self) -> bool:
|
|
return self._enum == PlatformEnum.CPU
|
|
|
|
def is_cuda_alike(self) -> bool:
|
|
"""Stateless version of :func:`torch.cuda.is_available`."""
|
|
return self._enum in (PlatformEnum.CUDA, PlatformEnum.ROCM)
|
|
|
|
@classmethod
|
|
def get_device_capability(
|
|
cls,
|
|
device_id: int = 0,
|
|
) -> Optional[DeviceCapability]:
|
|
"""Stateless version of :func:`torch.cuda.get_device_capability`."""
|
|
return None
|
|
|
|
@classmethod
|
|
def has_device_capability(
|
|
cls,
|
|
capability: Union[Tuple[int, int], int],
|
|
device_id: int = 0,
|
|
) -> bool:
|
|
"""
|
|
Test whether this platform is compatible with a device capability.
|
|
|
|
The ``capability`` argument can either be:
|
|
|
|
- A tuple ``(major, minor)``.
|
|
- An integer ``<major><minor>``. (See :meth:`DeviceCapability.to_int`)
|
|
"""
|
|
current_capability = cls.get_device_capability(device_id=device_id)
|
|
if current_capability is None:
|
|
return False
|
|
|
|
if isinstance(capability, tuple):
|
|
return current_capability >= capability
|
|
|
|
return current_capability.to_int() >= capability
|
|
|
|
@classmethod
|
|
def get_device_name(cls, device_id: int = 0) -> str:
|
|
"""Get the name of a device."""
|
|
raise NotImplementedError
|
|
|
|
@classmethod
|
|
def get_device_total_memory(cls, device_id: int = 0) -> int:
|
|
"""Get the total memory of a device in bytes."""
|
|
raise NotImplementedError
|
|
|
|
@classmethod
|
|
def inference_mode(cls):
|
|
"""A device-specific wrapper of `torch.inference_mode`.
|
|
|
|
This wrapper is recommended because some hardware backends such as TPU
|
|
do not support `torch.inference_mode`. In such a case, they will fall
|
|
back to `torch.no_grad` by overriding this method.
|
|
"""
|
|
return torch.inference_mode(mode=True)
|
|
|
|
|
|
class UnspecifiedPlatform(Platform):
|
|
_enum = PlatformEnum.UNSPECIFIED
|