Compare commits
6 Commits
e31bd69779
...
aa4b4992d1
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
aa4b4992d1 | ||
|
|
456380eed0 | ||
|
|
c6aa1b9c62 | ||
|
|
20aac5b212 | ||
|
|
048302bd4a | ||
|
|
15ad56a454 |
@@ -19,7 +19,7 @@ command:
|
||||
- --disable-log-requests
|
||||
- --disable-frontend-multiprocessing
|
||||
- --max-num-batched-tokens
|
||||
- '256'
|
||||
- '4096'
|
||||
- --enable-chunked-prefill
|
||||
- --max-seq-len-to-capture
|
||||
- '32768'
|
||||
@@ -49,4 +49,3 @@ env:
|
||||
value: '1'
|
||||
- name: PYTORCH_CUDA_ALLOC_CONF
|
||||
value: max_split_size_mb:512
|
||||
|
||||
|
||||
@@ -20,6 +20,12 @@ CAPACITY_ANCHOR = """\
|
||||
CAPACITY_REPLACEMENT = """\
|
||||
num_gpu_blocks = reserve_block_major_gpu_blocks(
|
||||
num_gpu_blocks, cache_block_size)
|
||||
# BI100: profiling with zero-tensor attention underestimates memory.
|
||||
# Hardcap at 5000 blocks (80K tokens) to prevent runtime OOM.
|
||||
if num_gpu_blocks > 5000:
|
||||
logger.warning(
|
||||
"[BI100] capping num_gpu_blocks: %d -> 5000", num_gpu_blocks)
|
||||
num_gpu_blocks = 5000
|
||||
num_gpu_blocks = max(num_gpu_blocks, 0)
|
||||
num_cpu_blocks = max(num_cpu_blocks, 0)
|
||||
"""
|
||||
|
||||
@@ -219,6 +219,11 @@ FALLBACK_METHOD = '''
|
||||
# Fallback: pure-math Q-tiling (original implementation)
|
||||
_Q_CHUNK = 256
|
||||
|
||||
# During profiling, skip expensive attention — return zeros.
|
||||
# Profiling only measures memory footprint, not output correctness.
|
||||
if os.environ.get("BI100_IN_STARTUP_PROFILE") == "1":
|
||||
return torch.zeros_like(query)
|
||||
|
||||
if (attn_metadata.query_start_loc is not None
|
||||
and len(attn_metadata.query_start_loc) == num_seqs + 1):
|
||||
q_lens = [
|
||||
|
||||
@@ -58,7 +58,7 @@ class CustomChatCompletionMessageParam(TypedDict, total=False):
|
||||
|
||||
class OpenAIBaseModel(BaseModel):
|
||||
# OpenAI API does not allow extra fields
|
||||
model_config = ConfigDict(extra="forbid")
|
||||
model_config = ConfigDict(extra="allow")
|
||||
|
||||
|
||||
class ErrorResponse(OpenAIBaseModel):
|
||||
|
||||
Reference in New Issue
Block a user