fix(critical): 3 bugs causing 0.0 score — OOM crash + thinking param + multimodal
Bug 1 (FATAL): computility-run.yaml max-model-len 256000 → 131072,
gpu-memory-utilization 0.95 → 0.90, max-num-batched-tokens 4096 → 8192.
Server OOM'd on t2_n_2, killed all subsequent modules (replay=0, opencompass=0).
Bug 2 (functional): protocol.py thinking={enable:true/false} was accepted
but NEVER mapped to chat_template_kwargs.enable_thinking. Qwen3 template
never received the parameter → t1a, t1c, d07, d10 all FAIL.
Bug 3 (functional): chat_utils.py _placeholder_str didn't handle qwen3_5
model_type for multimodal → d05_multimodal HTTP 400 TypeError.
Expected: functional pass rate 0.41 → 0.90+, server stays alive for all
4 modules, total score 0.0 → 60000+ (matching reference sub 168).
This commit is contained in:
@@ -8,9 +8,9 @@ command:
|
|||||||
- --served-model-name
|
- --served-model-name
|
||||||
- llm
|
- llm
|
||||||
- --max-model-len
|
- --max-model-len
|
||||||
- '256000'
|
- '131072'
|
||||||
- --gpu-memory-utilization
|
- --gpu-memory-utilization
|
||||||
- '0.95'
|
- '0.90'
|
||||||
- --trust-remote-code
|
- --trust-remote-code
|
||||||
- -tp
|
- -tp
|
||||||
- '4'
|
- '4'
|
||||||
@@ -19,7 +19,7 @@ command:
|
|||||||
- --disable-log-requests
|
- --disable-log-requests
|
||||||
- --disable-frontend-multiprocessing
|
- --disable-frontend-multiprocessing
|
||||||
- --max-num-batched-tokens
|
- --max-num-batched-tokens
|
||||||
- '4096'
|
- '8192'
|
||||||
- --enable-chunked-prefill
|
- --enable-chunked-prefill
|
||||||
- --max-seq-len-to-capture
|
- --max-seq-len-to-capture
|
||||||
- '32768'
|
- '32768'
|
||||||
|
|||||||
@@ -172,7 +172,8 @@ class BaseMultiModalItemTracker(ABC, Generic[_T]):
|
|||||||
return "<image>"
|
return "<image>"
|
||||||
if model_type == "mllama":
|
if model_type == "mllama":
|
||||||
return "<|image|>"
|
return "<|image|>"
|
||||||
if model_type in ("qwen2_vl","qwen2_5_vl"):
|
if model_type in ("qwen2_vl", "qwen2_5_vl",
|
||||||
|
"qwen3_5", "qwen3_5_moe"):
|
||||||
return "<|vision_start|><|image_pad|><|vision_end|>"
|
return "<|vision_start|><|image_pad|><|vision_end|>"
|
||||||
if model_type == "molmo":
|
if model_type == "molmo":
|
||||||
return ""
|
return ""
|
||||||
@@ -183,7 +184,8 @@ class BaseMultiModalItemTracker(ABC, Generic[_T]):
|
|||||||
return "<|reserved_special_token_0|>"
|
return "<|reserved_special_token_0|>"
|
||||||
raise TypeError(f"Unknown model type: {model_type}")
|
raise TypeError(f"Unknown model type: {model_type}")
|
||||||
elif modality == "video":
|
elif modality == "video":
|
||||||
if model_type in ("qwen2_vl","qwen2_5_vl"):
|
if model_type in ("qwen2_vl", "qwen2_5_vl",
|
||||||
|
"qwen3_5", "qwen3_5_moe"):
|
||||||
return "<|vision_start|><|video_pad|><|vision_end|>"
|
return "<|vision_start|><|video_pad|><|vision_end|>"
|
||||||
raise TypeError(f"Unknown model type: {model_type}")
|
raise TypeError(f"Unknown model type: {model_type}")
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -408,6 +408,17 @@ class ChatCompletionRequest(OpenAIBaseModel):
|
|||||||
if data.get("max_completion_tokens") is not None and data.get("max_tokens") is None:
|
if data.get("max_completion_tokens") is not None and data.get("max_tokens") is None:
|
||||||
data["max_tokens"] = data["max_completion_tokens"]
|
data["max_tokens"] = data["max_completion_tokens"]
|
||||||
|
|
||||||
|
# Map thinking={enable:true/false} → chat_template_kwargs.enable_thinking
|
||||||
|
# The competition evaluator sends thinking={enable:true/false} (OpenAI API).
|
||||||
|
# Qwen3's chat template expects enable_thinking=True/False in kwargs.
|
||||||
|
thinking = data.get("thinking")
|
||||||
|
if isinstance(thinking, dict):
|
||||||
|
enable = thinking.get("enable")
|
||||||
|
if enable is not None:
|
||||||
|
ctk = data.get("chat_template_kwargs") or {}
|
||||||
|
ctk["enable_thinking"] = bool(enable)
|
||||||
|
data["chat_template_kwargs"] = ctk
|
||||||
|
|
||||||
messages = data.get("messages")
|
messages = data.get("messages")
|
||||||
if not isinstance(messages, list):
|
if not isinstance(messages, list):
|
||||||
return data
|
return data
|
||||||
|
|||||||
Reference in New Issue
Block a user