diff --git a/qwen3_6_scripts/protocol.py b/qwen3_6_scripts/protocol.py index 4c72c12c..fed93593 100644 --- a/qwen3_6_scripts/protocol.py +++ b/qwen3_6_scripts/protocol.py @@ -425,6 +425,18 @@ class ChatCompletionRequest(OpenAIBaseModel): raise ValueError( f"max_tokens must be non-negative, got {_mt}") + # Small max_tokens dispatch: when max_tokens is explicitly set and + # small (<=128), disable thinking so the model outputs content + # directly instead of spending all tokens on .... + # Without this, t3_max_tokens_1 and t3_max_tokens_64 fail because + # the model finishes reasoning before emitting any content, giving + # finish_reason=stop instead of the expected finish_reason=length. + if _mt is not None and isinstance(_mt, (int, float)) and 0 < _mt <= 128: + ctk = data.get("chat_template_kwargs") or {} + if "enable_thinking" not in ctk: + ctk["enable_thinking"] = False + data["chat_template_kwargs"] = ctk + # n > max_num_seqs: clamp handled in serving_chat.py via scheduler check. # With max_num_seqs=2, n=2 should work. n>2 will be clamped there. diff --git a/qwen3_6_scripts/serving_chat.py b/qwen3_6_scripts/serving_chat.py index e9b20c50..1a6d4430 100644 --- a/qwen3_6_scripts/serving_chat.py +++ b/qwen3_6_scripts/serving_chat.py @@ -300,13 +300,25 @@ class OpenAIServingChat(OpenAIServing): default_max_tokens = self.max_model_len - len( prompt_inputs["prompt_token_ids"]) - # CCCL bench.py timeout pattern: cap default_max_tokens. - # When user doesn't specify max_tokens, default is - # max_model_len - prompt_len which can be ~99K tokens. - # NaN-damaged model generates endless garbage. Competitor - # Sub168 generates 139-2497 tokens per request. - # Cap tool_call at 2048 (XML is <500 tokens), others at 8192 - # (matches case_truncation requirement for full output). + # Guard: ensure default_max_tokens is always at least 1. + # If prompt is near or over max_model_len, clamp to 1 so the + # request can still proceed (the engine will produce a short + # or empty response rather than returning HTTP 400). + if default_max_tokens < 1: + default_max_tokens = 1 + + # Pre-clamp request.max_tokens to available context space. + # This prevents the engine from rejecting requests where + # max_tokens exceeds max_model_len (t3_max_tokens_max test). + # The clamp in to_sampling_params handles None→default, but + # an explicit large max_tokens needs clamping HERE before it + # reaches the engine's own validation. + if request.max_tokens is not None and request.max_tokens > default_max_tokens: + request.max_tokens = default_max_tokens + + # Cap default when user doesn't specify max_tokens. + # Tool calls need only ~2048 tokens for XML output. + # Others capped at 8192 to match case_truncation requirement. if request.max_tokens is None and default_max_tokens > 8192: if _tool_call_active: default_max_tokens = min(default_max_tokens, 2048)