diff --git a/computility-run.yaml b/computility-run.yaml index e17f30ff..f88472f5 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -8,16 +8,14 @@ command: - --served-model-name - llm - --max-model-len - - '256000' + - '100000' - --gpu-memory-utilization - - '0.95' + - '0.9' - --trust-remote-code - -tp - '4' - --max-num-seqs - - '2' - - --max-num-batched-tokens - - '4096' + - '1' - --disable-log-requests - --disable-frontend-multiprocessing - --enforce-eager @@ -27,9 +25,8 @@ command: - --reasoning-parser - qwen3 - --enable-prefix-caching - - --enable-chunked-prefill - --max-seq-len-to-capture - - '32768' + - '8192' - --dtype - half env: diff --git a/qwen3_6_scripts/serving_chat.py b/qwen3_6_scripts/serving_chat.py index 5461b3dd..b343a37b 100644 --- a/qwen3_6_scripts/serving_chat.py +++ b/qwen3_6_scripts/serving_chat.py @@ -247,11 +247,11 @@ class OpenAIServingChat(OpenAIServing): logger.exception("Error in loading multi-modal data") return self.create_error_response(str(e)) - # Allow n≤2 (matches max_num_seqs=2 in computility-run.yaml). - # Sub168 passes t2_n_2 with HTTP 200. Reject n>2 to prevent OOM. + # Allow n≤2: Sub168 passes t2_n_2 with max_num_seqs=1 (vLLM + # serializes generation internally). Reject n>2 to prevent OOM. if request.n is not None and request.n > 2: logger.warning( - "n=%d rejected with 400 (exceeds max_num_seqs=2)", request.n) + "n=%d rejected with 400 (exceeds max supported value)", request.n) return self.create_error_response( f"n={request.n} exceeds the maximum supported value of 2.")