From 80fa1fe78102ced96bd4872cfacf5b052cea517d Mon Sep 17 00:00:00 2001 From: Claude Date: Sat, 8 Aug 2026 11:01:52 +0000 Subject: [PATCH] arch(CRITICAL): match Sub168 proven engine config exactly MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Sub168 scored 60194.6 with these exact params: - max_model_len=100000 (was 256000) - max_num_seqs=1 (was 2 → caused crash cascade) - gpu_memory_utilization=0.9 (was 0.95) - chunked_prefill=disabled (was enabled) - max_num_batched_tokens=default (was 4096) - max_seq_len_to_capture=8192 (was 32768) Root cause of Sub508/509 0-score: engine crash at t2_n_2 with max_num_seqs=2 caused Connection Refused cascade. --- computility-run.yaml | 11 ++++------- qwen3_6_scripts/serving_chat.py | 6 +++--- 2 files changed, 7 insertions(+), 10 deletions(-) diff --git a/computility-run.yaml b/computility-run.yaml index e17f30ff..f88472f5 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -8,16 +8,14 @@ command: - --served-model-name - llm - --max-model-len - - '256000' + - '100000' - --gpu-memory-utilization - - '0.95' + - '0.9' - --trust-remote-code - -tp - '4' - --max-num-seqs - - '2' - - --max-num-batched-tokens - - '4096' + - '1' - --disable-log-requests - --disable-frontend-multiprocessing - --enforce-eager @@ -27,9 +25,8 @@ command: - --reasoning-parser - qwen3 - --enable-prefix-caching - - --enable-chunked-prefill - --max-seq-len-to-capture - - '32768' + - '8192' - --dtype - half env: diff --git a/qwen3_6_scripts/serving_chat.py b/qwen3_6_scripts/serving_chat.py index 5461b3dd..b343a37b 100644 --- a/qwen3_6_scripts/serving_chat.py +++ b/qwen3_6_scripts/serving_chat.py @@ -247,11 +247,11 @@ class OpenAIServingChat(OpenAIServing): logger.exception("Error in loading multi-modal data") return self.create_error_response(str(e)) - # Allow n≤2 (matches max_num_seqs=2 in computility-run.yaml). - # Sub168 passes t2_n_2 with HTTP 200. Reject n>2 to prevent OOM. + # Allow n≤2: Sub168 passes t2_n_2 with max_num_seqs=1 (vLLM + # serializes generation internally). Reject n>2 to prevent OOM. if request.n is not None and request.n > 2: logger.warning( - "n=%d rejected with 400 (exceeds max_num_seqs=2)", request.n) + "n=%d rejected with 400 (exceeds max supported value)", request.n) return self.create_error_response( f"n={request.n} exceeds the maximum supported value of 2.")