From d6dfff091877dbd498b4d2d60c2bc10acffbc69f Mon Sep 17 00:00:00 2001 From: wanglin <2281216234@qq.com> Date: Thu, 1 Oct 2026 23:29:16 +0800 Subject: [PATCH] v2: conservative max-num-seqs=8 + batched-tokens=16384, revert dtype/256K (v1 OOM crash) --- computility-run.yaml | 64 +++++++++++++++++++++----------------------- 1 file changed, 31 insertions(+), 33 deletions(-) diff --git a/computility-run.yaml b/computility-run.yaml index e301cdd..b4f96c8 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -1,36 +1,34 @@ concurrency: 1 command: - - python3 - - -m - - vllm.entrypoints.openai.api_server - - --model - - /model - - --served-model-name - - llm - - --max-model-len - - '256000' - - --gpu-memory-utilization - - '0.95' - - --trust-remote-code - - -tp - - '4' - - --max-num-seqs - - '8' - - --disable-log-requests - - --disable-frontend-multiprocessing - - --max-num-batched-tokens - - '16384' - - --enable-chunked-prefill - - --max-seq-len-to-capture - - '32768' - - --enable-auto-tool-choice - - --tool-call-parser - - qwen3_coder - - --reasoning-parser - - qwen3 - - --enable-prefix-caching - - --dtype - - half + - python3 + - -m + - vllm.entrypoints.openai.api_server + - --model + - /model + - --served-model-name + - llm + - --max-model-len + - '100000' + - --gpu-memory-utilization + - '0.9' + - --trust-remote-code + - -tp + - '4' + - --max-num-seqs + - '8' + - --disable-log-requests + - --disable-frontend-multiprocessing + - --max-num-batched-tokens + - '16384' + - --enable-chunked-prefill + - --max-seq-len-to-capture + - '32768' + - --enable-auto-tool-choice + - --tool-call-parser + - qwen3_coder + - --reasoning-parser + - qwen3 + - --enable-prefix-caching env: - - name: VLLM_ENGINE_ITERATION_TIMEOUT_S - value: 3600 \ No newline at end of file + - name: VLLM_ENGINE_ITERATION_TIMEOUT_S + value: 3600