From 941c42a04f336ecbce194b0c96248062f1b06d2a Mon Sep 17 00:00:00 2001 From: wanglin <2281216234@qq.com> Date: Fri, 2 Oct 2026 19:53:25 +0800 Subject: [PATCH] =?UTF-8?q?v3:=20=E5=AF=B9=E9=BD=90=E6=88=90=E5=8A=9F?= =?UTF-8?q?=E9=85=8D=E7=BD=AE=20=E2=80=94=20=E8=A1=A5=20--enforce-eager=20?= =?UTF-8?q?=E4=B8=8E=20corex=20=E8=BF=90=E8=A1=8C=E5=BA=93=E7=8E=AF?= =?UTF-8?q?=E5=A2=83=E5=8F=98=E9=87=8F,=20max-model-len=20256K=20/=20gmu?= =?UTF-8?q?=200.95=20/=20seqs=204=20/=20bt=208192?= MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit v3: 对齐成功配置 — 补 --enforce-eager 与 corex 运行库环境变量, max-model-len 256K / gmu 0.95 / seqs 4 / bt 8192 --- computility-run.yaml | 83 +++++++++++++++++++++++++++----------------- 1 file changed, 52 insertions(+), 31 deletions(-) diff --git a/computility-run.yaml b/computility-run.yaml index b4f96c8..a929c2a 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -1,34 +1,55 @@ concurrency: 1 command: - - python3 - - -m - - vllm.entrypoints.openai.api_server - - --model - - /model - - --served-model-name - - llm - - --max-model-len - - '100000' - - --gpu-memory-utilization - - '0.9' - - --trust-remote-code - - -tp - - '4' - - --max-num-seqs - - '8' - - --disable-log-requests - - --disable-frontend-multiprocessing - - --max-num-batched-tokens - - '16384' - - --enable-chunked-prefill - - --max-seq-len-to-capture - - '32768' - - --enable-auto-tool-choice - - --tool-call-parser - - qwen3_coder - - --reasoning-parser - - qwen3 - - --enable-prefix-caching + - python3 + - -m + - vllm.entrypoints.openai.api_server + - --model + - /model + - --served-model-name + - llm + - --max-model-len + - '256000' + - --gpu-memory-utilization + - '0.95' + - --trust-remote-code + - -tp + - '4' + - --max-num-seqs + - '4' + - --disable-log-requests + - --disable-frontend-multiprocessing + - --max-num-batched-tokens + - '8192' + - --enable-chunked-prefill + - --max-seq-len-to-capture + - '32768' + - --enforce-eager + - --enable-auto-tool-choice + - --tool-call-parser + - qwen3_coder + - --reasoning-parser + - qwen3 + - --enable-prefix-caching + - --dtype + - half env: - - name: VLLM_ENGINE_ITERATION_TIMEOUT_S - value: 3600 + - name: VLLM_ENGINE_ITERATION_TIMEOUT_S + value: 3600 + - name: VLLM_ATTENTION_BACKEND + value: XFORMERS + - name: ENABLE_CUSTOM_IPC + value: 1 + - name: PYTHONPATH + value: /usr/local/corex/lib/python3/dist-packages:/usr/local/corex/lib64/python3/dist-packages + - name: LD_LIBRARY_PATH + value: /usr/local/corex/lib64:/usr/local/openmpi/lib + - name: VLLM_COREX_FA2_LIBRARY + value: /usr/local/corex/lib64/libcorex_fa2.so + - name: VLLM_COREX_GDN_LIBRARY + value: /usr/local/corex/lib64/libcorex_gdn.so + - name: VLLM_COREX_MOE_LIBRARY + value: /usr/local/corex/lib64/libcorex_moe.so + - name: VLLM_REQUEST_METRICS_FILE + value: /tmp/vllm-request-metrics.jsonl + - name: VLLM_CACHE_BLOCK_SIZE + value: 16