concurrency: 1 command: - bash - -c - >- python3 /workspace/qwen3_6_scripts/patch_chat_template.py /model 2>&1 || echo '[runtime] chat template patch failed'; exec python3 -m vllm.entrypoints.openai.api_server --model /model --served-model-name llm --max-model-len 131072 --gpu-memory-utilization 0.92 --trust-remote-code -tp 4 --max-num-seqs 2 --disable-log-requests --disable-frontend-multiprocessing --max-num-batched-tokens 4096 --enable-chunked-prefill --max-seq-len-to-capture 32768 --enable-auto-tool-choice --tool-call-parser qwen3_coder --reasoning-parser qwen3 --enable-prefix-caching --enforce-eager --dtype half env: - name: VLLM_ENGINE_ITERATION_TIMEOUT_S value: '3600' - name: BI100_MAX_NUM_SEQS value: '2' # --- MoE kernel selection --- - name: BI100_MOE_COREX_DIRECT_ROUTED value: '1' - name: BI100_MOE_COREX_TOPK_SOFTMAX value: '1' # --- GDN kernel selection --- - name: BI100_GDN_COREX_PACKED_DECODE value: '1' # --- Hybrid KV/GDN cache --- - name: BI100_HYBRID_KV_ACCOUNTING value: full_attention - name: BI100_GDN_CACHE_POLICY value: admission64 - name: BI100_GDN_RESTORE_MODE value: hybrid64 # --- Image fetch timeout (container network) --- - name: VLLM_IMAGE_FETCH_TIMEOUT value: '10'