1. Dockerfile: 去掉ex_engine COPY和所有CUDA编译RUN步骤 - 只剩1个RUN: patch_ops.sh部署预编译.so和serving层 2. patch_ops.sh: exit 2 → exit 0, 跳过所有CUDA编译 - VLLM_ROOT找不到时不再abort - 去掉build_moe_topk/build_unified_bridge/py_compile 3. computility-run.yaml: 恢复comp168参数 - max_model_len: 80000 → 100000 - gpu_memory_utilization: 0.95 → 0.90 - 去掉 --max-num-batched-tokens --enable-chunked-prefill
15 lines
628 B
Docker
15 lines
628 B
Docker
FROM git.modelhub.org.cn:9443/enginex-iluvatar/bi100-3.2.3-x86-ubuntu20.04-py3.10-poc-llm-infer:v1.2.3
|
|
|
|
RUN mkdir -p /workspace
|
|
WORKDIR /workspace/
|
|
|
|
# Copy all our engine patches + prebuilt .so
|
|
COPY ./qwen3_6_scripts /workspace/qwen3_6_scripts
|
|
COPY ./computility-run.yaml /workspace/computility-run.yaml
|
|
|
|
# Single patch step — NO CUDA compilation during docker build
|
|
# All .so are prebuilt and bundled in qwen3_6_scripts/prebuilt/
|
|
RUN chmod +x /workspace/qwen3_6_scripts/patch_ops.sh && \
|
|
bash /workspace/qwen3_6_scripts/patch_ops.sh 2>&1 | tee /workspace/patch_ops.log ; \
|
|
echo "[Dockerfile] patch_ops exit code: $?"
|