Files
enginex-mlu370-compat/Dockerfile

44 lines
3.0 KiB
Docker
Raw Permalink Blame History

This file contains ambiguous Unicode characters

This file contains Unicode characters that might be confused with other characters. If you think that this is intentional, you can safely ignore this warning. Use the Escape button to reveal them.

# enginex-mlu370-compat
#
# 寒武纪 MLU370-X4(Cambricon_mlu-370-x4)· 文本生成 · vLLM「兼容增强」引擎。
#
# 与 enginex-bi100-compat 同源同逻辑,只换基础镜像。
# 做法照搬社区已上线引擎的模式:**把镜像内的 vllm 二进制换成本仓 wrapper**,
# 平台原样下发 `vllm serve /model --port 8000 ...` 就会流经 wrapper,无需平台改任何命令。
#
# 基础镜像来源(10-03 刚从平台 vllm-customized 框架的 build-config 里挖到,CI 拉得到):
# harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-20260204
# 之前找 Cambricon 基础镜像只看到官方仓 Dockerfile 里的 `combricon_vllm_mlu:v1.0_0510`
# (无 registry 前缀的内部名,CI 拉不到)而搁置;harbor 路径解锁后才能建。
#
# 为什么选 mlu-370-x4:该卡只有 vllm-mlu / vllm / vllm-customized 三个框架,
# **没有任何 tokenizer/compat 类引擎**;而本人该卡 3 个失败里 2 个是 R2(模型无 chat_template,
# 而 vllm / vllm-mlu 在该卡默认走 chat 路径)。
FROM harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-20260204
LABEL org.opencontainers.image.title="enginex-mlu370-compat" \
org.opencontainers.image.description="Cambricon mlu-370-x4 vLLM engine with preflight compatibility patches (R2/R3/R3b/R4/R7)" \
com.modelhubxc.engine.target-card="Cambricon_mlu-370-x4" \
com.modelhubxc.engine.framework="vllm" \
com.modelhubxc.engine.task-type="text-generation" \
com.modelhubxc.engine.base-image="harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-20260204"
COPY preflight.py /opt/
COPY detect_tokenizer.py /opt/
COPY vllm_wrapper.sh /opt/
COPY shims_ixformer_layers.py /opt/shims_src/
RUN set -eux && mkdir -p /opt/shims/ixformer/contrib/vllm/layers && for d in /opt/shims/ixformer /opt/shims/ixformer/contrib /opt/shims/ixformer/contrib/vllm /opt/shims/ixformer/contrib/vllm/layers; do printf '# ModelHub XC compat shim package\n' > "$d/__init__.py"; done && cp /opt/shims_src/shims_ixformer_layers.py /opt/shims/ixformer/contrib/vllm/layers/__init__.py && rm -rf /opt/shims_src && chmod +x /opt/vllm_wrapper.sh /opt/preflight.py && python3 -c "import ast,io;[ast.parse(io.open(f,encoding='utf-8').read()) for f in ['/opt/preflight.py','/opt/detect_tokenizer.py','/opt/shims/ixformer/contrib/vllm/layers/__init__.py']]" && bash -n /opt/vllm_wrapper.sh && echo "[enginex-mlu370-compat] preflight syntax OK"
ENV PYTHONPATH=/opt/shims:${PYTHONPATH}
ENV MODEL_DIR=/model
# 关键一步:把 vllm 二进制换成 wrapper(真身改名 vllm_real),
# 平台下发的 `vllm serve ...` 因此流经本引擎的启动前修补。
RUN set -eux && VLLM_BIN="$(command -v vllm || echo /usr/local/corex/lib64/python3/dist-packages/bin/vllm)" && \
if [ ! -f "$VLLM_BIN" ]; then echo "vllm binary not found at $VLLM_BIN"; exit 1; fi && \
mv "$VLLM_BIN" "${VLLM_BIN}_real" && \
ln -s /opt/vllm_wrapper.sh "$VLLM_BIN" && \
echo "[enginex-mlu370-compat] vllm -> wrapper installed"