commit 391302289de5c0946cb1623561ac896be69d1638 Author: wanglin <2281216234@qq.com> Date: Sat Oct 3 18:29:26 2026 +0800 enginex-bi100-compat v1: preflight 兼容增强引擎(R2/R3/R3b/R4/R7),结构与引擎修复模式同型,25 项自测全通过 diff --git a/.gitignore b/.gitignore new file mode 100644 index 0000000..99756a6 --- /dev/null +++ b/.gitignore @@ -0,0 +1,3 @@ +__pycache__/ +*.pyc +/tmp/ diff --git a/Dockerfile b/Dockerfile new file mode 100644 index 0000000..9a43f22 --- /dev/null +++ b/Dockerfile @@ -0,0 +1,44 @@ +# enginex-bi100-compat +# +# 天数智芯 天垓100(Iluvatar_bi-100)· 文本生成 · vLLM「兼容增强」引擎。 +# +# 不动模型、不动卡,只把「引擎镜像层」的结构性不兼容在启动前修掉。 +# 结构与社区已上线引擎 EngineX-Sunrise/enginex-S2-vllm-fix-tokenizer 完全同型 +# (Dockerfile + entrypoint.sh + 修补脚本 + README),该模式已在曦望 S2 生产验证。 +# +# 修什么(全部有日志实证,见 README.md): +# R3 extra_special_tokens 是 list -> transformers 崩(社区已上线修复,已合并) +# R3b tokenizer_class 是坏类名 -> 加载异常(社区已上线修复,已合并) +# R2 模型没有 chat_template -> /v1/chat/completions 400/空输出 +# R4 architectures 是镜像未注册类名 -> KeyError / MODEL_NOT_SUPPORTED +# R7 镜像缺 ixformer.contrib.vllm.layers -> ModuleNotFoundError(MoE) +# +# 为什么选 bi-100:本人 37 个验证失败里,bi-100 单卡 10 例为全平台最集中; +# 且该卡社区通过率约 51%(15 张卡最低),修好收益最大。 + +FROM git.modelhub.org.cn:9443/enginex-iluvatar/bi100-3.2.3-x86-ubuntu20.04-py3.10-poc-llm-infer:v1.2.3 + +LABEL org.opencontainers.image.title="enginex-bi100-compat" \ + org.opencontainers.image.description="Iluvatar bi-100 vLLM text-generation engine with preflight compatibility patches (R2/R3/R3b/R4/R7)" \ + com.modelhubxc.engine.target-card="Iluvatar_bi-100" \ + com.modelhubxc.engine.framework="vllm" \ + com.modelhubxc.engine.task-type="text-generation" \ + com.modelhubxc.engine.baseline="EngineX-Iluvatar/enginex-vllm-bi100-qwen36" \ + com.modelhubxc.engine.pattern="EngineX-Sunrise/enginex-S2-vllm-fix-tokenizer" + +# 修补脚本 + R7 纯 PyTorch shim(都只是小文本文件,无权重) +COPY preflight.py /opt/ +COPY entrypoint.sh /opt/ +COPY detect_tokenizer.py /opt/ +COPY shims/ /opt/shims/ + +RUN chmod +x /opt/entrypoint.sh /opt/preflight.py \ + && python3 -c "import ast,io;[ast.parse(io.open(f,encoding='utf-8').read()) for f in ['/opt/preflight.py','/opt/detect_tokenizer.py','/opt/shims/ixformer/contrib/vllm/layers/__init__.py']]" \ + && bash -n /opt/entrypoint.sh \ + && echo "[enginex-bi100-compat] preflight syntax OK" + +# R7:让 shims 全局可 import(平台若覆盖 entrypoint 也仍然生效) +ENV PYTHONPATH=/opt/shims:${PYTHONPATH} +ENV MODEL_DIR=/model + +ENTRYPOINT ["/opt/entrypoint.sh"] diff --git a/computility-run.yaml b/computility-run.yaml new file mode 100644 index 0000000..e420255 --- /dev/null +++ b/computility-run.yaml @@ -0,0 +1,31 @@ +# enginex-bi100-compat 的启动配置 +# +# 与平台 bi-100 × vllm × text-generation 的原生 build-config 逐项一致, +# 只把 command 换成「先过 preflight 再 exec 原命令」: +# python3 /workspace/compat/serve.py -- <原 command...> +# +# 这样平台下发的任何参数(端口 / max-model-len / tp 等)都原样传给 vLLM, +# preflight 只做「追加」,不做「改写」,行为可预期、可回退。 +concurrency: 1 +command: + - python3 + - /workspace/compat/serve.py + - -- + - vllm + - serve + - /model + - --port + - '80' + - --served-model-name + - llm + - --max-model-len + - '4096' + - --gpu-memory-utilization + - '0.9' + - --enforce-eager + - --trust-remote-code + - -tp + - '1' +env: + - name: PYTHONPATH + value: /workspace/compat/shims:/workspace/compat diff --git a/detect_tokenizer.py b/detect_tokenizer.py new file mode 100644 index 0000000..ede8122 --- /dev/null +++ b/detect_tokenizer.py @@ -0,0 +1,36 @@ +import os +import json + + +def detect(model_dir): + """判定 tokenizer 类型:fast / sentencepiece / bpe / unknown + + 与社区已上线引擎 EngineX-Sunrise/enginex-S2-vllm-fix-tokenizer 的 + detect_tokenizer.py 保持一致(同一判定口径,便于两套引擎交叉验证)。 + """ + cfg_path = os.path.join(model_dir, "tokenizer_config.json") + cls = "" + if os.path.exists(cfg_path): + try: + with open(cfg_path, encoding="utf-8") as f: + cls = (json.load(f) or {}).get("tokenizer_class", "") or "" + except Exception: # noqa: BLE001 + cls = "" + try: + files = set(os.listdir(model_dir)) + except Exception: # noqa: BLE001 + files = set() + + if "tokenizer.json" in files: + return "fast", cls + if "tokenizer.model" in files: + return "sentencepiece", cls + if "vocab.json" in files and "merges.txt" in files: + return "bpe", cls + return "unknown", cls + + +if __name__ == "__main__": + import sys + t, c = detect(sys.argv[1] if len(sys.argv) > 1 else "/model") + print(t, c) diff --git a/entrypoint.sh b/entrypoint.sh new file mode 100644 index 0000000..3b670ee --- /dev/null +++ b/entrypoint.sh @@ -0,0 +1,44 @@ +#!/bin/bash +# enginex-bi100-compat 入口:启动前修补 -> exec vllm serve +# +# 与社区已上线引擎 EngineX-Sunrise/enginex-S2-vllm-fix-tokenizer 同一模式 +# (detect -> fix -> `exec vllm serve "$MODEL_DIR" $EXTRA "$@"`), +# 只是把「只修 tokenizer」升级成「修 tokenizer + chat_template + 架构 + 缺模块」。 +# +# 平台会把 GPU 数、端口、max-model-len 等参数作为 "$@" 传进来,原样透传。 +set -u + +MODEL_DIR=${MODEL_DIR:-${1:-/model}} +# 若第一个参数不是目录,则认为是平台传入的 vllm 参数,MODEL_DIR 仍用默认 /model +if [ $# -gt 0 ] && [ -d "$1" ]; then + MODEL_DIR="$1" + shift +fi + +FIX_LOG=/tmp/mhxc_preflight.json +echo "[entrypoint] model dir: $MODEL_DIR" +echo "[entrypoint] args: $*" + +EXTRA="" +if python3 /opt/preflight.py --model "$MODEL_DIR" --out "$FIX_LOG" >/tmp/mhxc_preflight.out 2>/tmp/mhxc_preflight.err; then + # 从 JSON 里取 extra_args(用 python 解析,避免依赖 jq) + EXTRA=$(python3 - "$FIX_LOG" <<'PY' +import json, sys, shlex +try: + with open(sys.argv[1], encoding="utf-8") as f: + d = json.load(f) + print(" ".join(shlex.quote(a) for a in (d.get("extra_args") or []))) +except Exception as e: + print("") +PY +) + echo "[entrypoint] preflight extra args: ${EXTRA:-(无)}" + sed 's/^/[entrypoint] preflight: /' /tmp/mhxc_preflight.err 2>/dev/null || true +else + echo "[entrypoint] preflight 执行失败,按原命令继续(不阻断启动)" + sed 's/^/[entrypoint] preflight: /' /tmp/mhxc_preflight.err 2>/dev/null || true +fi + +echo "[entrypoint] starting vllm..." +# shellcheck disable=SC2086 +exec vllm serve "$MODEL_DIR" $EXTRA "$@"