diff --git a/Dockerfile b/Dockerfile index 663bcfc..567bff5 100644 --- a/Dockerfile +++ b/Dockerfile @@ -1,4 +1,25 @@ -FROM harbor.4pd.io/modelhubxc/enginex-sunrise/vllm-fix-tokenizer:v1.1.1 +FROM registry.maas.sunrise-ai.com/public/vllm:S2-v1.1.1 +ENV LD_LIBRARY_PATH=/usr/local/pccl/lib:\ +/usr/local/tangrt/targets/linux-x86_64/lib:\ +/usr/local/tangrt/targets/linux-x86_64/lib/stub:\ +/root/pt200/gcc-11.3.0/install/lib64:\ +/root:/root/gcc-11.5.0/lib64:\ +/usr/local/pccl/lib:\ +/usr/local/tangrt/targets/linux-x86_64/lib:\ +/usr/local/tangrt/targets/linux-x86_64/lib/stub:\ +/usr/local/tangrt/lib/linux-x86_64:\ +/root/pt200/gcc-11.3.0/install/lib64:\ +/root:\ +/usr/lib64:\ +/usr/local/lib/python3.10/site-packages/torch/lib +ENV TORCH_DEVICE_BACKEND_AUTOLOAD=0 +ENV PATH=/root/gcc-11.5.0/bin:/usr/local/sbin:/usr/local/bin:/usr/sbin:/usr/bin:/sbin:/bin +ENV PYTHONPATH=/sunrise_code/vllm:/sunrise_code/sunrise_vllm:/usr/local/lib/python3.10/site-packages: +RUN ln -sf /usr/local/bin/python3.10 /usr/bin/python3 +COPY fix_tokenizer.py /opt/ +COPY detect_tokenizer.py /opt/ +COPY entrypoint.sh /opt/ WORKDIR /model COPY . /model +RUN chmod +x /opt/entrypoint.sh ENTRYPOINT ["/opt/entrypoint.sh"] diff --git a/detect_tokenizer.py b/detect_tokenizer.py new file mode 100644 index 0000000..c0e7b3e --- /dev/null +++ b/detect_tokenizer.py @@ -0,0 +1,25 @@ +import os +import json + +def detect(model_dir): + cfg_path = os.path.join(model_dir, "tokenizer_config.json") + + if os.path.exists(cfg_path): + with open(cfg_path) as f: + cfg = json.load(f) + cls = cfg.get("tokenizer_class", "") + else: + cls = "" + + files = os.listdir(model_dir) + + if "tokenizer.json" in files: + return "fast", cls + + if "tokenizer.model" in files: + return "sentencepiece", cls + + if "vocab.json" in files and "merges.txt" in files: + return "bpe", cls + + return "unknown", cls diff --git a/entrypoint.sh b/entrypoint.sh new file mode 100644 index 0000000..07308d3 --- /dev/null +++ b/entrypoint.sh @@ -0,0 +1,39 @@ +#!/bin/bash +set -e + +MODEL_DIR=${1:-/model} +shift || true + +FIX_TOKENIZER_DIR=/tmp/fixed_tokenizer +AUTO_FIX=${AUTO_FIX_TOKENIZER:-auto} + +echo "[entrypoint] model dir: $MODEL_DIR" + +NEED_FIX=0 + +if [ "$AUTO_FIX" = "1" ] || [ "$AUTO_FIX" = "true" ]; then + NEED_FIX=1 +elif [ "$AUTO_FIX" = "auto" ]; then + if [ -f "$MODEL_DIR/tokenizer_config.json" ]; then + if grep -q "TokenizersBackend\|TiktokenTokenizer" "$MODEL_DIR/tokenizer_config.json"; then + NEED_FIX=1 + fi + # 检测 extra_special_tokens 是否为 list 格式 + if grep -q '"extra_special_tokens":\s*\[' "$MODEL_DIR/tokenizer_config.json"; then + NEED_FIX=1 + fi + fi +fi + +if [ $NEED_FIX -eq 1 ]; then + echo "[entrypoint] fixing tokenizer..." + python3 /opt/fix_tokenizer.py + TOKENIZER_ARG="--tokenizer $FIX_TOKENIZER_DIR" +else + echo "[entrypoint] tokenizer OK, skip fix" + TOKENIZER_ARG="" +fi + +echo "[entrypoint] starting vllm..." + +exec vllm serve "$MODEL_DIR" $TOKENIZER_ARG "$@" diff --git a/fix_tokenizer.py b/fix_tokenizer.py new file mode 100644 index 0000000..67563c2 --- /dev/null +++ b/fix_tokenizer.py @@ -0,0 +1,72 @@ +import os +import shutil +import json +from detect_tokenizer import detect + +MODEL_DIR = os.environ.get("MODEL_DIR", "/model") +OUT_DIR = os.environ.get("FIX_TOKENIZER_DIR", "/tmp/fixed_tokenizer") + +os.makedirs(OUT_DIR, exist_ok=True) + +def copy_if_exists(name): + src = os.path.join(MODEL_DIR, name) + if os.path.exists(src): + shutil.copy(src, OUT_DIR) + +# 复制所有可能相关文件 +for f in [ + "tokenizer.json", + "tokenizer_config.json", + "special_tokens_map.json", + "vocab.json", + "merges.txt", + "tokenizer.model", +]: + copy_if_exists(f) + +typ, orig_cls = detect(MODEL_DIR) + +cfg_path = os.path.join(OUT_DIR, "tokenizer_config.json") + +if os.path.exists(cfg_path): + with open(cfg_path) as f: + cfg = json.load(f) +else: + cfg = {} + +# ===== 自动修复策略 ===== +if typ == "fast": + cfg["tokenizer_class"] = "PreTrainedTokenizerFast" + cfg["from_slow"] = False + cfg.pop("backend", None) + +elif typ == "sentencepiece": + cfg["tokenizer_class"] = "LlamaTokenizer" + +elif typ == "bpe": + cfg["tokenizer_class"] = "GPT2TokenizerFast" + +else: + cfg["tokenizer_class"] = "PreTrainedTokenizerFast" + +# 特殊 case 修复 +bad_classes = [ + "TokenizersBackend", + "TiktokenTokenizer", +] + +if orig_cls in bad_classes: + print(f"[fix] override bad tokenizer_class: {orig_cls} → {cfg['tokenizer_class']}") + print(f"[fix] override from_slow: {cfg['from_slow']}") + +# 修复 extra_special_tokens: list → dict 格式 +if "extra_special_tokens" in cfg and isinstance(cfg["extra_special_tokens"], list): + orig_list = cfg["extra_special_tokens"] + cfg["extra_special_tokens"] = {token: token for token in orig_list} + print(f"[fix] converted extra_special_tokens from list ({len(orig_list)} items) to dict format") + +# 写回 +with open(cfg_path, "w") as f: + json.dump(cfg, f) + +print(f"[fix_tokenizer] done → {OUT_DIR}")