From 0a784858283fc0539313c42e675af04eb937f676 Mon Sep 17 00:00:00 2001 From: huni Date: Mon, 24 Aug 2026 18:07:08 +0800 Subject: [PATCH] add fix_tokenizer.py --- fix_tokenizer.py | 71 ++++++++++++++++++++++++++++++++++++++++++++++++ 1 file changed, 71 insertions(+) create mode 100644 fix_tokenizer.py diff --git a/fix_tokenizer.py b/fix_tokenizer.py new file mode 100644 index 0000000..8e012a3 --- /dev/null +++ b/fix_tokenizer.py @@ -0,0 +1,71 @@ +import os +import shutil +import json +from detect_tokenizer import detect + +MODEL_DIR = os.environ.get("MODEL_DIR", "/model") +OUT_DIR = os.environ.get("FIX_TOKENIZER_DIR", "/tmp/fixed_tokenizer") + +os.makedirs(OUT_DIR, exist_ok=True) + +def copy_if_exists(name): + src = os.path.join(MODEL_DIR, name) + if os.path.exists(src): + shutil.copy(src, OUT_DIR) + +# 复制所有可能相关文件 +for f in [ + "tokenizer.json", + "tokenizer_config.json", + "special_tokens_map.json", + "vocab.json", + "merges.txt", + "tokenizer.model", +]: + copy_if_exists(f) + +typ, orig_cls = detect(MODEL_DIR) + +cfg_path = os.path.join(OUT_DIR, "tokenizer_config.json") + +if os.path.exists(cfg_path): + with open(cfg_path) as f: + cfg = json.load(f) +else: + cfg = {} + +# ===== 自动修复策略 ===== +if typ == "fast": + cfg["tokenizer_class"] = "PreTrainedTokenizerFast" + cfg["from_slow"] = False + cfg.pop("backend", None) + +elif typ == "sentencepiece": + cfg["tokenizer_class"] = "LlamaTokenizer" + +elif typ == "bpe": + cfg["tokenizer_class"] = "GPT2TokenizerFast" + +else: + cfg["tokenizer_class"] = "PreTrainedTokenizerFast" + +# 特殊 case 修复 +bad_classes = [ + "TokenizersBackend", + "TiktokenTokenizer", +] + +if orig_cls in bad_classes: + print(f"[fix] override bad tokenizer_class: {orig_cls} → {cfg['tokenizer_class']}") + +# 修复 extra_special_tokens: list → dict 格式 +if "extra_special_tokens" in cfg and isinstance(cfg["extra_special_tokens"], list): + orig_list = cfg["extra_special_tokens"] + cfg["extra_special_tokens"] = {token: token for token in orig_list} + print(f"[fix] converted extra_special_tokens from list ({len(orig_list)} items) to dict format") + +# 写回 +with open(cfg_path, "w") as f: + json.dump(cfg, f) + +print(f"[fix_tokenizer] done → {OUT_DIR}")