diff --git a/computility-run.yaml b/computility-run.yaml index d1cf2a34..474d38aa 100644 --- a/computility-run.yaml +++ b/computility-run.yaml @@ -49,5 +49,4 @@ env: - name: BI100_GDN_CACHE_POLICY value: admission64 - name: BI100_GDN_RESTORE_MODE - value: hybrid64 -# rebuild + value: hybrid64 \ No newline at end of file diff --git a/qwen3_6_scripts/patch_chat_template.py b/qwen3_6_scripts/patch_chat_template.py new file mode 100644 index 00000000..d62cf5a0 --- /dev/null +++ b/qwen3_6_scripts/patch_chat_template.py @@ -0,0 +1,63 @@ +#!/usr/bin/env python3 +""" +patch_chat_template.py — Fix non-thinking mode '!!!!!' output + +Root cause: Qwen3.5-MoE's chat_template adds '\n\n\n\n' +when enable_thinking=false. This empty think block causes the model to +degenerate into outputting nothing but '!'. + +Fix: Remove the empty think block so the model generates directly. +""" +import json +import sys +import os + + +def patch_tokenizer_config(model_path: str) -> bool: + config_path = os.path.join(model_path, "tokenizer_config.json") + if not os.path.isfile(config_path): + print(f"[patch_chat_template] {config_path} not found") + return False + + with open(config_path, "r") as f: + config = json.load(f) + + template = config.get("chat_template", "") + if not template: + print("[patch_chat_template] No chat_template found") + return False + + # After json.load, \n in the JSON string becomes actual newline chars. + # The template content uses Jinja2 syntax with literal '\n\n\n\n' + # which in the Python string is: '\\n\\n\\n\\n' + # (because it's a Jinja string literal, not a Python string) + + # Look for the pattern: when enable_thinking=false, outputs empty think block + target = "{{- '\\n\\n\\n\\n' }}" + replacement = "{{- '' }}" + + if target in template: + template = template.replace(target, replacement, 1) + config["chat_template"] = template + with open(config_path, "w") as f: + json.dump(config, f, ensure_ascii=False, indent=2) + print("[patch_chat_template] ✓ Removed empty block for non-thinking mode") + return True + + # Fallback: check if already patched + if "enable_thinking is false" in template and target not in template: + print("[patch_chat_template] Already patched or different format") + return True + + print(f"[patch_chat_template] WARNING: Could not find target pattern") + # Debug: show what's actually there + idx = template.find("enable_thinking is false") + if idx >= 0: + print(f"[patch_chat_template] Context: {repr(template[idx:idx+150])}") + return False + + +if __name__ == "__main__": + model_path = sys.argv[1] if len(sys.argv) > 1 else os.environ.get("MODEL_PATH", "/model") + success = patch_tokenizer_config(model_path) + sys.exit(0 if success else 1) \ No newline at end of file diff --git a/qwen3_6_scripts/patch_ops.sh b/qwen3_6_scripts/patch_ops.sh index 63ff99b7..5ffc8fc4 100755 --- a/qwen3_6_scripts/patch_ops.sh +++ b/qwen3_6_scripts/patch_ops.sh @@ -487,6 +487,15 @@ if [[ -d "${EX_ENGINE_DIR}/python" ]]; then echo "[patch_ops] deployed $(ls -1 "${EX_PY_DIR}"/*.py 2>/dev/null | wc -l) Python modules → ${EX_PY_DIR}/" fi +build_stage "patching chat template for non-thinking mode" +MODEL_DIR="${MODEL_DIR:-/model}" +if [ -f "${MODEL_DIR}/tokenizer_config.json" ]; then + python3 ./patch_chat_template.py "${MODEL_DIR}" || \ + echo "[patch_ops] WARNING: chat template patch failed" +else + echo "[patch_ops] WARNING: ${MODEL_DIR}/tokenizer_config.json not found" +fi + build_stage "compiling submission Python sources" find . -path './wheels' -prune -o -name '*.py' -print0 | xargs -0 python3 -m py_compile