fix(runtime): launch_server.py强制覆盖所有vllm路径后启动
根因:patch_ops.sh部署到VLLM_ROOT(lib64),但python3 -m vllm走的是 /usr/local/corex/lib/python3/dist-packages/vllm/(未被覆盖的路径) 导致基础镜像原版api_server.py运行,不识别qwen3_coder/reasoning-parser 修复:launch_server.py在import前遍历sys.path所有vllm安装, 用shutil.copy2强制覆盖api_server/cli_args/serving_chat等 然后from vllm.entrypoints.openai.api_server import *启动
This commit is contained in:
@@ -1,8 +1,7 @@
|
||||
concurrency: 1
|
||||
command:
|
||||
- python3
|
||||
- -m
|
||||
- vllm.entrypoints.openai.api_server
|
||||
- /workspace/qwen3_6_scripts/launch_server.py
|
||||
- --model
|
||||
- /model
|
||||
- --served-model-name
|
||||
|
||||
49
qwen3_6_scripts/launch_server.py
Normal file
49
qwen3_6_scripts/launch_server.py
Normal file
@@ -0,0 +1,49 @@
|
||||
#!/usr/bin/env python3
|
||||
"""launch_server.py — Ensure our patched api_server.py runs, not the base image's.
|
||||
|
||||
Patches the RUNNING vllm install's api_server.py/cli_args.py in-place before
|
||||
importing, then delegates to the standard vllm api_server main().
|
||||
"""
|
||||
import os, sys, shutil
|
||||
|
||||
def _force_patch():
|
||||
"""Copy our files over ALL vllm installs found on sys.path."""
|
||||
src_dir = os.path.dirname(os.path.abspath(__file__))
|
||||
patched = set()
|
||||
|
||||
for p in sys.path:
|
||||
api = os.path.join(p, "vllm", "entrypoints", "openai", "api_server.py")
|
||||
if os.path.isfile(api) and api not in patched:
|
||||
for f in ["api_server.py", "cli_args.py", "serving_chat.py",
|
||||
"protocol.py", "serving_tokenization.py"]:
|
||||
src = os.path.join(src_dir, f)
|
||||
dst = os.path.join(p, "vllm", "entrypoints", "openai", f)
|
||||
if os.path.isfile(src):
|
||||
shutil.copy2(src, dst)
|
||||
# chat_utils
|
||||
cu_src = os.path.join(src_dir, "chat_utils.py")
|
||||
cu_dst = os.path.join(p, "vllm", "entrypoints", "chat_utils.py")
|
||||
if os.path.isfile(cu_src):
|
||||
shutil.copy2(cu_src, cu_dst)
|
||||
# reasoning
|
||||
reason_src = os.path.join(src_dir, "reasoning")
|
||||
reason_dst = os.path.join(p, "vllm", "reasoning")
|
||||
if os.path.isdir(reason_src):
|
||||
shutil.copytree(reason_src, reason_dst, dirs_exist_ok=True)
|
||||
# tool parser
|
||||
tp_src = os.path.join(src_dir, "qwen3coder_tool_parser.py")
|
||||
tp_dst = os.path.join(p, "vllm", "entrypoints", "openai",
|
||||
"tool_parsers", "qwen3coder_tool_parser.py")
|
||||
if os.path.isfile(tp_src) and os.path.isdir(os.path.dirname(tp_dst)):
|
||||
shutil.copy2(tp_src, tp_dst)
|
||||
patched.add(api)
|
||||
|
||||
if patched:
|
||||
print(f"[launch] Force-patched {len(patched)} vllm installs", file=sys.stderr)
|
||||
else:
|
||||
print("[launch] WARNING: no vllm installs found to patch", file=sys.stderr)
|
||||
|
||||
_force_patch()
|
||||
|
||||
# Now run the standard vllm api_server with our patched version
|
||||
from vllm.entrypoints.openai.api_server import *
|
||||
Reference in New Issue
Block a user