Files
vllm_agent_strategy/app/domain/vllm_configs.py

450 lines
11 KiB
Python
Raw Normal View History

2026-07-21 16:04:18 +08:00
from __future__ import annotations
# Templates are copied from config部分更新.py and use __MODEL_ID__ tokens instead of
# str.format placeholders so YAML inline maps like {name: ..., value: ...} stay intact.
VLLM_MRV100_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/iluvatar/llm-infer-iluvatar-mr:v0
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm
storage: gpfs
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
max_model_len: 4096
lang: en
temperature: 0.4
repetition_penalty: 1.1
top_p: 0.9
sut_config:
gpu_num: 1
values:
command: [vllm, serve, /model, --port, '20644', --served-model-name, llm, --max-model-len,
'4096', --gpu-memory-utilization, '0.9', --enforce-eager, --trust-remote-code,
-tp, '1']
ref_config:
gpu_num: 1
values:
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
'4096', --enforce-eager, --trust-remote-code, -tp, '1']
"""
VLLM_S4000_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-mthreads/vllm-musa-qy2-py310:v0.8.4-release
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm
storage: gpfs
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
api: completion
lang: en
max_model_len: 4096
max_tokens: 1024
temperature: 0.7
repetition_penalty: 1.1
top_p: 0.9
sut_config:
gpu_num: 1
values:
command:
- vllm
- serve
- /model
- --port
- '8000'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --dtype
- auto
- --gpu-memory-utilization
- '0.95'
- -tp
- '1'
- --enforce-eager
- --trust-remote-code
ref_config:
gpu_num: 1
values:
command:
- vllm
- serve
- /model
- --port
- '8000'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --dtype
- auto
- --gpu-memory-utilization
- '0.95'
- -tp
- '1'
- --enforce-eager
- --trust-remote-code
"""
VLLM_910B_CONFIG = """
framework: vllm
docker_image: git.modelhub.org.cn:9443/enginex-ascend/vllm-ascend:v0.11.0rc0
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
storage: gpfs
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
api: completion
max_tokens: 1024
temperature: 0.7
repetition_penalty: 1.2
top_p: 0.9
lang: zh
max_model_len: 2048
sut_config:
gpu_num: 1
values:
command:
- vllm
- serve
- /model
- --port
- '8000'
- --served-model-name
- llm
- --max-model-len
- '2048'
- --dtype
- auto
- --gpu-memory-utilization
- '0.95'
- -tp
- '1'
- --enforce-eager
- --trust-remote-code
ref_config:
gpu_num: 1
values:
command:
- vllm
- serve
- /model
- --port
- '8000'
- --served-model-name
- llm
- --max-model-len
- '2048'
- --dtype
- auto
- --gpu-memory-utilization
- '0.95'
- -tp
- '1'
- --enforce-eager
- --trust-remote-code
"""
VLLM_BI100_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-iluvatar-bi100/vllm:0.6.3
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm
storage: gpfs
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
lang: en
max_model_len: 4096
sut_config:
gpu_num: 1
environment:
MAX_MODEL_LEN: "4096"
values:
command: [/workspace/launch_service, --port, '80']
ref_config:
gpu_num: 1
environment:
MAX_MODEL_LEN: "4096"
values:
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
'4096', --enforce-eager, --trust-remote-code, -tp, '1']
"""
VLLM_BI150_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/sunruoxi/enginex-iluvatar-bi150-vllm-fix-tokenizer:v0.8.3
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
framework: vllm_fix_tokenizer
nv_framework: vllm_fix_tokenizer
lang: en
max_model_len: 4096
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
sut_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
ref_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
"""
VLLM_C500_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-metax/vllm:0.9.1
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm
storage: gpfs
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
api: completion
lang: en
max_model_len: 4096
max_tokens: 1024
temperature: 0.7
repetition_penalty: 1.1
top_p: 0.9
sut_config:
gpu_num: 1
values:
command: [/opt/conda/bin/vllm, serve, /model, --port, '20644', --served-model-name,
llm, --max-model-len, '4096', --gpu-memory-utilization, '0.9', -tp, '1', --enforce-eager,
--trust-remote-code]
ref_config:
gpu_num: 1
values:
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
'4096', -tp, '1', --enforce-eager, --trust-remote-code]
"""
VLLM_K100_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/i-peixingyu/k100-vllm-patched-v2.0:v2.0.0
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
framework: vllm-patch-tokenizer
nv_framework: vllm_fix_tokenizer
max_model_len: 4096
api: completion
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
sut_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- --port
- '20644'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
ref_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
"""
VLLM_MLU370_X4_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-torch2.5.0-torchmlu1.24.1-ubuntu22.04-py310
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm-mlu
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
sut_config:
values:
gpu_num: 1
env:
- {name: MAX_MODEL_LEN, value: 4096}
command: [vllm, serve, /model, --port, '8000', --served-model-name, llm, --max-model-len,
'4096', --trust-remote-code, --dtype, float16]
ref_config:
values:
cpu_num: 2
gpu_num: 1
env:
- {name: MAX_MODEL_LEN, value: 4096}
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
'4096', --trust-remote-code, --dtype, float16]
"""
VLLM_MLU370_X8_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-20260204
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
framework: vllm-mlu
api: completion
lang: en
temperature: 0.4
repetition_penalty: 1.1
top_p: 0.9
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
sut_config:
gpu_num: 1
values:
env:
- {name: MAX_MODEL_LEN, value: 4096}
- {name: VLLM_ALLOW_LONG_MAX_MODEL_LEN, value: '1'}
command: [vllm, serve, /model, --port, '8000', --served-model-name, llm, --max-model-len,
'4096', --trust-remote-code, --dtype, float16]
ref_config:
cpu_num: 2
gpu_num: 1
values:
env:
- {name: MAX_MODEL_LEN, value: 4096}
- {name: VLLM_ALLOW_LONG_MAX_MODEL_LEN, value: '1'}
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
'4096', --trust-remote-code, --dtype, float16]
"""
VLLM_P800_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/zheng/vllm-kunlunxin-p-800-tokenizer-patch:v1.0
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
framework: vllm_tokenizer_patch
nv_framework: vllm_fix_tokenizer
api: completion
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
max_model_len: 4096
sut_config:
gpu_num: 1
values:
command:
- vllm
- serve
- /model
- --port
- '8000'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --gpu-memory-utilization
- '0.9'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
ref_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
"""
VLLM_166M_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/sunruoxi/enginex-llm-biren166m-fix-tokenizer:v26.01
nv_docker_image: harbor.4pd.io/modelhubxc/enginex-nvidia/vllm:0.11.0-patch-tokenizer
modelhub_options:
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
mountPoint: /model
framework: vllm_fix_tokenizer
nv_framework: vllm_fix_tokenizer
lang: en
max_model_len: 4096
sut_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
ref_config:
gpu_num: 1
values:
command:
- /opt/entrypoint.sh
- /model
- --port
- '80'
- --served-model-name
- llm
- --max-model-len
- '4096'
- --enforce-eager
- --trust-remote-code
- -tp
- '1'
"""
VLLM_GPU_CONFIG_DICT = {
"Ascend_910-b3": VLLM_910B_CONFIG,
"Ascend_910-b4": VLLM_910B_CONFIG,
"Mthreads_s4000": VLLM_S4000_CONFIG,
"hygon_k100-ai": VLLM_K100_CONFIG,
"Iluvatar_mrv-100": VLLM_MRV100_CONFIG,
"Iluvatar_bi-100": VLLM_BI100_CONFIG,
"Iluvatar_bi-150": VLLM_BI150_CONFIG,
"MetaX_c-500": VLLM_C500_CONFIG,
"Cambricon_mlu-370-x4": VLLM_MLU370_X4_CONFIG,
"Cambricon_mlu-370-x8": VLLM_MLU370_X8_CONFIG,
"Kunlunxin_p-800": VLLM_P800_CONFIG,
"Biren_166m": VLLM_166M_CONFIG,
}
def supported_gpu_types() -> set[str]:
return set(VLLM_GPU_CONFIG_DICT)
def gen_vllm_config(gpu_type: str, model_id: str, max_model_len: int = 1024) -> str:
if gpu_type not in VLLM_GPU_CONFIG_DICT:
raise ValueError(f"Unsupported vLLM gpu_type: {gpu_type}")
return VLLM_GPU_CONFIG_DICT[gpu_type].replace("__MODEL_ID__", model_id)