450 lines
11 KiB
Python
450 lines
11 KiB
Python
from __future__ import annotations
|
|
|
|
# Templates are copied from config部分更新.py and use __MODEL_ID__ tokens instead of
|
|
# str.format placeholders so YAML inline maps like {name: ..., value: ...} stay intact.
|
|
|
|
VLLM_MRV100_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/iluvatar/llm-infer-iluvatar-mr:v0
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm
|
|
storage: gpfs
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
max_model_len: 4096
|
|
lang: en
|
|
temperature: 0.4
|
|
repetition_penalty: 1.1
|
|
top_p: 0.9
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command: [vllm, serve, /model, --port, '20644', --served-model-name, llm, --max-model-len,
|
|
'4096', --gpu-memory-utilization, '0.9', --enforce-eager, --trust-remote-code,
|
|
-tp, '1']
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
|
|
'4096', --enforce-eager, --trust-remote-code, -tp, '1']
|
|
"""
|
|
|
|
VLLM_S4000_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-mthreads/vllm-musa-qy2-py310:v0.8.4-release
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm
|
|
storage: gpfs
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
api: completion
|
|
lang: en
|
|
max_model_len: 4096
|
|
max_tokens: 1024
|
|
temperature: 0.7
|
|
repetition_penalty: 1.1
|
|
top_p: 0.9
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- vllm
|
|
- serve
|
|
- /model
|
|
- --port
|
|
- '8000'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --dtype
|
|
- auto
|
|
- --gpu-memory-utilization
|
|
- '0.95'
|
|
- -tp
|
|
- '1'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- vllm
|
|
- serve
|
|
- /model
|
|
- --port
|
|
- '8000'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --dtype
|
|
- auto
|
|
- --gpu-memory-utilization
|
|
- '0.95'
|
|
- -tp
|
|
- '1'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
"""
|
|
|
|
VLLM_910B_CONFIG = """
|
|
framework: vllm
|
|
docker_image: git.modelhub.org.cn:9443/enginex-ascend/vllm-ascend:v0.11.0rc0
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
storage: gpfs
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
api: completion
|
|
max_tokens: 1024
|
|
temperature: 0.7
|
|
repetition_penalty: 1.2
|
|
top_p: 0.9
|
|
lang: zh
|
|
max_model_len: 2048
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- vllm
|
|
- serve
|
|
- /model
|
|
- --port
|
|
- '8000'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '2048'
|
|
- --dtype
|
|
- auto
|
|
- --gpu-memory-utilization
|
|
- '0.95'
|
|
- -tp
|
|
- '1'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- vllm
|
|
- serve
|
|
- /model
|
|
- --port
|
|
- '8000'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '2048'
|
|
- --dtype
|
|
- auto
|
|
- --gpu-memory-utilization
|
|
- '0.95'
|
|
- -tp
|
|
- '1'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
"""
|
|
|
|
VLLM_BI100_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-iluvatar-bi100/vllm:0.6.3
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm
|
|
storage: gpfs
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
lang: en
|
|
max_model_len: 4096
|
|
sut_config:
|
|
gpu_num: 1
|
|
environment:
|
|
MAX_MODEL_LEN: "4096"
|
|
values:
|
|
command: [/workspace/launch_service, --port, '80']
|
|
ref_config:
|
|
gpu_num: 1
|
|
environment:
|
|
MAX_MODEL_LEN: "4096"
|
|
values:
|
|
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
|
|
'4096', --enforce-eager, --trust-remote-code, -tp, '1']
|
|
"""
|
|
|
|
VLLM_BI150_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/sunruoxi/enginex-iluvatar-bi150-vllm-fix-tokenizer:v0.8.3
|
|
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
|
|
framework: vllm_fix_tokenizer
|
|
nv_framework: vllm_fix_tokenizer
|
|
lang: en
|
|
max_model_len: 4096
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
"""
|
|
|
|
VLLM_C500_CONFIG = """docker_image: git.modelhub.org.cn:9443/enginex-metax/vllm:0.9.1
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm
|
|
storage: gpfs
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
api: completion
|
|
lang: en
|
|
max_model_len: 4096
|
|
max_tokens: 1024
|
|
temperature: 0.7
|
|
repetition_penalty: 1.1
|
|
top_p: 0.9
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command: [/opt/conda/bin/vllm, serve, /model, --port, '20644', --served-model-name,
|
|
llm, --max-model-len, '4096', --gpu-memory-utilization, '0.9', -tp, '1', --enforce-eager,
|
|
--trust-remote-code]
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
|
|
'4096', -tp, '1', --enforce-eager, --trust-remote-code]
|
|
"""
|
|
|
|
VLLM_K100_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/i-peixingyu/k100-vllm-patched-v2.0:v2.0.0
|
|
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
|
|
framework: vllm-patch-tokenizer
|
|
nv_framework: vllm_fix_tokenizer
|
|
max_model_len: 4096
|
|
api: completion
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- --port
|
|
- '20644'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
"""
|
|
|
|
VLLM_MLU370_X4_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-torch2.5.0-torchmlu1.24.1-ubuntu22.04-py310
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm-mlu
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
sut_config:
|
|
values:
|
|
gpu_num: 1
|
|
env:
|
|
- {name: MAX_MODEL_LEN, value: 4096}
|
|
command: [vllm, serve, /model, --port, '8000', --served-model-name, llm, --max-model-len,
|
|
'4096', --trust-remote-code, --dtype, float16]
|
|
ref_config:
|
|
values:
|
|
cpu_num: 2
|
|
gpu_num: 1
|
|
env:
|
|
- {name: MAX_MODEL_LEN, value: 4096}
|
|
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
|
|
'4096', --trust-remote-code, --dtype, float16]
|
|
"""
|
|
|
|
VLLM_MLU370_X8_CONFIG = """docker_image: harbor.4pd.io/hardcore-tech/cambricon-mlu370-pytorch:v25.01-20260204
|
|
nv_docker_image: harbor.4pd.io/dooke/vllm/vllm/vllm-openai:v0.11.0
|
|
framework: vllm-mlu
|
|
api: completion
|
|
lang: en
|
|
temperature: 0.4
|
|
repetition_penalty: 1.1
|
|
top_p: 0.9
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
env:
|
|
- {name: MAX_MODEL_LEN, value: 4096}
|
|
- {name: VLLM_ALLOW_LONG_MAX_MODEL_LEN, value: '1'}
|
|
command: [vllm, serve, /model, --port, '8000', --served-model-name, llm, --max-model-len,
|
|
'4096', --trust-remote-code, --dtype, float16]
|
|
ref_config:
|
|
cpu_num: 2
|
|
gpu_num: 1
|
|
values:
|
|
env:
|
|
- {name: MAX_MODEL_LEN, value: 4096}
|
|
- {name: VLLM_ALLOW_LONG_MAX_MODEL_LEN, value: '1'}
|
|
command: [vllm, serve, /model, --port, '80', --served-model-name, llm, --max-model-len,
|
|
'4096', --trust-remote-code, --dtype, float16]
|
|
"""
|
|
|
|
VLLM_P800_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/zheng/vllm-kunlunxin-p-800-tokenizer-patch:v1.0
|
|
nv_docker_image: harbor-contest.4pd.io/sunruoxi/vllm-openai-fix-tokenizer:v0.11.0
|
|
framework: vllm_tokenizer_patch
|
|
nv_framework: vllm_fix_tokenizer
|
|
api: completion
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
max_model_len: 4096
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- vllm
|
|
- serve
|
|
- /model
|
|
- --port
|
|
- '8000'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --gpu-memory-utilization
|
|
- '0.9'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
"""
|
|
|
|
VLLM_166M_CONFIG = """docker_image: harbor.4pd.io/modelhubxc/sunruoxi/enginex-llm-biren166m-fix-tokenizer:v26.01
|
|
nv_docker_image: harbor.4pd.io/modelhubxc/enginex-nvidia/vllm:0.11.0-patch-tokenizer
|
|
modelhub_options:
|
|
srcRelativePath: leaderboard/modelHubXC/__MODEL_ID__
|
|
mountPoint: /model
|
|
framework: vllm_fix_tokenizer
|
|
nv_framework: vllm_fix_tokenizer
|
|
lang: en
|
|
max_model_len: 4096
|
|
sut_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
ref_config:
|
|
gpu_num: 1
|
|
values:
|
|
command:
|
|
- /opt/entrypoint.sh
|
|
- /model
|
|
- --port
|
|
- '80'
|
|
- --served-model-name
|
|
- llm
|
|
- --max-model-len
|
|
- '4096'
|
|
- --enforce-eager
|
|
- --trust-remote-code
|
|
- -tp
|
|
- '1'
|
|
"""
|
|
|
|
VLLM_GPU_CONFIG_DICT = {
|
|
"Ascend_910-b3": VLLM_910B_CONFIG,
|
|
"Ascend_910-b4": VLLM_910B_CONFIG,
|
|
"Mthreads_s4000": VLLM_S4000_CONFIG,
|
|
"hygon_k100-ai": VLLM_K100_CONFIG,
|
|
"Iluvatar_mrv-100": VLLM_MRV100_CONFIG,
|
|
"Iluvatar_bi-100": VLLM_BI100_CONFIG,
|
|
"Iluvatar_bi-150": VLLM_BI150_CONFIG,
|
|
"MetaX_c-500": VLLM_C500_CONFIG,
|
|
"Cambricon_mlu-370-x4": VLLM_MLU370_X4_CONFIG,
|
|
"Cambricon_mlu-370-x8": VLLM_MLU370_X8_CONFIG,
|
|
"Kunlunxin_p-800": VLLM_P800_CONFIG,
|
|
"Biren_166m": VLLM_166M_CONFIG,
|
|
}
|
|
|
|
|
|
def supported_gpu_types() -> set[str]:
|
|
return set(VLLM_GPU_CONFIG_DICT)
|
|
|
|
|
|
def gen_vllm_config(gpu_type: str, model_id: str, max_model_len: int = 1024) -> str:
|
|
if gpu_type not in VLLM_GPU_CONFIG_DICT:
|
|
raise ValueError(f"Unsupported vLLM gpu_type: {gpu_type}")
|
|
return VLLM_GPU_CONFIG_DICT[gpu_type].replace("__MODEL_ID__", model_id)
|