Compare commits
9 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
207d44b8f2 | ||
|
|
1d33394dac | ||
|
|
1811a66aee | ||
|
|
12bf6d637f | ||
|
|
85c456afe0 | ||
|
|
b3b52848c2 | ||
|
|
d6e9f2e7a0 | ||
|
|
821edbc8f5 | ||
|
|
5e2df178e1 |
79
main.py
79
main.py
@@ -121,12 +121,12 @@ def init_db():
|
|||||||
# ModelScope 搜索
|
# ModelScope 搜索
|
||||||
# ============================================================
|
# ============================================================
|
||||||
|
|
||||||
def search_models(keyword: str, limit: int = 100) -> list:
|
def search_models(keyword: str, limit: int = 50) -> list:
|
||||||
"""从 ModelScope 搜索模型"""
|
"""从 ModelScope 搜索模型"""
|
||||||
url = "https://modelscope.cn/openapi/v1/models"
|
url = "https://modelscope.cn/openapi/v1/models"
|
||||||
params = {
|
params = {
|
||||||
'search': keyword,
|
'search': keyword,
|
||||||
'page_size': limit,
|
'page_size': min(limit, 50), # API 上限 50
|
||||||
'page_number': 1,
|
'page_number': 1,
|
||||||
'sort': 'downloads',
|
'sort': 'downloads',
|
||||||
}
|
}
|
||||||
@@ -135,7 +135,11 @@ def search_models(keyword: str, limit: int = 100) -> list:
|
|||||||
headers={'User-Agent': 'Mozilla/5.0'})
|
headers={'User-Agent': 'Mozilla/5.0'})
|
||||||
data = resp.json()
|
data = resp.json()
|
||||||
if data.get('success'):
|
if data.get('success'):
|
||||||
return data.get('data', {}).get('models', [])
|
models = data.get('data', {}).get('models', [])
|
||||||
|
log(f" 搜索 [{keyword}]: status={resp.status_code} models={len(models)}")
|
||||||
|
return models
|
||||||
|
else:
|
||||||
|
log(f" 搜索 [{keyword}]: success=false, data={str(data)[:200]}")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
log(f" 搜索失败 [{keyword}]: {e}")
|
log(f" 搜索失败 [{keyword}]: {e}")
|
||||||
return []
|
return []
|
||||||
@@ -191,7 +195,8 @@ def check_platform_verify(model_id: str) -> dict:
|
|||||||
resp = requests.get(url, headers=headers, params={'modelId': model_id}, timeout=10)
|
resp = requests.get(url, headers=headers, params={'modelId': model_id}, timeout=10)
|
||||||
data = resp.json()
|
data = resp.json()
|
||||||
if data.get('code') == 0:
|
if data.get('code') == 0:
|
||||||
return data.get('data', {}).get('verifyResult', {})
|
result = data.get('data', {}).get('verifyResult', {})
|
||||||
|
return result if isinstance(result, dict) else {}
|
||||||
except Exception:
|
except Exception:
|
||||||
pass
|
pass
|
||||||
return {}
|
return {}
|
||||||
@@ -238,16 +243,15 @@ def check_queue_available(gpu: str, token: str) -> int:
|
|||||||
|
|
||||||
|
|
||||||
def build_config_params(gpu: str) -> str:
|
def build_config_params(gpu: str) -> str:
|
||||||
"""构建 YAML 配置"""
|
"""构建 YAML 配置 - 完全匹配平台自动生成的格式"""
|
||||||
config = GPU_CONFIGS.get(gpu, {})
|
config = GPU_CONFIGS.get(gpu, {})
|
||||||
framework = config.get('framework', 'vllm')
|
framework = config.get('framework', 'vllm')
|
||||||
docker_image = config.get('docker_image', '')
|
|
||||||
|
|
||||||
if framework == 'llama.cpp':
|
if framework == 'llama.cpp':
|
||||||
|
# hygon_k100-ai 使用 llama.cpp
|
||||||
params = {
|
params = {
|
||||||
'framework': 'llama.cpp',
|
'framework': 'llama.cpp',
|
||||||
'docker_image': docker_image,
|
'nv_framework': 'llama.cpp',
|
||||||
'nv_docker_image': docker_image,
|
|
||||||
'api': 'completion',
|
'api': 'completion',
|
||||||
'max_tokens': 1024,
|
'max_tokens': 1024,
|
||||||
'temperature': 0.7,
|
'temperature': 0.7,
|
||||||
@@ -255,21 +259,21 @@ def build_config_params(gpu: str) -> str:
|
|||||||
'top_p': 0.9,
|
'top_p': 0.9,
|
||||||
'lang': 'zh',
|
'lang': 'zh',
|
||||||
'max_model_len': 4096,
|
'max_model_len': 4096,
|
||||||
'modelFormat': 'GGUF',
|
|
||||||
'sut_config': {
|
'sut_config': {
|
||||||
'values': {
|
|
||||||
'gpu_num': 1,
|
'gpu_num': 1,
|
||||||
|
'values': {
|
||||||
'command': [
|
'command': [
|
||||||
'/bin/bash', '-ic',
|
'llama-server', '--model', '/model', '--alias', 'llm',
|
||||||
'llama-server --model /model --alias llm --threads 20 '
|
'--threads', '20', '--n-gpu-layers', '999', '--prio', '3',
|
||||||
'--n-gpu-layers 999 --prio 3 --min_p 0.01 '
|
'--min_p', '0.01', '--ctx-size', '4096',
|
||||||
'--ctx-size 4096 --host 0.0.0.0 --port 8000 --jinja --flash-attn off'
|
'--host', '0.0.0.0', '--port', '8000',
|
||||||
|
'--jinja', '--flash-attn', 'off',
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
'ref_config': {
|
'ref_config': {
|
||||||
|
'gpu_num': 1,
|
||||||
'values': {
|
'values': {
|
||||||
'cpu_num': 2, 'gpu_num': 1,
|
|
||||||
'command': [
|
'command': [
|
||||||
'llama-server', '--model', '/model', '--alias', 'llm',
|
'llama-server', '--model', '/model', '--alias', 'llm',
|
||||||
'--threads', '20', '--n-gpu-layers', '999',
|
'--threads', '20', '--n-gpu-layers', '999',
|
||||||
@@ -277,13 +281,12 @@ def build_config_params(gpu: str) -> str:
|
|||||||
]
|
]
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
'model': 'llm',
|
|
||||||
}
|
}
|
||||||
else:
|
else:
|
||||||
|
# vllm (P800, Ascend, MetaX)
|
||||||
params = {
|
params = {
|
||||||
'framework': 'vllm',
|
'framework': 'vllm',
|
||||||
'docker_image': docker_image,
|
'nv_framework': 'vllm',
|
||||||
'nv_docker_image': docker_image,
|
|
||||||
'api': 'completion',
|
'api': 'completion',
|
||||||
'max_tokens': 1024,
|
'max_tokens': 1024,
|
||||||
'temperature': 0.7,
|
'temperature': 0.7,
|
||||||
@@ -291,21 +294,9 @@ def build_config_params(gpu: str) -> str:
|
|||||||
'top_p': 0.9,
|
'top_p': 0.9,
|
||||||
'lang': 'zh',
|
'lang': 'zh',
|
||||||
'max_model_len': 2048,
|
'max_model_len': 2048,
|
||||||
'modelFormat': 'HuggingFace',
|
|
||||||
'sut_config': {
|
'sut_config': {
|
||||||
'values': {
|
|
||||||
'gpu_num': 1,
|
'gpu_num': 1,
|
||||||
'command': [
|
|
||||||
'/bin/bash', '-ic',
|
|
||||||
'vllm serve /model --port 8000 --served-model-name llm '
|
|
||||||
'--max-model-len 2048 --dtype auto --gpu-memory-utilization 0.95 '
|
|
||||||
'-tp 1 --enforce-eager --trust-remote-code'
|
|
||||||
]
|
|
||||||
}
|
|
||||||
},
|
|
||||||
'ref_config': {
|
|
||||||
'values': {
|
'values': {
|
||||||
'cpu_num': 2, 'gpu_num': 1,
|
|
||||||
'command': [
|
'command': [
|
||||||
'vllm', 'serve', '/model', '--port', '8000',
|
'vllm', 'serve', '/model', '--port', '8000',
|
||||||
'--served-model-name', 'llm', '--max-model-len', '2048',
|
'--served-model-name', 'llm', '--max-model-len', '2048',
|
||||||
@@ -314,9 +305,18 @@ def build_config_params(gpu: str) -> str:
|
|||||||
]
|
]
|
||||||
}
|
}
|
||||||
},
|
},
|
||||||
'model': 'llm',
|
'ref_config': {
|
||||||
|
'gpu_num': 1,
|
||||||
|
'values': {
|
||||||
|
'command': [
|
||||||
|
'vllm', 'serve', '/model', '--port', '80',
|
||||||
|
'--served-model-name', 'llm', '--max-model-len', '4096',
|
||||||
|
'--enforce-eager', '--trust-remote-code', '-tp', '1',
|
||||||
|
]
|
||||||
}
|
}
|
||||||
return yaml.dump(params, default_flow_style=False, allow_unicode=True)
|
},
|
||||||
|
}
|
||||||
|
return yaml.dump(params, default_flow_style=False, allow_unicode=True, width=1000)
|
||||||
|
|
||||||
|
|
||||||
def submit_model(model_url: str, gpu: str, token: str) -> tuple:
|
def submit_model(model_url: str, gpu: str, token: str) -> tuple:
|
||||||
@@ -728,9 +728,22 @@ def main():
|
|||||||
log(f"[ModelHub提交] ❌ error={e}")
|
log(f"[ModelHub提交] ❌ error={e}")
|
||||||
|
|
||||||
log("=" * 50)
|
log("=" * 50)
|
||||||
log("连通性测试完成,请查看上方结果")
|
log("连通性测试完成")
|
||||||
log("=" * 50)
|
log("=" * 50)
|
||||||
|
|
||||||
|
# 开始正式提交流程
|
||||||
|
log("\n自动触发提交流程...")
|
||||||
|
try:
|
||||||
|
state['running'] = True
|
||||||
|
state['last_run'] = datetime.now().isoformat()
|
||||||
|
count = run_pipeline(submit_limit=30)
|
||||||
|
state['last_result'] = {'submitted': count, 'success': True}
|
||||||
|
except Exception as e:
|
||||||
|
log(f"流程异常: {traceback.format_exc()}")
|
||||||
|
state['last_result'] = {'error': str(e), 'success': False}
|
||||||
|
finally:
|
||||||
|
state['running'] = False
|
||||||
|
|
||||||
threading.Thread(target=_startup_test, daemon=True).start()
|
threading.Thread(target=_startup_test, daemon=True).start()
|
||||||
|
|
||||||
while not shutdown_requested:
|
while not shutdown_requested:
|
||||||
|
|||||||
Reference in New Issue
Block a user