7 Commits

Author SHA1 Message Date
9eb9b6b67f Fix frozen settings page limit override
Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-22 18:06:55 +08:00
702b436622 Force crawler page limits from code
Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-22 17:59:44 +08:00
9065fcffac Fix strategy scheduling and crawler limits
Co-Authored-By: Claude <noreply@anthropic.com>
2026-07-22 17:57:03 +08:00
820fa12fdb chore: narrow GPU targets, increase crawler pages to 200 2026-07-22 09:17:08 +08:00
16562b595f fix: change CONTRIBUTORS to hengyan 2026-07-22 09:02:42 +08:00
a7efcbff76 chore: limit crawler pages to 20 2026-07-22 08:58:15 +08:00
5cd21c1128 debug: add submission logging 2026-07-21 18:24:41 +08:00
7 changed files with 37 additions and 14 deletions

View File

@@ -142,7 +142,7 @@ class ModelHubClient:
message = body.get("message", "")
if code == 0 and message == "ok":
result = "success"
elif code == 40000 and "正在验证中" in message:
elif "请勿重复提交" in message or "已经被验证成功" in message or "正在验证中" in message:
result = "conflict"
elif code == 60007:
result = "queue_full"

View File

@@ -22,6 +22,8 @@ class Downloader:
active_self_slot=active,
next_poll_at=future_seconds(self.poll_interval_seconds) if active else None,
)
if status == "SUCCESS":
changed += self.repo.mark_waiting_download_tasks_ready(download["model_id"], download["source"])
changed += 1
return changed

View File

@@ -92,8 +92,6 @@ class StrategyLoop:
released = self.submitter.release_queue_full_backoff()
ready = self.mark_ready_tasks()
submitted = self.submitter.submit_due()
started_downloads = 0
if submitted == 0 and ready == 0:
started_downloads = self.downloader.maybe_start_self_downloads()
return {
"crawled_not_adapted": crawler_stats["not_adapted"],

View File

@@ -1,6 +1,9 @@
from app.clients.modelhub import ModelHubClient
from app.domain.vllm_configs import gen_vllm_config
from app.storage.repositories import Repository, future_seconds
import logging
LOG = logging.getLogger(__name__)
class Submitter:
@@ -13,6 +16,7 @@ class Submitter:
for task in self.repo.due_tasks(("ready_to_submit",), limit=limit):
config = gen_vllm_config(task["gpu_type"], task["model_id"], task["max_model_len"])
result = self.client.create_contest_task(task["model_id"], task["gpu_type"], config)
LOG.info("submit %s / %s / %s -> %s (%s)", task["model_id"], task["gpu_type"], task["source"], result.result, result.message)
self.repo.record_submission_attempt(
task["id"],
task["model_id"],

View File

@@ -81,11 +81,11 @@ class Settings:
target_task_level: str = "文本生成"
modelhub_page_size: int = 100
crawler_refresh_seconds: int = 3600
crawler_max_pages: int = 0
crawler_max_pages: int = 400
enable_download_success_crawler: bool = True
download_success_page_size: int = 50
download_success_max_pages: int = 0
download_success_max_pages: int = 100
config_file_loaded: str = ""
@@ -140,10 +140,10 @@ def load_settings(require_secrets: bool = True) -> Settings:
target_task_level=str(_value(config, "TARGET_TASK_LEVEL", "文本生成")),
modelhub_page_size=_int_value(config, "MODELHUB_PAGE_SIZE", 100),
crawler_refresh_seconds=_int_value(config, "CRAWLER_REFRESH_SECONDS", 3600),
crawler_max_pages=_int_value(config, "CRAWLER_MAX_PAGES", 0),
crawler_max_pages=max(_int_value(config, "CRAWLER_MAX_PAGES", 400), 400),
enable_download_success_crawler=_bool_value(config, "ENABLE_DOWNLOAD_SUCCESS_CRAWLER", True),
download_success_page_size=_int_value(config, "DOWNLOAD_SUCCESS_PAGE_SIZE", 50),
download_success_max_pages=_int_value(config, "DOWNLOAD_SUCCESS_MAX_PAGES", 0),
download_success_max_pages=max(_int_value(config, "DOWNLOAD_SUCCESS_MAX_PAGES", 100), 100),
config_file_loaded=loaded_path,
)
if require_secrets:

View File

@@ -233,6 +233,25 @@ class Repository:
)
return list(cursor.fetchall())
def mark_waiting_download_tasks_ready(self, model_id: str, source: str, code: str | None = None, message: str | None = None) -> int:
now = utcnow()
cur = self.conn.execute(
"""
UPDATE tasks SET
status = 'ready_to_submit',
last_error_code = ?,
last_error_message = ?,
next_attempt_at = NULL,
updated_at = ?
WHERE model_id = ?
AND status = 'waiting_download'
""",
(code, message, now, model_id),
)
self.conn.commit()
self.upsert_download(model_id, source, "SUCCESS", "self", False, code, message)
return cur.rowcount
def record_submission_attempt(
self,
task_id: int,

View File

@@ -4,8 +4,8 @@
"CONFTEST_API_TOKEN": "0f3c1fd38fbac493cf3760ce4aaeac90",
"EMAIL": "i-hengyan@4paradigm.com",
"USER_ID": "83",
"CONTRIBUTORS": "i-hengyan",
"STRATEGY_ID": "280bc4a7-a460-4778-bf74-89ccd3162bf5",
"CONTRIBUTORS": "hengyan",
"STRATEGY_ID": "a2514012-5ac8-4187-a3dd-8d74d8582827",
"MODELHUB_URL": "https://modelhub.org.cn",
"MODELHUB_API_BASE": "https://modelhub.org.cn/api",
@@ -25,18 +25,18 @@
"DOWNLOAD_POLL_INTERVAL_SECONDS": 300,
"CRAWLER_REFRESH_SECONDS": 3600,
"DEFAULT_GPU_ALIASES": "910b,k100,p800,166m,bi100,bi150,c500,s4000,mrv100,mlu370-x4,mlu370-x8",
"DEFAULT_GPU_ALIASES": "c500,s4000,166m,k100,p800",
"DEFAULT_MAX_MODEL_LEN": 1024,
"ENABLE_NOT_ADAPTED_CRAWLER": true,
"TARGET_MACHINE_NAMES": "MTT S4000,Hygon K100,Kunlunxin P800",
"TARGET_MACHINE_NAMES": "MetaX C500,MTT S4000,Biren 166M,Hygon K100,Kunlunxin P800",
"TARGET_TASK_LEVEL": "文本生成",
"MODELHUB_PAGE_SIZE": 100,
"CRAWLER_MAX_PAGES": 0,
"CRAWLER_MAX_PAGES": 200,
"ENABLE_DOWNLOAD_SUCCESS_CRAWLER": true,
"DOWNLOAD_SUCCESS_PAGE_SIZE": 50,
"DOWNLOAD_SUCCESS_MAX_PAGES": 0,
"DOWNLOAD_SUCCESS_MAX_PAGES": 20,
"BOUNTY_ALLOW_DIRECT_SUBMIT": true,
"AUTO_LOAD_SEED_PATH": ""