diff --git a/app/clients/modelhub.py b/app/clients/modelhub.py index 1d1b7b1..65db964 100644 --- a/app/clients/modelhub.py +++ b/app/clients/modelhub.py @@ -142,7 +142,7 @@ class ModelHubClient: message = body.get("message", "") if code == 0 and message == "ok": result = "success" - elif code == 40000 and "正在验证中" in message: + elif "请勿重复提交" in message or "已经被验证成功" in message or "正在验证中" in message: result = "conflict" elif code == 60007: result = "queue_full" diff --git a/app/scheduler/downloader.py b/app/scheduler/downloader.py index fd8bbe7..31854c8 100644 --- a/app/scheduler/downloader.py +++ b/app/scheduler/downloader.py @@ -22,6 +22,8 @@ class Downloader: active_self_slot=active, next_poll_at=future_seconds(self.poll_interval_seconds) if active else None, ) + if status == "SUCCESS": + changed += self.repo.mark_waiting_download_tasks_ready(download["model_id"], download["source"]) changed += 1 return changed diff --git a/app/scheduler/loop.py b/app/scheduler/loop.py index 644ef73..6f8c513 100644 --- a/app/scheduler/loop.py +++ b/app/scheduler/loop.py @@ -92,9 +92,7 @@ class StrategyLoop: released = self.submitter.release_queue_full_backoff() ready = self.mark_ready_tasks() submitted = self.submitter.submit_due() - started_downloads = 0 - if submitted == 0 and ready == 0: - started_downloads = self.downloader.maybe_start_self_downloads() + started_downloads = self.downloader.maybe_start_self_downloads() return { "crawled_not_adapted": crawler_stats["not_adapted"], "crawled_download_success": crawler_stats["download_success"], diff --git a/app/settings.py b/app/settings.py index 9d27cc2..699aa74 100644 --- a/app/settings.py +++ b/app/settings.py @@ -81,11 +81,11 @@ class Settings: target_task_level: str = "文本生成" modelhub_page_size: int = 100 crawler_refresh_seconds: int = 3600 - crawler_max_pages: int = 0 + crawler_max_pages: int = 400 enable_download_success_crawler: bool = True download_success_page_size: int = 50 - download_success_max_pages: int = 0 + download_success_max_pages: int = 100 config_file_loaded: str = "" @@ -140,10 +140,10 @@ def load_settings(require_secrets: bool = True) -> Settings: target_task_level=str(_value(config, "TARGET_TASK_LEVEL", "文本生成")), modelhub_page_size=_int_value(config, "MODELHUB_PAGE_SIZE", 100), crawler_refresh_seconds=_int_value(config, "CRAWLER_REFRESH_SECONDS", 3600), - crawler_max_pages=_int_value(config, "CRAWLER_MAX_PAGES", 0), + crawler_max_pages=_int_value(config, "CRAWLER_MAX_PAGES", 400), enable_download_success_crawler=_bool_value(config, "ENABLE_DOWNLOAD_SUCCESS_CRAWLER", True), download_success_page_size=_int_value(config, "DOWNLOAD_SUCCESS_PAGE_SIZE", 50), - download_success_max_pages=_int_value(config, "DOWNLOAD_SUCCESS_MAX_PAGES", 0), + download_success_max_pages=_int_value(config, "DOWNLOAD_SUCCESS_MAX_PAGES", 100), config_file_loaded=loaded_path, ) if require_secrets: diff --git a/app/storage/repositories.py b/app/storage/repositories.py index d97995c..c558740 100644 --- a/app/storage/repositories.py +++ b/app/storage/repositories.py @@ -233,6 +233,25 @@ class Repository: ) return list(cursor.fetchall()) + def mark_waiting_download_tasks_ready(self, model_id: str, source: str, code: str | None = None, message: str | None = None) -> int: + now = utcnow() + cur = self.conn.execute( + """ + UPDATE tasks SET + status = 'ready_to_submit', + last_error_code = ?, + last_error_message = ?, + next_attempt_at = NULL, + updated_at = ? + WHERE model_id = ? + AND status = 'waiting_download' + """, + (code, message, now, model_id), + ) + self.conn.commit() + self.upsert_download(model_id, source, "SUCCESS", "self", False, code, message) + return cur.rowcount + def record_submission_attempt( self, task_id: int,