refactor: retain decision state instead of full logs
This commit is contained in:
16
README.md
16
README.md
@@ -413,11 +413,23 @@ success-based transformer selection. Missing framework/task/profile fields in hi
|
|||||||
recovered from the durable task ledger. The evidence and category breakdown are
|
recovered from the durable task ledger. The evidence and category breakdown are
|
||||||
documented in `docs/failure-analysis-2026-09-04.md`.
|
documented in `docs/failure-analysis-2026-09-04.md`.
|
||||||
|
|
||||||
|
Version `2026.09.04.2` switches persistence from forensic-log retention to
|
||||||
|
decision-state retention after the historical audit was incorporated. New raw
|
||||||
|
outcomes and submission intents are no longer uploaded to monthly archive
|
||||||
|
branches. The durable checkpoint keeps cumulative success/failure statistics;
|
||||||
|
the hot state keeps all pending outcomes and active intents, 300 compact recent
|
||||||
|
outcomes, 300 compact terminal intents, every ledger row belonging to an active
|
||||||
|
task plus 500 recent ledger rows, and 50 crash summaries. Failure evidence text,
|
||||||
|
signed log URLs, duplicate submission events, and resolved history outside
|
||||||
|
those windows are discarded after their category, compatibility block, memory
|
||||||
|
observation, or safe configuration vector has been extracted. Existing archive
|
||||||
|
branches are left untouched but are no longer read or updated.
|
||||||
|
|
||||||
## Deploy
|
## Deploy
|
||||||
|
|
||||||
Create a tag and submit the repository URL plus tag in "我的适配智能体".
|
Create a tag and submit the repository URL plus tag in "我的适配智能体".
|
||||||
|
|
||||||
```bash
|
```bash
|
||||||
git tag -a agent-v30 -m "ModelHub agent 2026.09.04.1"
|
git tag -a agent-v31 -m "ModelHub agent 2026.09.04.2"
|
||||||
git push origin main agent-v30
|
git push origin main agent-v31
|
||||||
```
|
```
|
||||||
|
|||||||
@@ -68,16 +68,20 @@ The most useful structured codes were `MODEL_NOT_SUPPORTED` (1,700),
|
|||||||
|
|
||||||
## Durable-state corrections
|
## Durable-state corrections
|
||||||
|
|
||||||
- Hot intent history retains unresolved intents plus 200 recent terminal
|
- Hot intent history retains unresolved intents plus 300 compact recent
|
||||||
intents; older terminal attempts are gzip archived by month.
|
terminal intents; older terminal attempts are discarded after extraction.
|
||||||
- Community raw samples are capped locally at 200 and excluded from the hot Git
|
- Community raw samples are capped locally at 200 and excluded from the hot Git
|
||||||
snapshot. The aggregated GPU/framework statistics remain durable.
|
snapshot. The aggregated GPU/framework statistics remain durable.
|
||||||
- Model/GPU official-capability cache is bounded to the 1,500 newest entries.
|
- Model/GPU official-capability cache is bounded to the 750 newest entries.
|
||||||
- Outcome compaction starts at 1,000 rows instead of 2,000.
|
- Outcome compaction starts at 500 rows and retains 300 compact recent samples.
|
||||||
- An unchanged snapshot produces no Git commit. A failed push retries the exact
|
- An unchanged snapshot produces no Git commit. A failed push retries the exact
|
||||||
same commit and generation instead of creating a new generation every minute.
|
same commit and generation instead of creating a new generation every minute.
|
||||||
- Submission intent batches default to 100, reducing Git transactions while
|
- Submission intent batches default to 100, reducing Git transactions while
|
||||||
preserving write-ahead recovery.
|
preserving write-ahead recovery.
|
||||||
|
|
||||||
These changes keep full forensic evidence in cold archive branches while making
|
After this audit was completed, version `2026.09.04.2` changed the ongoing
|
||||||
the hot branch small enough for quick restart and reliable server-side unpack.
|
retention model to `decision_state_only`. The extracted aggregate statistics,
|
||||||
|
compatibility rules, memory observations, active recovery state, and compact
|
||||||
|
recent samples remain durable, but new full outcome/intent archives and the
|
||||||
|
duplicated event stream are no longer produced. Previously created archive
|
||||||
|
branches remain untouched and are not needed during startup.
|
||||||
|
|||||||
@@ -19,7 +19,6 @@ from history_stats import (
|
|||||||
build_empty_pre_submit_report,
|
build_empty_pre_submit_report,
|
||||||
count_submissions_for_day,
|
count_submissions_for_day,
|
||||||
load_ledger,
|
load_ledger,
|
||||||
update_history_archive,
|
|
||||||
)
|
)
|
||||||
from market_intelligence import (
|
from market_intelligence import (
|
||||||
DEFAULT_FETCH_WORKERS,
|
DEFAULT_FETCH_WORKERS,
|
||||||
@@ -1142,7 +1141,6 @@ def run_submission(
|
|||||||
preflight_advisor.set_feedback_stats(None)
|
preflight_advisor.set_feedback_stats(None)
|
||||||
|
|
||||||
updated_after = determine_updated_after(args, now)
|
updated_after = determine_updated_after(args, now)
|
||||||
history_begin = now - timedelta(days=args.stats_window_days)
|
|
||||||
day_start = now.replace(hour=0, minute=0, second=0, microsecond=0)
|
day_start = now.replace(hour=0, minute=0, second=0, microsecond=0)
|
||||||
# Count today's submissions from the local ledger (avoids expensive paginated API call)
|
# Count today's submissions from the local ledger (avoids expensive paginated API call)
|
||||||
daily_snapshot = count_submissions_for_day(tasks=[], ledger_entries=ledger_entries, day_start=day_start, day_end=now)
|
daily_snapshot = count_submissions_for_day(tasks=[], ledger_entries=ledger_entries, day_start=day_start, day_end=now)
|
||||||
@@ -1186,7 +1184,7 @@ def run_submission(
|
|||||||
if platform_available_slots is not None:
|
if platform_available_slots is not None:
|
||||||
remaining_daily_quota = min(remaining_daily_quota, platform_available_slots)
|
remaining_daily_quota = min(remaining_daily_quota, platform_available_slots)
|
||||||
|
|
||||||
history_report_reason = "history_archive_skipped" if getattr(args, "skip_history_archive", False) else "history_archive_only_mode"
|
history_report_reason = "decision_state_only"
|
||||||
if args.daily_target > 0 and remaining_daily_quota <= 0:
|
if args.daily_target > 0 and remaining_daily_quota <= 0:
|
||||||
archived_history: list[dict[str, Any]] = []
|
archived_history: list[dict[str, Any]] = []
|
||||||
report = build_empty_pre_submit_report(
|
report = build_empty_pre_submit_report(
|
||||||
@@ -1322,20 +1320,7 @@ def run_submission(
|
|||||||
)
|
)
|
||||||
strategy_summary = strategy_manager.summary()
|
strategy_summary = strategy_manager.summary()
|
||||||
|
|
||||||
if getattr(args, "skip_history_archive", False):
|
archived_history = []
|
||||||
archived_history = []
|
|
||||||
else:
|
|
||||||
history_tasks = modelhub_client.list_tasks(
|
|
||||||
page_size=50,
|
|
||||||
only_mine=True,
|
|
||||||
begin_time=history_begin,
|
|
||||||
end_time=now,
|
|
||||||
)
|
|
||||||
archived_history = update_history_archive(
|
|
||||||
history_archive_path,
|
|
||||||
history_tasks,
|
|
||||||
limit=getattr(args, "history_archive_limit", 5000),
|
|
||||||
)
|
|
||||||
|
|
||||||
report = build_empty_pre_submit_report(
|
report = build_empty_pre_submit_report(
|
||||||
window_days=args.stats_window_days,
|
window_days=args.stats_window_days,
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ from common import parse_datetime, read_json, utc_now, write_json
|
|||||||
|
|
||||||
OFFICIAL_CAPABILITY_VERSION = 1
|
OFFICIAL_CAPABILITY_VERSION = 1
|
||||||
DEFAULT_OFFICIAL_CAPABILITIES_PATH = Path(".modelhub_state/official_capabilities.json")
|
DEFAULT_OFFICIAL_CAPABILITIES_PATH = Path(".modelhub_state/official_capabilities.json")
|
||||||
DEFAULT_MODEL_GPU_CACHE_LIMIT = 1500
|
DEFAULT_MODEL_GPU_CACHE_LIMIT = 750
|
||||||
|
|
||||||
|
|
||||||
class OfficialCapabilityUnavailable(RuntimeError):
|
class OfficialCapabilityUnavailable(RuntimeError):
|
||||||
|
|||||||
@@ -3,12 +3,10 @@ from __future__ import annotations
|
|||||||
from collections import defaultdict
|
from collections import defaultdict
|
||||||
from concurrent.futures import ThreadPoolExecutor, as_completed
|
from concurrent.futures import ThreadPoolExecutor, as_completed
|
||||||
from datetime import datetime, timedelta
|
from datetime import datetime, timedelta
|
||||||
import gzip
|
|
||||||
import json
|
import json
|
||||||
import os
|
import os
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from typing import Any
|
from typing import Any
|
||||||
import uuid
|
|
||||||
|
|
||||||
from architecture_compatibility import (
|
from architecture_compatibility import (
|
||||||
DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS,
|
DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS,
|
||||||
@@ -29,15 +27,47 @@ from task_registry import task_type_from_history_task
|
|||||||
DEFAULT_OUTCOMES_PATH = Path("outcomes/submissions.jsonl")
|
DEFAULT_OUTCOMES_PATH = Path("outcomes/submissions.jsonl")
|
||||||
DEFAULT_OUTCOME_CHECKPOINT_PATH = Path(".modelhub_state/outcome_checkpoint.json")
|
DEFAULT_OUTCOME_CHECKPOINT_PATH = Path(".modelhub_state/outcome_checkpoint.json")
|
||||||
DEFAULT_RECENT_OUTCOMES_PATH = Path(".modelhub_state/recent_outcomes.jsonl")
|
DEFAULT_RECENT_OUTCOMES_PATH = Path(".modelhub_state/recent_outcomes.jsonl")
|
||||||
DEFAULT_ARCHIVE_PENDING_DIR = Path(".modelhub_state/archive_pending/outcomes")
|
|
||||||
OUTCOME_CHECKPOINT_VERSION = 1
|
OUTCOME_CHECKPOINT_VERSION = 1
|
||||||
DEFAULT_OUTCOME_COMPACT_THRESHOLD = 1000
|
DEFAULT_OUTCOME_COMPACT_THRESHOLD = 500
|
||||||
DEFAULT_RECENT_OUTCOME_LIMIT = 1000
|
DEFAULT_RECENT_OUTCOME_LIMIT = 300
|
||||||
FAILURE_ENRICHMENT_LIMIT = 40
|
FAILURE_ENRICHMENT_LIMIT = 40
|
||||||
FAILURE_ENRICHMENT_WORKERS = 4
|
FAILURE_ENRICHMENT_WORKERS = 4
|
||||||
FAILURE_ENRICHMENT_MAX_ATTEMPTS = 3
|
FAILURE_ENRICHMENT_MAX_ATTEMPTS = 3
|
||||||
TERMINAL_TASK_STATUSES = {"success", "failed", "error", "cancelled", "completed"}
|
TERMINAL_TASK_STATUSES = {"success", "failed", "error", "cancelled", "completed"}
|
||||||
|
|
||||||
|
DECISION_RECORD_FIELDS = {
|
||||||
|
"taskId",
|
||||||
|
"modelId",
|
||||||
|
"targetGpu",
|
||||||
|
"framework",
|
||||||
|
"taskType",
|
||||||
|
"submitTime",
|
||||||
|
"lastSyncTime",
|
||||||
|
"status",
|
||||||
|
"verifyResult",
|
||||||
|
"outcome",
|
||||||
|
"failReason",
|
||||||
|
"modelProfile",
|
||||||
|
"failureCode",
|
||||||
|
"failureCategory",
|
||||||
|
"failureScope",
|
||||||
|
"failureAction",
|
||||||
|
"failureDeterministic",
|
||||||
|
"failureNeedsLlm",
|
||||||
|
"failureClassificationReason",
|
||||||
|
"failureObservedGpuMemoryGiB",
|
||||||
|
"failureUnsupportedArchitectures",
|
||||||
|
"failureUnsupportedModelTypes",
|
||||||
|
"failureDetectedFramework",
|
||||||
|
"failureEnrichmentAttempts",
|
||||||
|
"failureEnrichmentError",
|
||||||
|
"platformFailure",
|
||||||
|
"policyCancelled",
|
||||||
|
"policyCancellationReasons",
|
||||||
|
"policyCancelledAt",
|
||||||
|
"policyCancellationResolvedAsSuccess",
|
||||||
|
}
|
||||||
|
|
||||||
|
|
||||||
def _now_iso() -> str:
|
def _now_iso() -> str:
|
||||||
return utc_now().isoformat()
|
return utc_now().isoformat()
|
||||||
@@ -50,12 +80,10 @@ class OutcomeTracker:
|
|||||||
*,
|
*,
|
||||||
checkpoint_path: Path | str = DEFAULT_OUTCOME_CHECKPOINT_PATH,
|
checkpoint_path: Path | str = DEFAULT_OUTCOME_CHECKPOINT_PATH,
|
||||||
recent_path: Path | str = DEFAULT_RECENT_OUTCOMES_PATH,
|
recent_path: Path | str = DEFAULT_RECENT_OUTCOMES_PATH,
|
||||||
archive_pending_dir: Path | str = DEFAULT_ARCHIVE_PENDING_DIR,
|
|
||||||
) -> None:
|
) -> None:
|
||||||
self.path = Path(path)
|
self.path = Path(path)
|
||||||
self.checkpoint_path = Path(checkpoint_path)
|
self.checkpoint_path = Path(checkpoint_path)
|
||||||
self.recent_path = Path(recent_path)
|
self.recent_path = Path(recent_path)
|
||||||
self.archive_pending_dir = Path(archive_pending_dir)
|
|
||||||
self._records: list[dict[str, Any]] = []
|
self._records: list[dict[str, Any]] = []
|
||||||
self._recent_records: list[dict[str, Any]] = read_jsonl(self.recent_path)
|
self._recent_records: list[dict[str, Any]] = read_jsonl(self.recent_path)
|
||||||
self._checkpoint: dict[str, Any] = self._load_checkpoint()
|
self._checkpoint: dict[str, Any] = self._load_checkpoint()
|
||||||
@@ -65,6 +93,14 @@ class OutcomeTracker:
|
|||||||
self._failure_llm_classifier: LLMAssistedClassifier | None = None
|
self._failure_llm_classifier: LLMAssistedClassifier | None = None
|
||||||
|
|
||||||
self._records = read_jsonl(self.path)
|
self._records = read_jsonl(self.path)
|
||||||
|
compact_recent = sorted(
|
||||||
|
(self._decision_record(record) for record in self._recent_records),
|
||||||
|
key=_outcome_record_timestamp,
|
||||||
|
reverse=True,
|
||||||
|
)[: self._recent_limit()]
|
||||||
|
if compact_recent != self._recent_records:
|
||||||
|
self._recent_records = compact_recent
|
||||||
|
write_jsonl(self.recent_path, self._recent_records)
|
||||||
self._rebuild_indexes()
|
self._rebuild_indexes()
|
||||||
|
|
||||||
self._compact_if_needed(force=not bool(self._checkpoint) and len(self._records) > self._compact_threshold())
|
self._compact_if_needed(force=not bool(self._checkpoint) and len(self._records) > self._compact_threshold())
|
||||||
@@ -100,6 +136,21 @@ class OutcomeTracker:
|
|||||||
return {}
|
return {}
|
||||||
if not isinstance(payload, dict) or int(payload.get("version") or 0) != OUTCOME_CHECKPOINT_VERSION:
|
if not isinstance(payload, dict) or int(payload.get("version") or 0) != OUTCOME_CHECKPOINT_VERSION:
|
||||||
return {}
|
return {}
|
||||||
|
changed = False
|
||||||
|
if "archiveShards" in payload:
|
||||||
|
payload.pop("archiveShards", None)
|
||||||
|
changed = True
|
||||||
|
if "archivedRecords" in payload:
|
||||||
|
payload["summarizedRecords"] = max(
|
||||||
|
int(payload.get("summarizedRecords") or 0),
|
||||||
|
int(payload.pop("archivedRecords") or 0),
|
||||||
|
)
|
||||||
|
changed = True
|
||||||
|
if payload.get("storageMode") != "decision_state_only":
|
||||||
|
payload["storageMode"] = "decision_state_only"
|
||||||
|
changed = True
|
||||||
|
if changed:
|
||||||
|
write_json(self.checkpoint_path, payload)
|
||||||
return payload
|
return payload
|
||||||
|
|
||||||
@property
|
@property
|
||||||
@@ -117,28 +168,18 @@ class OutcomeTracker:
|
|||||||
return sorted(by_key.values(), key=_outcome_record_timestamp, reverse=True)[: self._recent_limit()]
|
return sorted(by_key.values(), key=_outcome_record_timestamp, reverse=True)[: self._recent_limit()]
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _archive_record(record: dict[str, Any]) -> dict[str, Any]:
|
def _decision_record(record: dict[str, Any]) -> dict[str, Any]:
|
||||||
return {
|
compact = {key: value for key, value in record.items() if key in DECISION_RECORD_FIELDS}
|
||||||
key: value
|
if (
|
||||||
for key, value in record.items()
|
record.get("outcome") == "failed"
|
||||||
if not any(
|
and not record.get("failureCategory")
|
||||||
marker in key.casefold()
|
and int(record.get("failureEnrichmentAttempts") or 0) < FAILURE_ENRICHMENT_MAX_ATTEMPTS
|
||||||
for marker in ("url", "token", "cookie", "authorization", "configparams", "response")
|
and record.get("logCosUrl")
|
||||||
)
|
):
|
||||||
}
|
# Retain a temporary signed log location only until bounded
|
||||||
|
# classification retries finish; never copy it to recent history.
|
||||||
def _write_archive_shard(self, records: list[dict[str, Any]]) -> str | None:
|
compact["logCosUrl"] = record.get("logCosUrl")
|
||||||
if not records:
|
return compact
|
||||||
return None
|
|
||||||
now = utc_now()
|
|
||||||
month = now.strftime("%Y-%m")
|
|
||||||
name = f"{now.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:10]}.jsonl.gz"
|
|
||||||
path = self.archive_pending_dir / month / name
|
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
with gzip.open(path, "wt", encoding="utf-8", compresslevel=6) as handle:
|
|
||||||
for record in records:
|
|
||||||
handle.write(json.dumps(self._archive_record(record), ensure_ascii=False, sort_keys=True) + "\n")
|
|
||||||
return f"{month}/{name}"
|
|
||||||
|
|
||||||
def _compact_if_needed(self, *, force: bool = False) -> bool:
|
def _compact_if_needed(self, *, force: bool = False) -> bool:
|
||||||
if not force and len(self._records) <= self._compact_threshold():
|
if not force and len(self._records) <= self._compact_threshold():
|
||||||
@@ -157,18 +198,14 @@ class OutcomeTracker:
|
|||||||
]
|
]
|
||||||
full_report = self.get_stats_report()
|
full_report = self.get_stats_report()
|
||||||
recent_records = self._combined_recent_terminal()
|
recent_records = self._combined_recent_terminal()
|
||||||
shard = self._write_archive_shard(removed)
|
summarized_total = max(0, int(full_report.get("totalRecords") or 0) - len(retained))
|
||||||
archived_total = max(0, int(full_report.get("totalRecords") or 0) - len(retained))
|
|
||||||
checkpoint_report = dict(full_report)
|
checkpoint_report = dict(full_report)
|
||||||
checkpoint_report.update(
|
checkpoint_report.update(
|
||||||
{
|
{
|
||||||
"totalRecords": archived_total,
|
"totalRecords": summarized_total,
|
||||||
"pendingRecords": 0,
|
"pendingRecords": 0,
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
previous_shards = list(self._checkpoint.get("archiveShards") or [])
|
|
||||||
if shard:
|
|
||||||
previous_shards.append(shard)
|
|
||||||
sync_times = [
|
sync_times = [
|
||||||
parse_datetime(record.get("lastSyncTime"))
|
parse_datetime(record.get("lastSyncTime"))
|
||||||
for record in [*removed, *retained]
|
for record in [*removed, *retained]
|
||||||
@@ -177,15 +214,15 @@ class OutcomeTracker:
|
|||||||
last_sync = max((value for value in sync_times if value is not None), default=None)
|
last_sync = max((value for value in sync_times if value is not None), default=None)
|
||||||
self._checkpoint = {
|
self._checkpoint = {
|
||||||
"version": OUTCOME_CHECKPOINT_VERSION,
|
"version": OUTCOME_CHECKPOINT_VERSION,
|
||||||
|
"storageMode": "decision_state_only",
|
||||||
"generatedAt": utc_now().isoformat(),
|
"generatedAt": utc_now().isoformat(),
|
||||||
"lastSyncTime": last_sync.isoformat() if last_sync else self._checkpoint.get("lastSyncTime"),
|
"lastSyncTime": last_sync.isoformat() if last_sync else self._checkpoint.get("lastSyncTime"),
|
||||||
"archivedRecords": archived_total,
|
"summarizedRecords": summarized_total,
|
||||||
"archiveShards": previous_shards[-200:],
|
|
||||||
"recentLimit": self._recent_limit(),
|
"recentLimit": self._recent_limit(),
|
||||||
"report": checkpoint_report,
|
"report": checkpoint_report,
|
||||||
}
|
}
|
||||||
self._records = retained
|
self._records = retained
|
||||||
self._recent_records = recent_records
|
self._recent_records = [self._decision_record(record) for record in recent_records]
|
||||||
write_json(self.checkpoint_path, self._checkpoint)
|
write_json(self.checkpoint_path, self._checkpoint)
|
||||||
write_jsonl(self.recent_path, self._recent_records)
|
write_jsonl(self.recent_path, self._recent_records)
|
||||||
write_jsonl(self.path, self._records)
|
write_jsonl(self.path, self._records)
|
||||||
@@ -265,7 +302,7 @@ class OutcomeTracker:
|
|||||||
# platform history API uses an inclusive time cursor, so the first page
|
# platform history API uses an inclusive time cursor, so the first page
|
||||||
# after a restart can contain terminal tasks already represented by the
|
# after a restart can contain terminal tasks already represented by the
|
||||||
# checkpoint. Remembering their task IDs prevents double-counting
|
# checkpoint. Remembering their task IDs prevents double-counting
|
||||||
# without loading the cold archive.
|
# without retaining full historical rows.
|
||||||
indexed_records = [*self._recent_records, *self._records]
|
indexed_records = [*self._recent_records, *self._records]
|
||||||
for record in indexed_records:
|
for record in indexed_records:
|
||||||
task_id = record.get("taskId")
|
task_id = record.get("taskId")
|
||||||
@@ -890,7 +927,11 @@ class OutcomeTracker:
|
|||||||
merged[field] = combined
|
merged[field] = combined
|
||||||
|
|
||||||
merged["totals"] = _merge_stat_items(baseline.get("totals"), report.get("totals"))
|
merged["totals"] = _merge_stat_items(baseline.get("totals"), report.get("totals"))
|
||||||
merged["totalRecords"] = int(self._checkpoint.get("archivedRecords") or 0) + len(self._records)
|
merged["totalRecords"] = int(
|
||||||
|
self._checkpoint.get("summarizedRecords")
|
||||||
|
or self._checkpoint.get("archivedRecords")
|
||||||
|
or 0
|
||||||
|
) + len(self._records)
|
||||||
merged["terminalRecords"] = int((merged.get("totals") or {}).get("total") or 0)
|
merged["terminalRecords"] = int((merged.get("totals") or {}).get("total") or 0)
|
||||||
merged["pendingRecords"] = sum(1 for record in self._records if record.get("outcome") == "pending")
|
merged["pendingRecords"] = sum(1 for record in self._records if record.get("outcome") == "pending")
|
||||||
merged["policyCancelledRecords"] = int(baseline.get("policyCancelledRecords") or 0) + sum(
|
merged["policyCancelledRecords"] = int(baseline.get("policyCancelledRecords") or 0) + sum(
|
||||||
@@ -960,7 +1001,12 @@ class OutcomeTracker:
|
|||||||
return merged
|
return merged
|
||||||
|
|
||||||
def save(self) -> None:
|
def save(self) -> None:
|
||||||
local_records = list(self._records)
|
local_records = [
|
||||||
|
self._decision_record(record)
|
||||||
|
if record.get("outcome") in {"success", "failed", "policy_cancelled"}
|
||||||
|
else record
|
||||||
|
for record in self._records
|
||||||
|
]
|
||||||
|
|
||||||
def merge(existing: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
def merge(existing: list[dict[str, Any]]) -> list[dict[str, Any]]:
|
||||||
merged = list(existing)
|
merged = list(existing)
|
||||||
@@ -973,7 +1019,12 @@ class OutcomeTracker:
|
|||||||
merged.append(record)
|
merged.append(record)
|
||||||
continue
|
continue
|
||||||
merged[existing_index] = _prefer_newer_outcome(merged[existing_index], record)
|
merged[existing_index] = _prefer_newer_outcome(merged[existing_index], record)
|
||||||
return merged
|
return [
|
||||||
|
self._decision_record(record)
|
||||||
|
if record.get("outcome") in {"success", "failed", "policy_cancelled"}
|
||||||
|
else record
|
||||||
|
for record in merged
|
||||||
|
]
|
||||||
|
|
||||||
self._records = update_jsonl(self.path, merge)
|
self._records = update_jsonl(self.path, merge)
|
||||||
self._rebuild_indexes()
|
self._rebuild_indexes()
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ from market_intelligence import (
|
|||||||
DEFAULT_THROUGHPUT_WINDOW_HOURS,
|
DEFAULT_THROUGHPUT_WINDOW_HOURS,
|
||||||
)
|
)
|
||||||
from modelhub_client import DEFAULT_CAPACITY_STATE_PATH, ModelHubClient, ModelHubClientPool
|
from modelhub_client import DEFAULT_CAPACITY_STATE_PATH, ModelHubClient, ModelHubClientPool
|
||||||
from outcome_tracker import DEFAULT_OUTCOMES_PATH, OutcomeTracker
|
from outcome_tracker import DEFAULT_OUTCOMES_PATH, DEFAULT_RECENT_OUTCOME_LIMIT, OutcomeTracker
|
||||||
from official_capabilities import DEFAULT_OFFICIAL_CAPABILITIES_PATH
|
from official_capabilities import DEFAULT_OFFICIAL_CAPABILITIES_PATH
|
||||||
from queue_cleanup import cleanup_certain_oom_tasks
|
from queue_cleanup import cleanup_certain_oom_tasks
|
||||||
from routing_engine import DEFAULT_ROUTING_STATE_PATH
|
from routing_engine import DEFAULT_ROUTING_STATE_PATH
|
||||||
@@ -33,6 +33,8 @@ from runner_common import DEFAULT_KEY_PATH, ensure_tokens
|
|||||||
from state_sync import (
|
from state_sync import (
|
||||||
DEFAULT_BATCH_SIZE,
|
DEFAULT_BATCH_SIZE,
|
||||||
DEFAULT_BRANCH,
|
DEFAULT_BRANCH,
|
||||||
|
DEFAULT_LEDGER_RECORDS,
|
||||||
|
DEFAULT_RECENT_TERMINAL_INTENTS,
|
||||||
DEFAULT_REMOTE,
|
DEFAULT_REMOTE,
|
||||||
StateGitSync,
|
StateGitSync,
|
||||||
load_state_git_credentials,
|
load_state_git_credentials,
|
||||||
@@ -571,6 +573,12 @@ def run_poll_loop(
|
|||||||
STATS_PRINT_INTERVAL = 10
|
STATS_PRINT_INTERVAL = 10
|
||||||
|
|
||||||
log(f"[poll] version={AGENT_VERSION} poll_run_dir={poll_run_dir}")
|
log(f"[poll] version={AGENT_VERSION} poll_run_dir={poll_run_dir}")
|
||||||
|
log(
|
||||||
|
"[state-retention] mode=decision_state_only full_archive=disabled "
|
||||||
|
f"recent_outcomes={DEFAULT_RECENT_OUTCOME_LIMIT} "
|
||||||
|
f"recent_intents={DEFAULT_RECENT_TERMINAL_INTENTS} "
|
||||||
|
f"ledger_recent={DEFAULT_LEDGER_RECORDS}"
|
||||||
|
)
|
||||||
log(
|
log(
|
||||||
f"[poll] target={base_args.daily_target} dry_run={str(bool(base_args.dry_run)).lower()} "
|
f"[poll] target={base_args.daily_target} dry_run={str(bool(base_args.dry_run)).lower()} "
|
||||||
f"poll_interval={base_args.poll_interval_seconds}s idle_interval={base_args.idle_interval_seconds}s"
|
f"poll_interval={base_args.poll_interval_seconds}s idle_interval={base_args.idle_interval_seconds}s"
|
||||||
|
|||||||
@@ -25,11 +25,24 @@ from version import AGENT_VERSION
|
|||||||
STATE_SCHEMA_VERSION = 1
|
STATE_SCHEMA_VERSION = 1
|
||||||
DEFAULT_REMOTE = "https://dev.modelhub.org.cn/CoolBoy/submmit.git"
|
DEFAULT_REMOTE = "https://dev.modelhub.org.cn/CoolBoy/submmit.git"
|
||||||
DEFAULT_BRANCH = "agent-state"
|
DEFAULT_BRANCH = "agent-state"
|
||||||
DEFAULT_ARCHIVE_BRANCH_PREFIX = "agent-archive"
|
|
||||||
DEFAULT_BATCH_SIZE = 100
|
DEFAULT_BATCH_SIZE = 100
|
||||||
DEFAULT_RETENTION_DAYS = 30
|
|
||||||
DEFAULT_HISTORY_DEPTH = 200
|
DEFAULT_HISTORY_DEPTH = 200
|
||||||
DEFAULT_RECENT_TERMINAL_INTENTS = 200
|
DEFAULT_RECENT_TERMINAL_INTENTS = 300
|
||||||
|
DEFAULT_LEDGER_RECORDS = 500
|
||||||
|
DEFAULT_CRASH_RECORDS = 50
|
||||||
|
|
||||||
|
STATE_OUTCOME_FIELDS = {
|
||||||
|
"taskId", "modelId", "targetGpu", "framework", "taskType", "submitTime",
|
||||||
|
"lastSyncTime", "status", "verifyResult", "outcome", "failReason",
|
||||||
|
"modelProfile", "failureCode", "failureCategory", "failureScope",
|
||||||
|
"failureAction", "failureDeterministic", "failureNeedsLlm",
|
||||||
|
"failureClassificationReason", "failureObservedGpuMemoryGiB",
|
||||||
|
"failureUnsupportedArchitectures", "failureUnsupportedModelTypes",
|
||||||
|
"failureDetectedFramework", "failureEnrichmentAttempts",
|
||||||
|
"failureEnrichmentError", "platformFailure", "policyCancelled",
|
||||||
|
"policyCancellationReasons", "policyCancelledAt",
|
||||||
|
"policyCancellationResolvedAsSuccess",
|
||||||
|
}
|
||||||
|
|
||||||
# Only these runtime files may cross the trust boundary into the state branch.
|
# Only these runtime files may cross the trust boundary into the state branch.
|
||||||
# Credentials, raw stdout, downloaded archives and run directories are excluded.
|
# Credentials, raw stdout, downloaded archives and run directories are excluded.
|
||||||
@@ -134,7 +147,6 @@ class StateGitSync:
|
|||||||
remote: str = DEFAULT_REMOTE,
|
remote: str = DEFAULT_REMOTE,
|
||||||
branch: str = DEFAULT_BRANCH,
|
branch: str = DEFAULT_BRANCH,
|
||||||
batch_size: int = DEFAULT_BATCH_SIZE,
|
batch_size: int = DEFAULT_BATCH_SIZE,
|
||||||
retention_days: int = DEFAULT_RETENTION_DAYS,
|
|
||||||
history_depth: int = DEFAULT_HISTORY_DEPTH,
|
history_depth: int = DEFAULT_HISTORY_DEPTH,
|
||||||
log_fn=None,
|
log_fn=None,
|
||||||
) -> None:
|
) -> None:
|
||||||
@@ -143,7 +155,6 @@ class StateGitSync:
|
|||||||
self.remote = remote
|
self.remote = remote
|
||||||
self.branch = branch
|
self.branch = branch
|
||||||
self.batch_size = max(1, min(100, int(batch_size)))
|
self.batch_size = max(1, min(100, int(batch_size)))
|
||||||
self.retention_days = max(1, int(retention_days))
|
|
||||||
self.history_depth = max(2, int(history_depth))
|
self.history_depth = max(2, int(history_depth))
|
||||||
self.log = log_fn or (lambda message: print(message, flush=True))
|
self.log = log_fn or (lambda message: print(message, flush=True))
|
||||||
self.writer_id = uuid.uuid4().hex
|
self.writer_id = uuid.uuid4().hex
|
||||||
@@ -219,97 +230,6 @@ class StateGitSync:
|
|||||||
oid = result.refs.get(f"refs/heads/{branch}".encode("utf-8"))
|
oid = result.refs.get(f"refs/heads/{branch}".encode("utf-8"))
|
||||||
return oid.decode("ascii") if oid else None
|
return oid.decode("ascii") if oid else None
|
||||||
|
|
||||||
def _sync_archive_pending(self) -> None:
|
|
||||||
pending_root = self.state_dir / "archive_pending"
|
|
||||||
files = sorted(path for path in pending_root.glob("*/*/*.jsonl.gz") if path.is_file())
|
|
||||||
if not files:
|
|
||||||
return
|
|
||||||
by_month: dict[str, list[Path]] = {}
|
|
||||||
for path in files:
|
|
||||||
by_month.setdefault(path.parent.name, []).append(path)
|
|
||||||
for month, month_files in sorted(by_month.items()):
|
|
||||||
branch = f"{DEFAULT_ARCHIVE_BRANCH_PREFIX}-{month}"
|
|
||||||
remote_oid = self._remote_oid_for(branch)
|
|
||||||
parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-archive-"))
|
|
||||||
workspace = parent / "archive"
|
|
||||||
try:
|
|
||||||
if remote_oid:
|
|
||||||
porcelain.clone(
|
|
||||||
self.remote,
|
|
||||||
workspace,
|
|
||||||
branch=branch,
|
|
||||||
depth=1,
|
|
||||||
checkout=True,
|
|
||||||
errstream=io.BytesIO(),
|
|
||||||
**self._auth_kwargs(),
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
workspace.mkdir(parents=True)
|
|
||||||
repo = porcelain.init(workspace)
|
|
||||||
repo.refs.set_symbolic_ref(b"HEAD", f"refs/heads/{branch}".encode("utf-8"))
|
|
||||||
repo = Repo(str(workspace))
|
|
||||||
manifest_path = workspace / "manifest.json"
|
|
||||||
try:
|
|
||||||
manifest = read_json(manifest_path)
|
|
||||||
except (FileNotFoundError, ValueError, TypeError):
|
|
||||||
manifest = {}
|
|
||||||
archived = manifest.get("files") if isinstance(manifest.get("files"), dict) else {}
|
|
||||||
for source in month_files:
|
|
||||||
archive_kind = source.parent.parent.name
|
|
||||||
destination = workspace / archive_kind / source.name
|
|
||||||
destination.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
shutil.copy2(source, destination)
|
|
||||||
archived[f"{archive_kind}/{source.name}"] = {
|
|
||||||
"sha256": _sha256_file(destination),
|
|
||||||
"bytes": destination.stat().st_size,
|
|
||||||
}
|
|
||||||
write_json(
|
|
||||||
manifest_path,
|
|
||||||
{
|
|
||||||
"schemaVersion": 1,
|
|
||||||
"month": month,
|
|
||||||
"updatedAt": _utc_now().isoformat(),
|
|
||||||
"files": archived,
|
|
||||||
},
|
|
||||||
)
|
|
||||||
manifest_path.with_name(f".{manifest_path.name}.lock").unlink(missing_ok=True)
|
|
||||||
porcelain.add(repo)
|
|
||||||
status = porcelain.status(repo)
|
|
||||||
if any(status.staged.get(kind) for kind in ("add", "delete", "modify")):
|
|
||||||
porcelain.commit(
|
|
||||||
repo,
|
|
||||||
message=f"archive: durable records {month}".encode("utf-8"),
|
|
||||||
author=self._author,
|
|
||||||
committer=self._author,
|
|
||||||
)
|
|
||||||
if self._remote_oid_for(branch) != remote_oid:
|
|
||||||
raise StateSyncError(f"archive branch {branch} changed remotely")
|
|
||||||
porcelain.push(
|
|
||||||
repo,
|
|
||||||
self.remote,
|
|
||||||
refspecs=f"HEAD:refs/heads/{branch}",
|
|
||||||
force=True,
|
|
||||||
outstream=io.BytesIO(),
|
|
||||||
errstream=io.BytesIO(),
|
|
||||||
**self._auth_kwargs(),
|
|
||||||
)
|
|
||||||
if self._remote_oid_for(branch) != repo.head().decode("ascii"):
|
|
||||||
raise StateSyncError(f"archive branch {branch} verification failed")
|
|
||||||
for source in month_files:
|
|
||||||
source.unlink(missing_ok=True)
|
|
||||||
self.log(
|
|
||||||
f"[archive-sync] branch={branch} shards={len(month_files)} "
|
|
||||||
f"files_total={len(archived)} status=ok"
|
|
||||||
)
|
|
||||||
finally:
|
|
||||||
shutil.rmtree(parent, ignore_errors=True)
|
|
||||||
|
|
||||||
def _sync_archive_pending_safely(self) -> None:
|
|
||||||
try:
|
|
||||||
self._sync_archive_pending()
|
|
||||||
except Exception as exc:
|
|
||||||
self.log(f"[archive-sync] status=deferred reason={_safe_text(exc)}")
|
|
||||||
|
|
||||||
def _create_workspace(self, remote_oid: str | None) -> None:
|
def _create_workspace(self, remote_oid: str | None) -> None:
|
||||||
parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-state-"))
|
parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-state-"))
|
||||||
workspace = parent / "state"
|
workspace = parent / "state"
|
||||||
@@ -383,7 +303,7 @@ class StateGitSync:
|
|||||||
for row in [*read_jsonl(source), *read_jsonl(destination)]:
|
for row in [*read_jsonl(source), *read_jsonl(destination)]:
|
||||||
key = json.dumps(row, ensure_ascii=False, sort_keys=True)
|
key = json.dumps(row, ensure_ascii=False, sort_keys=True)
|
||||||
merged[key] = row
|
merged[key] = row
|
||||||
rows = sorted(merged.values(), key=lambda row: str(row.get("at") or ""))[-200:]
|
rows = sorted(merged.values(), key=lambda row: str(row.get("at") or ""))[-DEFAULT_CRASH_RECORDS:]
|
||||||
write_jsonl(temporary, rows)
|
write_jsonl(temporary, rows)
|
||||||
else:
|
else:
|
||||||
shutil.copy2(source, temporary)
|
shutil.copy2(source, temporary)
|
||||||
@@ -409,30 +329,8 @@ class StateGitSync:
|
|||||||
self._expected_remote_oid = None
|
self._expected_remote_oid = None
|
||||||
return self.restore()
|
return self.restore()
|
||||||
|
|
||||||
def _archive_intents(self, records: list[dict[str, Any]]) -> str | None:
|
|
||||||
if not records:
|
|
||||||
return None
|
|
||||||
now = _utc_now()
|
|
||||||
month = now.strftime("%Y-%m")
|
|
||||||
name = f"{now.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:10]}.jsonl.gz"
|
|
||||||
path = self.state_dir / "archive_pending" / "attempts" / month / name
|
|
||||||
path.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
import gzip
|
|
||||||
|
|
||||||
with gzip.open(path, "wt", encoding="utf-8", compresslevel=6) as handle:
|
|
||||||
for record in records:
|
|
||||||
safe = {
|
|
||||||
key: value
|
|
||||||
for key, value in record.items()
|
|
||||||
if not any(marker in key.casefold() for marker in ("token", "password", "authorization"))
|
|
||||||
}
|
|
||||||
handle.write(json.dumps(safe, ensure_ascii=False, sort_keys=True) + "\n")
|
|
||||||
return f"{month}/{name}"
|
|
||||||
|
|
||||||
def _compact_intents(self) -> int:
|
def _compact_intents(self) -> int:
|
||||||
records = read_jsonl(self.intents_path)
|
records = read_jsonl(self.intents_path)
|
||||||
if len(records) <= DEFAULT_RECENT_TERMINAL_INTENTS:
|
|
||||||
return 0
|
|
||||||
active_statuses = {"pending", "submitted", "recovered_active"}
|
active_statuses = {"pending", "submitted", "recovered_active"}
|
||||||
active = [row for row in records if str(row.get("status") or "") in active_statuses]
|
active = [row for row in records if str(row.get("status") or "") in active_statuses]
|
||||||
terminal = [row for row in records if str(row.get("status") or "") not in active_statuses]
|
terminal = [row for row in records if str(row.get("status") or "") not in active_statuses]
|
||||||
@@ -441,45 +339,32 @@ class StateGitSync:
|
|||||||
reverse=True,
|
reverse=True,
|
||||||
)
|
)
|
||||||
retained_terminal = terminal[:DEFAULT_RECENT_TERMINAL_INTENTS]
|
retained_terminal = terminal[:DEFAULT_RECENT_TERMINAL_INTENTS]
|
||||||
archived = terminal[DEFAULT_RECENT_TERMINAL_INTENTS:]
|
discarded = terminal[DEFAULT_RECENT_TERMINAL_INTENTS:]
|
||||||
if not archived:
|
compact_fields = {
|
||||||
|
"intentId", "batchId", "status", "createdAt", "completedAt",
|
||||||
|
"repoId", "targetGpu", "taskType", "framework", "configSource",
|
||||||
|
"configFingerprint", "safeConfigVector", "taskId",
|
||||||
|
}
|
||||||
|
compacted = [
|
||||||
|
*active,
|
||||||
|
*(
|
||||||
|
{key: value for key, value in row.items() if key in compact_fields}
|
||||||
|
for row in retained_terminal
|
||||||
|
),
|
||||||
|
]
|
||||||
|
if compacted == records:
|
||||||
return 0
|
return 0
|
||||||
self._archive_intents(archived)
|
write_jsonl(self.intents_path, compacted)
|
||||||
write_jsonl(self.intents_path, [*active, *retained_terminal])
|
|
||||||
self.log(
|
self.log(
|
||||||
f"[state-compact] intents_archived={len(archived)} "
|
f"[state-compact] mode=decision_state_only intents_discarded={len(discarded)} "
|
||||||
f"active={len(active)} recent_terminal={len(retained_terminal)}"
|
f"active={len(active)} recent_terminal={len(retained_terminal)}"
|
||||||
)
|
)
|
||||||
return len(archived)
|
return len(discarded)
|
||||||
|
|
||||||
def _event_files(self) -> list[Path]:
|
def _event_files(self) -> list[Path]:
|
||||||
event_dir = self.state_dir / "events"
|
# The recovery intent WAL already records both transitions. Persisting
|
||||||
if not event_dir.exists():
|
# a second event stream doubled state without improving recovery.
|
||||||
return []
|
return []
|
||||||
cutoff = (_utc_now() - timedelta(days=self.retention_days)).date()
|
|
||||||
result: list[Path] = []
|
|
||||||
for path in sorted(event_dir.glob("*.jsonl")):
|
|
||||||
try:
|
|
||||||
event_day = datetime.strptime(path.stem, "%Y-%m-%d").date()
|
|
||||||
except ValueError:
|
|
||||||
continue
|
|
||||||
if event_day >= cutoff:
|
|
||||||
result.append(path)
|
|
||||||
else:
|
|
||||||
path.unlink(missing_ok=True)
|
|
||||||
return result
|
|
||||||
|
|
||||||
def _append_event(self, event: dict[str, Any]) -> None:
|
|
||||||
now = _utc_now()
|
|
||||||
path = self.state_dir / "events" / f"{now.date().isoformat()}.jsonl"
|
|
||||||
existing = read_jsonl(path)
|
|
||||||
sanitized = {key: value for key, value in event.items() if key not in {"configParams", "token", "password"}}
|
|
||||||
sanitized["at"] = sanitized.get("at") or now.isoformat()
|
|
||||||
sanitized["eventId"] = sanitized.get("eventId") or uuid.uuid4().hex
|
|
||||||
if "reason" in sanitized:
|
|
||||||
sanitized["reason"] = _safe_text(sanitized["reason"])
|
|
||||||
existing.append(sanitized)
|
|
||||||
write_jsonl(path, existing)
|
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _intent(candidate: dict[str, Any], batch_id: str) -> dict[str, Any]:
|
def _intent(candidate: dict[str, Any], batch_id: str) -> dict[str, Any]:
|
||||||
@@ -508,8 +393,6 @@ class StateGitSync:
|
|||||||
intents = [self._intent(candidate, batch_id) for candidate in candidates]
|
intents = [self._intent(candidate, batch_id) for candidate in candidates]
|
||||||
records.extend(intents)
|
records.extend(intents)
|
||||||
write_jsonl(self.intents_path, records)
|
write_jsonl(self.intents_path, records)
|
||||||
for intent in intents:
|
|
||||||
self._append_event({**intent, "event": "submission_intent"})
|
|
||||||
if not self.sync("intent"):
|
if not self.sync("intent"):
|
||||||
return None
|
return None
|
||||||
return batch_id
|
return batch_id
|
||||||
@@ -539,7 +422,6 @@ class StateGitSync:
|
|||||||
intent["completedAt"] = _utc_now().isoformat()
|
intent["completedAt"] = _utc_now().isoformat()
|
||||||
intent["taskId"] = result.get("taskId")
|
intent["taskId"] = result.get("taskId")
|
||||||
intent["reason"] = _safe_text(result.get("reason")) if result.get("reason") else None
|
intent["reason"] = _safe_text(result.get("reason")) if result.get("reason") else None
|
||||||
self._append_event({**intent, "event": "submission_result"})
|
|
||||||
write_jsonl(self.intents_path, records)
|
write_jsonl(self.intents_path, records)
|
||||||
return self.sync("result")
|
return self.sync("result")
|
||||||
|
|
||||||
@@ -634,27 +516,9 @@ class StateGitSync:
|
|||||||
intent["completedAt"] = now.isoformat()
|
intent["completedAt"] = now.isoformat()
|
||||||
elif intent.get("status") == "pending":
|
elif intent.get("status") == "pending":
|
||||||
unresolved += 1
|
unresolved += 1
|
||||||
retention_cutoff = now - timedelta(days=self.retention_days)
|
# Terminal intents are compacted by count and reduced to decision
|
||||||
retained: list[dict[str, Any]] = []
|
# fields during snapshot creation. No full historical rows are kept.
|
||||||
expired: list[dict[str, Any]] = []
|
write_jsonl(self.intents_path, intents)
|
||||||
for intent in intents:
|
|
||||||
completed_text = intent.get("completedAt")
|
|
||||||
if not completed_text:
|
|
||||||
retained.append(intent)
|
|
||||||
continue
|
|
||||||
try:
|
|
||||||
completed_at = datetime.fromisoformat(str(completed_text).replace("Z", "+00:00"))
|
|
||||||
except ValueError:
|
|
||||||
retained.append(intent)
|
|
||||||
continue
|
|
||||||
if completed_at.tzinfo is None:
|
|
||||||
completed_at = completed_at.replace(tzinfo=timezone.utc)
|
|
||||||
if completed_at >= retention_cutoff:
|
|
||||||
retained.append(intent)
|
|
||||||
else:
|
|
||||||
expired.append(intent)
|
|
||||||
self._archive_intents(expired)
|
|
||||||
write_jsonl(self.intents_path, retained)
|
|
||||||
self.record_active_tasks(enriched)
|
self.record_active_tasks(enriched)
|
||||||
return {"active": len(enriched), "reconciled": reconciled, "unresolved": unresolved}
|
return {"active": len(enriched), "reconciled": reconciled, "unresolved": unresolved}
|
||||||
|
|
||||||
@@ -694,13 +558,13 @@ class StateGitSync:
|
|||||||
elif relative == ".modelhub_state/official_capabilities.json":
|
elif relative == ".modelhub_state/official_capabilities.json":
|
||||||
payload = read_json(source)
|
payload = read_json(source)
|
||||||
cache = payload.get("modelGpuTaskTypes") if isinstance(payload, dict) else None
|
cache = payload.get("modelGpuTaskTypes") if isinstance(payload, dict) else None
|
||||||
if isinstance(cache, dict) and len(cache) > 1500:
|
if isinstance(cache, dict) and len(cache) > 750:
|
||||||
ordered = sorted(
|
ordered = sorted(
|
||||||
cache.items(),
|
cache.items(),
|
||||||
key=lambda pair: str((pair[1] or {}).get("updatedAt") or ""),
|
key=lambda pair: str((pair[1] or {}).get("updatedAt") or ""),
|
||||||
reverse=True,
|
reverse=True,
|
||||||
)
|
)
|
||||||
payload["modelGpuTaskTypes"] = dict(ordered[:1500])
|
payload["modelGpuTaskTypes"] = dict(ordered[:750])
|
||||||
write_json(destination, payload)
|
write_json(destination, payload)
|
||||||
elif relative in {"outcomes/submissions.jsonl", ".modelhub_state/recent_outcomes.jsonl"}:
|
elif relative in {"outcomes/submissions.jsonl", ".modelhub_state/recent_outcomes.jsonl"}:
|
||||||
sanitized_outcomes: list[dict[str, Any]] = []
|
sanitized_outcomes: list[dict[str, Any]] = []
|
||||||
@@ -709,13 +573,33 @@ class StateGitSync:
|
|||||||
{
|
{
|
||||||
key: value
|
key: value
|
||||||
for key, value in row.items()
|
for key, value in row.items()
|
||||||
if not any(
|
if key in STATE_OUTCOME_FIELDS
|
||||||
marker in key.casefold()
|
|
||||||
for marker in ("url", "token", "cookie", "authorization", "configparams")
|
|
||||||
)
|
|
||||||
}
|
}
|
||||||
)
|
)
|
||||||
|
if relative == ".modelhub_state/recent_outcomes.jsonl":
|
||||||
|
sanitized_outcomes = sanitized_outcomes[:300]
|
||||||
write_jsonl(destination, sanitized_outcomes)
|
write_jsonl(destination, sanitized_outcomes)
|
||||||
|
elif relative == "ledger/submissions.jsonl":
|
||||||
|
ledger_rows = read_jsonl(source)
|
||||||
|
active_ids = {
|
||||||
|
str(row.get("taskId"))
|
||||||
|
for row in read_jsonl(self.active_tasks_path)
|
||||||
|
if row.get("taskId") is not None
|
||||||
|
}
|
||||||
|
selected_by_task: dict[str, dict[str, Any]] = {}
|
||||||
|
anonymous: list[dict[str, Any]] = []
|
||||||
|
for row in [
|
||||||
|
*(item for item in ledger_rows if str(item.get("taskId") or "") in active_ids),
|
||||||
|
*ledger_rows[-DEFAULT_LEDGER_RECORDS:],
|
||||||
|
]:
|
||||||
|
task_id = str(row.get("taskId") or "")
|
||||||
|
if task_id:
|
||||||
|
selected_by_task[task_id] = row
|
||||||
|
else:
|
||||||
|
anonymous.append(row)
|
||||||
|
write_jsonl(destination, [*selected_by_task.values(), *anonymous[-20:]])
|
||||||
|
elif relative == ".modelhub_state/worker_crashes.jsonl":
|
||||||
|
write_jsonl(destination, read_jsonl(source)[-DEFAULT_CRASH_RECORDS:])
|
||||||
else:
|
else:
|
||||||
shutil.copy2(source, destination)
|
shutil.copy2(source, destination)
|
||||||
checksums[relative] = _sha256_file(destination)
|
checksums[relative] = _sha256_file(destination)
|
||||||
@@ -766,7 +650,6 @@ class StateGitSync:
|
|||||||
self.last_error = None
|
self.last_error = None
|
||||||
self.healthy = True
|
self.healthy = True
|
||||||
self.log(f"[state-sync] generation={self.generation} phase={phase} status=ok")
|
self.log(f"[state-sync] generation={self.generation} phase={phase} status=ok")
|
||||||
self._sync_archive_pending_safely()
|
|
||||||
return True
|
return True
|
||||||
|
|
||||||
def sync(self, phase: str) -> bool:
|
def sync(self, phase: str) -> bool:
|
||||||
@@ -790,7 +673,6 @@ class StateGitSync:
|
|||||||
):
|
):
|
||||||
self.healthy = True
|
self.healthy = True
|
||||||
self.last_error = None
|
self.last_error = None
|
||||||
self._sync_archive_pending_safely()
|
|
||||||
return True
|
return True
|
||||||
next_generation = self.generation + 1
|
next_generation = self.generation + 1
|
||||||
manifest = {
|
manifest = {
|
||||||
@@ -809,10 +691,6 @@ class StateGitSync:
|
|||||||
staged = any(status.staged.get(kind) for kind in ("add", "delete", "modify"))
|
staged = any(status.staged.get(kind) for kind in ("add", "delete", "modify"))
|
||||||
if not staged:
|
if not staged:
|
||||||
self.healthy = True
|
self.healthy = True
|
||||||
# A previous archive push may have been deferred while the
|
|
||||||
# hot snapshot was already current. Retry cold shards even
|
|
||||||
# when this cycle has no hot-state commit to publish.
|
|
||||||
self._sync_archive_pending_safely()
|
|
||||||
return True
|
return True
|
||||||
porcelain.commit(
|
porcelain.commit(
|
||||||
repo,
|
repo,
|
||||||
|
|||||||
@@ -1 +1 @@
|
|||||||
AGENT_VERSION = "2026.09.04.1"
|
AGENT_VERSION = "2026.09.04.2"
|
||||||
|
|||||||
@@ -1,7 +1,6 @@
|
|||||||
from __future__ import annotations
|
from __future__ import annotations
|
||||||
|
|
||||||
import os
|
import os
|
||||||
import gzip
|
|
||||||
import tempfile
|
import tempfile
|
||||||
import unittest
|
import unittest
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
@@ -17,7 +16,7 @@ MODULE_ROOT = ROOT / "modelhub_submmit_api"
|
|||||||
if str(MODULE_ROOT) not in sys.path:
|
if str(MODULE_ROOT) not in sys.path:
|
||||||
sys.path.insert(0, str(MODULE_ROOT))
|
sys.path.insert(0, str(MODULE_ROOT))
|
||||||
|
|
||||||
from common import read_jsonl, write_json, write_jsonl # noqa: E402
|
from common import read_json, read_jsonl, write_json, write_jsonl # noqa: E402
|
||||||
from config_optimizer import SafeConfigOptimizer # noqa: E402
|
from config_optimizer import SafeConfigOptimizer # noqa: E402
|
||||||
from hf_discovery import HuggingFaceDiscovery, parse_model_card_front_matter # noqa: E402
|
from hf_discovery import HuggingFaceDiscovery, parse_model_card_front_matter # noqa: E402
|
||||||
from official_capabilities import OfficialCapabilityRegistry # noqa: E402
|
from official_capabilities import OfficialCapabilityRegistry # noqa: E402
|
||||||
@@ -47,7 +46,7 @@ class OfficialClient:
|
|||||||
|
|
||||||
|
|
||||||
class SuperAgentTests(unittest.TestCase):
|
class SuperAgentTests(unittest.TestCase):
|
||||||
def test_outcome_history_compacts_to_checkpoint_recent_window_and_gzip_archive(self) -> None:
|
def test_outcome_history_compacts_to_decision_checkpoint_without_raw_archive(self) -> None:
|
||||||
with tempfile.TemporaryDirectory() as temporary_dir:
|
with tempfile.TemporaryDirectory() as temporary_dir:
|
||||||
root = Path(temporary_dir)
|
root = Path(temporary_dir)
|
||||||
outcomes = root / "outcomes.jsonl"
|
outcomes = root / "outcomes.jsonl"
|
||||||
@@ -82,7 +81,6 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
outcomes,
|
outcomes,
|
||||||
checkpoint_path=checkpoint,
|
checkpoint_path=checkpoint,
|
||||||
recent_path=recent,
|
recent_path=recent,
|
||||||
archive_pending_dir=archive,
|
|
||||||
)
|
)
|
||||||
self.assertTrue(tracker.has_durable_checkpoint)
|
self.assertTrue(tracker.has_durable_checkpoint)
|
||||||
self.assertEqual([], read_jsonl(outcomes))
|
self.assertEqual([], read_jsonl(outcomes))
|
||||||
@@ -90,19 +88,14 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
report = tracker.get_stats_report()
|
report = tracker.get_stats_report()
|
||||||
self.assertEqual(600, report["terminalRecords"])
|
self.assertEqual(600, report["terminalRecords"])
|
||||||
self.assertEqual(300, report["totals"]["successCount"])
|
self.assertEqual(300, report["totals"]["successCount"])
|
||||||
shard = next(archive.rglob("*.jsonl.gz"))
|
self.assertFalse(archive.exists())
|
||||||
import gzip
|
self.assertNotIn("logCosUrl", read_jsonl(recent)[0])
|
||||||
|
self.assertEqual("decision_state_only", read_json(checkpoint)["storageMode"])
|
||||||
with gzip.open(shard, "rt", encoding="utf-8") as handle:
|
|
||||||
archived_text = handle.read()
|
|
||||||
self.assertNotIn("logCosUrl", archived_text)
|
|
||||||
self.assertNotIn("token=hidden", archived_text)
|
|
||||||
|
|
||||||
restored = OutcomeTracker(
|
restored = OutcomeTracker(
|
||||||
outcomes,
|
outcomes,
|
||||||
checkpoint_path=checkpoint,
|
checkpoint_path=checkpoint,
|
||||||
recent_path=recent,
|
recent_path=recent,
|
||||||
archive_pending_dir=archive,
|
|
||||||
)
|
)
|
||||||
self.assertEqual(600, restored.get_stats_report()["terminalRecords"])
|
self.assertEqual(600, restored.get_stats_report()["terminalRecords"])
|
||||||
|
|
||||||
@@ -127,7 +120,6 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
outcomes,
|
outcomes,
|
||||||
checkpoint_path=checkpoint,
|
checkpoint_path=checkpoint,
|
||||||
recent_path=recent,
|
recent_path=recent,
|
||||||
archive_pending_dir=archive,
|
|
||||||
)
|
)
|
||||||
self.assertEqual(601, restarted.get_stats_report()["terminalRecords"])
|
self.assertEqual(601, restarted.get_stats_report()["terminalRecords"])
|
||||||
self.assertIn("599", restarted._by_task_id)
|
self.assertIn("599", restarted._by_task_id)
|
||||||
@@ -294,17 +286,6 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
restored_project / ".modelhub_state" / "worker_crashes.jsonl",
|
restored_project / ".modelhub_state" / "worker_crashes.jsonl",
|
||||||
[{"at": "2026-08-21T01:00:00+00:00", "exitCode": 1}],
|
[{"at": "2026-08-21T01:00:00+00:00", "exitCode": 1}],
|
||||||
)
|
)
|
||||||
pending_archive = (
|
|
||||||
project
|
|
||||||
/ ".modelhub_state"
|
|
||||||
/ "archive_pending"
|
|
||||||
/ "outcomes"
|
|
||||||
/ "2026-08"
|
|
||||||
/ "shard.jsonl.gz"
|
|
||||||
)
|
|
||||||
pending_archive.parent.mkdir(parents=True, exist_ok=True)
|
|
||||||
with gzip.open(pending_archive, "wt", encoding="utf-8") as handle:
|
|
||||||
handle.write('{"taskId":"archived"}\n')
|
|
||||||
credentials = {"username": "tester", "email": "tester@example.com", "password": "secret-value"}
|
credentials = {"username": "tester", "email": "tester@example.com", "password": "secret-value"}
|
||||||
manager = StateGitSync(
|
manager = StateGitSync(
|
||||||
project_root=project,
|
project_root=project,
|
||||||
@@ -332,9 +313,8 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
self.assertTrue(manager.sync("unchanged_cycle"))
|
self.assertTrue(manager.sync("unchanged_cycle"))
|
||||||
self.assertEqual(generation, manager.generation)
|
self.assertEqual(generation, manager.generation)
|
||||||
self.assertEqual(remote_head, manager._remote_oid())
|
self.assertEqual(remote_head, manager._remote_oid())
|
||||||
self.assertFalse(pending_archive.exists())
|
|
||||||
archive_refs = porcelain.ls_remote(str(remote)).refs
|
archive_refs = porcelain.ls_remote(str(remote)).refs
|
||||||
self.assertIn(b"refs/heads/agent-archive-2026-08", archive_refs)
|
self.assertNotIn(b"refs/heads/agent-archive-2026-08", archive_refs)
|
||||||
manager.close()
|
manager.close()
|
||||||
|
|
||||||
restored = StateGitSync(
|
restored = StateGitSync(
|
||||||
@@ -384,7 +364,7 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
self.assertEqual(pending_oid, manager._remote_oid())
|
self.assertEqual(pending_oid, manager._remote_oid())
|
||||||
manager.close()
|
manager.close()
|
||||||
|
|
||||||
def test_terminal_intents_are_bounded_and_archived(self) -> None:
|
def test_terminal_intents_are_bounded_and_reduced_to_decision_fields(self) -> None:
|
||||||
with tempfile.TemporaryDirectory() as temporary_dir:
|
with tempfile.TemporaryDirectory() as temporary_dir:
|
||||||
root = Path(temporary_dir)
|
root = Path(temporary_dir)
|
||||||
intent_path = root / ".modelhub_state" / "recovery_intents.jsonl"
|
intent_path = root / ".modelhub_state" / "recovery_intents.jsonl"
|
||||||
@@ -397,7 +377,7 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
"createdAt": f"2026-08-01T00:{index % 60:02d}:00+00:00",
|
"createdAt": f"2026-08-01T00:{index % 60:02d}:00+00:00",
|
||||||
"completedAt": f"2026-08-02T00:{index % 60:02d}:00+00:00",
|
"completedAt": f"2026-08-02T00:{index % 60:02d}:00+00:00",
|
||||||
}
|
}
|
||||||
for index in range(250)
|
for index in range(350)
|
||||||
]
|
]
|
||||||
+ [{"intentId": "pending", "status": "pending"}],
|
+ [{"intentId": "pending", "status": "pending"}],
|
||||||
)
|
)
|
||||||
@@ -409,11 +389,9 @@ class SuperAgentTests(unittest.TestCase):
|
|||||||
)
|
)
|
||||||
self.assertEqual(50, manager._compact_intents())
|
self.assertEqual(50, manager._compact_intents())
|
||||||
retained = read_jsonl(intent_path)
|
retained = read_jsonl(intent_path)
|
||||||
self.assertEqual(201, len(retained))
|
self.assertEqual(301, len(retained))
|
||||||
self.assertEqual(1, sum(row.get("status") == "pending" for row in retained))
|
self.assertEqual(1, sum(row.get("status") == "pending" for row in retained))
|
||||||
shard = next((root / ".modelhub_state" / "archive_pending" / "attempts").rglob("*.jsonl.gz"))
|
self.assertFalse((root / ".modelhub_state" / "archive_pending").exists())
|
||||||
with gzip.open(shard, "rt", encoding="utf-8") as handle:
|
|
||||||
self.assertEqual(50, len(handle.readlines()))
|
|
||||||
|
|
||||||
def test_failed_intent_push_returns_no_batch_id(self) -> None:
|
def test_failed_intent_push_returns_no_batch_id(self) -> None:
|
||||||
with tempfile.TemporaryDirectory() as temporary_dir:
|
with tempfile.TemporaryDirectory() as temporary_dir:
|
||||||
|
|||||||
Reference in New Issue
Block a user