refactor: retain decision state instead of full logs

This commit is contained in:
CoolBoy
2026-09-04 10:49:04 +08:00
parent ff73768537
commit d60551e130
9 changed files with 206 additions and 290 deletions

View File

@@ -413,11 +413,23 @@ success-based transformer selection. Missing framework/task/profile fields in hi
recovered from the durable task ledger. The evidence and category breakdown are recovered from the durable task ledger. The evidence and category breakdown are
documented in `docs/failure-analysis-2026-09-04.md`. documented in `docs/failure-analysis-2026-09-04.md`.
Version `2026.09.04.2` switches persistence from forensic-log retention to
decision-state retention after the historical audit was incorporated. New raw
outcomes and submission intents are no longer uploaded to monthly archive
branches. The durable checkpoint keeps cumulative success/failure statistics;
the hot state keeps all pending outcomes and active intents, 300 compact recent
outcomes, 300 compact terminal intents, every ledger row belonging to an active
task plus 500 recent ledger rows, and 50 crash summaries. Failure evidence text,
signed log URLs, duplicate submission events, and resolved history outside
those windows are discarded after their category, compatibility block, memory
observation, or safe configuration vector has been extracted. Existing archive
branches are left untouched but are no longer read or updated.
## Deploy ## Deploy
Create a tag and submit the repository URL plus tag in "我的适配智能体". Create a tag and submit the repository URL plus tag in "我的适配智能体".
```bash ```bash
git tag -a agent-v30 -m "ModelHub agent 2026.09.04.1" git tag -a agent-v31 -m "ModelHub agent 2026.09.04.2"
git push origin main agent-v30 git push origin main agent-v31
``` ```

View File

@@ -68,16 +68,20 @@ The most useful structured codes were `MODEL_NOT_SUPPORTED` (1,700),
## Durable-state corrections ## Durable-state corrections
- Hot intent history retains unresolved intents plus 200 recent terminal - Hot intent history retains unresolved intents plus 300 compact recent
intents; older terminal attempts are gzip archived by month. terminal intents; older terminal attempts are discarded after extraction.
- Community raw samples are capped locally at 200 and excluded from the hot Git - Community raw samples are capped locally at 200 and excluded from the hot Git
snapshot. The aggregated GPU/framework statistics remain durable. snapshot. The aggregated GPU/framework statistics remain durable.
- Model/GPU official-capability cache is bounded to the 1,500 newest entries. - Model/GPU official-capability cache is bounded to the 750 newest entries.
- Outcome compaction starts at 1,000 rows instead of 2,000. - Outcome compaction starts at 500 rows and retains 300 compact recent samples.
- An unchanged snapshot produces no Git commit. A failed push retries the exact - An unchanged snapshot produces no Git commit. A failed push retries the exact
same commit and generation instead of creating a new generation every minute. same commit and generation instead of creating a new generation every minute.
- Submission intent batches default to 100, reducing Git transactions while - Submission intent batches default to 100, reducing Git transactions while
preserving write-ahead recovery. preserving write-ahead recovery.
These changes keep full forensic evidence in cold archive branches while making After this audit was completed, version `2026.09.04.2` changed the ongoing
the hot branch small enough for quick restart and reliable server-side unpack. retention model to `decision_state_only`. The extracted aggregate statistics,
compatibility rules, memory observations, active recovery state, and compact
recent samples remain durable, but new full outcome/intent archives and the
duplicated event stream are no longer produced. Previously created archive
branches remain untouched and are not needed during startup.

View File

@@ -19,7 +19,6 @@ from history_stats import (
build_empty_pre_submit_report, build_empty_pre_submit_report,
count_submissions_for_day, count_submissions_for_day,
load_ledger, load_ledger,
update_history_archive,
) )
from market_intelligence import ( from market_intelligence import (
DEFAULT_FETCH_WORKERS, DEFAULT_FETCH_WORKERS,
@@ -1142,7 +1141,6 @@ def run_submission(
preflight_advisor.set_feedback_stats(None) preflight_advisor.set_feedback_stats(None)
updated_after = determine_updated_after(args, now) updated_after = determine_updated_after(args, now)
history_begin = now - timedelta(days=args.stats_window_days)
day_start = now.replace(hour=0, minute=0, second=0, microsecond=0) day_start = now.replace(hour=0, minute=0, second=0, microsecond=0)
# Count today's submissions from the local ledger (avoids expensive paginated API call) # Count today's submissions from the local ledger (avoids expensive paginated API call)
daily_snapshot = count_submissions_for_day(tasks=[], ledger_entries=ledger_entries, day_start=day_start, day_end=now) daily_snapshot = count_submissions_for_day(tasks=[], ledger_entries=ledger_entries, day_start=day_start, day_end=now)
@@ -1186,7 +1184,7 @@ def run_submission(
if platform_available_slots is not None: if platform_available_slots is not None:
remaining_daily_quota = min(remaining_daily_quota, platform_available_slots) remaining_daily_quota = min(remaining_daily_quota, platform_available_slots)
history_report_reason = "history_archive_skipped" if getattr(args, "skip_history_archive", False) else "history_archive_only_mode" history_report_reason = "decision_state_only"
if args.daily_target > 0 and remaining_daily_quota <= 0: if args.daily_target > 0 and remaining_daily_quota <= 0:
archived_history: list[dict[str, Any]] = [] archived_history: list[dict[str, Any]] = []
report = build_empty_pre_submit_report( report = build_empty_pre_submit_report(
@@ -1322,20 +1320,7 @@ def run_submission(
) )
strategy_summary = strategy_manager.summary() strategy_summary = strategy_manager.summary()
if getattr(args, "skip_history_archive", False): archived_history = []
archived_history = []
else:
history_tasks = modelhub_client.list_tasks(
page_size=50,
only_mine=True,
begin_time=history_begin,
end_time=now,
)
archived_history = update_history_archive(
history_archive_path,
history_tasks,
limit=getattr(args, "history_archive_limit", 5000),
)
report = build_empty_pre_submit_report( report = build_empty_pre_submit_report(
window_days=args.stats_window_days, window_days=args.stats_window_days,

View File

@@ -10,7 +10,7 @@ from common import parse_datetime, read_json, utc_now, write_json
OFFICIAL_CAPABILITY_VERSION = 1 OFFICIAL_CAPABILITY_VERSION = 1
DEFAULT_OFFICIAL_CAPABILITIES_PATH = Path(".modelhub_state/official_capabilities.json") DEFAULT_OFFICIAL_CAPABILITIES_PATH = Path(".modelhub_state/official_capabilities.json")
DEFAULT_MODEL_GPU_CACHE_LIMIT = 1500 DEFAULT_MODEL_GPU_CACHE_LIMIT = 750
class OfficialCapabilityUnavailable(RuntimeError): class OfficialCapabilityUnavailable(RuntimeError):

View File

@@ -3,12 +3,10 @@ from __future__ import annotations
from collections import defaultdict from collections import defaultdict
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta from datetime import datetime, timedelta
import gzip
import json import json
import os import os
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
import uuid
from architecture_compatibility import ( from architecture_compatibility import (
DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS, DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS,
@@ -29,15 +27,47 @@ from task_registry import task_type_from_history_task
DEFAULT_OUTCOMES_PATH = Path("outcomes/submissions.jsonl") DEFAULT_OUTCOMES_PATH = Path("outcomes/submissions.jsonl")
DEFAULT_OUTCOME_CHECKPOINT_PATH = Path(".modelhub_state/outcome_checkpoint.json") DEFAULT_OUTCOME_CHECKPOINT_PATH = Path(".modelhub_state/outcome_checkpoint.json")
DEFAULT_RECENT_OUTCOMES_PATH = Path(".modelhub_state/recent_outcomes.jsonl") DEFAULT_RECENT_OUTCOMES_PATH = Path(".modelhub_state/recent_outcomes.jsonl")
DEFAULT_ARCHIVE_PENDING_DIR = Path(".modelhub_state/archive_pending/outcomes")
OUTCOME_CHECKPOINT_VERSION = 1 OUTCOME_CHECKPOINT_VERSION = 1
DEFAULT_OUTCOME_COMPACT_THRESHOLD = 1000 DEFAULT_OUTCOME_COMPACT_THRESHOLD = 500
DEFAULT_RECENT_OUTCOME_LIMIT = 1000 DEFAULT_RECENT_OUTCOME_LIMIT = 300
FAILURE_ENRICHMENT_LIMIT = 40 FAILURE_ENRICHMENT_LIMIT = 40
FAILURE_ENRICHMENT_WORKERS = 4 FAILURE_ENRICHMENT_WORKERS = 4
FAILURE_ENRICHMENT_MAX_ATTEMPTS = 3 FAILURE_ENRICHMENT_MAX_ATTEMPTS = 3
TERMINAL_TASK_STATUSES = {"success", "failed", "error", "cancelled", "completed"} TERMINAL_TASK_STATUSES = {"success", "failed", "error", "cancelled", "completed"}
DECISION_RECORD_FIELDS = {
"taskId",
"modelId",
"targetGpu",
"framework",
"taskType",
"submitTime",
"lastSyncTime",
"status",
"verifyResult",
"outcome",
"failReason",
"modelProfile",
"failureCode",
"failureCategory",
"failureScope",
"failureAction",
"failureDeterministic",
"failureNeedsLlm",
"failureClassificationReason",
"failureObservedGpuMemoryGiB",
"failureUnsupportedArchitectures",
"failureUnsupportedModelTypes",
"failureDetectedFramework",
"failureEnrichmentAttempts",
"failureEnrichmentError",
"platformFailure",
"policyCancelled",
"policyCancellationReasons",
"policyCancelledAt",
"policyCancellationResolvedAsSuccess",
}
def _now_iso() -> str: def _now_iso() -> str:
return utc_now().isoformat() return utc_now().isoformat()
@@ -50,12 +80,10 @@ class OutcomeTracker:
*, *,
checkpoint_path: Path | str = DEFAULT_OUTCOME_CHECKPOINT_PATH, checkpoint_path: Path | str = DEFAULT_OUTCOME_CHECKPOINT_PATH,
recent_path: Path | str = DEFAULT_RECENT_OUTCOMES_PATH, recent_path: Path | str = DEFAULT_RECENT_OUTCOMES_PATH,
archive_pending_dir: Path | str = DEFAULT_ARCHIVE_PENDING_DIR,
) -> None: ) -> None:
self.path = Path(path) self.path = Path(path)
self.checkpoint_path = Path(checkpoint_path) self.checkpoint_path = Path(checkpoint_path)
self.recent_path = Path(recent_path) self.recent_path = Path(recent_path)
self.archive_pending_dir = Path(archive_pending_dir)
self._records: list[dict[str, Any]] = [] self._records: list[dict[str, Any]] = []
self._recent_records: list[dict[str, Any]] = read_jsonl(self.recent_path) self._recent_records: list[dict[str, Any]] = read_jsonl(self.recent_path)
self._checkpoint: dict[str, Any] = self._load_checkpoint() self._checkpoint: dict[str, Any] = self._load_checkpoint()
@@ -65,6 +93,14 @@ class OutcomeTracker:
self._failure_llm_classifier: LLMAssistedClassifier | None = None self._failure_llm_classifier: LLMAssistedClassifier | None = None
self._records = read_jsonl(self.path) self._records = read_jsonl(self.path)
compact_recent = sorted(
(self._decision_record(record) for record in self._recent_records),
key=_outcome_record_timestamp,
reverse=True,
)[: self._recent_limit()]
if compact_recent != self._recent_records:
self._recent_records = compact_recent
write_jsonl(self.recent_path, self._recent_records)
self._rebuild_indexes() self._rebuild_indexes()
self._compact_if_needed(force=not bool(self._checkpoint) and len(self._records) > self._compact_threshold()) self._compact_if_needed(force=not bool(self._checkpoint) and len(self._records) > self._compact_threshold())
@@ -100,6 +136,21 @@ class OutcomeTracker:
return {} return {}
if not isinstance(payload, dict) or int(payload.get("version") or 0) != OUTCOME_CHECKPOINT_VERSION: if not isinstance(payload, dict) or int(payload.get("version") or 0) != OUTCOME_CHECKPOINT_VERSION:
return {} return {}
changed = False
if "archiveShards" in payload:
payload.pop("archiveShards", None)
changed = True
if "archivedRecords" in payload:
payload["summarizedRecords"] = max(
int(payload.get("summarizedRecords") or 0),
int(payload.pop("archivedRecords") or 0),
)
changed = True
if payload.get("storageMode") != "decision_state_only":
payload["storageMode"] = "decision_state_only"
changed = True
if changed:
write_json(self.checkpoint_path, payload)
return payload return payload
@property @property
@@ -117,28 +168,18 @@ class OutcomeTracker:
return sorted(by_key.values(), key=_outcome_record_timestamp, reverse=True)[: self._recent_limit()] return sorted(by_key.values(), key=_outcome_record_timestamp, reverse=True)[: self._recent_limit()]
@staticmethod @staticmethod
def _archive_record(record: dict[str, Any]) -> dict[str, Any]: def _decision_record(record: dict[str, Any]) -> dict[str, Any]:
return { compact = {key: value for key, value in record.items() if key in DECISION_RECORD_FIELDS}
key: value if (
for key, value in record.items() record.get("outcome") == "failed"
if not any( and not record.get("failureCategory")
marker in key.casefold() and int(record.get("failureEnrichmentAttempts") or 0) < FAILURE_ENRICHMENT_MAX_ATTEMPTS
for marker in ("url", "token", "cookie", "authorization", "configparams", "response") and record.get("logCosUrl")
) ):
} # Retain a temporary signed log location only until bounded
# classification retries finish; never copy it to recent history.
def _write_archive_shard(self, records: list[dict[str, Any]]) -> str | None: compact["logCosUrl"] = record.get("logCosUrl")
if not records: return compact
return None
now = utc_now()
month = now.strftime("%Y-%m")
name = f"{now.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:10]}.jsonl.gz"
path = self.archive_pending_dir / month / name
path.parent.mkdir(parents=True, exist_ok=True)
with gzip.open(path, "wt", encoding="utf-8", compresslevel=6) as handle:
for record in records:
handle.write(json.dumps(self._archive_record(record), ensure_ascii=False, sort_keys=True) + "\n")
return f"{month}/{name}"
def _compact_if_needed(self, *, force: bool = False) -> bool: def _compact_if_needed(self, *, force: bool = False) -> bool:
if not force and len(self._records) <= self._compact_threshold(): if not force and len(self._records) <= self._compact_threshold():
@@ -157,18 +198,14 @@ class OutcomeTracker:
] ]
full_report = self.get_stats_report() full_report = self.get_stats_report()
recent_records = self._combined_recent_terminal() recent_records = self._combined_recent_terminal()
shard = self._write_archive_shard(removed) summarized_total = max(0, int(full_report.get("totalRecords") or 0) - len(retained))
archived_total = max(0, int(full_report.get("totalRecords") or 0) - len(retained))
checkpoint_report = dict(full_report) checkpoint_report = dict(full_report)
checkpoint_report.update( checkpoint_report.update(
{ {
"totalRecords": archived_total, "totalRecords": summarized_total,
"pendingRecords": 0, "pendingRecords": 0,
} }
) )
previous_shards = list(self._checkpoint.get("archiveShards") or [])
if shard:
previous_shards.append(shard)
sync_times = [ sync_times = [
parse_datetime(record.get("lastSyncTime")) parse_datetime(record.get("lastSyncTime"))
for record in [*removed, *retained] for record in [*removed, *retained]
@@ -177,15 +214,15 @@ class OutcomeTracker:
last_sync = max((value for value in sync_times if value is not None), default=None) last_sync = max((value for value in sync_times if value is not None), default=None)
self._checkpoint = { self._checkpoint = {
"version": OUTCOME_CHECKPOINT_VERSION, "version": OUTCOME_CHECKPOINT_VERSION,
"storageMode": "decision_state_only",
"generatedAt": utc_now().isoformat(), "generatedAt": utc_now().isoformat(),
"lastSyncTime": last_sync.isoformat() if last_sync else self._checkpoint.get("lastSyncTime"), "lastSyncTime": last_sync.isoformat() if last_sync else self._checkpoint.get("lastSyncTime"),
"archivedRecords": archived_total, "summarizedRecords": summarized_total,
"archiveShards": previous_shards[-200:],
"recentLimit": self._recent_limit(), "recentLimit": self._recent_limit(),
"report": checkpoint_report, "report": checkpoint_report,
} }
self._records = retained self._records = retained
self._recent_records = recent_records self._recent_records = [self._decision_record(record) for record in recent_records]
write_json(self.checkpoint_path, self._checkpoint) write_json(self.checkpoint_path, self._checkpoint)
write_jsonl(self.recent_path, self._recent_records) write_jsonl(self.recent_path, self._recent_records)
write_jsonl(self.path, self._records) write_jsonl(self.path, self._records)
@@ -265,7 +302,7 @@ class OutcomeTracker:
# platform history API uses an inclusive time cursor, so the first page # platform history API uses an inclusive time cursor, so the first page
# after a restart can contain terminal tasks already represented by the # after a restart can contain terminal tasks already represented by the
# checkpoint. Remembering their task IDs prevents double-counting # checkpoint. Remembering their task IDs prevents double-counting
# without loading the cold archive. # without retaining full historical rows.
indexed_records = [*self._recent_records, *self._records] indexed_records = [*self._recent_records, *self._records]
for record in indexed_records: for record in indexed_records:
task_id = record.get("taskId") task_id = record.get("taskId")
@@ -890,7 +927,11 @@ class OutcomeTracker:
merged[field] = combined merged[field] = combined
merged["totals"] = _merge_stat_items(baseline.get("totals"), report.get("totals")) merged["totals"] = _merge_stat_items(baseline.get("totals"), report.get("totals"))
merged["totalRecords"] = int(self._checkpoint.get("archivedRecords") or 0) + len(self._records) merged["totalRecords"] = int(
self._checkpoint.get("summarizedRecords")
or self._checkpoint.get("archivedRecords")
or 0
) + len(self._records)
merged["terminalRecords"] = int((merged.get("totals") or {}).get("total") or 0) merged["terminalRecords"] = int((merged.get("totals") or {}).get("total") or 0)
merged["pendingRecords"] = sum(1 for record in self._records if record.get("outcome") == "pending") merged["pendingRecords"] = sum(1 for record in self._records if record.get("outcome") == "pending")
merged["policyCancelledRecords"] = int(baseline.get("policyCancelledRecords") or 0) + sum( merged["policyCancelledRecords"] = int(baseline.get("policyCancelledRecords") or 0) + sum(
@@ -960,7 +1001,12 @@ class OutcomeTracker:
return merged return merged
def save(self) -> None: def save(self) -> None:
local_records = list(self._records) local_records = [
self._decision_record(record)
if record.get("outcome") in {"success", "failed", "policy_cancelled"}
else record
for record in self._records
]
def merge(existing: list[dict[str, Any]]) -> list[dict[str, Any]]: def merge(existing: list[dict[str, Any]]) -> list[dict[str, Any]]:
merged = list(existing) merged = list(existing)
@@ -973,7 +1019,12 @@ class OutcomeTracker:
merged.append(record) merged.append(record)
continue continue
merged[existing_index] = _prefer_newer_outcome(merged[existing_index], record) merged[existing_index] = _prefer_newer_outcome(merged[existing_index], record)
return merged return [
self._decision_record(record)
if record.get("outcome") in {"success", "failed", "policy_cancelled"}
else record
for record in merged
]
self._records = update_jsonl(self.path, merge) self._records = update_jsonl(self.path, merge)
self._rebuild_indexes() self._rebuild_indexes()

View File

@@ -25,7 +25,7 @@ from market_intelligence import (
DEFAULT_THROUGHPUT_WINDOW_HOURS, DEFAULT_THROUGHPUT_WINDOW_HOURS,
) )
from modelhub_client import DEFAULT_CAPACITY_STATE_PATH, ModelHubClient, ModelHubClientPool from modelhub_client import DEFAULT_CAPACITY_STATE_PATH, ModelHubClient, ModelHubClientPool
from outcome_tracker import DEFAULT_OUTCOMES_PATH, OutcomeTracker from outcome_tracker import DEFAULT_OUTCOMES_PATH, DEFAULT_RECENT_OUTCOME_LIMIT, OutcomeTracker
from official_capabilities import DEFAULT_OFFICIAL_CAPABILITIES_PATH from official_capabilities import DEFAULT_OFFICIAL_CAPABILITIES_PATH
from queue_cleanup import cleanup_certain_oom_tasks from queue_cleanup import cleanup_certain_oom_tasks
from routing_engine import DEFAULT_ROUTING_STATE_PATH from routing_engine import DEFAULT_ROUTING_STATE_PATH
@@ -33,6 +33,8 @@ from runner_common import DEFAULT_KEY_PATH, ensure_tokens
from state_sync import ( from state_sync import (
DEFAULT_BATCH_SIZE, DEFAULT_BATCH_SIZE,
DEFAULT_BRANCH, DEFAULT_BRANCH,
DEFAULT_LEDGER_RECORDS,
DEFAULT_RECENT_TERMINAL_INTENTS,
DEFAULT_REMOTE, DEFAULT_REMOTE,
StateGitSync, StateGitSync,
load_state_git_credentials, load_state_git_credentials,
@@ -571,6 +573,12 @@ def run_poll_loop(
STATS_PRINT_INTERVAL = 10 STATS_PRINT_INTERVAL = 10
log(f"[poll] version={AGENT_VERSION} poll_run_dir={poll_run_dir}") log(f"[poll] version={AGENT_VERSION} poll_run_dir={poll_run_dir}")
log(
"[state-retention] mode=decision_state_only full_archive=disabled "
f"recent_outcomes={DEFAULT_RECENT_OUTCOME_LIMIT} "
f"recent_intents={DEFAULT_RECENT_TERMINAL_INTENTS} "
f"ledger_recent={DEFAULT_LEDGER_RECORDS}"
)
log( log(
f"[poll] target={base_args.daily_target} dry_run={str(bool(base_args.dry_run)).lower()} " f"[poll] target={base_args.daily_target} dry_run={str(bool(base_args.dry_run)).lower()} "
f"poll_interval={base_args.poll_interval_seconds}s idle_interval={base_args.idle_interval_seconds}s" f"poll_interval={base_args.poll_interval_seconds}s idle_interval={base_args.idle_interval_seconds}s"

View File

@@ -25,11 +25,24 @@ from version import AGENT_VERSION
STATE_SCHEMA_VERSION = 1 STATE_SCHEMA_VERSION = 1
DEFAULT_REMOTE = "https://dev.modelhub.org.cn/CoolBoy/submmit.git" DEFAULT_REMOTE = "https://dev.modelhub.org.cn/CoolBoy/submmit.git"
DEFAULT_BRANCH = "agent-state" DEFAULT_BRANCH = "agent-state"
DEFAULT_ARCHIVE_BRANCH_PREFIX = "agent-archive"
DEFAULT_BATCH_SIZE = 100 DEFAULT_BATCH_SIZE = 100
DEFAULT_RETENTION_DAYS = 30
DEFAULT_HISTORY_DEPTH = 200 DEFAULT_HISTORY_DEPTH = 200
DEFAULT_RECENT_TERMINAL_INTENTS = 200 DEFAULT_RECENT_TERMINAL_INTENTS = 300
DEFAULT_LEDGER_RECORDS = 500
DEFAULT_CRASH_RECORDS = 50
STATE_OUTCOME_FIELDS = {
"taskId", "modelId", "targetGpu", "framework", "taskType", "submitTime",
"lastSyncTime", "status", "verifyResult", "outcome", "failReason",
"modelProfile", "failureCode", "failureCategory", "failureScope",
"failureAction", "failureDeterministic", "failureNeedsLlm",
"failureClassificationReason", "failureObservedGpuMemoryGiB",
"failureUnsupportedArchitectures", "failureUnsupportedModelTypes",
"failureDetectedFramework", "failureEnrichmentAttempts",
"failureEnrichmentError", "platformFailure", "policyCancelled",
"policyCancellationReasons", "policyCancelledAt",
"policyCancellationResolvedAsSuccess",
}
# Only these runtime files may cross the trust boundary into the state branch. # Only these runtime files may cross the trust boundary into the state branch.
# Credentials, raw stdout, downloaded archives and run directories are excluded. # Credentials, raw stdout, downloaded archives and run directories are excluded.
@@ -134,7 +147,6 @@ class StateGitSync:
remote: str = DEFAULT_REMOTE, remote: str = DEFAULT_REMOTE,
branch: str = DEFAULT_BRANCH, branch: str = DEFAULT_BRANCH,
batch_size: int = DEFAULT_BATCH_SIZE, batch_size: int = DEFAULT_BATCH_SIZE,
retention_days: int = DEFAULT_RETENTION_DAYS,
history_depth: int = DEFAULT_HISTORY_DEPTH, history_depth: int = DEFAULT_HISTORY_DEPTH,
log_fn=None, log_fn=None,
) -> None: ) -> None:
@@ -143,7 +155,6 @@ class StateGitSync:
self.remote = remote self.remote = remote
self.branch = branch self.branch = branch
self.batch_size = max(1, min(100, int(batch_size))) self.batch_size = max(1, min(100, int(batch_size)))
self.retention_days = max(1, int(retention_days))
self.history_depth = max(2, int(history_depth)) self.history_depth = max(2, int(history_depth))
self.log = log_fn or (lambda message: print(message, flush=True)) self.log = log_fn or (lambda message: print(message, flush=True))
self.writer_id = uuid.uuid4().hex self.writer_id = uuid.uuid4().hex
@@ -219,97 +230,6 @@ class StateGitSync:
oid = result.refs.get(f"refs/heads/{branch}".encode("utf-8")) oid = result.refs.get(f"refs/heads/{branch}".encode("utf-8"))
return oid.decode("ascii") if oid else None return oid.decode("ascii") if oid else None
def _sync_archive_pending(self) -> None:
pending_root = self.state_dir / "archive_pending"
files = sorted(path for path in pending_root.glob("*/*/*.jsonl.gz") if path.is_file())
if not files:
return
by_month: dict[str, list[Path]] = {}
for path in files:
by_month.setdefault(path.parent.name, []).append(path)
for month, month_files in sorted(by_month.items()):
branch = f"{DEFAULT_ARCHIVE_BRANCH_PREFIX}-{month}"
remote_oid = self._remote_oid_for(branch)
parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-archive-"))
workspace = parent / "archive"
try:
if remote_oid:
porcelain.clone(
self.remote,
workspace,
branch=branch,
depth=1,
checkout=True,
errstream=io.BytesIO(),
**self._auth_kwargs(),
)
else:
workspace.mkdir(parents=True)
repo = porcelain.init(workspace)
repo.refs.set_symbolic_ref(b"HEAD", f"refs/heads/{branch}".encode("utf-8"))
repo = Repo(str(workspace))
manifest_path = workspace / "manifest.json"
try:
manifest = read_json(manifest_path)
except (FileNotFoundError, ValueError, TypeError):
manifest = {}
archived = manifest.get("files") if isinstance(manifest.get("files"), dict) else {}
for source in month_files:
archive_kind = source.parent.parent.name
destination = workspace / archive_kind / source.name
destination.parent.mkdir(parents=True, exist_ok=True)
shutil.copy2(source, destination)
archived[f"{archive_kind}/{source.name}"] = {
"sha256": _sha256_file(destination),
"bytes": destination.stat().st_size,
}
write_json(
manifest_path,
{
"schemaVersion": 1,
"month": month,
"updatedAt": _utc_now().isoformat(),
"files": archived,
},
)
manifest_path.with_name(f".{manifest_path.name}.lock").unlink(missing_ok=True)
porcelain.add(repo)
status = porcelain.status(repo)
if any(status.staged.get(kind) for kind in ("add", "delete", "modify")):
porcelain.commit(
repo,
message=f"archive: durable records {month}".encode("utf-8"),
author=self._author,
committer=self._author,
)
if self._remote_oid_for(branch) != remote_oid:
raise StateSyncError(f"archive branch {branch} changed remotely")
porcelain.push(
repo,
self.remote,
refspecs=f"HEAD:refs/heads/{branch}",
force=True,
outstream=io.BytesIO(),
errstream=io.BytesIO(),
**self._auth_kwargs(),
)
if self._remote_oid_for(branch) != repo.head().decode("ascii"):
raise StateSyncError(f"archive branch {branch} verification failed")
for source in month_files:
source.unlink(missing_ok=True)
self.log(
f"[archive-sync] branch={branch} shards={len(month_files)} "
f"files_total={len(archived)} status=ok"
)
finally:
shutil.rmtree(parent, ignore_errors=True)
def _sync_archive_pending_safely(self) -> None:
try:
self._sync_archive_pending()
except Exception as exc:
self.log(f"[archive-sync] status=deferred reason={_safe_text(exc)}")
def _create_workspace(self, remote_oid: str | None) -> None: def _create_workspace(self, remote_oid: str | None) -> None:
parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-state-")) parent = Path(tempfile.mkdtemp(prefix="modelhub-agent-state-"))
workspace = parent / "state" workspace = parent / "state"
@@ -383,7 +303,7 @@ class StateGitSync:
for row in [*read_jsonl(source), *read_jsonl(destination)]: for row in [*read_jsonl(source), *read_jsonl(destination)]:
key = json.dumps(row, ensure_ascii=False, sort_keys=True) key = json.dumps(row, ensure_ascii=False, sort_keys=True)
merged[key] = row merged[key] = row
rows = sorted(merged.values(), key=lambda row: str(row.get("at") or ""))[-200:] rows = sorted(merged.values(), key=lambda row: str(row.get("at") or ""))[-DEFAULT_CRASH_RECORDS:]
write_jsonl(temporary, rows) write_jsonl(temporary, rows)
else: else:
shutil.copy2(source, temporary) shutil.copy2(source, temporary)
@@ -409,30 +329,8 @@ class StateGitSync:
self._expected_remote_oid = None self._expected_remote_oid = None
return self.restore() return self.restore()
def _archive_intents(self, records: list[dict[str, Any]]) -> str | None:
if not records:
return None
now = _utc_now()
month = now.strftime("%Y-%m")
name = f"{now.strftime('%Y%m%dT%H%M%SZ')}-{uuid.uuid4().hex[:10]}.jsonl.gz"
path = self.state_dir / "archive_pending" / "attempts" / month / name
path.parent.mkdir(parents=True, exist_ok=True)
import gzip
with gzip.open(path, "wt", encoding="utf-8", compresslevel=6) as handle:
for record in records:
safe = {
key: value
for key, value in record.items()
if not any(marker in key.casefold() for marker in ("token", "password", "authorization"))
}
handle.write(json.dumps(safe, ensure_ascii=False, sort_keys=True) + "\n")
return f"{month}/{name}"
def _compact_intents(self) -> int: def _compact_intents(self) -> int:
records = read_jsonl(self.intents_path) records = read_jsonl(self.intents_path)
if len(records) <= DEFAULT_RECENT_TERMINAL_INTENTS:
return 0
active_statuses = {"pending", "submitted", "recovered_active"} active_statuses = {"pending", "submitted", "recovered_active"}
active = [row for row in records if str(row.get("status") or "") in active_statuses] active = [row for row in records if str(row.get("status") or "") in active_statuses]
terminal = [row for row in records if str(row.get("status") or "") not in active_statuses] terminal = [row for row in records if str(row.get("status") or "") not in active_statuses]
@@ -441,45 +339,32 @@ class StateGitSync:
reverse=True, reverse=True,
) )
retained_terminal = terminal[:DEFAULT_RECENT_TERMINAL_INTENTS] retained_terminal = terminal[:DEFAULT_RECENT_TERMINAL_INTENTS]
archived = terminal[DEFAULT_RECENT_TERMINAL_INTENTS:] discarded = terminal[DEFAULT_RECENT_TERMINAL_INTENTS:]
if not archived: compact_fields = {
"intentId", "batchId", "status", "createdAt", "completedAt",
"repoId", "targetGpu", "taskType", "framework", "configSource",
"configFingerprint", "safeConfigVector", "taskId",
}
compacted = [
*active,
*(
{key: value for key, value in row.items() if key in compact_fields}
for row in retained_terminal
),
]
if compacted == records:
return 0 return 0
self._archive_intents(archived) write_jsonl(self.intents_path, compacted)
write_jsonl(self.intents_path, [*active, *retained_terminal])
self.log( self.log(
f"[state-compact] intents_archived={len(archived)} " f"[state-compact] mode=decision_state_only intents_discarded={len(discarded)} "
f"active={len(active)} recent_terminal={len(retained_terminal)}" f"active={len(active)} recent_terminal={len(retained_terminal)}"
) )
return len(archived) return len(discarded)
def _event_files(self) -> list[Path]: def _event_files(self) -> list[Path]:
event_dir = self.state_dir / "events" # The recovery intent WAL already records both transitions. Persisting
if not event_dir.exists(): # a second event stream doubled state without improving recovery.
return [] return []
cutoff = (_utc_now() - timedelta(days=self.retention_days)).date()
result: list[Path] = []
for path in sorted(event_dir.glob("*.jsonl")):
try:
event_day = datetime.strptime(path.stem, "%Y-%m-%d").date()
except ValueError:
continue
if event_day >= cutoff:
result.append(path)
else:
path.unlink(missing_ok=True)
return result
def _append_event(self, event: dict[str, Any]) -> None:
now = _utc_now()
path = self.state_dir / "events" / f"{now.date().isoformat()}.jsonl"
existing = read_jsonl(path)
sanitized = {key: value for key, value in event.items() if key not in {"configParams", "token", "password"}}
sanitized["at"] = sanitized.get("at") or now.isoformat()
sanitized["eventId"] = sanitized.get("eventId") or uuid.uuid4().hex
if "reason" in sanitized:
sanitized["reason"] = _safe_text(sanitized["reason"])
existing.append(sanitized)
write_jsonl(path, existing)
@staticmethod @staticmethod
def _intent(candidate: dict[str, Any], batch_id: str) -> dict[str, Any]: def _intent(candidate: dict[str, Any], batch_id: str) -> dict[str, Any]:
@@ -508,8 +393,6 @@ class StateGitSync:
intents = [self._intent(candidate, batch_id) for candidate in candidates] intents = [self._intent(candidate, batch_id) for candidate in candidates]
records.extend(intents) records.extend(intents)
write_jsonl(self.intents_path, records) write_jsonl(self.intents_path, records)
for intent in intents:
self._append_event({**intent, "event": "submission_intent"})
if not self.sync("intent"): if not self.sync("intent"):
return None return None
return batch_id return batch_id
@@ -539,7 +422,6 @@ class StateGitSync:
intent["completedAt"] = _utc_now().isoformat() intent["completedAt"] = _utc_now().isoformat()
intent["taskId"] = result.get("taskId") intent["taskId"] = result.get("taskId")
intent["reason"] = _safe_text(result.get("reason")) if result.get("reason") else None intent["reason"] = _safe_text(result.get("reason")) if result.get("reason") else None
self._append_event({**intent, "event": "submission_result"})
write_jsonl(self.intents_path, records) write_jsonl(self.intents_path, records)
return self.sync("result") return self.sync("result")
@@ -634,27 +516,9 @@ class StateGitSync:
intent["completedAt"] = now.isoformat() intent["completedAt"] = now.isoformat()
elif intent.get("status") == "pending": elif intent.get("status") == "pending":
unresolved += 1 unresolved += 1
retention_cutoff = now - timedelta(days=self.retention_days) # Terminal intents are compacted by count and reduced to decision
retained: list[dict[str, Any]] = [] # fields during snapshot creation. No full historical rows are kept.
expired: list[dict[str, Any]] = [] write_jsonl(self.intents_path, intents)
for intent in intents:
completed_text = intent.get("completedAt")
if not completed_text:
retained.append(intent)
continue
try:
completed_at = datetime.fromisoformat(str(completed_text).replace("Z", "+00:00"))
except ValueError:
retained.append(intent)
continue
if completed_at.tzinfo is None:
completed_at = completed_at.replace(tzinfo=timezone.utc)
if completed_at >= retention_cutoff:
retained.append(intent)
else:
expired.append(intent)
self._archive_intents(expired)
write_jsonl(self.intents_path, retained)
self.record_active_tasks(enriched) self.record_active_tasks(enriched)
return {"active": len(enriched), "reconciled": reconciled, "unresolved": unresolved} return {"active": len(enriched), "reconciled": reconciled, "unresolved": unresolved}
@@ -694,13 +558,13 @@ class StateGitSync:
elif relative == ".modelhub_state/official_capabilities.json": elif relative == ".modelhub_state/official_capabilities.json":
payload = read_json(source) payload = read_json(source)
cache = payload.get("modelGpuTaskTypes") if isinstance(payload, dict) else None cache = payload.get("modelGpuTaskTypes") if isinstance(payload, dict) else None
if isinstance(cache, dict) and len(cache) > 1500: if isinstance(cache, dict) and len(cache) > 750:
ordered = sorted( ordered = sorted(
cache.items(), cache.items(),
key=lambda pair: str((pair[1] or {}).get("updatedAt") or ""), key=lambda pair: str((pair[1] or {}).get("updatedAt") or ""),
reverse=True, reverse=True,
) )
payload["modelGpuTaskTypes"] = dict(ordered[:1500]) payload["modelGpuTaskTypes"] = dict(ordered[:750])
write_json(destination, payload) write_json(destination, payload)
elif relative in {"outcomes/submissions.jsonl", ".modelhub_state/recent_outcomes.jsonl"}: elif relative in {"outcomes/submissions.jsonl", ".modelhub_state/recent_outcomes.jsonl"}:
sanitized_outcomes: list[dict[str, Any]] = [] sanitized_outcomes: list[dict[str, Any]] = []
@@ -709,13 +573,33 @@ class StateGitSync:
{ {
key: value key: value
for key, value in row.items() for key, value in row.items()
if not any( if key in STATE_OUTCOME_FIELDS
marker in key.casefold()
for marker in ("url", "token", "cookie", "authorization", "configparams")
)
} }
) )
if relative == ".modelhub_state/recent_outcomes.jsonl":
sanitized_outcomes = sanitized_outcomes[:300]
write_jsonl(destination, sanitized_outcomes) write_jsonl(destination, sanitized_outcomes)
elif relative == "ledger/submissions.jsonl":
ledger_rows = read_jsonl(source)
active_ids = {
str(row.get("taskId"))
for row in read_jsonl(self.active_tasks_path)
if row.get("taskId") is not None
}
selected_by_task: dict[str, dict[str, Any]] = {}
anonymous: list[dict[str, Any]] = []
for row in [
*(item for item in ledger_rows if str(item.get("taskId") or "") in active_ids),
*ledger_rows[-DEFAULT_LEDGER_RECORDS:],
]:
task_id = str(row.get("taskId") or "")
if task_id:
selected_by_task[task_id] = row
else:
anonymous.append(row)
write_jsonl(destination, [*selected_by_task.values(), *anonymous[-20:]])
elif relative == ".modelhub_state/worker_crashes.jsonl":
write_jsonl(destination, read_jsonl(source)[-DEFAULT_CRASH_RECORDS:])
else: else:
shutil.copy2(source, destination) shutil.copy2(source, destination)
checksums[relative] = _sha256_file(destination) checksums[relative] = _sha256_file(destination)
@@ -766,7 +650,6 @@ class StateGitSync:
self.last_error = None self.last_error = None
self.healthy = True self.healthy = True
self.log(f"[state-sync] generation={self.generation} phase={phase} status=ok") self.log(f"[state-sync] generation={self.generation} phase={phase} status=ok")
self._sync_archive_pending_safely()
return True return True
def sync(self, phase: str) -> bool: def sync(self, phase: str) -> bool:
@@ -790,7 +673,6 @@ class StateGitSync:
): ):
self.healthy = True self.healthy = True
self.last_error = None self.last_error = None
self._sync_archive_pending_safely()
return True return True
next_generation = self.generation + 1 next_generation = self.generation + 1
manifest = { manifest = {
@@ -809,10 +691,6 @@ class StateGitSync:
staged = any(status.staged.get(kind) for kind in ("add", "delete", "modify")) staged = any(status.staged.get(kind) for kind in ("add", "delete", "modify"))
if not staged: if not staged:
self.healthy = True self.healthy = True
# A previous archive push may have been deferred while the
# hot snapshot was already current. Retry cold shards even
# when this cycle has no hot-state commit to publish.
self._sync_archive_pending_safely()
return True return True
porcelain.commit( porcelain.commit(
repo, repo,

View File

@@ -1 +1 @@
AGENT_VERSION = "2026.09.04.1" AGENT_VERSION = "2026.09.04.2"

View File

@@ -1,7 +1,6 @@
from __future__ import annotations from __future__ import annotations
import os import os
import gzip
import tempfile import tempfile
import unittest import unittest
from datetime import datetime, timezone from datetime import datetime, timezone
@@ -17,7 +16,7 @@ MODULE_ROOT = ROOT / "modelhub_submmit_api"
if str(MODULE_ROOT) not in sys.path: if str(MODULE_ROOT) not in sys.path:
sys.path.insert(0, str(MODULE_ROOT)) sys.path.insert(0, str(MODULE_ROOT))
from common import read_jsonl, write_json, write_jsonl # noqa: E402 from common import read_json, read_jsonl, write_json, write_jsonl # noqa: E402
from config_optimizer import SafeConfigOptimizer # noqa: E402 from config_optimizer import SafeConfigOptimizer # noqa: E402
from hf_discovery import HuggingFaceDiscovery, parse_model_card_front_matter # noqa: E402 from hf_discovery import HuggingFaceDiscovery, parse_model_card_front_matter # noqa: E402
from official_capabilities import OfficialCapabilityRegistry # noqa: E402 from official_capabilities import OfficialCapabilityRegistry # noqa: E402
@@ -47,7 +46,7 @@ class OfficialClient:
class SuperAgentTests(unittest.TestCase): class SuperAgentTests(unittest.TestCase):
def test_outcome_history_compacts_to_checkpoint_recent_window_and_gzip_archive(self) -> None: def test_outcome_history_compacts_to_decision_checkpoint_without_raw_archive(self) -> None:
with tempfile.TemporaryDirectory() as temporary_dir: with tempfile.TemporaryDirectory() as temporary_dir:
root = Path(temporary_dir) root = Path(temporary_dir)
outcomes = root / "outcomes.jsonl" outcomes = root / "outcomes.jsonl"
@@ -82,7 +81,6 @@ class SuperAgentTests(unittest.TestCase):
outcomes, outcomes,
checkpoint_path=checkpoint, checkpoint_path=checkpoint,
recent_path=recent, recent_path=recent,
archive_pending_dir=archive,
) )
self.assertTrue(tracker.has_durable_checkpoint) self.assertTrue(tracker.has_durable_checkpoint)
self.assertEqual([], read_jsonl(outcomes)) self.assertEqual([], read_jsonl(outcomes))
@@ -90,19 +88,14 @@ class SuperAgentTests(unittest.TestCase):
report = tracker.get_stats_report() report = tracker.get_stats_report()
self.assertEqual(600, report["terminalRecords"]) self.assertEqual(600, report["terminalRecords"])
self.assertEqual(300, report["totals"]["successCount"]) self.assertEqual(300, report["totals"]["successCount"])
shard = next(archive.rglob("*.jsonl.gz")) self.assertFalse(archive.exists())
import gzip self.assertNotIn("logCosUrl", read_jsonl(recent)[0])
self.assertEqual("decision_state_only", read_json(checkpoint)["storageMode"])
with gzip.open(shard, "rt", encoding="utf-8") as handle:
archived_text = handle.read()
self.assertNotIn("logCosUrl", archived_text)
self.assertNotIn("token=hidden", archived_text)
restored = OutcomeTracker( restored = OutcomeTracker(
outcomes, outcomes,
checkpoint_path=checkpoint, checkpoint_path=checkpoint,
recent_path=recent, recent_path=recent,
archive_pending_dir=archive,
) )
self.assertEqual(600, restored.get_stats_report()["terminalRecords"]) self.assertEqual(600, restored.get_stats_report()["terminalRecords"])
@@ -127,7 +120,6 @@ class SuperAgentTests(unittest.TestCase):
outcomes, outcomes,
checkpoint_path=checkpoint, checkpoint_path=checkpoint,
recent_path=recent, recent_path=recent,
archive_pending_dir=archive,
) )
self.assertEqual(601, restarted.get_stats_report()["terminalRecords"]) self.assertEqual(601, restarted.get_stats_report()["terminalRecords"])
self.assertIn("599", restarted._by_task_id) self.assertIn("599", restarted._by_task_id)
@@ -294,17 +286,6 @@ class SuperAgentTests(unittest.TestCase):
restored_project / ".modelhub_state" / "worker_crashes.jsonl", restored_project / ".modelhub_state" / "worker_crashes.jsonl",
[{"at": "2026-08-21T01:00:00+00:00", "exitCode": 1}], [{"at": "2026-08-21T01:00:00+00:00", "exitCode": 1}],
) )
pending_archive = (
project
/ ".modelhub_state"
/ "archive_pending"
/ "outcomes"
/ "2026-08"
/ "shard.jsonl.gz"
)
pending_archive.parent.mkdir(parents=True, exist_ok=True)
with gzip.open(pending_archive, "wt", encoding="utf-8") as handle:
handle.write('{"taskId":"archived"}\n')
credentials = {"username": "tester", "email": "tester@example.com", "password": "secret-value"} credentials = {"username": "tester", "email": "tester@example.com", "password": "secret-value"}
manager = StateGitSync( manager = StateGitSync(
project_root=project, project_root=project,
@@ -332,9 +313,8 @@ class SuperAgentTests(unittest.TestCase):
self.assertTrue(manager.sync("unchanged_cycle")) self.assertTrue(manager.sync("unchanged_cycle"))
self.assertEqual(generation, manager.generation) self.assertEqual(generation, manager.generation)
self.assertEqual(remote_head, manager._remote_oid()) self.assertEqual(remote_head, manager._remote_oid())
self.assertFalse(pending_archive.exists())
archive_refs = porcelain.ls_remote(str(remote)).refs archive_refs = porcelain.ls_remote(str(remote)).refs
self.assertIn(b"refs/heads/agent-archive-2026-08", archive_refs) self.assertNotIn(b"refs/heads/agent-archive-2026-08", archive_refs)
manager.close() manager.close()
restored = StateGitSync( restored = StateGitSync(
@@ -384,7 +364,7 @@ class SuperAgentTests(unittest.TestCase):
self.assertEqual(pending_oid, manager._remote_oid()) self.assertEqual(pending_oid, manager._remote_oid())
manager.close() manager.close()
def test_terminal_intents_are_bounded_and_archived(self) -> None: def test_terminal_intents_are_bounded_and_reduced_to_decision_fields(self) -> None:
with tempfile.TemporaryDirectory() as temporary_dir: with tempfile.TemporaryDirectory() as temporary_dir:
root = Path(temporary_dir) root = Path(temporary_dir)
intent_path = root / ".modelhub_state" / "recovery_intents.jsonl" intent_path = root / ".modelhub_state" / "recovery_intents.jsonl"
@@ -397,7 +377,7 @@ class SuperAgentTests(unittest.TestCase):
"createdAt": f"2026-08-01T00:{index % 60:02d}:00+00:00", "createdAt": f"2026-08-01T00:{index % 60:02d}:00+00:00",
"completedAt": f"2026-08-02T00:{index % 60:02d}:00+00:00", "completedAt": f"2026-08-02T00:{index % 60:02d}:00+00:00",
} }
for index in range(250) for index in range(350)
] ]
+ [{"intentId": "pending", "status": "pending"}], + [{"intentId": "pending", "status": "pending"}],
) )
@@ -409,11 +389,9 @@ class SuperAgentTests(unittest.TestCase):
) )
self.assertEqual(50, manager._compact_intents()) self.assertEqual(50, manager._compact_intents())
retained = read_jsonl(intent_path) retained = read_jsonl(intent_path)
self.assertEqual(201, len(retained)) self.assertEqual(301, len(retained))
self.assertEqual(1, sum(row.get("status") == "pending" for row in retained)) self.assertEqual(1, sum(row.get("status") == "pending" for row in retained))
shard = next((root / ".modelhub_state" / "archive_pending" / "attempts").rglob("*.jsonl.gz")) self.assertFalse((root / ".modelhub_state" / "archive_pending").exists())
with gzip.open(shard, "rt", encoding="utf-8") as handle:
self.assertEqual(50, len(handle.readlines()))
def test_failed_intent_push_returns_no_batch_id(self) -> None: def test_failed_intent_push_returns_no_batch_id(self) -> None:
with tempfile.TemporaryDirectory() as temporary_dir: with tempfile.TemporaryDirectory() as temporary_dir: