feat: learn framework architecture incompatibilities

This commit is contained in:
CoolBoy
2026-08-12 08:02:51 +08:00
parent d908706f9a
commit 615bcad124
10 changed files with 503 additions and 9 deletions

View File

@@ -110,6 +110,19 @@ idle card does not permanently poison otherwise successful evidence.
Candidate shortages expand the model search window; they never unlock an Candidate shortages expand the model search window; they never unlock an
unvetted GPU or framework. unvetted GPU or framework.
Failed-task archives are also classified conservatively. When ModelHub explicitly
says that the selected framework does not support the model or architecture, the
runner learns an exact GPU + framework + task type + architecture block from the
candidate repository's `config.json`. Repository names are never used as
architecture evidence. Exact `architectures` values take priority and
`model_type` is used only when `architectures` is absent; missing metadata does
not create a block. Generic unsupported operators, attention backends, GPU types,
quantization failures, and OOMs cannot enter this blacklist. A newer success for
the same exact combination clears the block, and otherwise it expires after 30
days. Set `MODELHUB_ARCHITECTURE_BLOCK_TTL_DAYS` to a value from 1 to 365 to
change that window. The stats report exposes `architectureCompatibilityBlocks`
and per-GPU/framework block counts.
Before a candidate reaches the submit queue, failure-informed preflight checks Before a candidate reaches the submit queue, failure-informed preflight checks
the actual ModelScope repository structure and file sizes. Non-GGUF text the actual ModelScope repository structure and file sizes. Non-GGUF text
frameworks require root-level config, weights, and tokenizer assets. The memory frameworks require root-level config, weights, and tokenizer assets. The memory
@@ -253,12 +266,16 @@ and startup cleanup, then limit minus 5 for scheduled dynamic cleanup.
Version `2026.08.12.1` protects running tasks from age-based cleanup and records Version `2026.08.12.1` protects running tasks from age-based cleanup and records
worker-initiated stops as `policy_cancelled`, excluding them from success-rate, worker-initiated stops as `policy_cancelled`, excluding them from success-rate,
failure-cooldown, and circuit-breaker evidence. failure-cooldown, and circuit-breaker evidence.
Version `2026.08.12.2` learns conservative, expiring GPU/framework/architecture
compatibility blocks only from explicit ModelHub failure text, matches candidate
`config.json` metadata instead of repository names, and lets newer success
evidence clear stale blocks.
## Deploy ## Deploy
Create a tag and submit the repository URL plus tag in "我的适配智能体". Create a tag and submit the repository URL plus tag in "我的适配智能体".
```bash ```bash
git tag agent-v19 git tag agent-v20
git push origin agent-v19 git push origin agent-v20
``` ```

View File

@@ -23,6 +23,7 @@ It currently supports:
- `history_stats.py`: online history aggregation, ranking, and warnings - `history_stats.py`: online history aggregation, ranking, and warnings
- `candidate_preflight.py`: deterministic repository, memory, context, and compatibility gates - `candidate_preflight.py`: deterministic repository, memory, context, and compatibility gates
- `failure_taxonomy.py`: deterministic/platform/semantic failure routing - `failure_taxonomy.py`: deterministic/platform/semantic failure routing
- `architecture_compatibility.py`: exact architecture identities and learned compatibility keys
- `llm_classifier.py`: offline-only experimental ambiguity-analysis helper - `llm_classifier.py`: offline-only experimental ambiguity-analysis helper
- `template_selector.py`: template lookup and GPU normalization - `template_selector.py`: template lookup and GPU normalization
- `task_registry.py`: task-type and framework selection rules - `task_registry.py`: task-type and framework selection rules
@@ -93,6 +94,11 @@ bash run_poll.sh --dry-run
validation and a confidence score at least 10% above the best incumbent. validation and a confidence score at least 10% above the best incumbent.
- Five consecutive local failures pause a GPU/framework pair for 12 hours; a - Five consecutive local failures pause a GPU/framework pair for 12 hours; a
sub-20% rate over the latest 20 terminal tasks pauses it for 6 hours. sub-20% rate over the latest 20 terminal tasks pauses it for 6 hours.
- An explicit "framework does not support this model/architecture" failure learns
a 30-day GPU + framework + task + architecture block. Architecture identity
comes from candidate `config.json` (`architectures`, with `model_type` only as
fallback), never from repository names. A newer success clears the block, and
generic unsupported backend/operator messages cannot create one.
- A strategy generation lasts exactly 200 platform-accepted submissions. Rejected API calls and - A strategy generation lasts exactly 200 platform-accepted submissions. Rejected API calls and
duplicates do not advance it. The next cycle refreshes platform history before submitting again. duplicates do not advance it. The next cycle refreshes platform history before submitting again.
- Strategy state is stored in `.modelhub_state/gpu_strategy.json`; a generation never recalculates - Strategy state is stored in `.modelhub_state/gpu_strategy.json`; a generation never recalculates

View File

@@ -0,0 +1,65 @@
from __future__ import annotations
from collections.abc import Iterable
from typing import Any
EXPLICIT_ARCHITECTURE_FAILURE_CATEGORY = "framework_architecture_unsupported"
EXPLICIT_ARCHITECTURE_FAILURE_ACTION = "block_gpu_framework_architecture"
EXPLICIT_ARCHITECTURE_FAILURE_REASON = "explicit_framework_model_unsupported"
DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS = 30
def architecture_profile(
model_type: Any,
architectures: Any,
) -> dict[str, Any] | None:
"""Build a stable, conservative architecture identity for feedback matching."""
normalized_architectures = _normalize_architectures(architectures)
normalized_model_type = _normalize(model_type)
if normalized_architectures:
return {
"matchType": "architectures",
"signature": "architectures:" + ",".join(normalized_architectures),
"architectures": normalized_architectures,
"modelType": normalized_model_type or None,
}
if normalized_model_type:
return {
"matchType": "model_type",
"signature": f"model_type:{normalized_model_type}",
"architectures": [],
"modelType": normalized_model_type,
}
return None
def architecture_compatibility_key(
target_gpu: Any,
framework: Any,
task_type: Any,
signature: Any,
) -> str | None:
parts = (
_normalize(target_gpu),
_normalize(framework),
_normalize(task_type),
_normalize(signature),
)
if not all(parts):
return None
return "|".join(parts)
def _normalize_architectures(value: Any) -> list[str]:
if isinstance(value, str):
values: Iterable[Any] = [value]
elif isinstance(value, (list, tuple, set)):
values = value
else:
values = []
return sorted({_normalize(item) for item in values if _normalize(item)})
def _normalize(value: Any) -> str:
return str(value or "").strip().casefold()

View File

@@ -8,6 +8,7 @@ from dataclasses import dataclass
from datetime import timedelta from datetime import timedelta
from typing import Any from typing import Any
from architecture_compatibility import architecture_compatibility_key, architecture_profile
from llm_classifier import LLMAssistedClassifier from llm_classifier import LLMAssistedClassifier
from models import ModelInspection from models import ModelInspection
from common import parse_datetime, utc_now from common import parse_datetime, utc_now
@@ -150,10 +151,16 @@ class CandidatePreflightAdvisor:
self._llm_blocks = 0 self._llm_blocks = 0
self._context_clamps = 0 self._context_clamps = 0
self._ambiguous = 0 self._ambiguous = 0
self._architecture_blocks_applied = 0
self._feedback_stats: dict[str, Any] = {} self._feedback_stats: dict[str, Any] = {}
self._architecture_compatibility_blocks: dict[str, dict[str, Any]] = {}
def set_feedback_stats(self, report: dict[str, Any] | None) -> None: def set_feedback_stats(self, report: dict[str, Any] | None) -> None:
self._feedback_stats = report if isinstance(report, dict) else {} self._feedback_stats = report if isinstance(report, dict) else {}
raw_blocks = self._feedback_stats.get("architectureCompatibilityBlocks") or {}
self._architecture_compatibility_blocks = (
raw_blocks if isinstance(raw_blocks, dict) else {}
)
for gpu, value in (self._feedback_stats.get("observedGpuMemoryGiB") or {}).items(): for gpu, value in (self._feedback_stats.get("observedGpuMemoryGiB") or {}).items():
try: try:
memory_gib = float(value) memory_gib = float(value)
@@ -190,6 +197,24 @@ class CandidatePreflightAdvisor:
with self._lock: with self._lock:
self._assessed += 1 self._assessed += 1
learned_block = self._matching_architecture_block(
inspection=inspection,
target_gpu=target_gpu,
framework=framework,
task_type=task_type,
)
if learned_block is not None:
metadata["architectureCompatibilityBlock"] = learned_block
with self._lock:
self._architecture_blocks_applied += 1
return self._hard_block(
config_params,
"preflight_learned_architecture_incompatible",
warnings,
ambiguous,
metadata,
)
# Empty file_paths means an injected/test inspection lacks structural # Empty file_paths means an injected/test inspection lacks structural
# metadata. Real discoveries with an empty tree already fail the weight # metadata. Real discoveries with an empty tree already fail the weight
# compatibility gate, so do not make this test/fallback state a blocker. # compatibility gate, so do not make this test/fallback state a blocker.
@@ -389,6 +414,33 @@ class CandidatePreflightAdvisor:
metadata=metadata, metadata=metadata,
) )
def _matching_architecture_block(
self,
*,
inspection: ModelInspection,
target_gpu: str,
framework: str,
task_type: str,
) -> dict[str, Any] | None:
profile = architecture_profile(inspection.model_type, inspection.architectures)
if profile is None:
return None
key = architecture_compatibility_key(
target_gpu,
framework,
task_type,
profile["signature"],
)
if key is None:
return None
block = self._architecture_compatibility_blocks.get(key)
if not isinstance(block, dict):
return None
expires_at = parse_datetime(block.get("expiresAt"))
if expires_at is None or expires_at <= utc_now():
return None
return dict(block)
def summary(self) -> dict[str, Any]: def summary(self) -> dict[str, Any]:
with self._lock: with self._lock:
summary = { summary = {
@@ -397,6 +449,10 @@ class CandidatePreflightAdvisor:
"hardBlocks": self._hard_blocks, "hardBlocks": self._hard_blocks,
"llmBlocks": self._llm_blocks, "llmBlocks": self._llm_blocks,
"ambiguousCandidates": self._ambiguous, "ambiguousCandidates": self._ambiguous,
"architectureCompatibilityBlocksLoaded": len(
self._architecture_compatibility_blocks
),
"architectureCompatibilityBlocksApplied": self._architecture_blocks_applied,
"contextLengthClamps": self._context_clamps, "contextLengthClamps": self._context_clamps,
"knownGpuMemoryGiB": dict(self.gpu_memory_gib), "knownGpuMemoryGiB": dict(self.gpu_memory_gib),
"gpuMemoryEvidence": dict(self.gpu_memory_evidence), "gpuMemoryEvidence": dict(self.gpu_memory_evidence),

View File

@@ -18,7 +18,8 @@ MAX_ERROR_REPORT_BYTES = 1_000_000
ERROR_LINE_PATTERN = re.compile( ERROR_LINE_PATTERN = re.compile(
r"(?:\b(?:[A-Za-z_]*(?:Error|Exception)|PREFLIGHT_[A-Z_]+|OOM)\b|" r"(?:\b(?:[A-Za-z_]*(?:Error|Exception)|PREFLIGHT_[A-Z_]+|OOM)\b|"
r"out of memory|not supported|unsupported|does not recognize|cannot|can.t|" r"out of memory|not supported|unsupported|does not recognize|cannot|can.t|"
r"not found|no such file|failed to|invalid|traceback|找不到空闲卡)", r"not found|no such file|failed to|invalid|traceback|"
r"找不到空闲卡|不支持|暂不支持|不兼容|请换用|请更换)",
re.IGNORECASE, re.IGNORECASE,
) )
@@ -74,10 +75,14 @@ def classify_failure_archive(
error_lines = _extract_error_lines(runtime_log) error_lines = _extract_error_lines(runtime_log)
report_code = str(report.get("code") or "").strip() or None report_code = str(report.get("code") or "").strip() or None
classification = classify_failure_report(report_code, error_lines) suggestion = str(report.get("suggestion") or "")[:500] or None
classification_inputs = [*error_lines]
if suggestion:
classification_inputs.append(suggestion)
classification = classify_failure_report(report_code, classification_inputs)
result: dict[str, Any] = { result: dict[str, Any] = {
"failureCode": report_code, "failureCode": report_code,
"failureSuggestion": str(report.get("suggestion") or "")[:500] or None, "failureSuggestion": suggestion,
"failureCategory": classification.category, "failureCategory": classification.category,
"failureScope": classification.scope, "failureScope": classification.scope,
"failureAction": classification.action, "failureAction": classification.action,

View File

@@ -4,6 +4,12 @@ import re
from dataclasses import asdict, dataclass from dataclasses import asdict, dataclass
from typing import Iterable from typing import Iterable
from architecture_compatibility import (
EXPLICIT_ARCHITECTURE_FAILURE_ACTION,
EXPLICIT_ARCHITECTURE_FAILURE_CATEGORY,
EXPLICIT_ARCHITECTURE_FAILURE_REASON,
)
@dataclass(frozen=True) @dataclass(frozen=True)
class FailureClassification: class FailureClassification:
@@ -70,6 +76,42 @@ DETERMINISTIC_LOG_PATTERNS = (
(re.compile(r"config\.json.*(?:not found|no config)|Invalid repository ID or local directory", re.I), "MODEL_FILE_NOT_FOUND"), (re.compile(r"config\.json.*(?:not found|no config)|Invalid repository ID or local directory", re.I), "MODEL_FILE_NOT_FOUND"),
) )
# These patterns deliberately require both framework and model/architecture
# semantics. Generic messages such as "attention backend not supported" or
# "GPU type not supported" must not create an architecture blacklist.
EXPLICIT_FRAMEWORK_MODEL_UNSUPPORTED_PATTERNS = (
re.compile(
r"(?:当前|该|此)(?:推理)?框架.{0,100}(?:不支持|暂不支持|无法支持|不兼容)"
r".{0,100}(?:该|此|当前)?(?:模型架构|模型|架构)",
re.I,
),
re.compile(
r"(?:该|此|当前)?(?:模型架构|模型|架构).{0,100}"
r"(?:不被|不受).{0,50}(?:当前|该|此)?(?:推理)?框架.{0,30}支持",
re.I,
),
re.compile(
r"\b(?:this|the|current)\s+framework\b.{0,100}"
r"\b(?:does\s+not|doesn't|cannot|can't)\s+support\b.{0,100}"
r"\b(?:model|model\s+architecture|architecture)\b",
re.I,
),
re.compile(
r"\b(?:model|model\s+architecture|architecture)\b.{0,100}"
r"\b(?:is\s+)?not\s+supported\s+by\b.{0,80}\bframework\b",
re.I,
),
)
MODEL_NOT_SUPPORTED_SUGGESTION_PATTERNS = (
re.compile(r"请(?:更换|换用|使用|选择).{0,50}(?:受支持|支持的)(?:模型|模型架构)", re.I),
re.compile(
r"\b(?:please\s+)?(?:change|switch|use|choose).{0,60}"
r"\b(?:a\s+)?supported\s+(?:model|model\s+architecture)\b",
re.I,
),
)
def classify_failure_report(report_code: str | None, log_lines: Iterable[str] = ()) -> FailureClassification: def classify_failure_report(report_code: str | None, log_lines: Iterable[str] = ()) -> FailureClassification:
code = str(report_code or "").strip().upper() code = str(report_code or "").strip().upper()
@@ -93,6 +135,18 @@ def classify_failure_report(report_code: str | None, log_lines: Iterable[str] =
for pattern, inferred_code in DETERMINISTIC_LOG_PATTERNS: for pattern, inferred_code in DETERMINISTIC_LOG_PATTERNS:
if pattern.search(text): if pattern.search(text):
return DETERMINISTIC_POLICIES[inferred_code] return DETERMINISTIC_POLICIES[inferred_code]
if any(pattern.search(text) for pattern in EXPLICIT_FRAMEWORK_MODEL_UNSUPPORTED_PATTERNS) or (
code == "MODEL_NOT_SUPPORTED"
and any(pattern.search(text) for pattern in MODEL_NOT_SUPPORTED_SUGGESTION_PATTERNS)
):
return FailureClassification(
EXPLICIT_ARCHITECTURE_FAILURE_CATEGORY,
"model_gpu_framework",
EXPLICIT_ARCHITECTURE_FAILURE_ACTION,
True,
False,
EXPLICIT_ARCHITECTURE_FAILURE_REASON,
)
if code in SEMANTIC_POLICIES: if code in SEMANTIC_POLICIES:
return SEMANTIC_POLICIES[code] return SEMANTIC_POLICIES[code]
return FailureClassification( return FailureClassification(

View File

@@ -3,9 +3,17 @@ from __future__ import annotations
from collections import defaultdict from collections import defaultdict
from concurrent.futures import ThreadPoolExecutor, as_completed from concurrent.futures import ThreadPoolExecutor, as_completed
from datetime import datetime, timedelta from datetime import datetime, timedelta
import os
from pathlib import Path from pathlib import Path
from typing import Any from typing import Any
from architecture_compatibility import (
DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS,
EXPLICIT_ARCHITECTURE_FAILURE_CATEGORY,
EXPLICIT_ARCHITECTURE_FAILURE_REASON,
architecture_compatibility_key,
architecture_profile,
)
from common import append_jsonl, parse_datetime, read_jsonl, update_jsonl, utc_now from common import append_jsonl, parse_datetime, read_jsonl, update_jsonl, utc_now
from failure_log_inspector import fetch_and_classify_failure_log from failure_log_inspector import fetch_and_classify_failure_log
from history_stats import classify_failure, is_failure, is_success from history_stats import classify_failure, is_failure, is_success
@@ -245,7 +253,8 @@ class OutcomeTracker:
return failed_at >= now - timedelta(hours=max(0, int(cooldown_hours))) return failed_at >= now - timedelta(hours=max(0, int(cooldown_hours)))
def get_stats_report(self) -> dict[str, Any]: def get_stats_report(self) -> dict[str, Any]:
now = _now_iso() now_datetime = utc_now()
now = now_datetime.isoformat()
terminal = [r for r in self._records if r.get("outcome") in {"success", "failed"}] terminal = [r for r in self._records if r.get("outcome") in {"success", "failed"}]
gpu_groups: dict[str, list[dict[str, Any]]] = defaultdict(list) gpu_groups: dict[str, list[dict[str, Any]]] = defaultdict(list)
@@ -344,6 +353,17 @@ class OutcomeTracker:
previous = observed_gpu_memory.get(gpu) previous = observed_gpu_memory.get(gpu)
observed_gpu_memory[gpu] = min(previous, memory_gib) if previous else memory_gib observed_gpu_memory[gpu] = min(previous, memory_gib) if previous else memory_gib
architecture_block_ttl_days = _architecture_block_ttl_days()
architecture_blocks = _build_architecture_compatibility_blocks(
terminal,
now=now_datetime,
ttl_days=architecture_block_ttl_days,
)
architecture_blocks_by_gpu_framework: dict[str, int] = defaultdict(int)
for block in architecture_blocks.values():
combination = f"{block['targetGpu']}|{block['framework']}"
architecture_blocks_by_gpu_framework[combination] += 1
return { return {
"generatedAt": now, "generatedAt": now,
"totalRecords": len(self._records), "totalRecords": len(self._records),
@@ -356,6 +376,12 @@ class OutcomeTracker:
"recentCombinationStats": recent_combination_stats, "recentCombinationStats": recent_combination_stats,
"profileCombinationStats": profile_combination_stats, "profileCombinationStats": profile_combination_stats,
"recentProfileCombinationStats": recent_profile_combination_stats, "recentProfileCombinationStats": recent_profile_combination_stats,
"architectureCompatibilityBlocks": architecture_blocks,
"architectureCompatibilitySummary": {
"activeBlockCount": len(architecture_blocks),
"ttlDays": architecture_block_ttl_days,
"byGpuFramework": dict(architecture_blocks_by_gpu_framework),
},
"observedGpuMemoryGiB": observed_gpu_memory, "observedGpuMemoryGiB": observed_gpu_memory,
"totals": _summarize(terminal), "totals": _summarize(terminal),
"warnings": warnings, "warnings": warnings,
@@ -489,6 +515,118 @@ def _summarize(records: list[dict[str, Any]]) -> dict[str, Any]:
} }
def _architecture_block_ttl_days() -> int:
raw = os.getenv("MODELHUB_ARCHITECTURE_BLOCK_TTL_DAYS")
if raw is None:
return DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS
try:
value = int(raw)
except ValueError:
return DEFAULT_ARCHITECTURE_BLOCK_TTL_DAYS
return min(365, max(1, value))
def _build_architecture_compatibility_blocks(
records: list[dict[str, Any]],
*,
now: datetime,
ttl_days: int,
) -> dict[str, dict[str, Any]]:
successes: dict[str, list[tuple[datetime, dict[str, Any]]]] = defaultdict(list)
failures: dict[str, list[tuple[datetime, dict[str, Any], dict[str, Any]]]] = defaultdict(list)
cutoff = now - timedelta(days=max(1, int(ttl_days)))
for record in records:
profile_data = record.get("modelProfile")
if not isinstance(profile_data, dict):
continue
profile = architecture_profile(
profile_data.get("modelType"),
profile_data.get("architectures"),
)
if profile is None:
continue
target_gpu = str(record.get("targetGpu") or "").strip()
framework = str(record.get("framework") or "").strip()
task_type = str(record.get("taskType") or "").strip()
key = architecture_compatibility_key(
target_gpu,
framework,
task_type,
profile["signature"],
)
event_time = (
parse_datetime(record.get("submitTime"))
or parse_datetime(record.get("lastSyncTime"))
)
if key is None or event_time is None:
continue
if record.get("outcome") == "success":
successes[key].append((event_time, record))
continue
if not _is_explicit_architecture_failure(record) or event_time < cutoff:
continue
failures[key].append((event_time, record, profile))
blocks: dict[str, dict[str, Any]] = {}
for key, failure_events in failures.items():
failure_events.sort(key=lambda item: item[0])
latest_failure_at, latest_failure, latest_profile = failure_events[-1]
success_events = successes.get(key) or []
latest_success_at = max((item[0] for item in success_events), default=None)
if latest_success_at is not None and latest_success_at >= latest_failure_at:
continue
effective_failures = [
item
for item in failure_events
if latest_success_at is None or item[0] > latest_success_at
]
expires_at = latest_failure_at + timedelta(days=max(1, int(ttl_days)))
if expires_at <= now:
continue
blocks[key] = {
"targetGpu": str(latest_failure.get("targetGpu") or ""),
"framework": str(latest_failure.get("framework") or ""),
"taskType": str(latest_failure.get("taskType") or ""),
"matchType": latest_profile["matchType"],
"architectureSignature": latest_profile["signature"],
"architectures": latest_profile["architectures"],
"modelType": latest_profile["modelType"],
"evidenceCount": len(effective_failures),
"latestFailureAt": latest_failure_at.isoformat(),
"latestSuccessfulAt": latest_success_at.isoformat() if latest_success_at else None,
"expiresAt": expires_at.isoformat(),
"sourceTaskIds": _bounded_unique(
record.get("taskId") for _, record, _ in reversed(effective_failures)
),
"sourceModelIds": _bounded_unique(
record.get("modelId") for _, record, _ in reversed(effective_failures)
),
}
return blocks
def _is_explicit_architecture_failure(record: dict[str, Any]) -> bool:
return bool(
record.get("outcome") == "failed"
and record.get("failureCategory") == EXPLICIT_ARCHITECTURE_FAILURE_CATEGORY
and record.get("failureDeterministic") is True
and record.get("failureClassificationReason") == EXPLICIT_ARCHITECTURE_FAILURE_REASON
)
def _bounded_unique(values: Any, limit: int = 5) -> list[str]:
result: list[str] = []
for value in values:
rendered = str(value or "").strip()
if not rendered or rendered in result:
continue
result.append(rendered)
if len(result) >= limit:
break
return result
def _is_platform_failure(record: dict[str, Any]) -> bool: def _is_platform_failure(record: dict[str, Any]) -> bool:
if record.get("outcome") != "failed": if record.get("outcome") != "failed":
return False return False

View File

@@ -428,7 +428,8 @@ def run_poll_loop(
f"success={totals.get('successCount', 0)} " f"success={totals.get('successCount', 0)} "
f"failed={totals.get('failureCount', 0)} " f"failed={totals.get('failureCount', 0)} "
f"success_rate={totals.get('successRate', 0):.3f} " f"success_rate={totals.get('successRate', 0):.3f} "
f"failure_rate={totals.get('failureRate', 0):.3f}" f"failure_rate={totals.get('failureRate', 0):.3f} "
f"architecture_blocks={len(stats.get('architectureCompatibilityBlocks') or {})}"
) )
except Exception as exc: except Exception as exc:
log(f"[poll] cycle={cycles} outcome_stats_error={exc}") log(f"[poll] cycle={cycles} outcome_stats_error={exc}")

View File

@@ -1 +1 @@
AGENT_VERSION = "2026.08.12.1" AGENT_VERSION = "2026.08.12.2"

View File

@@ -6,7 +6,7 @@ import sys
import tempfile import tempfile
import unittest import unittest
import zipfile import zipfile
from datetime import datetime, timezone from datetime import datetime, timedelta, timezone
from pathlib import Path from pathlib import Path
from unittest.mock import patch from unittest.mock import patch
@@ -475,6 +475,41 @@ class CandidatePreflightTests(unittest.TestCase):
self.assertTrue(model.needs_llm) self.assertTrue(model.needs_llm)
self.assertTrue(oom.deterministic) self.assertTrue(oom.deterministic)
def test_explicit_framework_model_error_is_deterministic_architecture_feedback(self) -> None:
classifier = FailureClassifier()
result = classify_failure_archive(
make_failure_archive(
"MODEL_NOT_SUPPORTED",
"",
"该框架不支持该模型,请换用支持的模型",
),
llm_classifier=classifier, # type: ignore[arg-type]
)
self.assertEqual("framework_architecture_unsupported", result["failureCategory"])
self.assertEqual("block_gpu_framework_architecture", result["failureAction"])
self.assertTrue(result["failureDeterministic"])
self.assertFalse(result["failureNeedsLlm"])
self.assertEqual(0, classifier.calls)
def test_generic_unsupported_backend_does_not_create_architecture_feedback(self) -> None:
classification = classify_failure_report(
"ATTENTION_NOT_SUPPORTED",
["Flash attention backend is not supported on this GPU"],
)
self.assertEqual("attention_backend", classification.category)
self.assertFalse(classification.deterministic)
def test_structured_oom_takes_priority_over_architecture_wording(self) -> None:
classification = classify_failure_report(
"PREFLIGHT_OOM",
["该框架不支持该模型,请换用支持的模型"],
)
self.assertEqual("memory_capacity", classification.category)
self.assertEqual("structured_oom", classification.reason)
def test_failure_archive_uses_deterministic_platform_signature_without_llm(self) -> None: def test_failure_archive_uses_deterministic_platform_signature_without_llm(self) -> None:
classifier = FailureClassifier() classifier = FailureClassifier()
result = classify_failure_archive( result = classify_failure_archive(
@@ -511,6 +546,123 @@ class CandidatePreflightTests(unittest.TestCase):
self.assertFalse(result["failureNeedsLlm"]) self.assertFalse(result["failureNeedsLlm"])
self.assertEqual(1, classifier.calls) self.assertEqual(1, classifier.calls)
def test_explicit_failure_learns_exact_gpu_framework_architecture_block(self) -> None:
now = datetime.now(timezone.utc)
with tempfile.TemporaryDirectory() as temporary_dir:
tracker = OutcomeTracker(Path(temporary_dir) / "outcomes.jsonl")
tracker.record_submission(
"owner/source-model",
"Biren_166m",
"vllm",
"text-generation",
"task-architecture-failure",
(now - timedelta(minutes=5)).isoformat(),
model_profile={
"modelType": "qwen2",
"architectures": ["Qwen2ForCausalLM"],
},
)
tracker._records[0].update( # noqa: SLF001
{
"outcome": "failed",
"failureCategory": "framework_architecture_unsupported",
"failureAction": "block_gpu_framework_architecture",
"failureDeterministic": True,
"failureClassificationReason": "explicit_framework_model_unsupported",
}
)
report = tracker.get_stats_report()
key = "biren_166m|vllm|text-generation|architectures:qwen2forcausallm"
self.assertIn(key, report["architectureCompatibilityBlocks"])
self.assertEqual(
1,
report["architectureCompatibilitySummary"]["activeBlockCount"],
)
advisor = CandidatePreflightAdvisor(gpu_memory_gib={})
advisor.set_feedback_stats(report)
exact_architecture = ModelInspection(
repo_id="different-name/no-string-match-needed",
model_config={
"model_type": "qwen2",
"architectures": ["Qwen2ForCausalLM"],
},
)
blocked = advisor.assess(
inspection=exact_architecture,
task_type="text-generation",
target_gpu="Biren_166m",
framework="vllm",
config_params="",
)
other_framework = advisor.assess(
inspection=exact_architecture,
task_type="text-generation",
target_gpu="Biren_166m",
framework="mindie",
config_params="",
)
other_architecture = advisor.assess(
inspection=ModelInspection(
repo_id="owner/other",
model_config={
"model_type": "qwen2",
"architectures": ["Qwen2ForSequenceClassification"],
},
),
task_type="text-generation",
target_gpu="Biren_166m",
framework="vllm",
config_params="",
)
self.assertFalse(blocked.allowed)
self.assertEqual("preflight_learned_architecture_incompatible", blocked.reason)
self.assertTrue(other_framework.allowed)
self.assertTrue(other_architecture.allowed)
self.assertEqual(1, advisor.summary()["architectureCompatibilityBlocksApplied"])
def test_later_success_clears_learned_architecture_block(self) -> None:
now = datetime.now(timezone.utc)
profile = {
"modelType": "qwen2",
"architectures": ["Qwen2ForCausalLM"],
}
with tempfile.TemporaryDirectory() as temporary_dir:
tracker = OutcomeTracker(Path(temporary_dir) / "outcomes.jsonl")
tracker.record_submission(
"owner/failed",
"gpu",
"vllm",
"text-generation",
"task-failed",
(now - timedelta(hours=2)).isoformat(),
model_profile=profile,
)
tracker._records[0].update( # noqa: SLF001
{
"outcome": "failed",
"failureCategory": "framework_architecture_unsupported",
"failureDeterministic": True,
"failureClassificationReason": "explicit_framework_model_unsupported",
}
)
tracker.record_submission(
"owner/succeeded",
"gpu",
"vllm",
"text-generation",
"task-success",
(now - timedelta(hours=1)).isoformat(),
model_profile=profile,
)
tracker._records[1]["outcome"] = "success" # noqa: SLF001
report = tracker.get_stats_report()
self.assertEqual({}, report["architectureCompatibilityBlocks"])
def test_outcome_sync_enriches_failure_and_excludes_platform_fault_from_feedback(self) -> None: def test_outcome_sync_enriches_failure_and_excludes_platform_fault_from_feedback(self) -> None:
with tempfile.TemporaryDirectory() as temporary_dir: with tempfile.TemporaryDirectory() as temporary_dir:
path = Path(temporary_dir) / "outcomes.jsonl" path = Path(temporary_dir) / "outcomes.jsonl"