690 lines
28 KiB
Python
690 lines
28 KiB
Python
from __future__ import annotations
|
||
|
||
import io
|
||
import json
|
||
import sys
|
||
import tempfile
|
||
import unittest
|
||
import zipfile
|
||
from datetime import datetime, timezone
|
||
from pathlib import Path
|
||
from unittest.mock import patch
|
||
|
||
|
||
PACKAGE_DIR = Path(__file__).resolve().parents[1] / "modelhub_submmit_api"
|
||
if str(PACKAGE_DIR) in sys.path:
|
||
sys.path.remove(str(PACKAGE_DIR))
|
||
sys.path.insert(0, str(PACKAGE_DIR))
|
||
|
||
from candidate_preflight import CandidatePreflightAdvisor, clamp_context_length # noqa: E402
|
||
from common import read_jsonl # noqa: E402
|
||
from failure_log_inspector import classify_failure_archive # noqa: E402
|
||
from failure_taxonomy import classify_failure_report # noqa: E402
|
||
from hf_discovery import inspect_repo_tree # noqa: E402
|
||
from llm_classifier import LLMAssistedClassifier, _parse_json_object # noqa: E402
|
||
from models import ModelInspection # noqa: E402
|
||
from outcome_tracker import OutcomeTracker # noqa: E402
|
||
from runner_common import ensure_dashscope_key # noqa: E402
|
||
|
||
|
||
class DenyingClassifier:
|
||
enabled = True
|
||
|
||
def classify(self, **_kwargs): # noqa: ANN003
|
||
return {"decision": "deny", "confidence": 0.95, "reason": "unsupported_custom_arch"}
|
||
|
||
@staticmethod
|
||
def blocks(decision): # noqa: ANN001
|
||
return decision["decision"] == "deny" and decision["confidence"] >= 0.85
|
||
|
||
@staticmethod
|
||
def summary():
|
||
return {"enabled": True}
|
||
|
||
|
||
class FailureClassifier:
|
||
enabled = True
|
||
|
||
def __init__(self) -> None:
|
||
self.calls = 0
|
||
|
||
def classify_failure(self, **_kwargs): # noqa: ANN003
|
||
self.calls += 1
|
||
return {
|
||
"category": "custom_architecture_runtime",
|
||
"scope": "model_gpu",
|
||
"action": "avoid_exact_profile",
|
||
"confidence": 0.91,
|
||
"reason": "unsupported_remote_code",
|
||
"evidence": ["unsupported architecture"],
|
||
}
|
||
|
||
|
||
class TaskClient:
|
||
def __init__(self, tasks): # noqa: ANN001
|
||
self.tasks = tasks
|
||
|
||
def list_tasks(self, **_kwargs): # noqa: ANN003
|
||
return self.tasks
|
||
|
||
|
||
def make_failure_archive(code: str, runtime_log: str, suggestion: str = "") -> bytes:
|
||
target = io.BytesIO()
|
||
with zipfile.ZipFile(target, "w", compression=zipfile.ZIP_DEFLATED) as bundle:
|
||
bundle.writestr(
|
||
"error_report_raw.json",
|
||
json.dumps({"code": code, "suggestion": suggestion}, ensure_ascii=False),
|
||
)
|
||
bundle.writestr("pod_runtime_log.txt", runtime_log)
|
||
return target.getvalue()
|
||
|
||
|
||
class CandidatePreflightTests(unittest.TestCase):
|
||
def test_repo_inspection_preserves_sizes_and_structured_config_fields(self) -> None:
|
||
inspection = inspect_repo_tree(
|
||
"owner/model",
|
||
[
|
||
{"Path": "config.json", "Type": "blob", "Size": 100},
|
||
{"Path": "tokenizer.json", "Type": "blob", "Size": 200},
|
||
{"Path": "model-1.safetensors", "Type": "blob", "Size": 3_000},
|
||
{"Path": "model-2.safetensors", "Type": "blob", "Size": 4_000},
|
||
{"Path": "pytorch_model.bin", "Type": "blob", "Size": 9_000},
|
||
],
|
||
)
|
||
inspection = ModelInspection(
|
||
**{**inspection.__dict__, "model_config": {"model_type": "qwen2", "architectures": ["Qwen2ForCausalLM"]}}
|
||
)
|
||
|
||
self.assertTrue(inspection.has_root_config)
|
||
self.assertTrue(inspection.has_root_tokenizer)
|
||
self.assertEqual(7_000, inspection.estimated_load_bytes("vllm"))
|
||
self.assertEqual("qwen2", inspection.model_type)
|
||
self.assertEqual(["Qwen2ForCausalLM"], inspection.architectures)
|
||
|
||
def test_indexed_nested_weight_shards_are_included_in_load_size(self) -> None:
|
||
inspection = ModelInspection(
|
||
repo_id="owner/sharded",
|
||
file_paths=[
|
||
"config.json",
|
||
"model.safetensors.index.json",
|
||
"shards/model-1.safetensors",
|
||
"shards/model-2.safetensors",
|
||
],
|
||
file_sizes={
|
||
"shards/model-1.safetensors": 3_000,
|
||
"shards/model-2.safetensors": 4_000,
|
||
},
|
||
weight_files=[
|
||
"shards/model-1.safetensors",
|
||
"shards/model-2.safetensors",
|
||
],
|
||
)
|
||
self.assertEqual(7_000, inspection.estimated_load_bytes("vllm"))
|
||
|
||
def test_non_gguf_missing_root_config_is_blocked_but_gguf_is_allowed(self) -> None:
|
||
advisor = CandidatePreflightAdvisor(gpu_memory_gib={})
|
||
standard = ModelInspection(
|
||
repo_id="owner/model",
|
||
file_paths=["model.safetensors", "tokenizer.json"],
|
||
weight_files=["model.safetensors"],
|
||
)
|
||
blocked = advisor.assess(
|
||
inspection=standard,
|
||
task_type="text-generation",
|
||
target_gpu="unknown",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
gguf = ModelInspection(
|
||
repo_id="owner/gguf",
|
||
file_paths=["model-q4_0.gguf"],
|
||
file_sizes={"model-q4_0.gguf": 100},
|
||
gguf_files=["model-q4_0.gguf"],
|
||
selected_gguf="model-q4_0.gguf",
|
||
)
|
||
allowed = advisor.assess(
|
||
inspection=gguf,
|
||
task_type="text-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="llamacpp",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
|
||
self.assertFalse(blocked.allowed)
|
||
self.assertEqual("preflight_missing_root_config", blocked.reason)
|
||
self.assertTrue(allowed.allowed)
|
||
|
||
def test_text_specific_root_layout_rule_does_not_reject_diffusers_layout(self) -> None:
|
||
inspection = ModelInspection(
|
||
repo_id="owner/diffusion-model",
|
||
file_paths=["model_index.json", "unet/model.safetensors"],
|
||
file_sizes={"model_index.json": 100, "unet/model.safetensors": 1_000},
|
||
weight_files=["unet/model.safetensors"],
|
||
)
|
||
assessment = CandidatePreflightAdvisor(gpu_memory_gib={}).assess(
|
||
inspection=inspection,
|
||
task_type="text-to-image-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="diffusers",
|
||
config_params="{}",
|
||
)
|
||
self.assertTrue(assessment.allowed)
|
||
|
||
def test_new_gpu_without_capacity_evidence_is_deferred(self) -> None:
|
||
inspection = ModelInspection(
|
||
repo_id="owner/model",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={"config.json": 1, "tokenizer.json": 1, "model.safetensors": 10},
|
||
weight_files=["model.safetensors"],
|
||
model_config={"model_type": "qwen2"},
|
||
)
|
||
assessment = CandidatePreflightAdvisor().assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="future_gpu_without_evidence",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("preflight_gpu_memory_unknown", assessment.reason)
|
||
|
||
def test_predicted_model_load_memory_blocks_known_oom(self) -> None:
|
||
gib = 1024**3
|
||
inspection = ModelInspection(
|
||
repo_id="owner/large",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={
|
||
"config.json": 0,
|
||
"tokenizer.json": 0,
|
||
"model.safetensors": 30 * gib,
|
||
},
|
||
weight_files=["model.safetensors"],
|
||
model_config={"model_type": "llama", "architectures": ["LlamaForCausalLM"]},
|
||
)
|
||
assessment = CandidatePreflightAdvisor(gpu_memory_gib={"test": 32}).assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="test",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("preflight_predicted_oom", assessment.reason)
|
||
self.assertEqual(36.0, assessment.metadata["estimatedRequiredGiB"])
|
||
|
||
def test_memory_gate_uses_full_repository_size_like_modelhub_preflight(self) -> None:
|
||
gib = 1024**3
|
||
inspection = ModelInspection(
|
||
repo_id="owner/duplicate-formats",
|
||
file_paths=[
|
||
"config.json",
|
||
"tokenizer.json",
|
||
"model.safetensors",
|
||
"pytorch_model.bin",
|
||
],
|
||
file_sizes={
|
||
"config.json": 0,
|
||
"tokenizer.json": 0,
|
||
"model.safetensors": 20 * gib,
|
||
"pytorch_model.bin": 40 * gib,
|
||
},
|
||
weight_files=["model.safetensors", "pytorch_model.bin"],
|
||
model_config={"model_type": "qwen2"},
|
||
)
|
||
assessment = CandidatePreflightAdvisor(gpu_memory_gib={"gpu": 64}).assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="gpu",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertEqual(20 * gib, inspection.estimated_load_bytes("vllm"))
|
||
self.assertEqual(60 * gib, inspection.repository_size_bytes)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("recursive_repository_on_disk", assessment.metadata["memorySizingBasis"])
|
||
|
||
def test_p800_verified_capacity_accepts_80_gib_repository_boundary(self) -> None:
|
||
gib = 1024**3
|
||
|
||
def assess(size_gib: int):
|
||
inspection = ModelInspection(
|
||
repo_id=f"owner/model-{size_gib}",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={
|
||
"config.json": 0,
|
||
"tokenizer.json": 0,
|
||
"model.safetensors": size_gib * gib,
|
||
},
|
||
weight_files=["model.safetensors"],
|
||
model_config={"model_type": "qwen2"},
|
||
)
|
||
return CandidatePreflightAdvisor().assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="Kunlunxin_p-800",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
|
||
self.assertTrue(assess(80).allowed)
|
||
self.assertFalse(assess(81).allowed)
|
||
|
||
def test_modelhub_observed_memory_overrides_published_capacity(self) -> None:
|
||
gib = 1024**3
|
||
inspection = ModelInspection(
|
||
repo_id="owner/platform-observed",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={
|
||
"config.json": 0,
|
||
"tokenizer.json": 0,
|
||
"model.safetensors": 45 * gib,
|
||
},
|
||
weight_files=["model.safetensors"],
|
||
model_config={"model_type": "qwen2"},
|
||
)
|
||
advisor = CandidatePreflightAdvisor()
|
||
self.assertTrue(
|
||
advisor.assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
).allowed
|
||
)
|
||
|
||
advisor.set_feedback_stats({"observedGpuMemoryGiB": {"Biren_166m": 48}})
|
||
assessment = advisor.assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("preflight_predicted_oom", assessment.reason)
|
||
self.assertEqual("local_modelhub_preflight_oom", assessment.metadata["gpuMemoryEvidence"]["source"])
|
||
|
||
def test_context_length_is_clamped_in_yaml_inline_and_list_forms(self) -> None:
|
||
source = (
|
||
"max_model_len: 4096\n"
|
||
"command: [vllm, --max-model-len, '4096']\n"
|
||
"args:\n - --max-model-len\n - '4096'\n"
|
||
"env: [{name: MAX_MODEL_LEN, value: 4096}]\n"
|
||
'ref: {"max_seq_len": 4096}\n'
|
||
)
|
||
rendered, changed = clamp_context_length(source, 1024)
|
||
self.assertTrue(changed)
|
||
self.assertNotIn("4096", rendered)
|
||
self.assertGreaterEqual(rendered.count("1024"), 5)
|
||
|
||
def test_only_high_confidence_llm_deny_blocks_ambiguous_profile(self) -> None:
|
||
inspection = ModelInspection(
|
||
repo_id="owner/custom",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={"config.json": 1, "tokenizer.json": 1, "model.safetensors": 10},
|
||
weight_files=["model.safetensors"],
|
||
model_config={
|
||
"architectures": ["CustomGenerationArchitecture"],
|
||
"auto_map": {"AutoModel": "model.CustomModel"},
|
||
},
|
||
)
|
||
assessment = CandidatePreflightAdvisor(
|
||
llm_classifier=DenyingClassifier(), # type: ignore[arg-type]
|
||
gpu_memory_gib={},
|
||
).assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("llm_high_confidence_incompatible:unsupported_custom_arch", assessment.reason)
|
||
|
||
def test_five_recent_profile_failures_open_temporary_preflight_circuit(self) -> None:
|
||
inspection = ModelInspection(
|
||
repo_id="owner/repeated",
|
||
file_paths=["config.json", "tokenizer.json", "model.safetensors"],
|
||
file_sizes={"config.json": 1, "tokenizer.json": 1, "model.safetensors": 10},
|
||
weight_files=["model.safetensors"],
|
||
model_config={"model_type": "custom", "architectures": ["CustomForCausalLM"]},
|
||
)
|
||
key = "Biren_166m|vllm|text-generation|custom|none"
|
||
advisor = CandidatePreflightAdvisor(gpu_memory_gib={})
|
||
advisor.set_feedback_stats(
|
||
{
|
||
"recentProfileCombinationStats": {
|
||
key: {
|
||
"consecutiveFailures": 5,
|
||
"lastTerminalAt": datetime.now(timezone.utc).isoformat(),
|
||
}
|
||
}
|
||
}
|
||
)
|
||
assessment = advisor.assess(
|
||
inspection=inspection,
|
||
task_type="text-generation",
|
||
target_gpu="Biren_166m",
|
||
framework="vllm",
|
||
config_params="max_model_len: 4096",
|
||
)
|
||
self.assertFalse(assessment.allowed)
|
||
self.assertEqual("preflight_recent_profile_failure_circuit", assessment.reason)
|
||
|
||
def test_outcome_tracker_groups_feedback_by_model_profile(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
tracker = OutcomeTracker(Path(temporary_dir) / "outcomes.jsonl")
|
||
tracker.record_submission(
|
||
"owner/model",
|
||
"gpu",
|
||
"vllm",
|
||
"text-generation",
|
||
"task-1",
|
||
datetime.now(timezone.utc).isoformat(),
|
||
model_profile={"modelType": "qwen3", "quantizationMethod": "awq"},
|
||
)
|
||
tracker._records[0]["outcome"] = "failed" # noqa: SLF001
|
||
report = tracker.get_stats_report()
|
||
|
||
key = "gpu|vllm|text-generation|qwen3|awq"
|
||
self.assertEqual(1, report["profileCombinationStats"][key]["failureCount"])
|
||
self.assertEqual(1, report["recentProfileCombinationStats"][key]["consecutiveFailures"])
|
||
|
||
def test_policy_cancellation_is_excluded_from_failure_feedback(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
path = Path(temporary_dir) / "outcomes.jsonl"
|
||
tracker = OutcomeTracker(path)
|
||
tracker.record_submission(
|
||
"owner/model",
|
||
"gpu",
|
||
"vllm",
|
||
"text-generation",
|
||
"task-policy",
|
||
datetime.now(timezone.utc).isoformat(),
|
||
)
|
||
marked = tracker.mark_policy_cancellations(
|
||
[
|
||
{
|
||
"taskId": "task-policy",
|
||
"modelId": "owner/model",
|
||
"gpuType": "gpu",
|
||
"cleanupReasons": ["old_model_beyond_account_queue_threshold"],
|
||
}
|
||
]
|
||
)
|
||
tracker.save()
|
||
tracker.sync_from_api(
|
||
TaskClient(
|
||
[
|
||
{
|
||
"taskId": "task-policy",
|
||
"status": "cancelled",
|
||
"verifyResult": None,
|
||
}
|
||
]
|
||
) # type: ignore[arg-type]
|
||
)
|
||
report = tracker.get_stats_report()
|
||
|
||
self.assertEqual(1, marked)
|
||
self.assertEqual(1, report["policyCancelledRecords"])
|
||
self.assertEqual(0, report["terminalRecords"])
|
||
self.assertEqual({}, report["combinationStats"])
|
||
self.assertFalse(tracker.is_model_gpu_failed("owner/model", "gpu"))
|
||
|
||
def test_policy_cancellation_does_not_hide_a_racing_success(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
tracker = OutcomeTracker(Path(temporary_dir) / "outcomes.jsonl")
|
||
tracker.record_submission(
|
||
"owner/model",
|
||
"gpu",
|
||
"vllm",
|
||
"text-generation",
|
||
"task-success",
|
||
datetime.now(timezone.utc).isoformat(),
|
||
)
|
||
tracker.mark_policy_cancellations(
|
||
[{"taskId": "task-success", "modelId": "owner/model", "gpuType": "gpu"}]
|
||
)
|
||
tracker.sync_from_api(
|
||
TaskClient(
|
||
[
|
||
{
|
||
"taskId": "task-success",
|
||
"status": "success",
|
||
"verifyResult": 1,
|
||
}
|
||
]
|
||
) # type: ignore[arg-type]
|
||
)
|
||
report = tracker.get_stats_report()
|
||
|
||
self.assertEqual(0, report["policyCancelledRecords"])
|
||
self.assertEqual(1, report["totals"]["successCount"])
|
||
|
||
def test_failure_taxonomy_separates_platform_faults_from_model_faults(self) -> None:
|
||
platform = classify_failure_report(
|
||
"EXECUTE_EMPTY_RESULT",
|
||
["workspace/launch_service: /iluvatar/welcome.sh: No such file or directory"],
|
||
)
|
||
model = classify_failure_report("MODEL_NOT_SUPPORTED", [])
|
||
oom = classify_failure_report("PREFLIGHT_OOM", [])
|
||
self.assertEqual("platform_infrastructure", platform.category)
|
||
self.assertFalse(platform.needs_llm)
|
||
self.assertTrue(model.needs_llm)
|
||
self.assertTrue(oom.deterministic)
|
||
|
||
def test_failure_archive_uses_deterministic_platform_signature_without_llm(self) -> None:
|
||
classifier = FailureClassifier()
|
||
result = classify_failure_archive(
|
||
make_failure_archive(
|
||
"EXECUTE_EMPTY_RESULT",
|
||
"workspace/launch_service: /iluvatar/welcome.sh: No such file or directory",
|
||
),
|
||
llm_classifier=classifier, # type: ignore[arg-type]
|
||
)
|
||
self.assertEqual("platform_infrastructure", result["failureCategory"])
|
||
self.assertEqual(0, classifier.calls)
|
||
|
||
def test_failure_archive_extracts_modelhub_allocated_memory(self) -> None:
|
||
result = classify_failure_archive(
|
||
make_failure_archive(
|
||
"PREFLIGHT_OOM",
|
||
"PREFLIGHT_OOM: gpu_type=TEST, 1 × 48 GB = 48 GB available < 60 GB required",
|
||
)
|
||
)
|
||
self.assertEqual(48.0, result["failureObservedGpuMemoryGiB"])
|
||
|
||
def test_failure_archive_promotes_only_confident_llm_semantic_result(self) -> None:
|
||
classifier = FailureClassifier()
|
||
result = classify_failure_archive(
|
||
make_failure_archive(
|
||
"MODEL_NOT_SUPPORTED",
|
||
"ValueError: unsupported custom architecture",
|
||
),
|
||
task_context={"modelId": "owner/model", "targetGpu": "gpu"},
|
||
llm_classifier=classifier, # type: ignore[arg-type]
|
||
)
|
||
self.assertEqual("custom_architecture_runtime", result["failureCategory"])
|
||
self.assertEqual("model_gpu", result["failureScope"])
|
||
self.assertFalse(result["failureNeedsLlm"])
|
||
self.assertEqual(1, classifier.calls)
|
||
|
||
def test_outcome_sync_enriches_failure_and_excludes_platform_fault_from_feedback(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
path = Path(temporary_dir) / "outcomes.jsonl"
|
||
tracker = OutcomeTracker(path)
|
||
tracker.record_submission(
|
||
"owner/model",
|
||
"gpu",
|
||
"vllm",
|
||
"text-generation",
|
||
"task-1",
|
||
datetime.now(timezone.utc).isoformat(),
|
||
model_profile={"modelType": "qwen3", "quantizationMethod": "none"},
|
||
)
|
||
task = {
|
||
"taskId": "task-1",
|
||
"status": "failed",
|
||
"verifyResult": -1,
|
||
"logCosUrl": "https://logs.invalid/task-1.zip",
|
||
}
|
||
classification = {
|
||
"failureCategory": "platform_infrastructure",
|
||
"failureScope": "gpu_framework",
|
||
"failureAction": "retry_later",
|
||
"failureNeedsLlm": False,
|
||
}
|
||
with patch(
|
||
"outcome_tracker.fetch_and_classify_failure_log",
|
||
return_value=classification,
|
||
):
|
||
self.assertEqual(1, tracker.sync_from_api(TaskClient([task]))) # type: ignore[arg-type]
|
||
|
||
stats = tracker.get_stats_report()
|
||
combo = stats["combinationStats"]["gpu|vllm|text-generation"]
|
||
profile = stats["recentProfileCombinationStats"][
|
||
"gpu|vllm|text-generation|qwen3|none"
|
||
]
|
||
self.assertEqual(1, combo["failureCount"])
|
||
self.assertEqual(0, combo["attributableFailureCount"])
|
||
self.assertEqual(1, combo["platformFailureCount"])
|
||
self.assertEqual(0, profile["consecutiveFailures"])
|
||
self.assertFalse(tracker.is_model_gpu_failed("owner/model", "gpu"))
|
||
self.assertNotIn("logCosUrl", read_jsonl(path)[0])
|
||
|
||
def test_failed_log_enrichment_attempt_is_persisted_and_bounded(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
path = Path(temporary_dir) / "outcomes.jsonl"
|
||
tracker = OutcomeTracker(path)
|
||
tracker.record_submission(
|
||
"owner/model",
|
||
"gpu",
|
||
"vllm",
|
||
"text-generation",
|
||
"task-2",
|
||
datetime.now(timezone.utc).isoformat(),
|
||
model_profile={"modelType": "qwen3"},
|
||
)
|
||
task = {
|
||
"taskId": "task-2",
|
||
"status": "failed",
|
||
"verifyResult": -1,
|
||
"logCosUrl": "https://logs.invalid/task-2.zip",
|
||
}
|
||
with patch(
|
||
"outcome_tracker.fetch_and_classify_failure_log",
|
||
side_effect=TimeoutError("timed out"),
|
||
):
|
||
tracker.sync_from_api(TaskClient([task])) # type: ignore[arg-type]
|
||
|
||
record = read_jsonl(path)[0]
|
||
self.assertEqual(1, record["failureEnrichmentAttempts"])
|
||
self.assertIn("TimeoutError", record["failureEnrichmentError"])
|
||
self.assertIn("logCosUrl", record)
|
||
|
||
def test_llm_json_parser_accepts_fenced_json_only(self) -> None:
|
||
parsed = _parse_json_object('```json\n{"decision":"abstain","confidence":0}\n```')
|
||
self.assertEqual("abstain", parsed["decision"])
|
||
|
||
def test_qwen_payload_uses_json_mode_without_thinking_or_token_truncation(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
classifier = LLMAssistedClassifier(
|
||
endpoint="https://dashscope.aliyuncs.com/compatible-mode/v1",
|
||
model="qwen-flash",
|
||
api_key="test-key",
|
||
cache_path=Path(temporary_dir) / "cache.json",
|
||
)
|
||
payload = classifier._chat_payload( # noqa: SLF001
|
||
system_prompt="Return JSON",
|
||
user_payload={"error": "unknown"},
|
||
max_tokens=10,
|
||
)
|
||
self.assertEqual(
|
||
"https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||
classifier.endpoint,
|
||
)
|
||
self.assertEqual({"type": "json_object"}, payload["response_format"])
|
||
self.assertFalse(payload["enable_thinking"])
|
||
self.assertNotIn("max_tokens", payload)
|
||
|
||
def test_lowercase_dashscope_dotenv_enables_default_qwen_model(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
dotenv_path = Path(temporary_dir) / ".env"
|
||
dotenv_path.write_text("dashscope=test-secret\n", encoding="utf-8")
|
||
with patch.dict(
|
||
"os.environ",
|
||
{
|
||
"MODELHUB_QWEN_API_KEY": "",
|
||
"DASHSCOPE_API_KEY": "",
|
||
"MODELHUB_QWEN_MODEL": "",
|
||
"MODELHUB_QWEN_ENDPOINT": "",
|
||
},
|
||
):
|
||
self.assertTrue(ensure_dashscope_key(dotenv_path))
|
||
classifier = LLMAssistedClassifier(
|
||
endpoint=None,
|
||
model=None,
|
||
cache_path=Path(temporary_dir) / "cache.json",
|
||
)
|
||
|
||
self.assertTrue(classifier.enabled)
|
||
self.assertEqual("qwen3.7-flash", classifier.model)
|
||
self.assertEqual("test-secret", classifier.api_key)
|
||
self.assertEqual(
|
||
"https://dashscope.aliyuncs.com/compatible-mode/v1/chat/completions",
|
||
classifier.endpoint,
|
||
)
|
||
|
||
def test_qwen_candidate_gate_skips_low_value_quantization_only_review(self) -> None:
|
||
self.assertFalse(
|
||
LLMAssistedClassifier.should_review_candidate(
|
||
["quantization_compatibility:awq"]
|
||
)
|
||
)
|
||
self.assertTrue(
|
||
LLMAssistedClassifier.should_review_candidate(
|
||
["architecture_not_in_mature_baseline:qwen_future"]
|
||
)
|
||
)
|
||
|
||
def test_qwen_hourly_budget_prevents_unbounded_calls(self) -> None:
|
||
with tempfile.TemporaryDirectory() as temporary_dir:
|
||
classifier = LLMAssistedClassifier(
|
||
endpoint="http://localhost:8000/v1",
|
||
model="qwen-test",
|
||
max_calls_per_hour=1,
|
||
cache_path=Path(temporary_dir) / "cache.json",
|
||
)
|
||
inspection_a = ModelInspection(repo_id="owner/a")
|
||
inspection_b = ModelInspection(repo_id="owner/b")
|
||
decision = {
|
||
"decision": "abstain",
|
||
"confidence": 0.0,
|
||
"reason": "unknown",
|
||
"evidence": [],
|
||
}
|
||
with patch.object(classifier, "_request_decision", return_value=decision) as request:
|
||
first = classifier.classify(
|
||
inspection=inspection_a,
|
||
task_type="text-generation",
|
||
target_gpu="gpu",
|
||
framework="vllm",
|
||
ambiguous_reasons=["model_type_missing_or_unknown"],
|
||
)
|
||
second = classifier.classify(
|
||
inspection=inspection_b,
|
||
task_type="text-generation",
|
||
target_gpu="gpu",
|
||
framework="vllm",
|
||
ambiguous_reasons=["model_type_missing_or_unknown"],
|
||
)
|
||
self.assertEqual("live", first["source"])
|
||
self.assertEqual("rate_limit", second["source"])
|
||
self.assertEqual(1, request.call_count)
|
||
|
||
|
||
if __name__ == "__main__":
|
||
unittest.main()
|