state: generation 5173 (cycle)

This commit is contained in:
2026-09-10 13:07:42 +00:00
parent 8d08c3b6f9
commit e2f8eeab8a
10 changed files with 2146 additions and 2129 deletions

View File

@@ -243,7 +243,7 @@
"ascend_910-b4|vllm|text-generation|model_type:qwen3_5_moe": {
"architectureSignature": "model_type:qwen3_5_moe",
"architectures": [],
"evidenceCount": 2,
"evidenceCount": 1,
"expiresAt": "2026-10-08T03:33:22+00:00",
"framework": "vllm",
"latestFailureAt": "2026-09-08T03:33:22+00:00",
@@ -251,12 +251,10 @@
"matchType": "model_type",
"modelType": "qwen3_5_moe",
"sourceModelIds": [
"mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B",
"mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B"
"mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B"
],
"sourceTaskIds": [
"4478318",
"4481796"
"4478318"
],
"targetGpu": "Ascend_910-b4",
"taskType": "text-generation"
@@ -1313,7 +1311,7 @@
"taskType": "text-generation"
}
},
"generatedAt": "2026-09-10T13:05:07.556498+00:00",
"generatedAt": "2026-09-10T13:07:41.674717+00:00",
"summary": {
"activeBlockCount": 66,
"byGpuFramework": {

View File

@@ -11,7 +11,7 @@
"1": {
"complete": false,
"lastError": "ModelHubAPIError: 系统错误",
"listingErrors": 368,
"listingErrors": 369,
"nextPage": 1,
"recordsScanned": 0,
"uniqueRecords": 0
@@ -102,12 +102,12 @@
"cutoffAt": "2026-09-04T03:55:51.365685+00:00",
"failureLogsInspected": 0,
"mode": "incremental_decision_only",
"nextAccountIndex": 1,
"nextAccountIndex": 2,
"recordsScanned": 0,
"seenTaskIds": [],
"startedAt": "2026-09-04T03:55:51.365685+00:00",
"terminalRecords": 0,
"uniqueRecords": 0,
"updatedAt": "2026-09-10T13:05:07.533162+00:00",
"updatedAt": "2026-09-10T13:07:41.652111+00:00",
"version": 1
}

View File

@@ -416,7 +416,7 @@
}
},
"frameworkUpdatedAt": null,
"generatedAt": "2026-09-10T13:03:07.997350+00:00",
"generatedAt": "2026-09-10T13:06:14.747482+00:00",
"gpuStats": {
"Ascend_910-b3": {
"available": true,

File diff suppressed because it is too large Load Diff

View File

@@ -1,6 +1,6 @@
{
"generatedAt": "2026-09-10T12:49:34.524285+00:00",
"lastSyncTime": "2026-09-10T12:49:34.298351+00:00",
"generatedAt": "2026-09-10T13:06:08.640092+00:00",
"lastSyncTime": "2026-09-10T13:06:08.293484+00:00",
"recentLimit": 300,
"report": {
"architectureCompatibilityBlocks": {
@@ -247,7 +247,7 @@
"ascend_910-b4|vllm|text-generation|model_type:qwen3_5_moe": {
"architectureSignature": "model_type:qwen3_5_moe",
"architectures": [],
"evidenceCount": 2,
"evidenceCount": 1,
"expiresAt": "2026-10-08T03:33:22+00:00",
"framework": "vllm",
"latestFailureAt": "2026-09-08T03:33:22+00:00",
@@ -255,12 +255,10 @@
"matchType": "model_type",
"modelType": "qwen3_5_moe",
"sourceModelIds": [
"mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B",
"mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B"
"mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B"
],
"sourceTaskIds": [
"4478318",
"4481796"
"4478318"
],
"targetGpu": "Ascend_910-b4",
"taskType": "text-generation"
@@ -1598,11 +1596,11 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 7,
"failureBreakdown": {
"ambiguous_runtime": 15,
"ambiguous_runtime": 17,
"framework_architecture_unsupported": 7,
"参数/模板问题": 3
},
"failureCount": 25,
"failureCount": 27,
"failureRate": 1.0,
"framework": "vllm-mlu",
"pendingCount": 0,
@@ -1612,8 +1610,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 25,
"unresolvedFailureCount": 18
"total": 27,
"unresolvedFailureCount": 20
},
"Cambricon_mlu-370-x8|vllm|text-generation": {
"attributableFailureCount": 3,
@@ -2350,19 +2348,19 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 7,
"failureBreakdown": {
"ambiguous_runtime": 15,
"ambiguous_runtime": 17,
"framework_architecture_unsupported": 7,
"参数/模板问题": 3
},
"failureCount": 25,
"failureCount": 27,
"failureRate": 1.0,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 0,
"successRate": 0.0,
"total": 25,
"unresolvedFailureCount": 18
"total": 27,
"unresolvedFailureCount": 20
},
"vllm-patch-tokenizer": {
"attributableFailureCount": 3,
@@ -2445,7 +2443,7 @@
"unresolvedFailureCount": 1
}
},
"generatedAt": "2026-09-10T12:49:34.520093+00:00",
"generatedAt": "2026-09-10T13:06:08.636229+00:00",
"gpuSummaries": {
"Ascend_910-b3": {
"attributableFailureCount": 26,
@@ -2547,22 +2545,22 @@
"decisionSuccessRate": 0.1053,
"decisionTotal": 19,
"failureBreakdown": {
"ambiguous_runtime": 34,
"ambiguous_runtime": 36,
"framework_architecture_unsupported": 15,
"memory_capacity": 1,
"tokenizer_compatibility": 1,
"参数/模板问题": 18,
"验证失败": 22
},
"failureCount": 91,
"failureRate": 0.9785,
"failureCount": 93,
"failureRate": 0.9789,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 2,
"successRate": 0.0215,
"total": 93,
"unresolvedFailureCount": 74
"successRate": 0.0211,
"total": 95,
"unresolvedFailureCount": 76
},
"Iluvatar_bi-100": {
"attributableFailureCount": 0,
@@ -2841,9 +2839,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 3
"ambiguous_runtime": 5
},
"failureCount": 3,
"failureCount": 5,
"failureRate": 1.0,
"framework": "vllm-mlu",
"modelType": "gemma2",
@@ -2855,8 +2853,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 3,
"unresolvedFailureCount": 3
"total": 5,
"unresolvedFailureCount": 5
},
"Cambricon_mlu-370-x8|vllm-mlu|text-generation|llama|compressed-tensors": {
"attributableFailureCount": 0,
@@ -4152,15 +4150,15 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 4,
"failureBreakdown": {
"ambiguous_runtime": 11,
"ambiguous_runtime": 13,
"framework_architecture_unsupported": 4,
"参数/模板问题": 1
},
"failureCount": 16,
"failureCount": 18,
"failureRate": 1.0,
"framework": "vllm-mlu",
"lastPlatformFailureAt": null,
"lastTerminalAt": "2026-09-08T20:33:46.399743+00:00",
"lastTerminalAt": "2026-09-10T13:06:08.293453+00:00",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
@@ -4168,8 +4166,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 16,
"unresolvedFailureCount": 12
"total": 18,
"unresolvedFailureCount": 14
},
"Cambricon_mlu-370-x8|vllm|text-generation": {
"attributableFailureCount": 3,
@@ -4619,12 +4617,12 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 2
"ambiguous_runtime": 4
},
"failureCount": 2,
"failureCount": 4,
"failureRate": 1.0,
"framework": "vllm-mlu",
"lastTerminalAt": "2026-09-08T20:33:46.399718+00:00",
"lastTerminalAt": "2026-09-10T13:06:08.293453+00:00",
"modelType": "gemma2",
"pendingCount": 0,
"pendingRate": 0.0,
@@ -4634,8 +4632,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 2,
"unresolvedFailureCount": 2
"total": 4,
"unresolvedFailureCount": 4
},
"Cambricon_mlu-370-x8|vllm-mlu|text-generation|llama|compressed-tensors": {
"attributableFailureCount": 0,
@@ -5242,9 +5240,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 2
"ambiguous_runtime": 3
},
"failureCount": 2,
"failureCount": 3,
"failureRate": 1.0,
"framework": "vllm-mlu",
"loadSizeLog2Bucket": 32,
@@ -5257,8 +5255,32 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 2,
"unresolvedFailureCount": 2
"total": 3,
"unresolvedFailureCount": 3
},
"Cambricon_mlu-370-x8|vllm-mlu|text-generation|gemma2|compressed-tensors|33": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm-mlu",
"loadSizeLog2Bucket": 33,
"modelType": "gemma2",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Cambricon_mlu-370-x8|vllm-mlu|text-generation|llama|compressed-tensors|30": {
"attributableFailureCount": 0,
@@ -7031,15 +7053,15 @@
"unresolvedFailureCount": 0
}
},
"terminalRecords": 1586,
"totalRecords": 1673,
"terminalRecords": 1588,
"totalRecords": 1675,
"totals": {
"attributableFailureCount": 415,
"decisionFailureRate": 0.8906,
"decisionSuccessRate": 0.1094,
"decisionTotal": 466,
"failureBreakdown": {
"ambiguous_runtime": 346,
"ambiguous_runtime": 348,
"backend_operator": 35,
"framework_architecture_unsupported": 272,
"memory_capacity": 11,
@@ -7051,21 +7073,21 @@
"参数/模板问题": 98,
"验证失败": 673
},
"failureCount": 1535,
"failureRate": 0.9678,
"failureCount": 1537,
"failureRate": 0.9679,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 3,
"successCount": 51,
"successRate": 0.0322,
"total": 1586,
"unresolvedFailureCount": 1117
"successRate": 0.0321,
"total": 1588,
"unresolvedFailureCount": 1119
},
"warnings": [
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b3 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_mrv-100 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Vastai_va16 本地统计失败率偏高≥50%),建议重点关注。",
"GPU MetaX_c-500 本地统计失败率偏高≥50%),建议重点关注。",
@@ -7073,7 +7095,7 @@
"GPU hygon_k100-ai 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Mthreads_s4000 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。",
"组合 Iluvatar_mrv-100|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Biren_166m|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
@@ -7094,6 +7116,6 @@
]
},
"storageMode": "decision_state_only",
"summarizedRecords": 1673,
"summarizedRecords": 1675,
"version": 1
}

View File

@@ -259,6 +259,7 @@
{"failReason": "参数/模板问题", "framework": "vllm", "lastSyncTime": "2026-09-08T23:42:27.831489+00:00", "modelId": "empero-ai/Qwen3.8-9B-Distill", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-08T06:46:11.788912+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712827", "taskType": "text-generation", "verifyResult": null}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm", "lastSyncTime": "2026-09-08T21:18:07.599895+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "6db01b5037d5ed2db74875d2113f8582c97f34bd825637161e54174c9f2cabed", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683239637, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703969742, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703969742}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.787458+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712825", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm", "lastSyncTime": "2026-09-08T23:42:27.831458+00:00", "modelId": "callmezcc/Qwen3.8-27B-GPTQ-W4A16", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-08T06:46:11.785071+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712826", "taskType": "text-generation", "verifyResult": null}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-10T13:06:08.293453+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020563851, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020563851}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.601676+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712820", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["nemotron_h"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T20:33:46.399743+00:00", "modelId": "nota-ai/Nemotron-3.5-Lightning-30B-A3B-NVFP4-Global-Pruned-15", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 18974813860, "estimatedRequiredGiB": 21.229, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 18995666906, "modelscopeLicense": "other", "modelscopeParams": 15524066944, "modelscopeTags": ["license:other", "model_type:nemotron_h", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:nvidia", "custom_tag:pytorch", "custom_tag:nemotron-3.5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 18995666906}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.598902+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712818", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T20:33:46.399718+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3403624536, "estimatedRequiredGiB": 3.828, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 3425449433, "modelscopeLicense": "llama2", "modelscopeParams": 3204165888, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3425449433}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.594968+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712814", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-09T01:56:43.203825+00:00", "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.593738+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712817", "taskType": "text-generation", "verifyResult": -1}
@@ -266,6 +267,7 @@
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_mtp"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T18:40:55.416623+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-nvfp4", "modelProfile": {"architectures": [], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 238933307, "estimatedRequiredGiB": 0.297, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 265664321, "modelscopeLicense": "apache-2.0", "modelscopeParams": 106194432, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:nvfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 265664321}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.587867+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712816", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_mtp"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T23:49:45.832679+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-mxfp4", "modelProfile": {"architectures": [], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 225662258, "estimatedRequiredGiB": 0.282, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 252393272, "modelscopeLicense": "apache-2.0", "modelscopeParams": 79652352, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:mxfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 252393272}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.586261+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712822", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T16:13:44.313839+00:00", "modelId": "RedHatAI/starcoder2-7b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857298896, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860673720, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860673720}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.586014+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712812", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-10T13:06:08.293484+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367933, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367933}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.532675+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712815", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["rwkv7"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-09T20:13:30.044188+00:00", "modelId": "RWKV/RWKV7-1.5B-20260805", "modelProfile": {"architectures": ["Rwkv7ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3055418240, "estimatedRequiredGiB": 3.417, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "rwkv7", "modelscopeFileSize": 3057371034, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1527668736, "modelscopeTags": ["license:apache-2.0", "model_type:rwkv7", "library:safetensors", "task:text-generation", "custom_tag:rwkv", "custom_tag:rwkv7", "custom_tag:recurrent", "custom_tag:causal-lm", "custom_tag:conversational"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3057371034}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.531728+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712811", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T15:50:00.101683+00:00", "modelId": "RedHatAI/starcoder2-7b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7855165704, "estimatedRequiredGiB": 8.783, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7858540940, "modelscopeLicense": "other", "modelscopeParams": 7400416256, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7858540940}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:06.891127+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712805", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T15:41:33.598963+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:06.889991+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712799", "taskType": "text-generation", "verifyResult": -1}
@@ -296,5 +298,3 @@
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-08T01:27:33.503535+00:00", "modelId": "poolside/Laguna-XS-2.1-NVFP4", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T01:19:21+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4609109", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedArchitectures": ["NemotronHForCausalLM"], "framework": "", "lastSyncTime": "2026-09-08T01:18:10.323644+00:00", "modelId": "nv-community/NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T01:13:21+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4470644", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-08T01:18:10.323605+00:00", "modelId": "dickeyli/llmrec-onereason-8b-sh2-grpo4-step160", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T01:11:21+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4460690", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-08T00:41:43.539049+00:00", "modelId": "ysqlian/YZH_Model", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T00:35:21+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4481795", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm", "lastSyncTime": "2026-09-08T00:32:26.904386+00:00", "modelId": "mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T00:29:21+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4481796", "taskType": "text-generation", "verifyResult": -1}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -1,24 +1,24 @@
{
"agentVersion": "2026.09.04.4",
"checksums": {
".modelhub_state/architecture_compatibility_blacklist.json": "915bf473fac6e05c7becec282a99484f1cebf6c7bb27491f6d83dd5d6c38ca67",
".modelhub_state/architecture_history_backfill.json": "626112ac3aee0cadff964fc0e9af5a65cb4a59d373fe1802586c5e0f060169fb",
".modelhub_state/market_intelligence.json": "973172e953653c834a071b37991ef6b41c20464a375c73c5b5bc47b29d7a70d3",
".modelhub_state/official_capabilities.json": "b9fc60949355cc66ce6b1549fe1c3d54d22fc1b315decf1fe59bb93dec5686db",
".modelhub_state/outcome_checkpoint.json": "fa5a67accad87488c64fa25134439873f0f1c10648373fa47bc9db3941ca70a9",
".modelhub_state/architecture_compatibility_blacklist.json": "1a95a258c387bda97f4f0a14e3caaa688abc1bc1e1edf75e616255d2d757097a",
".modelhub_state/architecture_history_backfill.json": "851801a1a80c1071a4a03ed29a820d0ec466d121b5bff5498a81e5471723cf50",
".modelhub_state/market_intelligence.json": "bbb72166e998c7e756d4a82abab77d898494c3823790d4ea32cbac275ed06198",
".modelhub_state/official_capabilities.json": "da34ca42122fc1881944ecd9ec542ff59db886e86427a82b206c1368c15f5124",
".modelhub_state/outcome_checkpoint.json": "fa7339211afceed07207ee4ed475aba246ed10ead661593513e2faeabc92ac5b",
".modelhub_state/queue_cleanup_latest.json": "b9038630dc67feced29c6931cb43d6d94c28bb175e6dcf80fac9c625a52887d5",
".modelhub_state/recent_outcomes.jsonl": "edde5210dbc900b7a777583294e1f6bf9fbf1aadebd7f86b8df0e59991033546",
".modelhub_state/recovery_active_tasks.jsonl": "14a8bb5b06b67b3c9e31cf86bd7b22ce50cbd1f4f27cebb4430fe3cb46464642",
".modelhub_state/recovery_intents.jsonl": "7c99471204c53b66fe14699a6df06f315ba184397eb665ebff38faa65540a131",
".modelhub_state/recent_outcomes.jsonl": "234c4d6da4b2836721449edc47b8bac1ddc03584b8c74a4072d550d3752e6986",
".modelhub_state/recovery_active_tasks.jsonl": "bc17c5e93f49b4b4babe8e43dc4be8a96550ef1db8108b76fa2a3650d191c92e",
".modelhub_state/recovery_intents.jsonl": "a04a1934cf4ca394aff8ea5207ea3beef566d29f25acf7373bc4a39fcffb2741",
".modelhub_state/routing_intelligence.json": "f36cfaf9f9ff191164b0ce0a73a3ce4bbc3a2dc4589f252d583e33b427807fc0",
".modelhub_state/submission_exclusions.jsonl": "b2de1125bf1f3e0597cd50cc30b563a5bfee39459e843767163eb72101cebc03",
".modelhub_state/worker_crashes.jsonl": "d1c3f447ce0377ed5e820c2b5ff730d29f84a96af4fd14332b138f78c0ddde76",
"ledger/submissions.jsonl": "8a621da5637d481c7de4b4e3375f2a68cbdabf71fe459244a727034f36ab3eb9",
"outcomes/submissions.jsonl": "1d0e57a2f36689386249cde5821cd18163ee96d55d854cf03a46d503337b678b"
"outcomes/submissions.jsonl": "b1281959ac9ddf2c1932ccf3cdddbf7f81ec7fd1ad0539917e6d09fb9437a680"
},
"generation": 5172,
"generation": 5173,
"phase": "cycle",
"schemaVersion": 1,
"updatedAt": "2026-09-10T13:05:07.681334+00:00",
"updatedAt": "2026-09-10T13:07:42.286399+00:00",
"writerId": "58ff5dc2a2d64b119ad1e4560700f977"
}

View File

@@ -249,8 +249,6 @@
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614275+00:00", "modelId": "OpenBMB/MiniCPM5-2B", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5033557096, "estimatedRequiredGiB": 5.637, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5043805854, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5043805854}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:06.799047+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712803", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614470+00:00", "modelId": "RedHatAI/starcoder2-3b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3484485728, "estimatedRequiredGiB": 3.898, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 3487786133, "modelscopeLicense": "other", "modelscopeParams": 3181366272, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3487786133}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:06.808048+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712807", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614439+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293335560, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295942065, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295942065}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:06.810137+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712809", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614401+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367933, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367933}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:11.532675+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712815", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614371+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020563851, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020563851}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:11.601676+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712820", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614292+00:00", "modelId": "inferencerlabs/DeepSeek-V4-Flash-MTP-DSpark-MLX", "modelProfile": {"architectures": [], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 12994906286, "estimatedRequiredGiB": 14.53, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "deepseek_v4_dspark", "modelscopeFileSize": 13001289422, "modelscopeLicense": null, "modelscopeParams": 4276397927, "modelscopeTags": ["model_type:deepseek_v4_dspark", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13001289422}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:11.590454+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712813", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614454+00:00", "modelId": "nanbeige/Nanbeige4.2-3B-GPTQ-Int8", "modelProfile": {"architectures": ["NanbeigeForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5192364968, "estimatedRequiredGiB": 5.827, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nanbeige", "modelscopeFileSize": 5214325469, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4169800704, "modelscopeTags": ["license:apache-2.0", "model_type:nanbeige", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:llm", "custom_tag:nanbeige"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 5214325469}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:11.606547+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712821", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T14:58:38.614444+00:00", "modelId": "XHToken/Spark-X2.5-4B-FP8", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4450260320, "estimatedRequiredGiB": 4.991, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 4465562659, "modelscopeLicense": "apache-2.0", "modelscopeParams": null, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "fp8", "repositoryOnDiskBytes": 4465562659}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T06:46:11.792714+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712824", "taskType": "text-generation", "verifyResult": null}