state: generation 4946 (cycle)

This commit is contained in:
2026-09-10 04:20:31 +00:00
parent 7ba2732fe6
commit 81a4177086
10 changed files with 2001 additions and 2007 deletions

View File

@@ -1325,7 +1325,7 @@
"taskType": "text-generation"
}
},
"generatedAt": "2026-09-10T04:16:39.764303+00:00",
"generatedAt": "2026-09-10T04:20:31.211083+00:00",
"summary": {
"activeBlockCount": 66,
"byGpuFramework": {

View File

@@ -51,7 +51,7 @@
"4": {
"complete": false,
"lastError": "ModelHubAPIError: 系统错误",
"listingErrors": 352,
"listingErrors": 353,
"nextPage": 1,
"recordsScanned": 0,
"uniqueRecords": 0
@@ -102,12 +102,12 @@
"cutoffAt": "2026-09-04T03:55:51.365685+00:00",
"failureLogsInspected": 0,
"mode": "incremental_decision_only",
"nextAccountIndex": 4,
"nextAccountIndex": 5,
"recordsScanned": 0,
"seenTaskIds": [],
"startedAt": "2026-09-04T03:55:51.365685+00:00",
"terminalRecords": 0,
"uniqueRecords": 0,
"updatedAt": "2026-09-10T04:16:39.739518+00:00",
"updatedAt": "2026-09-10T04:20:31.184978+00:00",
"version": 1
}

View File

@@ -416,7 +416,7 @@
}
},
"frameworkUpdatedAt": null,
"generatedAt": "2026-09-10T04:14:52.948890+00:00",
"generatedAt": "2026-09-10T04:17:49.082533+00:00",
"gpuStats": {
"Ascend_910-b3": {
"available": true,

File diff suppressed because it is too large Load Diff

View File

@@ -1,6 +1,6 @@
{
"generatedAt": "2026-09-10T04:09:30.753219+00:00",
"lastSyncTime": "2026-09-10T04:09:30.723646+00:00",
"generatedAt": "2026-09-10T04:17:40.484591+00:00",
"lastSyncTime": "2026-09-10T04:17:40.427917+00:00",
"recentLimit": 300,
"report": {
"architectureCompatibilityBlocks": {
@@ -1612,9 +1612,9 @@
"failureBreakdown": {
"ambiguous_runtime": 13,
"framework_architecture_unsupported": 7,
"参数/模板问题": 2
"参数/模板问题": 3
},
"failureCount": 22,
"failureCount": 23,
"failureRate": 1.0,
"framework": "vllm-mlu",
"pendingCount": 0,
@@ -1624,8 +1624,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 22,
"unresolvedFailureCount": 15
"total": 23,
"unresolvedFailureCount": 16
},
"Cambricon_mlu-370-x8|vllm|text-generation": {
"attributableFailureCount": 3,
@@ -2362,17 +2362,17 @@
"failureBreakdown": {
"ambiguous_runtime": 13,
"framework_architecture_unsupported": 7,
"参数/模板问题": 2
"参数/模板问题": 3
},
"failureCount": 22,
"failureCount": 23,
"failureRate": 1.0,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 0,
"successRate": 0.0,
"total": 22,
"unresolvedFailureCount": 15
"total": 23,
"unresolvedFailureCount": 16
},
"vllm-patch-tokenizer": {
"attributableFailureCount": 3,
@@ -2455,7 +2455,7 @@
"unresolvedFailureCount": 1
}
},
"generatedAt": "2026-09-10T04:09:30.749128+00:00",
"generatedAt": "2026-09-10T04:17:40.458056+00:00",
"gpuSummaries": {
"Ascend_910-b3": {
"attributableFailureCount": 26,
@@ -2561,18 +2561,18 @@
"framework_architecture_unsupported": 15,
"memory_capacity": 1,
"tokenizer_compatibility": 1,
"参数/模板问题": 17,
"参数/模板问题": 18,
"验证失败": 22
},
"failureCount": 83,
"failureRate": 0.9765,
"failureCount": 84,
"failureRate": 0.9767,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"successCount": 2,
"successRate": 0.0235,
"total": 85,
"unresolvedFailureCount": 66
"successRate": 0.0233,
"total": 86,
"unresolvedFailureCount": 67
},
"Iluvatar_bi-100": {
"attributableFailureCount": 0,
@@ -4118,9 +4118,10 @@
"decisionTotal": 4,
"failureBreakdown": {
"ambiguous_runtime": 10,
"framework_architecture_unsupported": 4
"framework_architecture_unsupported": 4,
"参数/模板问题": 1
},
"failureCount": 14,
"failureCount": 15,
"failureRate": 1.0,
"framework": "vllm-mlu",
"lastPlatformFailureAt": null,
@@ -4132,8 +4133,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 14,
"unresolvedFailureCount": 10
"total": 15,
"unresolvedFailureCount": 11
},
"Cambricon_mlu-370-x8|vllm|text-generation": {
"attributableFailureCount": 3,
@@ -6833,8 +6834,8 @@
"unresolvedFailureCount": 0
}
},
"terminalRecords": 1495,
"totalRecords": 1582,
"terminalRecords": 1496,
"totalRecords": 1583,
"totals": {
"attributableFailureCount": 398,
"decisionFailureRate": 0.8904,
@@ -6850,24 +6851,24 @@
"repository_structure": 5,
"runtime_memory": 5,
"tokenizer_compatibility": 33,
"参数/模板问题": 94,
"参数/模板问题": 95,
"验证失败": 673
},
"failureCount": 1446,
"failureCount": 1447,
"failureRate": 0.9672,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 2,
"successCount": 49,
"successRate": 0.0328,
"total": 1495,
"unresolvedFailureCount": 1046
"total": 1496,
"unresolvedFailureCount": 1047
},
"warnings": [
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b3 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_mrv-100 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Vastai_va16 本地统计失败率偏高≥50%),建议重点关注。",
"GPU MetaX_c-500 本地统计失败率偏高≥50%),建议重点关注。",
@@ -6875,7 +6876,7 @@
"GPU hygon_k100-ai 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Mthreads_s4000 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"组合 Iluvatar_mrv-100|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Biren_166m|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
@@ -6896,6 +6897,6 @@
]
},
"storageMode": "decision_state_only",
"summarizedRecords": 1582,
"summarizedRecords": 1583,
"version": 1
}

View File

@@ -190,6 +190,7 @@
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["nemotron_h"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T20:33:46.399743+00:00", "modelId": "nota-ai/Nemotron-3.5-Lightning-30B-A3B-NVFP4-Global-Pruned-15", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 18974813860, "estimatedRequiredGiB": 21.229, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 18995666906, "modelscopeLicense": "other", "modelscopeParams": 15524066944, "modelscopeTags": ["license:other", "model_type:nemotron_h", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:nvidia", "custom_tag:pytorch", "custom_tag:nemotron-3.5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 18995666906}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.598902+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712818", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T20:33:46.399718+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3403624536, "estimatedRequiredGiB": 3.828, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 3425449433, "modelscopeLicense": "llama2", "modelscopeParams": 3204165888, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3425449433}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.594968+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712814", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-09T01:56:43.203825+00:00", "modelId": "RedHatAI/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615358, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615358}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.593738+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712817", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "参数/模板问题", "framework": "vllm-mlu", "lastSyncTime": "2026-09-10T04:17:40.427917+00:00", "modelId": "RWKV/RWKV7-7.2B-20260805", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-08T06:46:11.589203+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712819", "taskType": "text-generation", "verifyResult": null}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_mtp"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T18:40:55.416623+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-nvfp4", "modelProfile": {"architectures": [], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 238933307, "estimatedRequiredGiB": 0.297, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 265664321, "modelscopeLicense": "apache-2.0", "modelscopeParams": 106194432, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:nvfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 265664321}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.587867+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712816", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_mtp"], "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T23:49:45.832679+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-mxfp4", "modelProfile": {"architectures": [], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 225662258, "estimatedRequiredGiB": 0.282, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 252393272, "modelscopeLicense": "apache-2.0", "modelscopeParams": 79652352, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:mxfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 252393272}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.586261+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712822", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-mlu", "lastSyncTime": "2026-09-08T16:13:44.313839+00:00", "modelId": "RedHatAI/starcoder2-7b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857298896, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860673720, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860673720}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T06:46:11.586014+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4712812", "taskType": "text-generation", "verifyResult": -1}
@@ -297,4 +298,3 @@
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-07T09:00:01.243313+00:00", "modelId": "dickeyli/llmrec-onereason-8b-sh2-grpo4-step160", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-07T08:55:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4460896", "taskType": "text-generation", "verifyResult": 1}
{"failReason": "参数/模板问题", "framework": "", "lastSyncTime": "2026-09-07T09:00:01.243276+00:00", "modelId": "nightmedia/Mistral-Nemo-Instruct-2407-12B-Thinking-HI-Claude-Opus-High-Reasoning-mxfp4-mlx", "modelProfile": {}, "outcome": "failed", "status": "failed", "submitTime": "2026-09-07T08:43:54+00:00", "targetGpu": "Vastai_va16", "taskId": "4079294", "taskType": "text-generation", "verifyResult": null}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-07T08:32:22.399378+00:00", "modelId": "microsoft/Dayhoff-3b-GR-HM-c", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T08:31:21+00:00", "targetGpu": "Vastai_va16", "taskId": "4079222", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-07T08:32:22.399402+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a8", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T08:09:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4083445", "taskType": "text-generation", "verifyResult": -1}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff