state: generation 5158 (cycle)
This commit is contained in:
@@ -1313,7 +1313,7 @@
|
||||
"taskType": "text-generation"
|
||||
}
|
||||
},
|
||||
"generatedAt": "2026-09-10T12:22:24.663691+00:00",
|
||||
"generatedAt": "2026-09-10T12:25:24.097614+00:00",
|
||||
"summary": {
|
||||
"activeBlockCount": 66,
|
||||
"byGpuFramework": {
|
||||
|
||||
@@ -19,7 +19,7 @@
|
||||
"10": {
|
||||
"complete": false,
|
||||
"lastError": "ModelHubAPIError: 系统错误",
|
||||
"listingErrors": 366,
|
||||
"listingErrors": 367,
|
||||
"nextPage": 1,
|
||||
"recordsScanned": 0,
|
||||
"uniqueRecords": 0
|
||||
@@ -102,12 +102,12 @@
|
||||
"cutoffAt": "2026-09-04T03:55:51.365685+00:00",
|
||||
"failureLogsInspected": 0,
|
||||
"mode": "incremental_decision_only",
|
||||
"nextAccountIndex": 10,
|
||||
"nextAccountIndex": 11,
|
||||
"recordsScanned": 0,
|
||||
"seenTaskIds": [],
|
||||
"startedAt": "2026-09-04T03:55:51.365685+00:00",
|
||||
"terminalRecords": 0,
|
||||
"uniqueRecords": 0,
|
||||
"updatedAt": "2026-09-10T12:22:24.633883+00:00",
|
||||
"updatedAt": "2026-09-10T12:25:24.073876+00:00",
|
||||
"version": 1
|
||||
}
|
||||
|
||||
@@ -416,7 +416,7 @@
|
||||
}
|
||||
},
|
||||
"frameworkUpdatedAt": null,
|
||||
"generatedAt": "2026-09-10T12:20:49.640714+00:00",
|
||||
"generatedAt": "2026-09-10T12:23:33.507923+00:00",
|
||||
"gpuStats": {
|
||||
"Ascend_910-b3": {
|
||||
"available": true,
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"generatedAt": "2026-09-10T12:15:14.576026+00:00",
|
||||
"lastSyncTime": "2026-09-10T12:15:14.346623+00:00",
|
||||
"generatedAt": "2026-09-10T12:23:25.924304+00:00",
|
||||
"lastSyncTime": "2026-09-10T12:23:25.501745+00:00",
|
||||
"recentLimit": 300,
|
||||
"report": {
|
||||
"architectureCompatibilityBlocks": {
|
||||
@@ -1868,10 +1868,10 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 29,
|
||||
"ambiguous_runtime": 30,
|
||||
"参数/模板问题": 1
|
||||
},
|
||||
"failureCount": 30,
|
||||
"failureCount": 31,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"pendingCount": 0,
|
||||
@@ -1881,8 +1881,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 30,
|
||||
"unresolvedFailureCount": 30
|
||||
"total": 31,
|
||||
"unresolvedFailureCount": 31
|
||||
},
|
||||
"MetaX_c-500|unknown|text-generation": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -2145,12 +2145,12 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 44,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 52,
|
||||
"ambiguous_runtime": 55,
|
||||
"framework_architecture_unsupported": 38,
|
||||
"memory_capacity": 1,
|
||||
"model_load": 5
|
||||
},
|
||||
"failureCount": 96,
|
||||
"failureCount": 99,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"pendingCount": 0,
|
||||
@@ -2160,8 +2160,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Vastai_va16",
|
||||
"taskType": "text-generation",
|
||||
"total": 96,
|
||||
"unresolvedFailureCount": 52
|
||||
"total": 99,
|
||||
"unresolvedFailureCount": 55
|
||||
},
|
||||
"hygon_k100-ai|llamacpp|text-generation": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -2228,19 +2228,19 @@
|
||||
"unresolvedFailureCount": 8
|
||||
},
|
||||
"hygon_k100-ai|vllm|text-generation": {
|
||||
"attributableFailureCount": 35,
|
||||
"attributableFailureCount": 36,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 35,
|
||||
"decisionTotal": 36,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 15,
|
||||
"ambiguous_runtime": 16,
|
||||
"framework_architecture_unsupported": 27,
|
||||
"memory_capacity": 1,
|
||||
"model_load": 3,
|
||||
"repository_structure": 3,
|
||||
"runtime_memory": 1
|
||||
"runtime_memory": 2
|
||||
},
|
||||
"failureCount": 50,
|
||||
"failureCount": 52,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"pendingCount": 0,
|
||||
@@ -2250,8 +2250,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "hygon_k100-ai",
|
||||
"taskType": "text-generation",
|
||||
"total": 50,
|
||||
"unresolvedFailureCount": 15
|
||||
"total": 52,
|
||||
"unresolvedFailureCount": 16
|
||||
}
|
||||
},
|
||||
"frameworkSummaries": {
|
||||
@@ -2318,31 +2318,31 @@
|
||||
"unresolvedFailureCount": 844
|
||||
},
|
||||
"vllm": {
|
||||
"attributableFailureCount": 235,
|
||||
"attributableFailureCount": 236,
|
||||
"decisionFailureRate": 0.9958,
|
||||
"decisionSuccessRate": 0.0042,
|
||||
"decisionTotal": 236,
|
||||
"decisionTotal": 237,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 167,
|
||||
"ambiguous_runtime": 171,
|
||||
"backend_operator": 29,
|
||||
"framework_architecture_unsupported": 169,
|
||||
"memory_capacity": 6,
|
||||
"model_load": 18,
|
||||
"platform_infrastructure": 1,
|
||||
"repository_structure": 5,
|
||||
"runtime_memory": 5,
|
||||
"runtime_memory": 6,
|
||||
"tokenizer_compatibility": 3,
|
||||
"参数/模板问题": 24
|
||||
},
|
||||
"failureCount": 427,
|
||||
"failureCount": 432,
|
||||
"failureRate": 0.9977,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 1,
|
||||
"successCount": 1,
|
||||
"successRate": 0.0023,
|
||||
"total": 428,
|
||||
"unresolvedFailureCount": 191
|
||||
"total": 433,
|
||||
"unresolvedFailureCount": 195
|
||||
},
|
||||
"vllm-mlu": {
|
||||
"attributableFailureCount": 7,
|
||||
@@ -2409,22 +2409,22 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 37,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 30,
|
||||
"ambiguous_runtime": 31,
|
||||
"framework_architecture_unsupported": 4,
|
||||
"model_load": 1,
|
||||
"platform_infrastructure": 2,
|
||||
"tokenizer_compatibility": 32,
|
||||
"参数/模板问题": 7
|
||||
},
|
||||
"failureCount": 76,
|
||||
"failureCount": 77,
|
||||
"failureRate": 1.0,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 2,
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"total": 76,
|
||||
"unresolvedFailureCount": 37
|
||||
"total": 77,
|
||||
"unresolvedFailureCount": 38
|
||||
},
|
||||
"vllm_tokenizer_patch": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -2445,7 +2445,7 @@
|
||||
"unresolvedFailureCount": 1
|
||||
}
|
||||
},
|
||||
"generatedAt": "2026-09-10T12:15:14.571193+00:00",
|
||||
"generatedAt": "2026-09-10T12:23:25.917637+00:00",
|
||||
"gpuSummaries": {
|
||||
"Ascend_910-b3": {
|
||||
"attributableFailureCount": 26,
|
||||
@@ -2636,20 +2636,20 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 1,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 61,
|
||||
"ambiguous_runtime": 62,
|
||||
"memory_capacity": 1,
|
||||
"参数/模板问题": 1,
|
||||
"验证失败": 23
|
||||
},
|
||||
"failureCount": 86,
|
||||
"failureCount": 87,
|
||||
"failureRate": 1.0,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"total": 86,
|
||||
"unresolvedFailureCount": 85
|
||||
"total": 87,
|
||||
"unresolvedFailureCount": 86
|
||||
},
|
||||
"MetaX_c-500": {
|
||||
"attributableFailureCount": 42,
|
||||
@@ -2730,47 +2730,47 @@
|
||||
"decisionSuccessRate": 0.2167,
|
||||
"decisionTotal": 60,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 53,
|
||||
"ambiguous_runtime": 56,
|
||||
"framework_architecture_unsupported": 40,
|
||||
"memory_capacity": 1,
|
||||
"model_load": 6,
|
||||
"参数/模板问题": 13,
|
||||
"验证失败": 130
|
||||
},
|
||||
"failureCount": 243,
|
||||
"failureRate": 0.9492,
|
||||
"failureCount": 246,
|
||||
"failureRate": 0.9498,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"successCount": 13,
|
||||
"successRate": 0.0508,
|
||||
"total": 256,
|
||||
"unresolvedFailureCount": 196
|
||||
"successRate": 0.0502,
|
||||
"total": 259,
|
||||
"unresolvedFailureCount": 199
|
||||
},
|
||||
"hygon_k100-ai": {
|
||||
"attributableFailureCount": 38,
|
||||
"attributableFailureCount": 39,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 38,
|
||||
"decisionTotal": 39,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 17,
|
||||
"ambiguous_runtime": 18,
|
||||
"framework_architecture_unsupported": 30,
|
||||
"memory_capacity": 1,
|
||||
"model_load": 3,
|
||||
"repository_structure": 3,
|
||||
"runtime_memory": 1,
|
||||
"runtime_memory": 2,
|
||||
"参数/模板问题": 8,
|
||||
"验证失败": 27
|
||||
},
|
||||
"failureCount": 90,
|
||||
"failureCount": 92,
|
||||
"failureRate": 1.0,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"total": 90,
|
||||
"unresolvedFailureCount": 52
|
||||
"total": 92,
|
||||
"unresolvedFailureCount": 53
|
||||
}
|
||||
},
|
||||
"observedGpuMemoryGiB": {
|
||||
@@ -3276,9 +3276,9 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 1
|
||||
"ambiguous_runtime": 2
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureCount": 2,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"modelType": "gemma2",
|
||||
@@ -3290,8 +3290,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"unresolvedFailureCount": 1
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 2
|
||||
},
|
||||
"Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|llama|compressed-tensors": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -4309,9 +4309,9 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 14
|
||||
"ambiguous_runtime": 15
|
||||
},
|
||||
"failureCount": 14,
|
||||
"failureCount": 15,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"lastPlatformFailureAt": null,
|
||||
@@ -4323,8 +4323,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 14,
|
||||
"unresolvedFailureCount": 14
|
||||
"total": 15,
|
||||
"unresolvedFailureCount": 15
|
||||
},
|
||||
"MetaX_c-500|unknown|text-generation": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -4545,7 +4545,7 @@
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"lastPlatformFailureAt": null,
|
||||
"lastTerminalAt": "2026-09-10T10:12:25.704447+00:00",
|
||||
"lastTerminalAt": "2026-09-10T12:23:25.501701+00:00",
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
@@ -4556,31 +4556,6 @@
|
||||
"total": 20,
|
||||
"unresolvedFailureCount": 20
|
||||
},
|
||||
"hygon_k100-ai|llamacpp|text-generation": {
|
||||
"attributableFailureCount": 0,
|
||||
"consecutiveFailures": 0,
|
||||
"consecutivePlatformFailures": 0,
|
||||
"decisionFailureRate": 0.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 2
|
||||
},
|
||||
"failureCount": 2,
|
||||
"failureRate": 1.0,
|
||||
"framework": "llamacpp",
|
||||
"lastPlatformFailureAt": null,
|
||||
"lastTerminalAt": "2026-09-07T23:53:16.902630+00:00",
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "hygon_k100-ai",
|
||||
"taskType": "text-generation",
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 2
|
||||
},
|
||||
"hygon_k100-ai|vllm-patch-tokenizer|text-generation": {
|
||||
"attributableFailureCount": 1,
|
||||
"consecutiveFailures": 1,
|
||||
@@ -4614,16 +4589,17 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 7,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 12,
|
||||
"framework_architecture_unsupported": 5,
|
||||
"ambiguous_runtime": 13,
|
||||
"framework_architecture_unsupported": 4,
|
||||
"model_load": 1,
|
||||
"repository_structure": 1
|
||||
"repository_structure": 1,
|
||||
"runtime_memory": 1
|
||||
},
|
||||
"failureCount": 19,
|
||||
"failureCount": 20,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"lastPlatformFailureAt": null,
|
||||
"lastTerminalAt": "2026-09-10T10:21:30.006897+00:00",
|
||||
"lastTerminalAt": "2026-09-10T12:23:25.501677+00:00",
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
@@ -4631,8 +4607,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "hygon_k100-ai",
|
||||
"taskType": "text-generation",
|
||||
"total": 19,
|
||||
"unresolvedFailureCount": 12
|
||||
"total": 20,
|
||||
"unresolvedFailureCount": 13
|
||||
}
|
||||
},
|
||||
"recentProfileCombinationStats": {
|
||||
@@ -4911,6 +4887,31 @@
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|gemma2|compressed-tensors": {
|
||||
"attributableFailureCount": 0,
|
||||
"consecutiveFailures": 0,
|
||||
"decisionFailureRate": 0.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 1
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"lastTerminalAt": "2026-09-10T12:23:25.501745+00:00",
|
||||
"modelType": "gemma2",
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"quantizationMethod": "compressed-tensors",
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"unresolvedFailureCount": 1
|
||||
},
|
||||
"Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|llama|compressed-tensors": {
|
||||
"attributableFailureCount": 0,
|
||||
"consecutiveFailures": 0,
|
||||
@@ -5885,9 +5886,9 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 1
|
||||
"ambiguous_runtime": 2
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureCount": 2,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"loadSizeLog2Bucket": 33,
|
||||
@@ -5900,8 +5901,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"unresolvedFailureCount": 1
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 2
|
||||
},
|
||||
"Kunlunxin_p-800|vllm_fix_tokenizer|text-generation|llama|compressed-tensors|28": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -7030,41 +7031,41 @@
|
||||
"unresolvedFailureCount": 0
|
||||
}
|
||||
},
|
||||
"terminalRecords": 1577,
|
||||
"totalRecords": 1664,
|
||||
"terminalRecords": 1583,
|
||||
"totalRecords": 1670,
|
||||
"totals": {
|
||||
"attributableFailureCount": 413,
|
||||
"decisionFailureRate": 0.8901,
|
||||
"decisionSuccessRate": 0.1099,
|
||||
"decisionTotal": 464,
|
||||
"attributableFailureCount": 414,
|
||||
"decisionFailureRate": 0.8903,
|
||||
"decisionSuccessRate": 0.1097,
|
||||
"decisionTotal": 465,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 339,
|
||||
"ambiguous_runtime": 344,
|
||||
"backend_operator": 35,
|
||||
"framework_architecture_unsupported": 272,
|
||||
"memory_capacity": 11,
|
||||
"model_load": 47,
|
||||
"platform_infrastructure": 3,
|
||||
"repository_structure": 5,
|
||||
"runtime_memory": 5,
|
||||
"runtime_memory": 6,
|
||||
"tokenizer_compatibility": 38,
|
||||
"参数/模板问题": 98,
|
||||
"验证失败": 673
|
||||
},
|
||||
"failureCount": 1526,
|
||||
"failureRate": 0.9677,
|
||||
"failureCount": 1532,
|
||||
"failureRate": 0.9678,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 3,
|
||||
"successCount": 51,
|
||||
"successRate": 0.0323,
|
||||
"total": 1577,
|
||||
"unresolvedFailureCount": 1110
|
||||
"successRate": 0.0322,
|
||||
"total": 1583,
|
||||
"unresolvedFailureCount": 1115
|
||||
},
|
||||
"warnings": [
|
||||
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Ascend_910-b3 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Iluvatar_mrv-100 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Sunrise_pt-200-x1 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Iluvatar_bi-150 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Ascend_910-b4 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Vastai_va16 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU MetaX_c-500 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
@@ -7072,7 +7073,7 @@
|
||||
"GPU hygon_k100-ai 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Cambricon_mlu-370-x4 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Mthreads_s4000 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Iluvatar_bi-150 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"GPU Sunrise_pt-200-x1 本地统计失败率偏高(≥50%),建议重点关注。",
|
||||
"组合 Iluvatar_mrv-100|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
|
||||
"组合 Cambricon_mlu-370-x8|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
|
||||
"组合 Biren_166m|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
|
||||
@@ -7093,6 +7094,6 @@
|
||||
]
|
||||
},
|
||||
"storageMode": "decision_state_only",
|
||||
"summarizedRecords": 1664,
|
||||
"summarizedRecords": 1670,
|
||||
"version": 1
|
||||
}
|
||||
|
||||
@@ -1,3 +1,8 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-10T12:23:25.501701+00:00", "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-W4-Group128-A16-Test", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:23:21+00:00", "targetGpu": "Vastai_va16", "taskId": "4471971", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-10T12:23:25.501713+00:00", "modelId": "nm-testing/llama3-8b-w8_channel-a8_tensor-compressed", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:19:21+00:00", "targetGpu": "Vastai_va16", "taskId": "4471969", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "runtime_memory", "failureAction": "lower_memory_risk_or_reject_combination", "failureCategory": "runtime_memory", "failureClassificationReason": "structured_device_oom", "failureCode": "DEVICE_OOM", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu", "framework": "vllm", "lastSyncTime": "2026-09-10T12:23:25.501677+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:15:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4493612", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-10T12:23:25.501722+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:15:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4493610", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-10T12:23:25.501730+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8a8-dynamic-token-v2-asym", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:15:21+00:00", "targetGpu": "Vastai_va16", "taskId": "4471967", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "", "lastSyncTime": "2026-09-10T12:15:14.346623+00:00", "modelId": "mlx-community/Qwen3.6-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:09:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4463766", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "", "lastSyncTime": "2026-09-10T12:06:09.109223+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8a16-per-channel", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:05:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4523306", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-10T12:06:09.109198+00:00", "modelId": "ysqlian/YZH_Model", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-10T12:03:22+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4490278", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -158,6 +163,7 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T14:21:21.949441+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683239637, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703969742, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703969742}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.998598+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729524", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T13:57:00.114030+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-mxfp4", "modelProfile": {"architectures": [], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 225662258, "estimatedRequiredGiB": 0.282, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 252393272, "modelscopeLicense": "apache-2.0", "modelscopeParams": 79652352, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:mxfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 252393272}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.995104+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729526", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T20:13:30.044094+00:00", "modelId": "mlx-community/Qwen3.8-27B-MTP-nvfp4", "modelProfile": {"architectures": [], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 238933307, "estimatedRequiredGiB": 0.297, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_mtp", "modelscopeFileSize": 265664321, "modelscopeLicense": "apache-2.0", "modelscopeParams": 106194432, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_mtp", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-vlm", "custom_tag:qwen3_5_mtp", "custom_tag:qwen3.8", "custom_tag:qwen3.8-27b", "custom_tag:qwen", "custom_tag:mtp", "custom_tag:speculative-decoding", "custom_tag:draft-model", "custom_tag:nvfp4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 265664321}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.985375+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729522", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T12:23:25.501745+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020563851, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020563851}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.913002+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729521", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T16:50:57.939999+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_fix_tokenizer", "lastTerminalAt": "2026-09-07T06:07:37.712079+00:00", "modelType": "starcoder2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Kunlunxin_p-800", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 17788663591}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.889698+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729518", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T11:30:49.443166+00:00", "modelId": "RedHatAI/Phi-3-small-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3SmallForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8630134648, "estimatedRequiredGiB": 9.647, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3small", "modelscopeFileSize": 8632060880, "modelscopeLicense": "mit", "modelscopeParams": 7803314176, "modelscopeTags": ["license:mit", "model_type:phi3small", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8632060880}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.888188+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729519", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-10T02:58:51.546388+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293356304, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295851911, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295851911}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T23:02:51.847566+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729520", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -292,9 +298,3 @@
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm", "lastSyncTime": "2026-09-08T00:32:26.904386+00:00", "modelId": "mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T00:29:21+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4481796", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "", "lastSyncTime": "2026-09-08T00:22:46.993284+00:00", "modelId": "ysqlian/YZH_Model", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-08T00:13:21+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4482440", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm", "lastSyncTime": "2026-09-07T23:53:16.902664+00:00", "modelId": "mlx-community/Qwen3.5-35B-A3B-OptiQ-4bit-REAP-19B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:51:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4479115", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "llamacpp", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "llamacpp", "lastSyncTime": "2026-09-07T23:53:16.902630+00:00", "modelId": "ewinregirgojr/MiniCPM5-1B-Agentic-Tooluse-v3-GGUF", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:47:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "3990087", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": null, "framework": "", "lastSyncTime": "2026-09-07T23:53:16.902654+00:00", "modelId": "RedHatAI/Llama-2-7b-chat-quantized.w8a8", "modelProfile": {}, "outcome": "success", "status": "success", "submitTime": "2026-09-07T23:43:22+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4080853", "taskType": "text-generation", "verifyResult": 1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "vllm", "lastSyncTime": "2026-09-07T23:43:05.101200+00:00", "modelId": "mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:41:22+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4531788", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "", "lastSyncTime": "2026-09-07T23:43:05.101227+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:35:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4332996", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5_moe"], "framework": "", "lastSyncTime": "2026-09-07T23:33:45.147699+00:00", "modelId": "mlx-community/Qwen-AgentWorld-35B-A3B-oQ4", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:31:21+00:00", "targetGpu": "Iluvatar_mrv-100", "taskId": "4080039", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "missing_structured_error_code", "failureCode": null, "failureDetectedFramework": "llamacpp", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "llamacpp", "lastSyncTime": "2026-09-07T23:33:45.147733+00:00", "modelId": "kurakurai/Luth-LFM2-700M-GGUF", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-07T23:31:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "3986792", "taskType": "text-generation", "verifyResult": -1}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,24 +1,24 @@
|
||||
{
|
||||
"agentVersion": "2026.09.04.4",
|
||||
"checksums": {
|
||||
".modelhub_state/architecture_compatibility_blacklist.json": "6d75d019e604066ca4da96a36fc9b92c814eabc55921b4d941ded07f82b71b09",
|
||||
".modelhub_state/architecture_history_backfill.json": "e7a4ede4dfe4243c8fd4706df650acfd01b53828c995c94b867f7cf046014056",
|
||||
".modelhub_state/market_intelligence.json": "46248957e827a179881ac404e928773d73f7dd9a15f21d06e070d5a595b86cd7",
|
||||
".modelhub_state/official_capabilities.json": "e8e0c7e0de801023892c8dc901c9907f85d53865ba0ad906b9443ab1bf776a46",
|
||||
".modelhub_state/outcome_checkpoint.json": "19c82fc3ed0bbec6649a7b306f79c755b2d17a18aab1a8e388d84ca6ff383083",
|
||||
".modelhub_state/architecture_compatibility_blacklist.json": "6a9fa17b61d838dc7f9a71884c2387c4f4ce9a1e8d427b1d814d03970ec664a1",
|
||||
".modelhub_state/architecture_history_backfill.json": "ae3b888cc14ee4be068e3cdbb622a0555a949eadb78157357d4db71be95b163a",
|
||||
".modelhub_state/market_intelligence.json": "1e39e12eecc3b8887f0cd135cdb1888b682740ca95c563c40bb797a2612bee68",
|
||||
".modelhub_state/official_capabilities.json": "5d39f59a464781beba4ce539338afb0d6dbe5a41021265f64be668d35d39ca15",
|
||||
".modelhub_state/outcome_checkpoint.json": "e2d6d4e2e03343baface59172753e2107d9b1071c1edee6515e5cb8d0ee16554",
|
||||
".modelhub_state/queue_cleanup_latest.json": "b9038630dc67feced29c6931cb43d6d94c28bb175e6dcf80fac9c625a52887d5",
|
||||
".modelhub_state/recent_outcomes.jsonl": "45ba37ec5777429da69b691166827644a7b83eaa6ffe42c2d4244d2a73a78c6d",
|
||||
".modelhub_state/recovery_active_tasks.jsonl": "ded87b4b5919ac33f02c27882d2c639f5444c3dd923923434c0027093700c4d0",
|
||||
".modelhub_state/recovery_intents.jsonl": "9c8d17ad30f0963b11d9a5d4e81699a75f241c6e7015fd9c2ed0c36424773a39",
|
||||
".modelhub_state/recent_outcomes.jsonl": "5c42a62e2f1eb6eba9a917b8a42a911fe987bb994ca97d17c1f26f91f106ae09",
|
||||
".modelhub_state/recovery_active_tasks.jsonl": "d043fadc304d6c14a059ba4ea366d86d43b04ccad91f0e7f7df05e40157f9a82",
|
||||
".modelhub_state/recovery_intents.jsonl": "d5b10c98f11d1259d12e879a563560a7c1bbdb454505e4deac13cfd934572884",
|
||||
".modelhub_state/routing_intelligence.json": "f36cfaf9f9ff191164b0ce0a73a3ce4bbc3a2dc4589f252d583e33b427807fc0",
|
||||
".modelhub_state/submission_exclusions.jsonl": "b2de1125bf1f3e0597cd50cc30b563a5bfee39459e843767163eb72101cebc03",
|
||||
".modelhub_state/worker_crashes.jsonl": "d1c3f447ce0377ed5e820c2b5ff730d29f84a96af4fd14332b138f78c0ddde76",
|
||||
"ledger/submissions.jsonl": "8a621da5637d481c7de4b4e3375f2a68cbdabf71fe459244a727034f36ab3eb9",
|
||||
"outcomes/submissions.jsonl": "3dc7b0773d134ec492c2a72ee218f706b6cf17ad01cb3d1e32a072f37b533130"
|
||||
"outcomes/submissions.jsonl": "249bd947bbece33d1e7d20b546c76ca577966a52b7f8d45fcbf882be0353517a"
|
||||
},
|
||||
"generation": 5157,
|
||||
"generation": 5158,
|
||||
"phase": "cycle",
|
||||
"schemaVersion": 1,
|
||||
"updatedAt": "2026-09-10T12:22:24.807819+00:00",
|
||||
"updatedAt": "2026-09-10T12:25:24.888853+00:00",
|
||||
"writerId": "58ff5dc2a2d64b119ad1e4560700f977"
|
||||
}
|
||||
|
||||
@@ -391,7 +391,6 @@
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504496+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8818103892, "estimatedRequiredGiB": 9.883, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8842796317, "modelscopeLicense": "mit", "modelscopeParams": 6728625904, "modelscopeTags": ["license:mit", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 8842796317}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.831733+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729515", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504512+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293335560, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295942065, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295942065}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.830682+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729514", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504565+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367933, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367933}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.824770+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729512", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504647+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020563851, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020563851}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.913002+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729521", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504528+00:00", "modelId": "siliconflow/Hunyuan-A13B-Instruct-SF-FP8", "modelProfile": {"architectures": ["HunYuanMoEV1ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1700, "estimatedRequiredGiB": 90.469, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "hunyuan", "modelscopeFileSize": 80949912827, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 80393195968, "modelscopeTags": ["license:Apache License 2.0", "model_type:hunyuan", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 80949912827}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.924564+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729534", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504434+00:00", "modelId": "cyankiwi/Ornith-1.5-35B-A3B-AWQ-INT4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26119861184, "estimatedRequiredGiB": 29.243, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26165778261, "modelscopeLicense": "mit", "modelscopeParams": 35951822704, "modelscopeTags": ["license:mit", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26165778261}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.933574+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729525", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-09T07:05:13.504558+00:00", "modelId": "empero-ai/Qwen3.8-9B-Distill", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 19306305296, "estimatedRequiredGiB": 21.602, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 19329225494, "modelscopeLicense": "apache-2.0", "modelscopeParams": 9653104368, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:empero-ai", "custom_tag:qwen3.5", "custom_tag:qwen3.8", "custom_tag:distillation", "custom_tag:reasoning", "custom_tag:function-calling", "custom_tag:sft"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 19329225494}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-08T23:02:51.939534+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4729523", "taskType": "text-generation", "verifyResult": null}
|
||||
|
||||
Reference in New Issue
Block a user