state: generation 13301 (cycle)

This commit is contained in:
2026-09-23 02:39:48 +00:00
parent d500725e5f
commit cfe96b026f
10 changed files with 1991 additions and 1998 deletions

View File

@@ -3044,7 +3044,7 @@
"taskType": "text-generation"
}
},
"generatedAt": "2026-09-23T02:35:36.844851+00:00",
"generatedAt": "2026-09-23T02:39:33.342313+00:00",
"summary": {
"activeBlockCount": 153,
"byGpuFramework": {

View File

@@ -396,7 +396,7 @@
}
},
"frameworkUpdatedAt": null,
"generatedAt": "2026-09-23T02:38:28.451173+00:00",
"generatedAt": "2026-09-23T02:39:47.600785+00:00",
"gpuStats": {
"Ascend_910-b4": {
"available": true,

View File

@@ -1,5 +1,5 @@
{
"catalogUpdatedAt": "2026-09-23T02:38:28.451173+00:00",
"catalogUpdatedAt": "2026-09-23T02:39:47.600785+00:00",
"configuredTaskTypes": [
"text-generation"
],
@@ -56,7 +56,7 @@
"time-series-forecasting"
],
"errors": [],
"generatedAt": "2026-09-23T02:38:28.451173+00:00",
"generatedAt": "2026-09-23T02:39:47.600785+00:00",
"gpuCatalog": {
"Ascend_910-b3": {
"canVerify": true,
@@ -6693,6 +6693,6 @@
"updateTime": "2025-12-22 08:59:53"
}
],
"taskTreeUpdatedAt": "2026-09-23T02:38:28.451173+00:00",
"taskTreeUpdatedAt": "2026-09-23T02:39:47.600785+00:00",
"version": 1
}

View File

@@ -1,6 +1,6 @@
{
"generatedAt": "2026-09-23T02:35:36.757931+00:00",
"lastSyncTime": "2026-09-23T02:35:34.755698+00:00",
"generatedAt": "2026-09-23T02:39:33.263628+00:00",
"lastSyncTime": "2026-09-23T02:39:29.759281+00:00",
"recentLimit": 300,
"report": {
"architectureCompatibilityBlocks": {
@@ -3481,10 +3481,10 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 1,
"failureBreakdown": {
"ambiguous_runtime": 106,
"ambiguous_runtime": 107,
"memory_capacity": 1
},
"failureCount": 107,
"failureCount": 108,
"failureRate": 1.0,
"framework": "vllm_fix_tokenizer",
"pendingCount": 0,
@@ -3494,8 +3494,8 @@
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 107,
"unresolvedFailureCount": 106
"total": 108,
"unresolvedFailureCount": 107
},
"Biren_166m|vllm|text-generation": {
"attributableFailureCount": 44,
@@ -3700,14 +3700,14 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 12,
"failureBreakdown": {
"ambiguous_runtime": 34,
"ambiguous_runtime": 35,
"framework_architecture_unsupported": 8,
"model_load": 3,
"platform_infrastructure": 2,
"tokenizer_compatibility": 1,
"参数/模板问题": 5
},
"failureCount": 53,
"failureCount": 54,
"failureRate": 1.0,
"framework": "vllm-customized",
"pendingCount": 0,
@@ -3717,8 +3717,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 53,
"unresolvedFailureCount": 39
"total": 54,
"unresolvedFailureCount": 40
},
"Cambricon_mlu-370-x8|vllm-mlu|text-generation": {
"attributableFailureCount": 14,
@@ -5747,22 +5747,22 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 12,
"failureBreakdown": {
"ambiguous_runtime": 34,
"ambiguous_runtime": 35,
"framework_architecture_unsupported": 8,
"model_load": 3,
"platform_infrastructure": 2,
"tokenizer_compatibility": 1,
"参数/模板问题": 5
},
"failureCount": 53,
"failureCount": 54,
"failureRate": 1.0,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 2,
"successCount": 0,
"successRate": 0.0,
"total": 53,
"unresolvedFailureCount": 39
"total": 54,
"unresolvedFailureCount": 40
},
"vllm-mlu": {
"attributableFailureCount": 25,
@@ -5840,7 +5840,7 @@
"decisionSuccessRate": 0.0357,
"decisionTotal": 140,
"failureBreakdown": {
"ambiguous_runtime": 299,
"ambiguous_runtime": 300,
"backend_operator": 9,
"framework_architecture_unsupported": 29,
"memory_capacity": 10,
@@ -5849,15 +5849,15 @@
"tokenizer_compatibility": 82,
"参数/模板问题": 13
},
"failureCount": 466,
"failureCount": 467,
"failureRate": 0.9894,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 19,
"successCount": 5,
"successRate": 0.0106,
"total": 471,
"unresolvedFailureCount": 312
"total": 472,
"unresolvedFailureCount": 313
},
"vllm_tokenizer_patch": {
"attributableFailureCount": 107,
@@ -5888,7 +5888,7 @@
"unresolvedFailureCount": 175
}
},
"generatedAt": "2026-09-23T02:35:36.744395+00:00",
"generatedAt": "2026-09-23T02:39:33.251720+00:00",
"gpuSummaries": {
"Ascend_910-b3": {
"attributableFailureCount": 102,
@@ -5950,7 +5950,7 @@
"decisionSuccessRate": 0.1179,
"decisionTotal": 212,
"failureBreakdown": {
"ambiguous_runtime": 249,
"ambiguous_runtime": 250,
"backend_operator": 4,
"context_length": 10,
"framework_architecture_unsupported": 120,
@@ -5964,15 +5964,15 @@
"日志缺失": 62,
"验证失败": 27
},
"failureCount": 741,
"failureCount": 742,
"failureRate": 0.9674,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 2,
"successCount": 25,
"successRate": 0.0326,
"total": 766,
"unresolvedFailureCount": 552
"total": 767,
"unresolvedFailureCount": 553
},
"Cambricon_mlu-370-x4": {
"attributableFailureCount": 723,
@@ -6009,7 +6009,7 @@
"decisionSuccessRate": 0.1892,
"decisionTotal": 74,
"failureBreakdown": {
"ambiguous_runtime": 141,
"ambiguous_runtime": 142,
"framework_architecture_unsupported": 48,
"memory_capacity": 4,
"model_load": 5,
@@ -6018,15 +6018,15 @@
"参数/模板问题": 41,
"验证失败": 22
},
"failureCount": 266,
"failureRate": 0.95,
"failureCount": 267,
"failureRate": 0.9502,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 2,
"successCount": 14,
"successRate": 0.05,
"total": 280,
"unresolvedFailureCount": 204
"successRate": 0.0498,
"total": 281,
"unresolvedFailureCount": 205
},
"Iluvatar_bi-100": {
"attributableFailureCount": 194,
@@ -9230,9 +9230,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 11
"ambiguous_runtime": 12
},
"failureCount": 11,
"failureCount": 12,
"failureRate": 1.0,
"framework": "vllm_fix_tokenizer",
"modelType": "qwen3_5",
@@ -9244,8 +9244,8 @@
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 11,
"unresolvedFailureCount": 11
"total": 12,
"unresolvedFailureCount": 12
},
"Biren_166m|vllm_fix_tokenizer|text-generation|qwen3|compressed-tensors": {
"attributableFailureCount": 0,
@@ -11188,9 +11188,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
"ambiguous_runtime": 2
},
"failureCount": 1,
"failureCount": 2,
"failureRate": 1.0,
"framework": "vllm-customized",
"modelType": "phi3",
@@ -11202,8 +11202,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
"total": 2,
"unresolvedFailureCount": 2
},
"Cambricon_mlu-370-x8|vllm-customized|text-generation|qwen2|compressed-tensors": {
"attributableFailureCount": 0,
@@ -20888,9 +20888,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 15
"ambiguous_runtime": 13
},
"failureCount": 15,
"failureCount": 13,
"failureRate": 1.0,
"framework": "transformers",
"lastPlatformFailureAt": null,
@@ -20902,8 +20902,8 @@
"successRate": 0.0,
"targetGpu": "Iluvatar_bi-100",
"taskType": "text-generation",
"total": 15,
"unresolvedFailureCount": 15
"total": 13,
"unresolvedFailureCount": 13
},
"Iluvatar_bi-150|transformers|text-generation": {
"attributableFailureCount": 5,
@@ -22230,9 +22230,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 6
"ambiguous_runtime": 7
},
"failureCount": 6,
"failureCount": 7,
"failureRate": 1.0,
"framework": "vllm_fix_tokenizer",
"lastTerminalAt": "2026-09-22T16:15:27.384896+00:00",
@@ -22245,8 +22245,8 @@
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 6,
"unresolvedFailureCount": 6
"total": 7,
"unresolvedFailureCount": 7
},
"Biren_166m|vllm_fix_tokenizer|text-generation|qwen3|compressed-tensors": {
"attributableFailureCount": 0,
@@ -22755,12 +22755,12 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
"ambiguous_runtime": 2
},
"failureCount": 1,
"failureCount": 2,
"failureRate": 1.0,
"framework": "vllm-customized",
"lastTerminalAt": "2026-09-22T13:13:03.464336+00:00",
"lastTerminalAt": "2026-09-23T02:39:29.759268+00:00",
"modelType": "phi3",
"pendingCount": 0,
"pendingRate": 0.0,
@@ -22770,8 +22770,8 @@
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
"total": 2,
"unresolvedFailureCount": 2
},
"Cambricon_mlu-370-x8|vllm-customized|text-generation|qwen2|compressed-tensors": {
"attributableFailureCount": 0,
@@ -23080,9 +23080,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 3
"ambiguous_runtime": 2
},
"failureCount": 3,
"failureCount": 2,
"failureRate": 1.0,
"framework": "transformers",
"lastTerminalAt": "2026-09-21T05:42:03.443575+00:00",
@@ -23095,33 +23095,8 @@
"successRate": 0.0,
"targetGpu": "Iluvatar_bi-100",
"taskType": "text-generation",
"total": 3,
"unresolvedFailureCount": 3
},
"Iluvatar_bi-100|transformers|text-generation|qwen3_vl|none": {
"attributableFailureCount": 0,
"consecutiveFailures": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "transformers",
"lastTerminalAt": "2026-09-21T05:14:26.178418+00:00",
"modelType": "qwen3_vl",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "none",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Iluvatar_bi-100",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
"total": 2,
"unresolvedFailureCount": 2
},
"Iluvatar_bi-100|transformers|text-generation|sophia_hybrid|none": {
"attributableFailureCount": 0,
@@ -29046,9 +29021,9 @@
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 6
"ambiguous_runtime": 7
},
"failureCount": 6,
"failureCount": 7,
"failureRate": 1.0,
"framework": "vllm_fix_tokenizer",
"loadSizeLog2Bucket": 34,
@@ -29061,8 +29036,8 @@
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 6,
"unresolvedFailureCount": 6
"total": 7,
"unresolvedFailureCount": 7
},
"Biren_166m|vllm_fix_tokenizer|text-generation|qwen3_5|none|35": {
"attributableFailureCount": 0,
@@ -31888,6 +31863,30 @@
"total": 1,
"unresolvedFailureCount": 0
},
"Cambricon_mlu-370-x8|vllm-customized|text-generation|phi3|compressed-tensors|31": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm-customized",
"loadSizeLog2Bucket": 31,
"modelType": "phi3",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Cambricon_mlu-370-x8",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Cambricon_mlu-370-x8|vllm-customized|text-generation|phi3|compressed-tensors|32": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
@@ -45826,15 +45825,15 @@
"unresolvedFailureCount": 0
}
},
"terminalRecords": 16796,
"totalRecords": 17014,
"terminalRecords": 16798,
"totalRecords": 17016,
"totals": {
"attributableFailureCount": 6013,
"decisionFailureRate": 0.8622,
"decisionSuccessRate": 0.1378,
"decisionTotal": 6974,
"failureBreakdown": {
"ambiguous_runtime": 4283,
"ambiguous_runtime": 4285,
"architecture_compatibility": 212,
"attention_backend": 3,
"backend_operator": 115,
@@ -45850,20 +45849,19 @@
"日志缺失": 719,
"验证失败": 676
},
"failureCount": 15835,
"failureCount": 15837,
"failureRate": 0.9428,
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 931,
"successCount": 961,
"successRate": 0.0572,
"total": 16796,
"unresolvedFailureCount": 8891
"total": 16798,
"unresolvedFailureCount": 8893
},
"warnings": [
"GPU Iluvatar_mrv-100 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b3 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Biren_166m 本地统计失败率偏高≥50%),建议重点关注。",
"GPU MetaX_c-500 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x4 本地统计失败率偏高≥50%),建议重点关注。",
@@ -45872,8 +45870,9 @@
"GPU hygon_k100-ai 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Vastai_va16 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Mthreads_s4000 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Biren_166m 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-100 本地统计失败率偏高≥50%),建议重点关注。",
"组合 Iluvatar_bi-150|vllm_tokenizer_patch|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Sunrise_pt-200-x1|vllm|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。",
@@ -45897,6 +45896,7 @@
"组合 Ascend_910-b4|llamacpp|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 MetaX_c-500|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Iluvatar_bi-150|vllm_0_17_0_corex_4_4_0|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Ascend_910-b4|vllm_tokenizer_patch|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Sunrise_pt-200-x1|vllm_fix_tokenizer|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Sunrise_pt-200-x1|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
@@ -45910,11 +45910,10 @@
"组合 Mthreads_s4000|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 MetaX_c-500|vllm|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 hygon_k100-ai|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Ascend_910-b3|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。"
"组合 hygon_k100-ai|vllm|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。"
]
},
"storageMode": "decision_state_only",
"summarizedRecords": 17014,
"summarizedRecords": 17016,
"version": 1
}

View File

@@ -48,6 +48,7 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm_fix_tokenizer", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-22T01:15:43.966491+00:00", "modelId": "OpenBMB/MiniCPM5-2B-MLX", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1416035216, "estimatedRequiredGiB": 1.594, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1426233716, "modelscopeLicense": "apache-2.0", "modelscopeParams": 393390080, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1426233716}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T17:03:32.240792+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "5004389", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "", "lastSyncTime": "2026-09-21T16:50:30.559368+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:39:21+00:00", "targetGpu": "Biren_166m", "taskId": "4610379", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm", "lastSyncTime": "2026-09-21T16:50:30.559274+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:37:21+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4332638", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-23T02:39:29.759268+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4020349648, "estimatedRequiredGiB": 4.496, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4022916227, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4022916227}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:14:31.701642+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5003541", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-22T11:22:37.063897+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093203965, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm", "custom_tag:quantized", "custom_tag:8-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 3}, "failureCount": 3, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T15:02:24.876374+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 3, "unresolvedFailureCount": 3}, "repositoryOnDiskBytes": 9093203965}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:14:19.660180+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5003580", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-22T00:26:31.156547+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-5bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "3b6ff6a21da290a9b1ca3ac432aa93c3d06a816b4b694cc9a45180f81f107aa5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 804816628, "estimatedRequiredGiB": 0.905, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 809686744, "modelscopeLicense": "other", "modelscopeParams": 219544320, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 1, "consecutiveFailures": 1, "decisionFailureRate": 1.0, "decisionSuccessRate": 0.0, "decisionTotal": 1, "failureBreakdown": {"model_load": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-21T04:36:00.569307+00:00", "modelType": "lfm2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Iluvatar_bi-150", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 0}, "repositoryOnDiskBytes": 809686744}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:14:19.468977+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "5003540", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "transformers", "lastSyncTime": "2026-09-22T00:17:02.854932+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248516007, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 4}, "failureCount": 4, "failureRate": 1.0, "framework": "transformers", "lastTerminalAt": "2026-09-20T17:24:32.660406+00:00", "modelType": "lfm2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Iluvatar_bi-150", "taskType": "text-generation", "total": 4, "unresolvedFailureCount": 4}, "repositoryOnDiskBytes": 1248516007}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T16:04:25.561880+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "5003413", "taskType": "text-generation", "verifyResult": -1}
@@ -109,6 +110,7 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-22T01:48:12.547533+00:00", "modelId": "XHToken/Spark-X2.5-1.7B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3415340496, "estimatedRequiredGiB": 3.834, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 3430847031, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1707657216, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 3430847031}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T09:10:32.259004+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4997718", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "backend_operator", "failureAction": "prefer_other_proven_framework_or_gpu", "failureCategory": "backend_operator", "failureClassificationReason": "backend_version_dependent", "failureCode": "MISSING_OPERATOR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "gpu_framework", "framework": "vllm_0_17_0_corex_4_4_0", "lastSyncTime": "2026-09-21T18:39:29.750819+00:00", "modelId": "RedHatAI/starcoder2-15b-FP8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1cc16a7df03cf7b1485e2fda7e5c05e286a546d04e80bdd595c015004b3852a8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16564748304, "estimatedRequiredGiB": 18.516, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 16568142065, "modelscopeLicense": "other", "modelscopeParams": 15957889024, "modelscopeTags": ["license:other", "model_type:starcoder2", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_0_17_0_corex_4_4_0", "lastTerminalAt": "2026-09-20T12:58:28.353096+00:00", "modelType": "starcoder2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Iluvatar_bi-150", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 16568142065}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T09:05:59.142974+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4997663", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-22T21:35:04.154893+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8a8-dynamic-token-v2", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1231252716, "estimatedRequiredGiB": 1.378, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1233101396, "modelscopeLicense": null, "modelscopeParams": 1100048384, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 1233101396}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T08:46:11.874749+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4997349", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-23T02:39:29.759242+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952488643, "estimatedRequiredGiB": 33.497, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29972714678, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29972714678}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T08:43:01.453596+00:00", "targetGpu": "Biren_166m", "taskId": "4997304", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-22T21:35:04.154848+00:00", "modelId": "neuralmagic/SmolLM-1.7B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2014810120, "estimatedRequiredGiB": 2.256, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2018195521, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1812039680, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2018195521}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T08:39:50.946521+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4997284", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T17:41:26.757140+00:00", "modelId": "aisingapore/SEA-LION-v1-7B-IT-GPTQ", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "5227b42f4cdd7539bb62412569c7d9c8b3cc2e321cdbac3654c73c95c5bc8a7b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5469228368, "estimatedRequiredGiB": 6.118, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 5473985594, "modelscopeLicense": "mit", "modelscopeParams": 1922048000, "modelscopeTags": ["license:mit", "model_type:mpt", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5473985594}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T08:39:42.289781+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4997241", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["nemotron_h"], "framework": "vllm-customized", "lastSyncTime": "2026-09-22T13:37:53.563484+00:00", "modelId": "primitive-ai/Nemotron-3.5-Lightning-30B-A3B-mixed-INT4-INT8", "modelProfile": {"architectures": ["NemotronHForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20628596944, "estimatedRequiredGiB": 23.076, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "nemotron_h", "modelscopeFileSize": 20647674585, "modelscopeLicense": "other", "modelscopeParams": 33943909952, "modelscopeTags": ["license:other", "model_type:nemotron_h", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:int4", "custom_tag:int8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:mamba", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20647674585}, "outcome": "failed", "status": "success", "submitTime": "2026-09-21T08:37:39.043820+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4997240", "taskType": "text-generation", "verifyResult": -1}
@@ -296,5 +298,3 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "transformers", "lastSyncTime": "2026-09-21T05:42:03.443575+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16293473597, "estimatedRequiredGiB": 18.233, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 16314185171, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4665462000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:chat", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16314185171}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:09:24.169691+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986868", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "transformers", "lastSyncTime": "2026-09-21T05:17:51.457611+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683239637, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703971805, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703971805}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:09:23.997557+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986862", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "transformers", "lastSyncTime": "2026-09-21T05:17:51.457647+00:00", "modelId": "mlx-community/gemma-4-e4b-it-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7486327132, "estimatedRequiredGiB": 8.403, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 7518908360, "modelscopeLicense": "gemma", "modelscopeParams": 2227418442, "modelscopeTags": ["license:gemma", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7518908360}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:09:23.994993+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986861", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "failureUnsupportedModelTypes": ["qwen3_vl"], "framework": "transformers", "lastSyncTime": "2026-09-21T05:14:26.178418+00:00", "modelId": "BAAI/RoboBrain2.5-8B-MT", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918174, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918174}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:09:23.991772+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986860", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "transformers", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "transformers", "lastSyncTime": "2026-09-21T05:25:05.959914+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16293474987, "estimatedRequiredGiB": 18.232, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 16313701503, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4665462000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:chat", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16313701503}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:09:23.969433+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986859", "taskType": "text-generation", "verifyResult": -1}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -122,7 +122,6 @@
{"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/guaidao2/LFM2.5-2.6B-For-CTF", "modelId": "guaidao2/LFM2.5-2.6B-For-CTF", "submitTime": "2026-09-10T18:09:17.543816+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4764783", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-ascend-910-b3"}
{"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/guaidao2/LFM2.5-2.6B-For-CTF", "modelId": "guaidao2/LFM2.5-2.6B-For-CTF", "submitTime": "2026-09-10T18:25:40.316507+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4765091", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-ascend-910-b4"}
{"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/guaidao2/LFM2.5-2.6B-For-CTF", "modelId": "guaidao2/LFM2.5-2.6B-For-CTF", "submitTime": "2026-09-11T10:27:33.286927+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "4776571", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-iluvatar-bi-150"}
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/CohereLabs/tiny-aya-base-32K", "modelId": "CohereLabs/tiny-aya-base-32K", "submitTime": "2026-09-11T16:49:10.249505+00:00", "targetGpu": "Vastai_va16", "taskId": "4783129", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-vastai-va16"}
{"framework": "llamacpp", "modelAddress": "https://modelscope.cn/models/guaidao2/LFM2.5-2.6B-For-CTF", "modelId": "guaidao2/LFM2.5-2.6B-For-CTF", "submitTime": "2026-09-11T18:18:29.685555+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4784429", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-llamacpp-mthreads-s4000"}
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/CohereLabs/tiny-aya-base-32K", "modelId": "CohereLabs/tiny-aya-base-32K", "submitTime": "2026-09-11T18:51:22.774816+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4784981", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "submitTime": "2026-09-11T18:51:22.770505+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4784982", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
@@ -616,7 +615,6 @@
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/RedHatAI/Llama-3.2-3B-Instruct-FP8", "modelId": "RedHatAI/Llama-3.2-3B-Instruct-FP8", "submitTime": "2026-09-21T11:42:38.193150+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5000196", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
{"framework": "vllm-customized", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "submitTime": "2026-09-21T11:44:40.201702+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5000223", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-customized-cambricon-mlu-370-x8"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-mini-128k-instruct-FP8", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-FP8", "submitTime": "2026-09-21T11:57:55.153765+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5000419", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-FP8-Dynamic", "submitTime": "2026-09-21T11:57:55.143712+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000420", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"}
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-medium-128k-instruct-quantized.w4a16", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w4a16", "submitTime": "2026-09-21T11:57:55.174530+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5000425", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-2-7b-chat-quantized.w8a8", "modelId": "neuralmagic/Llama-2-7b-chat-quantized.w8a8", "submitTime": "2026-09-21T11:58:50.375505+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5000455", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
{"framework": "vllm-mlu", "modelAddress": "https://modelscope.cn/models/RedHatAI/SmolLM-1.7B-Instruct-quantized.w8a16", "modelId": "RedHatAI/SmolLM-1.7B-Instruct-quantized.w8a16", "submitTime": "2026-09-21T12:49:01.243460+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5001060", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-mlu-cambricon-mlu-370-x4"}

View File

@@ -2,24 +2,24 @@
"agentVersion": "2026.09.22.1",
"checksums": {
".modelhub_state/account_capacity.json": "930f9869793c3d1aa071c53f05b7cf82a4e372f002be88361d6d6189e27c12a1",
".modelhub_state/architecture_compatibility_blacklist.json": "d891b45df03e4a1e56761b86b5162b5d74120bec1c693327f9b2785286a71088",
".modelhub_state/architecture_compatibility_blacklist.json": "a95daeec6c8a78cc817fce3fe1ea7da45ab2c8d024bd7bdc5e75d12ba3a04452",
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
".modelhub_state/market_intelligence.json": "94bdfa9004e95a5a137483396660081b424ccc55da2b0cca6b108b6ac64c54b2",
".modelhub_state/official_capabilities.json": "81bb0a646456323f21ae44584289cc4a29653a1b63c1dae4474bcb313a137847",
".modelhub_state/outcome_checkpoint.json": "eb01f90f8b326f3cb5ea18888009e1ae273920b6ab85f2331e475a61609a3223",
".modelhub_state/market_intelligence.json": "b0177583cde82299c3367120ecbce5cec591c874877440fc4d1c55a92f286aee",
".modelhub_state/official_capabilities.json": "b536a457a415f7266afda92b38788bd2c111f89612cf14b7ce9e878ac8082f39",
".modelhub_state/outcome_checkpoint.json": "806c570058bb1af9560a71b1891f93167876293a1d8981600134344976619753",
".modelhub_state/queue_cleanup_latest.json": "abeedea8ce654c253250bcfd506d9b5eab392b53faf9867483b77b9a5c85233b",
".modelhub_state/recent_outcomes.jsonl": "be89f770ae4bdc0e9de6f4ee70ec569bf6dea08f70738c7d578a4ebea2bf310a",
".modelhub_state/recovery_active_tasks.jsonl": "10159aaebad20748935c0aabde549201d4b154b6e1fc11476f67418862a9ce1c",
".modelhub_state/recovery_intents.jsonl": "ddd029a6dbae152d255f2bda464b2c17d3523e15f7f6cca2ffc394735a45dc38",
".modelhub_state/recent_outcomes.jsonl": "742107c27fa4bac8b0df79e335eed2028fea1719e1ff953459208ee56c339b97",
".modelhub_state/recovery_active_tasks.jsonl": "bac8f0c51a1f1728a6dd549182e7e6976158931a9b7c0eb0ffcf80f88f2a9dff",
".modelhub_state/recovery_intents.jsonl": "5bb2f68b8cf9dfc31476561ae31bff166b7b5554a1da0e893ace9ec58162ecd8",
".modelhub_state/routing_intelligence.json": "39ec7b5e77e11b96887ed828d68b094ab47d4d4e68b51e1ea92f4df7a81ab3e3",
".modelhub_state/submission_exclusions.jsonl": "80315f370e9cc3cdadaf390e12573ef887794214779379c50b7b37b7631e8989",
".modelhub_state/worker_crashes.jsonl": "619eac16f716b2420672872ff4ce92d452f5fced4f96e8dbb37f9ce511b53c5b",
"ledger/submissions.jsonl": "1949681660efa39796ca1a4f1338205cd3a2d7ae211571a2b78e7fc663fcba49",
"outcomes/submissions.jsonl": "860e986a50ebcb1a9c0bd01cf90dca36eeb907a901d39e16e56191a0f3f38c0e"
"ledger/submissions.jsonl": "b97e3ee6089ccc093850c7e2d26ecd715079fefb580a266f9b30ef6c29a32188",
"outcomes/submissions.jsonl": "b06470d385200a00d73a98385d752172d09647d6ef320f89cfc109a32e96c638"
},
"generation": 13300,
"generation": 13301,
"phase": "cycle",
"schemaVersion": 1,
"updatedAt": "2026-09-23T02:38:29.104806+00:00",
"updatedAt": "2026-09-23T02:39:48.862453+00:00",
"writerId": "328f98096427428498653b568f0d5041"
}

View File

@@ -623,7 +623,6 @@
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559176+00:00", "modelId": "RedHatAI/gemma-2-27b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30775446656, "estimatedRequiredGiB": 34.419, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 30797371689, "modelscopeLicense": "gemma", "modelscopeParams": 28406776320, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 30797371689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:33:57.835795+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4997190", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T16:50:30.559299+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.742, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663548868, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 2}, "failureCount": 2, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T02:15:17.867294+00:00", "modelType": "lfm2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 2, "unresolvedFailureCount": 2}, "repositoryOnDiskBytes": 663548868}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:37:38.921478+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4997218", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559312+00:00", "modelId": "blue2star/Qwen-Image-2.1-PE-I2I-ComfyUI", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29306922460, "estimatedRequiredGiB": 32.775, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29326959151, "modelscopeLicense": "other", "modelscopeParams": 18821595084, "modelscopeTags": ["license:other", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:prompt-rewriting", "custom_tag:image-editing"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29326959151}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:43:01.432709+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4997305", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559340+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952488643, "estimatedRequiredGiB": 33.497, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29972714678, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29972714678}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:43:01.453596+00:00", "targetGpu": "Biren_166m", "taskId": "4997304", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T17:22:47.256894+00:00", "modelId": "blue2star/Qwen-Image-2.1-PE-I2I-ComfyUI", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29306922460, "estimatedRequiredGiB": 32.775, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29326959151, "modelscopeLicense": "other", "modelscopeParams": 18821595084, "modelscopeTags": ["license:other", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:prompt-rewriting", "custom_tag:image-editing"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29326959151}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:59:28.154388+00:00", "targetGpu": "Biren_166m", "taskId": "4997528", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T17:22:47.256966+00:00", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "167e80c9a74f5acd91819ac9a661546c8992cc5beb0bdd0a1dc9fc9659a5e778", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219876272, "estimatedRequiredGiB": 0.25, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223261617, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223261617}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T09:02:30.534692+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4997591", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T17:22:47.256900+00:00", "modelId": "nm-testing/Meta-Llama-3-8B-Instruct-W8A8-FP8-Channelwise-compressed-tensors", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084014480, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093198689, "modelscopeLicense": null, "modelscopeParams": 8030261248, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093198689}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T09:02:30.537080+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4997592", "taskType": "text-generation", "verifyResult": null}
@@ -694,7 +693,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T23:49:30.866162+00:00", "modelId": "aisingapore/SEA-LION-v1-7B", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "f49713b9fe8e3e7683ab2722544c85a44993bcd72b9421c1599f634e68b3a2f7", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361968, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008115847, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008115847}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T15:46:08.955762+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5003203", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T23:56:03.457453+00:00", "modelId": "amd/Qwen3.6-35B-A3B-w4a16-llmcompressor", "modelProfile": {"architectures": ["InstellaMoEForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 51304360808, "estimatedRequiredGiB": 57.349, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "deepseek_v3", "modelscopeFileSize": 20341994707, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107181936, "modelscopeTags": ["license:apache-2.0", "model_type:deepseek_v3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:quantized", "custom_tag:int4", "custom_tag:w4a16", "custom_tag:weight-only", "custom_tag:4-bit", "custom_tag:llm-compressor", "custom_tag:zendnn", "custom_tag:compressed-tensors", "custom_tag:amd", "custom_tag:cpu-inference"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 51315158961}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T15:54:32.713857+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5003279", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-22T00:11:13.160650+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14293356304, "estimatedRequiredGiB": 15.977, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14295851911, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14295851911}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T16:07:44.481428+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5003477", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-22T00:15:39.154168+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4020349648, "estimatedRequiredGiB": 4.496, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4022916227, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4022916227}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T16:14:31.701642+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "5003541", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-22T00:17:02.854906+00:00", "modelId": "aisingapore/SEA-LION-v1-7B-IT", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "0378395697c8345a5f2bfec5cd339a734a57d469359b2d1e41bd5325ee9d95e5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361928, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008164431, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008164431}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T16:16:37.962127+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5003581", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-22T00:17:02.854886+00:00", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785240496, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788612744, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788612744}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T16:16:38.115066+00:00", "targetGpu": "Mthreads_s4000", "taskId": "5003582", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-22T00:21:56.762829+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-DPO-v0.1-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737317601, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737317601}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T16:21:12.358143+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5003636", "taskType": "text-generation", "verifyResult": null}
@@ -975,7 +973,7 @@
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-22T22:29:34.567118+00:00", "modelId": "nkkbr/Mini-K3-1H-attn-1kda-3mla-nope-v2", "modelProfile": {"architectures": [], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1987222504, "estimatedRequiredGiB": 2.406, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": 2112106349, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2152567281}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-22T14:29:05.911870+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5020241", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-22T22:45:05.970820+00:00", "modelId": "nkkbr/Mini-K3-1H-attn-1kda-3mla-nope-v2", "modelProfile": {"architectures": [], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1987222504, "estimatedRequiredGiB": 2.406, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": 2112106349, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2152567281}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-22T14:44:16.051430+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5020425", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-23T00:45:45.504745+00:00", "modelId": "Youssofal/Qwen3.6-35B-A3B-MTPLX-Optimized-Balance", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29644061610, "estimatedRequiredGiB": 33.16, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 29671037772, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8030801776, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:speculative-decoding", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:qwen3-next", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:mixture-of-experts", "custom_tag:6-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29671037772}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-22T16:43:47.256113+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "5021799", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "nkkbr/Mini-K3-1H-kda-kernel-1-v2", "modelProfile": {"architectures": [], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2033754192, "estimatedRequiredGiB": 2.299, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": null, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2057327077}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-22T18:38:42.000009+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5023122", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-23T02:39:29.759281+00:00", "modelId": "nkkbr/Mini-K3-1H-kda-kernel-1-v2", "modelProfile": {"architectures": [], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2033754192, "estimatedRequiredGiB": 2.299, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": null, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2057327077}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-22T18:38:42.000009+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5023122", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "nkkbr/Mini-K3-1H-kda-kernel-16-v2", "modelProfile": {"architectures": [], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2036242688, "estimatedRequiredGiB": 2.291, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": null, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2049902907}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-22T18:54:21.886349+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5023399", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "nkkbr/Mini-K3-1H-kda-kernel-2-v2", "modelProfile": {"architectures": [], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2033920104, "estimatedRequiredGiB": 2.288, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mini_k3", "modelscopeFileSize": null, "modelscopeLicense": null, "modelscopeParams": null, "modelscopeTags": ["model_type:mini_k3", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:kimi-k3", "custom_tag:pretraining", "custom_tag:mixture-of-experts", "custom_tag:linear-attention", "custom_tag:architecture-ablation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2047574270}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-22T18:54:21.885307+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5023398", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": null, "modelId": "ProCreations/BetterWright-K2-Horizon-7B-Uno", "modelProfile": {"architectures": ["K2HorizonForCausalLM"], "configFingerprint": "36ec94e289a8da5fd70b98a2f401fc5159740daf9e2a076da4fd490f34bcbc0b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 19395158128, "estimatedRequiredGiB": 21.699, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "k2_horizon", "modelscopeFileSize": null, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4022513664, "modelscopeTags": ["license:apache-2.0", "model_type:k2_horizon", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:betterwright", "custom_tag:browser-agent", "custom_tag:web-agent", "custom_tag:tool-use", "custom_tag:function-calling", "custom_tag:diffusion-language-model", "custom_tag:uno", "custom_tag:k2-horizon", "custom_tag:full-finetune"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 19416072650}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-22T18:54:21.887542+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5023400", "taskType": "text-generation", "verifyResult": null}