state: generation 11228 (cycle)

This commit is contained in:
2026-09-21 05:48:31 +00:00
parent f022bae88e
commit e4287fae84
9 changed files with 1867 additions and 1867 deletions

View File

@@ -2607,7 +2607,7 @@
"taskType": "text-generation" "taskType": "text-generation"
} }
}, },
"generatedAt": "2026-09-21T05:44:50.999238+00:00", "generatedAt": "2026-09-21T05:48:19.717172+00:00",
"summary": { "summary": {
"activeBlockCount": 130, "activeBlockCount": 130,
"byGpuFramework": { "byGpuFramework": {

View File

@@ -425,7 +425,7 @@
} }
}, },
"frameworkUpdatedAt": null, "frameworkUpdatedAt": null,
"generatedAt": "2026-09-21T05:47:18.253530+00:00", "generatedAt": "2026-09-21T05:48:30.261513+00:00",
"gpuStats": { "gpuStats": {
"Ascend_910-b3": { "Ascend_910-b3": {
"available": true, "available": true,

View File

@@ -1,5 +1,5 @@
{ {
"catalogUpdatedAt": "2026-09-21T05:47:18.253530+00:00", "catalogUpdatedAt": "2026-09-21T05:48:30.261513+00:00",
"configuredTaskTypes": [ "configuredTaskTypes": [
"text-generation" "text-generation"
], ],
@@ -56,7 +56,7 @@
"time-series-forecasting" "time-series-forecasting"
], ],
"errors": [], "errors": [],
"generatedAt": "2026-09-21T05:47:18.253530+00:00", "generatedAt": "2026-09-21T05:48:30.261513+00:00",
"gpuCatalog": { "gpuCatalog": {
"Ascend_910-b3": { "Ascend_910-b3": {
"canVerify": true, "canVerify": true,
@@ -6500,6 +6500,6 @@
"updateTime": "2025-12-22 08:59:53" "updateTime": "2025-12-22 08:59:53"
} }
], ],
"taskTreeUpdatedAt": "2026-09-21T05:47:18.253530+00:00", "taskTreeUpdatedAt": "2026-09-21T05:48:30.261513+00:00",
"version": 1 "version": 1
} }

View File

@@ -1,6 +1,6 @@
{ {
"generatedAt": "2026-09-21T05:42:03.917217+00:00", "generatedAt": "2026-09-21T05:48:19.649042+00:00",
"lastSyncTime": "2026-09-21T05:42:03.443622+00:00", "lastSyncTime": "2026-09-21T05:48:19.353127+00:00",
"recentLimit": 300, "recentLimit": 300,
"report": { "report": {
"architectureCompatibilityBlocks": { "architectureCompatibilityBlocks": {
@@ -3040,23 +3040,23 @@
"decisionSuccessRate": 0.2679, "decisionSuccessRate": 0.2679,
"decisionTotal": 56, "decisionTotal": 56,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 8, "ambiguous_runtime": 9,
"framework_architecture_unsupported": 28, "framework_architecture_unsupported": 28,
"model_load": 11, "model_load": 11,
"tokenizer_compatibility": 2 "tokenizer_compatibility": 2
}, },
"failureCount": 49, "failureCount": 50,
"failureRate": 0.7656, "failureRate": 0.7692,
"framework": "vllm", "framework": "vllm",
"pendingCount": 0, "pendingCount": 0,
"pendingRate": 0.0, "pendingRate": 0.0,
"platformFailureCount": 0, "platformFailureCount": 0,
"successCount": 15, "successCount": 15,
"successRate": 0.2344, "successRate": 0.2308,
"targetGpu": "Biren_166m", "targetGpu": "Biren_166m",
"taskType": "text-generation", "taskType": "text-generation",
"total": 64, "total": 65,
"unresolvedFailureCount": 8 "unresolvedFailureCount": 9
}, },
"Cambricon_mlu-370-x4|unknown|text-generation": { "Cambricon_mlu-370-x4|unknown|text-generation": {
"attributableFailureCount": 695, "attributableFailureCount": 695,
@@ -5231,7 +5231,7 @@
"decisionSuccessRate": 0.0236, "decisionSuccessRate": 0.0236,
"decisionTotal": 3601, "decisionTotal": 3601,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 1469, "ambiguous_runtime": 1470,
"architecture_compatibility": 112, "architecture_compatibility": 112,
"attention_backend": 1, "attention_backend": 1,
"backend_operator": 84, "backend_operator": 84,
@@ -5245,15 +5245,15 @@
"tokenizer_compatibility": 410, "tokenizer_compatibility": 410,
"参数/模板问题": 41 "参数/模板问题": 41
}, },
"failureCount": 5891, "failureCount": 5892,
"failureRate": 0.9858, "failureRate": 0.9858,
"pendingCount": 0, "pendingCount": 0,
"pendingRate": 0.0, "pendingRate": 0.0,
"platformFailureCount": 865, "platformFailureCount": 865,
"successCount": 85, "successCount": 85,
"successRate": 0.0142, "successRate": 0.0142,
"total": 5976, "total": 5977,
"unresolvedFailureCount": 1510 "unresolvedFailureCount": 1511
}, },
"vllm-customized": { "vllm-customized": {
"attributableFailureCount": 3, "attributableFailureCount": 3,
@@ -5393,7 +5393,7 @@
"unresolvedFailureCount": 67 "unresolvedFailureCount": 67
} }
}, },
"generatedAt": "2026-09-21T05:42:03.905867+00:00", "generatedAt": "2026-09-21T05:48:19.638018+00:00",
"gpuSummaries": { "gpuSummaries": {
"Ascend_910-b3": { "Ascend_910-b3": {
"attributableFailureCount": 91, "attributableFailureCount": 91,
@@ -5455,7 +5455,7 @@
"decisionSuccessRate": 0.1214, "decisionSuccessRate": 0.1214,
"decisionTotal": 206, "decisionTotal": 206,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 138, "ambiguous_runtime": 139,
"backend_operator": 4, "backend_operator": 4,
"context_length": 10, "context_length": 10,
"framework_architecture_unsupported": 116, "framework_architecture_unsupported": 116,
@@ -5469,15 +5469,15 @@
"日志缺失": 62, "日志缺失": 62,
"验证失败": 27 "验证失败": 27
}, },
"failureCount": 624, "failureCount": 625,
"failureRate": 0.9615, "failureRate": 0.9615,
"pendingCount": 0, "pendingCount": 0,
"pendingRate": 0.0, "pendingRate": 0.0,
"platformFailureCount": 2, "platformFailureCount": 2,
"successCount": 25, "successCount": 25,
"successRate": 0.0385, "successRate": 0.0385,
"total": 649, "total": 650,
"unresolvedFailureCount": 441 "unresolvedFailureCount": 442
}, },
"Cambricon_mlu-370-x4": { "Cambricon_mlu-370-x4": {
"attributableFailureCount": 713, "attributableFailureCount": 713,
@@ -7444,9 +7444,10 @@
"decisionSuccessRate": 0.0, "decisionSuccessRate": 0.0,
"decisionTotal": 1, "decisionTotal": 1,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 1,
"model_load": 1 "model_load": 1
}, },
"failureCount": 1, "failureCount": 2,
"failureRate": 1.0, "failureRate": 1.0,
"framework": "vllm", "framework": "vllm",
"modelType": "phi3", "modelType": "phi3",
@@ -7458,8 +7459,8 @@
"successRate": 0.0, "successRate": 0.0,
"targetGpu": "Biren_166m", "targetGpu": "Biren_166m",
"taskType": "text-generation", "taskType": "text-generation",
"total": 1, "total": 2,
"unresolvedFailureCount": 0 "unresolvedFailureCount": 1
}, },
"Biren_166m|vllm|text-generation|plamo3|none": { "Biren_166m|vllm|text-generation|plamo3|none": {
"attributableFailureCount": 1, "attributableFailureCount": 1,
@@ -14935,13 +14936,13 @@
"decisionSuccessRate": 0.125, "decisionSuccessRate": 0.125,
"decisionTotal": 8, "decisionTotal": 8,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 7, "ambiguous_runtime": 8,
"framework_architecture_unsupported": 2, "framework_architecture_unsupported": 2,
"model_load": 4, "model_load": 4,
"tokenizer_compatibility": 1 "tokenizer_compatibility": 1
}, },
"failureCount": 14, "failureCount": 15,
"failureRate": 0.9333, "failureRate": 0.9375,
"framework": "vllm", "framework": "vllm",
"lastPlatformFailureAt": null, "lastPlatformFailureAt": null,
"lastTerminalAt": "2026-09-21T05:07:16.468545+00:00", "lastTerminalAt": "2026-09-21T05:07:16.468545+00:00",
@@ -14949,11 +14950,11 @@
"pendingRate": 0.0, "pendingRate": 0.0,
"platformFailureCount": 0, "platformFailureCount": 0,
"successCount": 1, "successCount": 1,
"successRate": 0.0667, "successRate": 0.0625,
"targetGpu": "Biren_166m", "targetGpu": "Biren_166m",
"taskType": "text-generation", "taskType": "text-generation",
"total": 15, "total": 16,
"unresolvedFailureCount": 7 "unresolvedFailureCount": 8
}, },
"Cambricon_mlu-370-x4|vllm|text-generation": { "Cambricon_mlu-370-x4|vllm|text-generation": {
"attributableFailureCount": 1, "attributableFailureCount": 1,
@@ -15632,31 +15633,6 @@
"total": 1, "total": 1,
"unresolvedFailureCount": 1 "unresolvedFailureCount": 1
}, },
"Ascend_910-b3|vllm_tokenizer_patch|text-generation|llama|awq": {
"attributableFailureCount": 0,
"consecutiveFailures": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm_tokenizer_patch",
"lastTerminalAt": "2026-09-20T16:44:23.353819+00:00",
"modelType": "llama",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "awq",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Ascend_910-b3",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Ascend_910-b3|vllm_tokenizer_patch|text-generation|llama|compressed-tensors": { "Ascend_910-b3|vllm_tokenizer_patch|text-generation|llama|compressed-tensors": {
"attributableFailureCount": 0, "attributableFailureCount": 0,
"consecutiveFailures": 0, "consecutiveFailures": 0,
@@ -16268,9 +16244,10 @@
"decisionSuccessRate": 0.0, "decisionSuccessRate": 0.0,
"decisionTotal": 1, "decisionTotal": 1,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 1,
"model_load": 1 "model_load": 1
}, },
"failureCount": 1, "failureCount": 2,
"failureRate": 1.0, "failureRate": 1.0,
"framework": "vllm", "framework": "vllm",
"lastTerminalAt": "2026-09-21T05:07:16.468553+00:00", "lastTerminalAt": "2026-09-21T05:07:16.468553+00:00",
@@ -16283,8 +16260,8 @@
"successRate": 0.0, "successRate": 0.0,
"targetGpu": "Biren_166m", "targetGpu": "Biren_166m",
"taskType": "text-generation", "taskType": "text-generation",
"total": 1, "total": 2,
"unresolvedFailureCount": 0 "unresolvedFailureCount": 1
}, },
"Biren_166m|vllm|text-generation|qwen2|compressed-tensors": { "Biren_166m|vllm|text-generation|qwen2|compressed-tensors": {
"attributableFailureCount": 0, "attributableFailureCount": 0,
@@ -21708,6 +21685,30 @@
"total": 1, "total": 1,
"unresolvedFailureCount": 0 "unresolvedFailureCount": 0
}, },
"Biren_166m|vllm|text-generation|phi3|compressed-tensors|33": {
"attributableFailureCount": 0,
"decisionFailureRate": 0.0,
"decisionSuccessRate": 0.0,
"decisionTotal": 0,
"failureBreakdown": {
"ambiguous_runtime": 1
},
"failureCount": 1,
"failureRate": 1.0,
"framework": "vllm",
"loadSizeLog2Bucket": 33,
"modelType": "phi3",
"pendingCount": 0,
"pendingRate": 0.0,
"platformFailureCount": 0,
"quantizationMethod": "compressed-tensors",
"successCount": 0,
"successRate": 0.0,
"targetGpu": "Biren_166m",
"taskType": "text-generation",
"total": 1,
"unresolvedFailureCount": 1
},
"Biren_166m|vllm|text-generation|plamo3|none|33": { "Biren_166m|vllm|text-generation|plamo3|none|33": {
"attributableFailureCount": 1, "attributableFailureCount": 1,
"decisionFailureRate": 1.0, "decisionFailureRate": 1.0,
@@ -32256,15 +32257,15 @@
"unresolvedFailureCount": 0 "unresolvedFailureCount": 0
} }
}, },
"terminalRecords": 16152, "terminalRecords": 16153,
"totalRecords": 16333, "totalRecords": 16334,
"totals": { "totals": {
"attributableFailureCount": 5854, "attributableFailureCount": 5854,
"decisionFailureRate": 0.8615, "decisionFailureRate": 0.8615,
"decisionSuccessRate": 0.1385, "decisionSuccessRate": 0.1385,
"decisionTotal": 6795, "decisionTotal": 6795,
"failureBreakdown": { "failureBreakdown": {
"ambiguous_runtime": 3880, "ambiguous_runtime": 3881,
"architecture_compatibility": 212, "architecture_compatibility": 212,
"attention_backend": 1, "attention_backend": 1,
"backend_operator": 102, "backend_operator": 102,
@@ -32280,31 +32281,31 @@
"日志缺失": 719, "日志缺失": 719,
"验证失败": 673 "验证失败": 673
}, },
"failureCount": 15211, "failureCount": 15212,
"failureRate": 0.9417, "failureRate": 0.9417,
"pendingCount": 0, "pendingCount": 0,
"pendingRate": 0.0, "pendingRate": 0.0,
"platformFailureCount": 923, "platformFailureCount": 923,
"successCount": 941, "successCount": 941,
"successRate": 0.0583, "successRate": 0.0583,
"total": 16152, "total": 16153,
"unresolvedFailureCount": 8434 "unresolvedFailureCount": 8435
}, },
"warnings": [ "warnings": [
"GPU Iluvatar_mrv-100 本地统计失败率偏高≥50%),建议重点关注。", "GPU Iluvatar_mrv-100 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Mthreads_s4000 本地统计失败率偏高≥50%),建议重点关注。", "GPU Mthreads_s4000 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Kunlunxin_p-800 本地统计失败率偏高≥50%),建议重点关注。", "GPU Kunlunxin_p-800 本地统计失败率偏高≥50%),建议重点关注。",
"GPU MetaX_c-500 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Biren_166m 本地统计失败率偏高≥50%),建议重点关注。", "GPU Biren_166m 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-100 本地统计失败率偏高≥50%),建议重点关注。", "GPU Iluvatar_bi-100 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。", "GPU Sunrise_pt-200-x1 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b3 本地统计失败率偏高≥50%),建议重点关注。", "GPU Ascend_910-b3 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Vastai_va16 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x4 本地统计失败率偏高≥50%),建议重点关注。", "GPU Cambricon_mlu-370-x4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。", "GPU Iluvatar_bi-150 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Cambricon_mlu-370-x8 本地统计失败率偏高≥50%),建议重点关注。",
"GPU hygon_k100-ai 本地统计失败率偏高≥50%),建议重点关注。", "GPU hygon_k100-ai 本地统计失败率偏高≥50%),建议重点关注。",
"GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。", "GPU Ascend_910-b4 本地统计失败率偏高≥50%),建议重点关注。",
"GPU MetaX_c-500 本地统计失败率偏高≥50%),建议重点关注。", "GPU Vastai_va16 本地统计失败率偏高≥50%),建议重点关注。",
"组合 Iluvatar_bi-150|vllm|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Iluvatar_bi-150|vllm|visual-multi-modal 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Sunrise_pt-200-x1|vllm_fix_tokenizer|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Sunrise_pt-200-x1|vllm_fix_tokenizer|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
"组合 Cambricon_mlu-370-x8|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。", "组合 Cambricon_mlu-370-x8|unknown|text-generation 近期失败集中,建议降低该 GPU+框架的提交优先级。",
@@ -32343,6 +32344,6 @@
] ]
}, },
"storageMode": "decision_state_only", "storageMode": "decision_state_only",
"summarizedRecords": 16333, "summarizedRecords": 16334,
"version": 1 "version": 1
} }

View File

@@ -77,6 +77,7 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T16:33:38.958026+00:00", "modelId": "YOYO-AI/Qwen2.5-14B-YOYO-V3", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:33:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079134", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T16:33:38.958026+00:00", "modelId": "YOYO-AI/Qwen2.5-14B-YOYO-V3", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:33:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079134", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T16:12:47.264764+00:00", "modelId": "bharatgenai/Param2-17B-A2.4B-Thinking", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:11:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4592334", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-20T16:12:47.264764+00:00", "modelId": "bharatgenai/Param2-17B-A2.4B-Thinking", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:11:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4592334", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T16:12:47.264793+00:00", "modelId": "KenDual3090tiNvlink/NVIDIA-Nemotron-Labs-3-Puzzle-75B-A9B-W4A16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:11:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4080000", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T16:12:47.264793+00:00", "modelId": "KenDual3090tiNvlink/NVIDIA-Nemotron-Labs-3-Puzzle-75B-A9B-W4A16", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T16:11:21+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4080000", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "quantization_error", "failureCode": "QUANTIZATION_ERROR", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-21T05:48:19.353127+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:26:25.535380+00:00", "targetGpu": "Biren_166m", "taskId": "4980428", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T00:42:10.765041+00:00", "modelId": "LiquidAI/LFM2.5-2.6B", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5394427456, "estimatedRequiredGiB": 6.049, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 5412393192, "modelscopeLicense": "other", "modelscopeParams": 2697198592, "modelscopeTags": ["license:other", "model_type:lfm2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5412393192}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:26:18.189644+00:00", "targetGpu": "Biren_166m", "taskId": "4980427", "taskType": "text-generation", "verifyResult": -1} {"failReason": "tokenizer_compatibility", "failureAction": "check_tokenizer_files_then_llm", "failureCategory": "tokenizer_compatibility", "failureClassificationReason": "broad_tokenizer_error", "failureCode": "TOKENIZER_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T00:42:10.765041+00:00", "modelId": "LiquidAI/LFM2.5-2.6B", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5394427456, "estimatedRequiredGiB": 6.049, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 5412393192, "modelscopeLicense": "other", "modelscopeParams": 2697198592, "modelscopeTags": ["license:other", "model_type:lfm2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 5412393192}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:26:18.189644+00:00", "targetGpu": "Biren_166m", "taskId": "4980427", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T03:02:20.757790+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.742, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663548868, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663548868}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:26:11.835960+00:00", "targetGpu": "Biren_166m", "taskId": "4980425", "taskType": "text-generation", "verifyResult": -1} {"failReason": "model_load", "failureAction": "inspect_root_exception_then_llm_if_unknown", "failureCategory": "model_load", "failureClassificationReason": "broad_load_error", "failureCode": "MODEL_LOAD_FAILED", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "model_gpu_framework", "framework": "vllm", "lastSyncTime": "2026-09-21T03:02:20.757790+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Instruct-MLX-4bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 658540250, "estimatedRequiredGiB": 0.742, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 663548868, "modelscopeLicense": "other", "modelscopeParams": 182975232, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 663548868}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:26:11.835960+00:00", "targetGpu": "Biren_166m", "taskId": "4980425", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T14:32:15.260773+00:00", "modelId": "AI-ModelScope/granite-20b-code-base", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:25:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079181", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm", "lastSyncTime": "2026-09-20T14:32:15.260773+00:00", "modelId": "AI-ModelScope/granite-20b-code-base", "modelProfile": {}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T14:25:21+00:00", "targetGpu": "MetaX_c-500", "taskId": "4079181", "taskType": "text-generation", "verifyResult": -1}
@@ -297,4 +298,3 @@
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T17:09:45.266185+00:00", "modelId": "sapientinc/HRM-Text-1B", "modelProfile": {"architectures": ["HrmTextForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2365606568, "estimatedRequiredGiB": 2.651, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "hrm_text", "modelscopeFileSize": 2371660433, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1182795264, "modelscopeTags": ["license:apache-2.0", "model_type:hrm_text", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:hrm", "custom_tag:hierarchical-reasoning", "custom_tag:prefix-lm", "custom_tag:pre-alignment", "custom_tag:non-chat", "custom_tag:non-instruction-tuned"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2371660433}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.693959+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969057", "taskType": "text-generation", "verifyResult": -1} {"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T17:09:45.266185+00:00", "modelId": "sapientinc/HRM-Text-1B", "modelProfile": {"architectures": ["HrmTextForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2365606568, "estimatedRequiredGiB": 2.651, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "hrm_text", "modelscopeFileSize": 2371660433, "modelscopeLicense": "apache-2.0", "modelscopeParams": 1182795264, "modelscopeTags": ["license:apache-2.0", "model_type:hrm_text", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:hrm", "custom_tag:hierarchical-reasoning", "custom_tag:prefix-lm", "custom_tag:pre-alignment", "custom_tag:non-chat", "custom_tag:non-instruction-tuned"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 2371660433}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.693959+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969057", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T22:43:47.754925+00:00", "modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.691589+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969060", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T22:43:47.754925+00:00", "modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.691589+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969060", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T16:54:50.062693+00:00", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.682339+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969056", "taskType": "text-generation", "verifyResult": -1} {"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDetectedFramework": "vllm", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T16:54:50.062693+00:00", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.682339+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969056", "taskType": "text-generation", "verifyResult": -1}
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "driver_error", "failureCode": "DRIVER_ERROR", "failureDetectedFramework": "vllm", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-20T16:44:23.353819+00:00", "modelId": "OpenBMB/MiniCPM5-2B-GPTQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2099615880, "estimatedRequiredGiB": 2.358, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2109841981, "modelscopeLicense": "apache-2.0", "modelscopeParams": 2516756480, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 2109841981}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T02:52:51.646422+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4969055", "taskType": "text-generation", "verifyResult": -1}

File diff suppressed because it is too large Load Diff

File diff suppressed because it is too large Load Diff

View File

@@ -2,24 +2,24 @@
"agentVersion": "2026.09.20.2", "agentVersion": "2026.09.20.2",
"checksums": { "checksums": {
".modelhub_state/account_capacity.json": "68d2a8390ad022f2ddcd30841f3860c4535387fa64cb46dc7507eb16837e798f", ".modelhub_state/account_capacity.json": "68d2a8390ad022f2ddcd30841f3860c4535387fa64cb46dc7507eb16837e798f",
".modelhub_state/architecture_compatibility_blacklist.json": "7c6a5042f83d88bdc7f6b5d7abc173b7e65f9d3b1bca7f429b5f11b9b53b2b37", ".modelhub_state/architecture_compatibility_blacklist.json": "c296cd89ece784254a18d9d017cd09429094b58619cf3851843a29017029d074",
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab", ".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
".modelhub_state/market_intelligence.json": "98c2ddb5af442b5f7fdb0875e9d980f2c3ca204cc39475f7cd1e7e94378ef1ca", ".modelhub_state/market_intelligence.json": "1bf8d8a55c8bb571554f9e163335c2cd49e0966c1ef55b8792a6c863b0f3f396",
".modelhub_state/official_capabilities.json": "a02f169d75c60818641686667090d703df4f5ba1de10192944ccc53f5099edca", ".modelhub_state/official_capabilities.json": "745b8ed1d9acde7f5db837861c93586b577dfa68ef48cb54347788ecf86f8e81",
".modelhub_state/outcome_checkpoint.json": "fe83dda8ab0f028190ceb44fe69da21ba9c106b66e9b2cbe9bcb9ec762781062", ".modelhub_state/outcome_checkpoint.json": "05bd137019f69f964b03fc04fb475bdaa68d08c68ead8adacb955e30f207ec1b",
".modelhub_state/queue_cleanup_latest.json": "0de105598c9478c6cb1c89a58c4794a2395e2c58f508a89a0e4c6fd6e5c312a7", ".modelhub_state/queue_cleanup_latest.json": "0de105598c9478c6cb1c89a58c4794a2395e2c58f508a89a0e4c6fd6e5c312a7",
".modelhub_state/recent_outcomes.jsonl": "af46d69887f9a3fe86d0e8d005ae92dd2578bae71c761211cab6a210a8fd036e", ".modelhub_state/recent_outcomes.jsonl": "eed5b73df0d3dd48e504ce46b1a34c1dae52aee702729f4e9173cea1898fc43c",
".modelhub_state/recovery_active_tasks.jsonl": "921c68203341a0a2a7ea4c328641b4305a4713b3f4371c6918111f8b9ef012dd", ".modelhub_state/recovery_active_tasks.jsonl": "8915ac0356f85cfd7501ce2b345bf35c114d669892261722deccc77da0c258ae",
".modelhub_state/recovery_intents.jsonl": "1047a916863a5ee480752e5d3fb5e5c3535a99b184ba525927049c07191bc44e", ".modelhub_state/recovery_intents.jsonl": "ca8c2f96270babe8a6fa6a988d8b0972e70d8163cfadceeb38e453bf6e131e4c",
".modelhub_state/routing_intelligence.json": "ea955c6b3c0bf7864fd2cf8ec51a66c344763482eb7715662096bcaceb384776", ".modelhub_state/routing_intelligence.json": "ea955c6b3c0bf7864fd2cf8ec51a66c344763482eb7715662096bcaceb384776",
".modelhub_state/submission_exclusions.jsonl": "80315f370e9cc3cdadaf390e12573ef887794214779379c50b7b37b7631e8989", ".modelhub_state/submission_exclusions.jsonl": "80315f370e9cc3cdadaf390e12573ef887794214779379c50b7b37b7631e8989",
".modelhub_state/worker_crashes.jsonl": "07505b69e0b050da5f90f8b046a3069d3e1ca2574ca39896bc7d8a66293167f7", ".modelhub_state/worker_crashes.jsonl": "07505b69e0b050da5f90f8b046a3069d3e1ca2574ca39896bc7d8a66293167f7",
"ledger/submissions.jsonl": "58a59537cb4094a283f35e457e90c8768740d6be2b2df3b02bdc91c002067f8a", "ledger/submissions.jsonl": "58a59537cb4094a283f35e457e90c8768740d6be2b2df3b02bdc91c002067f8a",
"outcomes/submissions.jsonl": "6143acbb07ef3dd3c4b44ee634d103f88d866aee7cf45d21f6e6c687ab0d4de3" "outcomes/submissions.jsonl": "4d8e82eeb81187a55377572dd6ec2377bc144f64e4348b797834bc7ca65c1c7e"
}, },
"generation": 11227, "generation": 11228,
"phase": "cycle", "phase": "cycle",
"schemaVersion": 1, "schemaVersion": 1,
"updatedAt": "2026-09-21T05:47:18.697592+00:00", "updatedAt": "2026-09-21T05:48:31.494347+00:00",
"writerId": "8b35139af6674067a339a670222d4b67" "writerId": "8b35139af6674067a339a670222d4b67"
} }

View File

@@ -833,7 +833,6 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960822+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26057989872, "estimatedRequiredGiB": 29.158, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26090095180, "modelscopeLicense": "apache-2.0", "modelscopeParams": 21416999792, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26090095180}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:29.832253+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980394", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960822+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26057989872, "estimatedRequiredGiB": 29.158, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26090095180, "modelscopeLicense": "apache-2.0", "modelscopeParams": 21416999792, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26090095180}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:29.832253+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980394", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960787+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 36679335352, "estimatedRequiredGiB": 41.028, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 36711610638, "modelscopeLicense": "apache-2.0", "modelscopeParams": 18339618304, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:pruning", "custom_tag:width-pruning", "custom_tag:zero-training", "custom_tag:qwen3_5", "custom_tag:gated-deltanet"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 36711610638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:29.955385+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980395", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960787+00:00", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 36679335352, "estimatedRequiredGiB": 41.028, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 36711610638, "modelscopeLicense": "apache-2.0", "modelscopeParams": 18339618304, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:pruning", "custom_tag:width-pruning", "custom_tag:zero-training", "custom_tag:qwen3_5", "custom_tag:gated-deltanet"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 36711610638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:29.955385+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980395", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960863+00:00", "modelId": "primitive-ai/Qwen3.8-27B-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 22210552448, "estimatedRequiredGiB": 24.849, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 22234091413, "modelscopeLicense": "apache-2.0", "modelscopeParams": 17463440388, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:speculative-decoding", "custom_tag:blackwell", "custom_tag:a100"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 22234091413}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:30.080805+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980396", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:27:36.960863+00:00", "modelId": "primitive-ai/Qwen3.8-27B-mixed-NVFP4-FP8", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 22210552448, "estimatedRequiredGiB": 24.849, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 22234091413, "modelscopeLicense": "apache-2.0", "modelscopeParams": 17463440388, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:fp8", "custom_tag:mixed-precision", "custom_tag:quantized", "custom_tag:speculative-decoding", "custom_tag:blackwell", "custom_tag:a100"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 22234091413}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:25:30.080805+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4980396", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "d341b8002cf39d3299fd429479e8c8e954583e49d4fec5a640ddc0f8f0cbd054", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T14:26:25.535380+00:00", "targetGpu": "Biren_166m", "taskId": "4980428", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-20T22:43:47.754827+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:28:11.858304+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4980451", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-20T22:43:47.754827+00:00", "modelId": "RedHatAI/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824831, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 2033824831}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:28:11.858304+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4980451", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:43:47.754941+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:30:11.079895+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4980489", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-20T22:43:47.754941+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "a494b1667fab53609b94cc853db1a12c8d3b9fcf98795ecbab3dd5eb3a3c77f6", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 48.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 48.0, "maximumRepositorySizeGiB": 40.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:30:11.079895+00:00", "targetGpu": "Mthreads_s4000", "taskId": "4980489", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T22:43:47.754934+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.5-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737318602, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737318602}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:32:45.916677+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4980504", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-20T22:43:47.754934+00:00", "modelId": "solidrust/Llama-3-8B-Instruct-v0.5-AWQ", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 5727954960, "estimatedRequiredGiB": 6.412, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 5737318602, "modelscopeLicense": "other", "modelscopeParams": 8030269440, "modelscopeTags": ["license:other", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:4-bit", "custom_tag:AWQ", "custom_tag:text-generation", "custom_tag:autotrain_compatible", "custom_tag:endpoints_compatible", "custom_tag:axolotl", "custom_tag:finetune", "custom_tag:facebook", "custom_tag:meta", "custom_tag:pytorch", "custom_tag:llama", "custom_tag:llama-3"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "awq", "repositoryOnDiskBytes": 5737318602}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T14:32:45.916677+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4980504", "taskType": "text-generation", "verifyResult": null}
@@ -985,7 +984,7 @@
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443534+00:00", "modelId": "EschaLabs/Qwen3.8-27B-Escha-W2", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11002488632, "estimatedRequiredGiB": 12.322, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 10176447713, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6340437856, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:safetensors", "task:text-generation", "custom_tag:qwen3", "custom_tag:2-bit", "custom_tag:quantization", "custom_tag:escha", "custom_tag:sglang", "custom_tag:code", "custom_tag:reasoning", "custom_tag:conversational"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "escha", "repositoryOnDiskBytes": 11025855102}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.183318+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4988062", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443534+00:00", "modelId": "EschaLabs/Qwen3.8-27B-Escha-W2", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11002488632, "estimatedRequiredGiB": 12.322, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 10176447713, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6340437856, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:safetensors", "task:text-generation", "custom_tag:qwen3", "custom_tag:2-bit", "custom_tag:quantization", "custom_tag:escha", "custom_tag:sglang", "custom_tag:code", "custom_tag:reasoning", "custom_tag:conversational"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "escha", "repositoryOnDiskBytes": 11025855102}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.183318+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4988062", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443569+00:00", "modelId": "OpenBMB/MiniCPM5-2B-MLX", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1416035216, "estimatedRequiredGiB": 1.594, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1426233716, "modelscopeLicense": "apache-2.0", "modelscopeParams": 393390080, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1426233716}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.189599+00:00", "targetGpu": "Biren_166m", "taskId": "4988061", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443569+00:00", "modelId": "OpenBMB/MiniCPM5-2B-MLX", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1416035216, "estimatedRequiredGiB": 1.594, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1426233716, "modelscopeLicense": "apache-2.0", "modelscopeParams": 393390080, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:minicpm", "custom_tag:minicpm5", "custom_tag:llama", "custom_tag:text-generation", "custom_tag:long-context", "custom_tag:tool-calling", "custom_tag:on-device", "custom_tag:edge-ai"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1426233716}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.189599+00:00", "targetGpu": "Biren_166m", "taskId": "4988061", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443526+00:00", "modelId": "neuralmagic/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367987, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367987}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.191312+00:00", "targetGpu": "Biren_166m", "taskId": "4988063", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:42:03.443526+00:00", "modelId": "neuralmagic/gemma-2-2b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385544296, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407367987, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407367987}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:37:47.191312+00:00", "targetGpu": "Biren_166m", "taskId": "4988063", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "pfnet/plamo-3-nict-8b-base", "modelProfile": {"architectures": ["Plamo3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16182734928, "estimatedRequiredGiB": 18.09, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "plamo3", "modelscopeFileSize": 16186391993, "modelscopeLicense": "other", "modelscopeParams": 8091348992, "modelscopeTags": ["license:other", "model_type:plamo3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16186391993}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T21:45:44.135645+00:00", "targetGpu": "Biren_166m", "taskId": "4988230", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T05:48:19.353103+00:00", "modelId": "pfnet/plamo-3-nict-8b-base", "modelProfile": {"architectures": ["Plamo3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16182734928, "estimatedRequiredGiB": 18.09, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "plamo3", "modelscopeFileSize": 16186391993, "modelscopeLicense": "other", "modelscopeParams": 8091348992, "modelscopeTags": ["license:other", "model_type:plamo3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16186391993}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:45:44.135645+00:00", "targetGpu": "Biren_166m", "taskId": "4988230", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": null, "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T21:47:29.378176+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4988291", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": null, "modelId": "neuralmagic/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124714, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124714}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T21:47:29.378176+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4988291", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T21:57:41.600515+00:00", "targetGpu": "Biren_166m", "taskId": "4988391", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": null, "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T21:57:41.600515+00:00", "targetGpu": "Biren_166m", "taskId": "4988391", "taskType": "text-generation", "verifyResult": null}
{"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T22:11:07.168563+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4988543", "taskType": "text-generation", "verifyResult": null} {"failReason": null, "framework": "vllm", "lastSyncTime": null, "modelId": "RedHatAI/gemma-4-12B-it-FP8-Dynamic", "modelProfile": {"architectures": ["Gemma4UnifiedForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15037393456, "estimatedRequiredGiB": 16.842, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4_unified", "modelscopeFileSize": 15069601989, "modelscopeLicense": "apache-2.0", "modelscopeParams": 12966363184, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4_unified", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "task:image-text-to-text", "custom_tag:fp8", "custom_tag:vllm", "custom_tag:llm-compressor", "custom_tag:compressed-tensors"], "modelscopeTasks": ["text-generation", "image-text-to-text"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 15069601989}, "outcome": "pending", "status": "pending", "submitTime": "2026-09-20T22:11:07.168563+00:00", "targetGpu": "Ascend_910-b4", "taskId": "4988543", "taskType": "text-generation", "verifyResult": null}