state: generation 13253 (cycle)
This commit is contained in:
@@ -2186,6 +2186,25 @@
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
"kunlunxin_p-800|vllm|text-generation|model_type:qwen3_5": {
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectures": [],
|
||||
"evidenceCount": 1,
|
||||
"expiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"framework": "vllm",
|
||||
"latestFailureAt": "2026-09-20T20:51:02.374696+00:00",
|
||||
"latestSuccessfulAt": null,
|
||||
"matchType": "model_type",
|
||||
"modelType": "qwen3_5",
|
||||
"sourceModelIds": [
|
||||
"VerStella/Qwen3.8-27B-heretic-ara-W8A8-DCU"
|
||||
],
|
||||
"sourceTaskIds": [
|
||||
"4987485"
|
||||
],
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
"kunlunxin_p-800|vllm|text-generation|model_type:qwen3_5_moe": {
|
||||
"architectureSignature": "model_type:qwen3_5_moe",
|
||||
"architectures": [],
|
||||
@@ -3023,9 +3042,9 @@
|
||||
"taskType": "text-generation"
|
||||
}
|
||||
},
|
||||
"generatedAt": "2026-09-23T01:29:16.689348+00:00",
|
||||
"generatedAt": "2026-09-23T01:33:15.985873+00:00",
|
||||
"summary": {
|
||||
"activeBlockCount": 152,
|
||||
"activeBlockCount": 153,
|
||||
"byGpuFramework": {
|
||||
"Ascend_910-b3|vllm": 13,
|
||||
"Ascend_910-b3|vllm_tokenizer_patch": 5,
|
||||
@@ -3040,7 +3059,7 @@
|
||||
"Iluvatar_bi-150|vllm": 8,
|
||||
"Iluvatar_bi-150|vllm_0_17_0_corex_4_4_0": 3,
|
||||
"Iluvatar_bi-150|vllm_tokenizer_patch": 2,
|
||||
"Kunlunxin_p-800|vllm": 4,
|
||||
"Kunlunxin_p-800|vllm": 5,
|
||||
"Kunlunxin_p-800|vllm_tokenizer_patch": 6,
|
||||
"MetaX_c-500|vllm": 7,
|
||||
"Mthreads_s4000|vllm": 9,
|
||||
|
||||
@@ -396,7 +396,7 @@
|
||||
}
|
||||
},
|
||||
"frameworkUpdatedAt": null,
|
||||
"generatedAt": "2026-09-23T01:32:11.627923+00:00",
|
||||
"generatedAt": "2026-09-23T01:33:35.507308+00:00",
|
||||
"gpuStats": {
|
||||
"Ascend_910-b4": {
|
||||
"available": true,
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
{
|
||||
"catalogUpdatedAt": "2026-09-23T01:32:11.627923+00:00",
|
||||
"catalogUpdatedAt": "2026-09-23T01:33:35.507308+00:00",
|
||||
"configuredTaskTypes": [
|
||||
"text-generation"
|
||||
],
|
||||
@@ -56,7 +56,7 @@
|
||||
"time-series-forecasting"
|
||||
],
|
||||
"errors": [],
|
||||
"generatedAt": "2026-09-23T01:32:11.627923+00:00",
|
||||
"generatedAt": "2026-09-23T01:33:35.507308+00:00",
|
||||
"gpuCatalog": {
|
||||
"Ascend_910-b3": {
|
||||
"canVerify": true,
|
||||
@@ -6669,6 +6669,6 @@
|
||||
"updateTime": "2025-12-22 08:59:53"
|
||||
}
|
||||
],
|
||||
"taskTreeUpdatedAt": "2026-09-23T01:32:11.627923+00:00",
|
||||
"taskTreeUpdatedAt": "2026-09-23T01:33:35.507308+00:00",
|
||||
"version": 1
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"generatedAt": "2026-09-23T01:29:16.600030+00:00",
|
||||
"lastSyncTime": "2026-09-23T01:29:14.657262+00:00",
|
||||
"generatedAt": "2026-09-23T01:33:35.419411+00:00",
|
||||
"lastSyncTime": "2026-09-23T01:33:35.361109+00:00",
|
||||
"recentLimit": 300,
|
||||
"report": {
|
||||
"architectureCompatibilityBlocks": {
|
||||
@@ -2190,6 +2190,25 @@
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
"kunlunxin_p-800|vllm|text-generation|model_type:qwen3_5": {
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectures": [],
|
||||
"evidenceCount": 1,
|
||||
"expiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"framework": "vllm",
|
||||
"latestFailureAt": "2026-09-20T20:51:02.374696+00:00",
|
||||
"latestSuccessfulAt": null,
|
||||
"matchType": "model_type",
|
||||
"modelType": "qwen3_5",
|
||||
"sourceModelIds": [
|
||||
"VerStella/Qwen3.8-27B-heretic-ara-W8A8-DCU"
|
||||
],
|
||||
"sourceTaskIds": [
|
||||
"4987485"
|
||||
],
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
"kunlunxin_p-800|vllm|text-generation|model_type:qwen3_5_moe": {
|
||||
"architectureSignature": "model_type:qwen3_5_moe",
|
||||
"architectures": [],
|
||||
@@ -3028,7 +3047,7 @@
|
||||
}
|
||||
},
|
||||
"architectureCompatibilitySummary": {
|
||||
"activeBlockCount": 152,
|
||||
"activeBlockCount": 153,
|
||||
"byGpuFramework": {
|
||||
"Ascend_910-b3|vllm": 13,
|
||||
"Ascend_910-b3|vllm_tokenizer_patch": 5,
|
||||
@@ -3043,7 +3062,7 @@
|
||||
"Iluvatar_bi-150|vllm": 8,
|
||||
"Iluvatar_bi-150|vllm_0_17_0_corex_4_4_0": 3,
|
||||
"Iluvatar_bi-150|vllm_tokenizer_patch": 2,
|
||||
"Kunlunxin_p-800|vllm": 4,
|
||||
"Kunlunxin_p-800|vllm": 5,
|
||||
"Kunlunxin_p-800|vllm_tokenizer_patch": 6,
|
||||
"MetaX_c-500|vllm": 7,
|
||||
"Mthreads_s4000|vllm": 9,
|
||||
@@ -4708,17 +4727,17 @@
|
||||
"unresolvedFailureCount": 19
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation": {
|
||||
"attributableFailureCount": 10,
|
||||
"attributableFailureCount": 11,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 10,
|
||||
"decisionTotal": 11,
|
||||
"failureBreakdown": {
|
||||
"backend_operator": 2,
|
||||
"framework_architecture_unsupported": 7,
|
||||
"framework_architecture_unsupported": 8,
|
||||
"model_load": 1,
|
||||
"参数/模板问题": 2
|
||||
},
|
||||
"failureCount": 12,
|
||||
"failureCount": 13,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"pendingCount": 0,
|
||||
@@ -4728,7 +4747,7 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 12,
|
||||
"total": 13,
|
||||
"unresolvedFailureCount": 2
|
||||
},
|
||||
"Kunlunxin_r-200-8f|unknown|text-generation": {
|
||||
@@ -5691,17 +5710,17 @@
|
||||
"unresolvedFailureCount": 6462
|
||||
},
|
||||
"vllm": {
|
||||
"attributableFailureCount": 3572,
|
||||
"attributableFailureCount": 3573,
|
||||
"decisionFailureRate": 0.9741,
|
||||
"decisionSuccessRate": 0.0259,
|
||||
"decisionTotal": 3667,
|
||||
"decisionTotal": 3668,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 1534,
|
||||
"architecture_compatibility": 112,
|
||||
"attention_backend": 3,
|
||||
"backend_operator": 86,
|
||||
"context_length": 161,
|
||||
"framework_architecture_unsupported": 1328,
|
||||
"framework_architecture_unsupported": 1329,
|
||||
"memory_capacity": 722,
|
||||
"model_load": 211,
|
||||
"platform_infrastructure": 866,
|
||||
@@ -5710,14 +5729,14 @@
|
||||
"tokenizer_compatibility": 413,
|
||||
"参数/模板问题": 48
|
||||
},
|
||||
"failureCount": 6020,
|
||||
"failureCount": 6021,
|
||||
"failureRate": 0.9845,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 866,
|
||||
"successCount": 95,
|
||||
"successRate": 0.0155,
|
||||
"total": 6115,
|
||||
"total": 6116,
|
||||
"unresolvedFailureCount": 1582
|
||||
},
|
||||
"vllm-customized": {
|
||||
@@ -5867,7 +5886,7 @@
|
||||
"unresolvedFailureCount": 173
|
||||
}
|
||||
},
|
||||
"generatedAt": "2026-09-23T01:29:16.587275+00:00",
|
||||
"generatedAt": "2026-09-23T01:33:35.406215+00:00",
|
||||
"gpuSummaries": {
|
||||
"Ascend_910-b3": {
|
||||
"attributableFailureCount": 102,
|
||||
@@ -6093,14 +6112,14 @@
|
||||
"unresolvedFailureCount": 717
|
||||
},
|
||||
"Kunlunxin_p-800": {
|
||||
"attributableFailureCount": 50,
|
||||
"decisionFailureRate": 0.9091,
|
||||
"decisionSuccessRate": 0.0909,
|
||||
"decisionTotal": 55,
|
||||
"attributableFailureCount": 51,
|
||||
"decisionFailureRate": 0.9107,
|
||||
"decisionSuccessRate": 0.0893,
|
||||
"decisionTotal": 56,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 208,
|
||||
"backend_operator": 10,
|
||||
"framework_architecture_unsupported": 22,
|
||||
"framework_architecture_unsupported": 23,
|
||||
"memory_capacity": 2,
|
||||
"model_load": 12,
|
||||
"platform_infrastructure": 1,
|
||||
@@ -6110,14 +6129,14 @@
|
||||
"日志缺失": 1,
|
||||
"验证失败": 23
|
||||
},
|
||||
"failureCount": 315,
|
||||
"failureCount": 316,
|
||||
"failureRate": 0.9844,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 1,
|
||||
"successCount": 5,
|
||||
"successRate": 0.0156,
|
||||
"total": 320,
|
||||
"total": 321,
|
||||
"unresolvedFailureCount": 264
|
||||
},
|
||||
"Kunlunxin_r-200-8f": {
|
||||
@@ -6306,7 +6325,7 @@
|
||||
"hygon_k100-ai": 64.0
|
||||
},
|
||||
"pendingRecords": 0,
|
||||
"policyCancelledRecords": 208,
|
||||
"policyCancelledRecords": 218,
|
||||
"profileCombinationStats": {
|
||||
"Ascend_910-b3|llamacpp|text-generation|smollm3|none": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -17496,14 +17515,14 @@
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|qwen3_5|compressed-tensors": {
|
||||
"attributableFailureCount": 1,
|
||||
"attributableFailureCount": 2,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 1,
|
||||
"decisionTotal": 2,
|
||||
"failureBreakdown": {
|
||||
"framework_architecture_unsupported": 1
|
||||
"framework_architecture_unsupported": 2
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureCount": 2,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"modelType": "qwen3_5",
|
||||
@@ -17515,7 +17534,7 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|qwen3_5|none": {
|
||||
@@ -21051,18 +21070,18 @@
|
||||
"unresolvedFailureCount": 1
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation": {
|
||||
"attributableFailureCount": 4,
|
||||
"consecutiveFailures": 4,
|
||||
"attributableFailureCount": 5,
|
||||
"consecutiveFailures": 5,
|
||||
"consecutivePlatformFailures": 0,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 4,
|
||||
"decisionTotal": 5,
|
||||
"failureBreakdown": {
|
||||
"backend_operator": 2,
|
||||
"framework_architecture_unsupported": 2,
|
||||
"framework_architecture_unsupported": 3,
|
||||
"参数/模板问题": 2
|
||||
},
|
||||
"failureCount": 6,
|
||||
"failureCount": 7,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"lastPlatformFailureAt": null,
|
||||
@@ -21074,7 +21093,7 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 6,
|
||||
"total": 7,
|
||||
"unresolvedFailureCount": 2
|
||||
},
|
||||
"MetaX_c-500|vllm|text-generation": {
|
||||
@@ -22015,9 +22034,9 @@
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 0,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 4
|
||||
"ambiguous_runtime": 3
|
||||
},
|
||||
"failureCount": 4,
|
||||
"failureCount": 3,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm_fix_tokenizer",
|
||||
"lastTerminalAt": "2026-09-22T10:22:21.952661+00:00",
|
||||
@@ -22030,8 +22049,8 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Biren_166m",
|
||||
"taskType": "text-generation",
|
||||
"total": 4,
|
||||
"unresolvedFailureCount": 4
|
||||
"total": 3,
|
||||
"unresolvedFailureCount": 3
|
||||
},
|
||||
"Biren_166m|vllm_fix_tokenizer|text-generation|qwen3_5_text|none": {
|
||||
"attributableFailureCount": 0,
|
||||
@@ -23959,6 +23978,31 @@
|
||||
"total": 1,
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|qwen3_5|compressed-tensors": {
|
||||
"attributableFailureCount": 1,
|
||||
"consecutiveFailures": 1,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 1,
|
||||
"failureBreakdown": {
|
||||
"framework_architecture_unsupported": 1
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"lastTerminalAt": "2026-09-23T01:33:13.863433+00:00",
|
||||
"modelType": "qwen3_5",
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 0,
|
||||
"quantizationMethod": "compressed-tensors",
|
||||
"successCount": 0,
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|spark2_5|none": {
|
||||
"attributableFailureCount": 1,
|
||||
"consecutiveFailures": 1,
|
||||
@@ -40992,14 +41036,14 @@
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|qwen3_5|compressed-tensors|34": {
|
||||
"attributableFailureCount": 1,
|
||||
"attributableFailureCount": 2,
|
||||
"decisionFailureRate": 1.0,
|
||||
"decisionSuccessRate": 0.0,
|
||||
"decisionTotal": 1,
|
||||
"decisionTotal": 2,
|
||||
"failureBreakdown": {
|
||||
"framework_architecture_unsupported": 1
|
||||
"framework_architecture_unsupported": 2
|
||||
},
|
||||
"failureCount": 1,
|
||||
"failureCount": 2,
|
||||
"failureRate": 1.0,
|
||||
"framework": "vllm",
|
||||
"loadSizeLog2Bucket": 34,
|
||||
@@ -41012,7 +41056,7 @@
|
||||
"successRate": 0.0,
|
||||
"targetGpu": "Kunlunxin_p-800",
|
||||
"taskType": "text-generation",
|
||||
"total": 1,
|
||||
"total": 2,
|
||||
"unresolvedFailureCount": 0
|
||||
},
|
||||
"Kunlunxin_p-800|vllm|text-generation|qwen3_5|none|33": {
|
||||
@@ -45565,20 +45609,20 @@
|
||||
"unresolvedFailureCount": 0
|
||||
}
|
||||
},
|
||||
"terminalRecords": 16784,
|
||||
"totalRecords": 16992,
|
||||
"terminalRecords": 16785,
|
||||
"totalRecords": 17003,
|
||||
"totals": {
|
||||
"attributableFailureCount": 6008,
|
||||
"attributableFailureCount": 6009,
|
||||
"decisionFailureRate": 0.8621,
|
||||
"decisionSuccessRate": 0.1379,
|
||||
"decisionTotal": 6969,
|
||||
"decisionTotal": 6970,
|
||||
"failureBreakdown": {
|
||||
"ambiguous_runtime": 4278,
|
||||
"architecture_compatibility": 212,
|
||||
"attention_backend": 3,
|
||||
"backend_operator": 113,
|
||||
"context_length": 322,
|
||||
"framework_architecture_unsupported": 2130,
|
||||
"framework_architecture_unsupported": 2131,
|
||||
"memory_capacity": 1203,
|
||||
"model_load": 525,
|
||||
"platform_infrastructure": 929,
|
||||
@@ -45589,14 +45633,14 @@
|
||||
"日志缺失": 719,
|
||||
"验证失败": 676
|
||||
},
|
||||
"failureCount": 15823,
|
||||
"failureCount": 15824,
|
||||
"failureRate": 0.9427,
|
||||
"pendingCount": 0,
|
||||
"pendingRate": 0.0,
|
||||
"platformFailureCount": 929,
|
||||
"successCount": 961,
|
||||
"successRate": 0.0573,
|
||||
"total": 16784,
|
||||
"total": 16785,
|
||||
"unresolvedFailureCount": 8886
|
||||
},
|
||||
"warnings": [
|
||||
@@ -45654,6 +45698,6 @@
|
||||
]
|
||||
},
|
||||
"storageMode": "decision_state_only",
|
||||
"summarizedRecords": 16992,
|
||||
"summarizedRecords": 17003,
|
||||
"version": 1
|
||||
}
|
||||
|
||||
@@ -14,13 +14,13 @@
|
||||
100
|
||||
],
|
||||
"accounts": 12,
|
||||
"activeScanned": 1073,
|
||||
"activeScanned": 1064,
|
||||
"ageCleanupMode": "admission_only",
|
||||
"agePolicySkipped": {
|
||||
"cleanupDisabled": true,
|
||||
"reason": "admission_only"
|
||||
},
|
||||
"architectureBlockCount": 152,
|
||||
"architectureBlockCount": 153,
|
||||
"architectureFrameworkCatalog": {
|
||||
"ascend_910-b3|text-generation": [
|
||||
"llamacpp",
|
||||
@@ -38,27 +38,11 @@
|
||||
"vllm_fix_tokenizer",
|
||||
"vllm_tokenizer_patch"
|
||||
],
|
||||
"cambricon_mlu-370-x4|text-generation": [
|
||||
"vllm",
|
||||
"vllm-customized",
|
||||
"vllm-mlu"
|
||||
],
|
||||
"cambricon_mlu-370-x8|text-generation": [
|
||||
"vllm",
|
||||
"vllm-customized",
|
||||
"vllm-mlu"
|
||||
],
|
||||
"hygon_k100-ai|text-generation": [
|
||||
"llamacpp",
|
||||
"vllm",
|
||||
"vllm-patch-tokenizer"
|
||||
],
|
||||
"iluvatar_bi-100|text-generation": [
|
||||
"transformers",
|
||||
"vllm",
|
||||
"vllm-patch-tokenizer",
|
||||
"vllm_fix_tokenizer"
|
||||
],
|
||||
"iluvatar_bi-150|text-generation": [
|
||||
"llamacpp",
|
||||
"transformers",
|
||||
@@ -67,19 +51,11 @@
|
||||
"vllm_fix_tokenizer",
|
||||
"vllm_tokenizer_patch"
|
||||
],
|
||||
"iluvatar_mrv-100|text-generation": [
|
||||
"transformers",
|
||||
"vllm"
|
||||
],
|
||||
"kunlunxin_p-800|text-generation": [
|
||||
"vllm",
|
||||
"vllm_fix_tokenizer",
|
||||
"vllm_tokenizer_patch"
|
||||
],
|
||||
"metax_c-500|text-generation": [
|
||||
"vllm",
|
||||
"vllm_tokenizer_patch"
|
||||
],
|
||||
"mthreads_s4000|text-generation": [
|
||||
"llamacpp",
|
||||
"vllm",
|
||||
@@ -89,15 +65,222 @@
|
||||
"sglang",
|
||||
"vllm",
|
||||
"vllm_fix_tokenizer"
|
||||
],
|
||||
"vastai_va16|text-generation": [
|
||||
"vllm",
|
||||
"vllm_fix_tokenizer"
|
||||
]
|
||||
},
|
||||
"architectureFrameworkCatalogErrors": {},
|
||||
"architectureIncompatibleCount": 0,
|
||||
"architectureIncompatibleTasks": [],
|
||||
"architectureIncompatibleCount": 10,
|
||||
"architectureIncompatibleTasks": [
|
||||
{
|
||||
"accountIndex": 1,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ornith-ai/Ornith-1.5-9B-NVFP4",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4985113,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 1,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987156,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 2,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4992046,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 5,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-6bit",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987155,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 5,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987486,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 9,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 5001319,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 10,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4984931,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 10,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4986237,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 11,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4984928,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 11,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4996858,
|
||||
"taskType": "text-generation"
|
||||
}
|
||||
],
|
||||
"architectureModelConfigErrors": {
|
||||
"GestaltLabs/Ornstein-3.5-9B-V2-GGUF": "HttpJsonError: HTTP request failed for https://modelscope.cn/models/GestaltLabs/Ornstein-3.5-9B-V2-GGUF/resolve/master/config.json (status=404)",
|
||||
"LiquidAI/LFM2.5-2.6B-DSpark-GGUF": "HttpJsonError: HTTP request failed for https://modelscope.cn/models/LiquidAI/LFM2.5-2.6B-DSpark-GGUF/resolve/master/config.json (status=500)",
|
||||
@@ -121,22 +304,263 @@
|
||||
"z-lab/Qwen3.8-27B-DFlash2-GGUF": "HttpJsonError: HTTP request failed for https://modelscope.cn/models/z-lab/Qwen3.8-27B-DFlash2-GGUF/resolve/master/config.json (status=404)"
|
||||
},
|
||||
"architectureModelConfigsComplete": 53,
|
||||
"architectureOnly": false,
|
||||
"architectureOnly": true,
|
||||
"architecturePolicySkipped": {
|
||||
"frameworkCatalogUnknown": 0,
|
||||
"frameworkContextUnknown": 59,
|
||||
"modelArchitectureUnknown": 136,
|
||||
"noMatchingBlock": 921,
|
||||
"noMatchingBlock": 902,
|
||||
"partiallyBlockedFrameworkSet": 16,
|
||||
"runningMatchedProtected": 0,
|
||||
"submissionContextMismatch": 0,
|
||||
"submissionContextUnknown": 0
|
||||
},
|
||||
"cancelledCount": 0,
|
||||
"cancelledTasks": [],
|
||||
"cancelledCount": 10,
|
||||
"cancelledTasks": [
|
||||
{
|
||||
"accountIndex": 1,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ornith-ai/Ornith-1.5-9B-NVFP4",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4985113,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 1,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987156,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 2,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4992046,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 5,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-6bit",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987155,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 5,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4987486,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 9,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 5001319,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 10,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4984931,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 10,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4986237,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 11,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4984928,
|
||||
"taskType": "text-generation"
|
||||
},
|
||||
{
|
||||
"accountIndex": 11,
|
||||
"architectureBlockEvidenceCount": 1,
|
||||
"architectureBlockExpiresAt": "2026-10-20T20:51:02.374696+00:00",
|
||||
"architectureMatchScope": "exact_framework",
|
||||
"architectureMatchType": "model_type",
|
||||
"architectureSignature": "model_type:qwen3_5",
|
||||
"architectureSignatures": [
|
||||
"model_type:qwen3_5"
|
||||
],
|
||||
"cleanupReasons": [
|
||||
"known_framework_architecture_incompatible"
|
||||
],
|
||||
"evaluatedFrameworks": [
|
||||
"vllm"
|
||||
],
|
||||
"framework": "vllm",
|
||||
"gpuType": "Kunlunxin_p-800",
|
||||
"modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality",
|
||||
"reason": "known_framework_architecture_incompatible",
|
||||
"status": "waiting",
|
||||
"taskId": 4996858,
|
||||
"taskType": "text-generation"
|
||||
}
|
||||
],
|
||||
"certainOomCount": 0,
|
||||
"certainOomTasks": [],
|
||||
"cleanupCandidateCount": 0,
|
||||
"cleanupCandidateCount": 10,
|
||||
"dryRun": false,
|
||||
"listingErrors": {},
|
||||
"modelAgeErrors": {},
|
||||
@@ -161,19 +585,17 @@
|
||||
],
|
||||
"oldOverflowCount": 0,
|
||||
"oldOverflowTasks": [],
|
||||
"policyCancelledRecorded": 0,
|
||||
"policyCancelledRecorded": 10,
|
||||
"policyNoLongerAppliesCount": 0,
|
||||
"policyNoLongerAppliesTasks": [],
|
||||
"recentModelDays": 7,
|
||||
"recentModelReserveSlots": 5,
|
||||
"repositorySizeErrors": {
|
||||
"empero-ai/Qwen3.8-9B-GGUF": "recursive_repository_size_incomplete"
|
||||
},
|
||||
"repositorySizesComplete": 371,
|
||||
"repositorySizeErrors": {},
|
||||
"repositorySizesComplete": 0,
|
||||
"skipped": {
|
||||
"fitsKnownCapacity": 1071,
|
||||
"fitsKnownCapacity": 0,
|
||||
"gpuCapacityUnknown": 0,
|
||||
"repositorySizeUnknown": 2
|
||||
"repositorySizeUnknown": 0
|
||||
},
|
||||
"stopErrors": [],
|
||||
"uniqueModels": 372
|
||||
|
||||
@@ -266,6 +266,7 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T22:22:17.565349+00:00", "modelId": "aisingapore/SEA-LION-v1-7B-IT", "modelProfile": {"architectures": ["MPTForCausalLM"], "configFingerprint": "d8df464812ff2fd3c46dc89b0fc6a96370e694fab8e52179f1ada2cdd0ea92b9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15003361928, "estimatedRequiredGiB": 16.773, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mpt", "modelscopeFileSize": 15008164431, "modelscopeLicense": "mit", "modelscopeParams": 7501651968, "modelscopeTags": ["license:mit", "model_type:mpt", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15008164431}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:55:39.990873+00:00", "targetGpu": "Biren_166m", "taskId": "4987554", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559353+00:00", "modelId": "cyankiwi/Apodex-1.1-mini-AWQ-INT4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 26119812080, "estimatedRequiredGiB": 29.233, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 26156887523, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35951822704, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agent", "custom_tag:apodex", "custom_tag:arxiv:2608.23283"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 26156887523}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:53:47.320072+00:00", "targetGpu": "Biren_166m", "taskId": "4987553", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T22:22:17.565358+00:00", "modelId": "primitive-ai/Nex-N2.5-mini-NVFP4", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 23926484304, "estimatedRequiredGiB": 26.777, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 23959625409, "modelscopeLicense": "apache-2.0", "modelscopeParams": 35107212656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:vllm", "custom_tag:compressed-tensors", "custom_tag:nvfp4", "custom_tag:quantized", "custom_tag:moe", "custom_tag:nex-n2.5", "custom_tag:agentic", "custom_tag:tool-calling", "custom_tag:single-gpu", "custom_tag:blackwell"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 23959625409}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:51:34.051986+00:00", "targetGpu": "Biren_166m", "taskId": "4987489", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "framework_architecture_unsupported", "failureAction": "block_gpu_framework_architecture", "failureCategory": "framework_architecture_unsupported", "failureClassificationReason": "explicit_framework_model_unsupported", "failureCode": "MODEL_NOT_SUPPORTED", "failureDeterministic": true, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": false, "failureScope": "model_gpu_framework", "failureUnsupportedModelTypes": ["qwen3_5"], "framework": "vllm", "lastSyncTime": "2026-09-23T01:33:13.863433+00:00", "modelId": "VerStella/Qwen3.8-27B-heretic-ara-W8A8-DCU", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 31228126952, "estimatedRequiredGiB": 34.934, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 31258377392, "modelscopeLicense": "apache-2.0", "modelscopeParams": 27781427952, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:w8a8", "custom_tag:int8", "custom_tag:quantized"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 31258377392}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:51:02.374696+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987485", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-21T19:45:40.057911+00:00", "modelId": "mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "modelProfile": {"architectures": ["Mistral3ForConditionalGeneration"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17367755740, "estimatedRequiredGiB": 19.429, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral3", "modelscopeFileSize": 17385072379, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4929899520, "modelscopeTags": ["license:apache-2.0", "model_type:mistral3", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:mistral", "custom_tag:mistral3", "custom_tag:devstral"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17385072379}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:48:13.967563+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4987442", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm-customized", "lastSyncTime": "2026-09-22T18:38:47.055915+00:00", "modelId": "neuralmagic/Qwen2-7B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8708787584, "estimatedRequiredGiB": 9.746, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 8720442075, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7615616512, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 8720442075}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:43:36.176091+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4987372", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:50:30.559306+00:00", "modelId": "JANGQ-AI/Osaurus-AppleScript-8B-JANG_4M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463599, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463599}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T20:39:13.053978+00:00", "targetGpu": "Biren_166m", "taskId": "4987256", "taskType": "text-generation", "verifyResult": -1}
|
||||
@@ -297,4 +298,3 @@
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T12:09:51.363854+00:00", "modelId": "mlx-community/Devstral-Small-2-24B-Instruct-2512-OptiQ-4bit", "modelProfile": {"architectures": ["Mistral3ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17367755740, "estimatedRequiredGiB": 19.429, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "mistral3", "modelscopeFileSize": 17385072379, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4929899520, "modelscopeTags": ["license:apache-2.0", "model_type:mistral3", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:mistral", "custom_tag:mistral3", "custom_tag:devstral"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17385072379}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T19:48:14.736283+00:00", "targetGpu": "Biren_166m", "taskId": "4986620", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T12:09:51.363816+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16293473597, "estimatedRequiredGiB": 18.233, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 16314185171, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4665462000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:chat", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16314185171}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T19:48:14.673114+00:00", "targetGpu": "Biren_166m", "taskId": "4986619", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T12:58:50.741528+00:00", "modelId": "mlx-community/Qwen3.5-122B-A10B-OptiQ-2bit-REAP-63B", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 25593158704, "estimatedRequiredGiB": 28.626, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 25613691403, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7626590448, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:expert-pruning", "custom_tag:reap", "custom_tag:moe", "custom_tag:optiq", "custom_tag:apple-silicon", "custom_tag:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 25613691403}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T19:48:14.653000+00:00", "targetGpu": "Biren_166m", "taskId": "4986616", "taskType": "text-generation", "verifyResult": -1}
|
||||
{"failReason": "ambiguous_runtime", "failureAction": "send_compact_profile_and_root_exception_to_llm", "failureCategory": "ambiguous_runtime", "failureClassificationReason": "execute_empty_result", "failureCode": "EXECUTE_EMPTY_RESULT", "failureDeterministic": false, "failureEnrichmentAttempts": 1, "failureEnrichmentError": null, "failureNeedsLlm": true, "failureScope": "unknown", "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T12:58:50.741607+00:00", "modelId": "mlx-community/KAT-Coder-V2.5-Dev-OptiQ-4bit", "modelProfile": {"architectures": ["Qwen3_5MoeForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 21930054875, "estimatedRequiredGiB": 24.532, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_moe", "modelscopeFileSize": 21950415614, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6024588416, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5_moe", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:optiq", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 21950415614}, "outcome": "failed", "status": "success", "submitTime": "2026-09-20T19:48:14.646752+00:00", "targetGpu": "Biren_166m", "taskId": "4986611", "taskType": "text-generation", "verifyResult": -1}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -488,10 +488,7 @@
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/JANGQ-AI/Hy3-preview-JANGTQ", "modelId": "JANGQ-AI/Hy3-preview-JANGTQ", "submitTime": "2026-09-20T17:43:28.088638+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984763", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-9b-it-quantized.w8a16", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w8a16", "submitTime": "2026-09-20T17:44:06.835549+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4984803", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-15b-quantized.w8a16", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "submitTime": "2026-09-20T17:49:20.875289+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4984851", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "submitTime": "2026-09-20T17:55:49.585284+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984928", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "submitTime": "2026-09-20T17:55:49.590451+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984931", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/Qwen2-0.5B-Instruct-quantized.w8a8", "modelId": "RedHatAI/Qwen2-0.5B-Instruct-quantized.w8a8", "submitTime": "2026-09-20T17:59:59.555224+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4985000", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ornith-ai/Ornith-1.5-9B-NVFP4", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "submitTime": "2026-09-20T18:09:15.235288+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4985113", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/neuralmagic/gemma-2-2b-quantized.w8a16", "modelId": "neuralmagic/gemma-2-2b-quantized.w8a16", "submitTime": "2026-09-20T18:09:27.989767+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4985115", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
|
||||
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "submitTime": "2026-09-20T18:39:41.567990+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4985663", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}
|
||||
{"framework": "vllm-mlu", "modelAddress": "https://modelscope.cn/models/inceptionai/Jais-2-8B-Chat", "modelId": "inceptionai/Jais-2-8B-Chat", "submitTime": "2026-09-20T18:39:53.164098+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4985692", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-mlu-cambricon-mlu-370-x8"}
|
||||
@@ -503,7 +500,6 @@
|
||||
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-15b-FP8", "modelId": "RedHatAI/starcoder2-15b-FP8", "submitTime": "2026-09-20T18:57:33.406739+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4985872", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "modelId": "neuralmagic/Qwen2-0.5B-Instruct-quantized.w8a8", "submitTime": "2026-09-20T19:01:17.970020+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4985910", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/inclusionAI/Ling-3.0-tiny", "modelId": "inclusionAI/Ling-3.0-tiny", "submitTime": "2026-09-20T19:01:32.459840+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4985911", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "submitTime": "2026-09-20T19:23:30.183489+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4986237", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/Meta-Llama-3-8B-Instruct-quantized.w8a8", "modelId": "RedHatAI/Meta-Llama-3-8B-Instruct-quantized.w8a8", "submitTime": "2026-09-20T19:30:14.636274+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986332", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/sbintuitions/sarashina2.2-3b-instruct-v0.1", "modelId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "submitTime": "2026-09-20T19:31:34.206245+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986333", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/aisingapore/Gemma-SEA-LION-v4-27B-IT", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT", "submitTime": "2026-09-20T19:46:10.235701+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986586", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-sunrise-pt-200-x1"}
|
||||
@@ -512,10 +508,7 @@
|
||||
{"framework": "transformers", "modelAddress": "https://modelscope.cn/models/BAAI/RoboBrain2.5-8B-NV", "modelId": "BAAI/RoboBrain2.5-8B-NV", "submitTime": "2026-09-20T20:09:23.973756+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986856", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-transformers-iluvatar-bi-100"}
|
||||
{"framework": "transformers", "modelAddress": "https://modelscope.cn/models/SyonLi/Qwen3-4B-Instruct-2507-Segmenter", "modelId": "SyonLi/Qwen3-4B-Instruct-2507-Segmenter", "submitTime": "2026-09-20T20:09:24.163788+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986867", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-transformers-iluvatar-bi-100"}
|
||||
{"framework": "transformers", "modelAddress": "https://modelscope.cn/models/mlx-community/LFM2.5-1.2B-Thinking-8bit", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-8bit", "submitTime": "2026-09-20T20:09:24.460163+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986875", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-transformers-iluvatar-bi-100"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ddalcu/Qwen3.8-27B-MLX-Serve-6bit", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-6bit", "submitTime": "2026-09-20T20:33:04.562192+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987155", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw", "modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw", "submitTime": "2026-09-20T20:33:04.573215+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987156", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "submitTime": "2026-09-20T20:41:37.494440+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4987371", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "submitTime": "2026-09-20T20:51:02.378387+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987486", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/JANGQ-AI/AppleScript-8B-JANG_4M", "modelId": "JANGQ-AI/AppleScript-8B-JANG_4M", "submitTime": "2026-09-20T21:13:06.834956+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987784", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "submitTime": "2026-09-20T21:22:47.667525+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4987928", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/aisingapore/Gemma-SEA-LION-v3-9B", "modelId": "aisingapore/Gemma-SEA-LION-v3-9B", "submitTime": "2026-09-20T21:26:40.552644+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4987963", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
@@ -562,7 +555,6 @@
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/starcoder2-3b-quantized.w8a16", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "submitTime": "2026-09-21T02:10:00.507387+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991869", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-it-quantized.w4a16", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w4a16", "submitTime": "2026-09-21T02:10:00.601936+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991870", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-9b-it-quantized.w4a16", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w4a16", "submitTime": "2026-09-21T02:10:00.705684+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991871", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16", "submitTime": "2026-09-21T02:21:28.285988+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4992046", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm-customized", "modelAddress": "https://modelscope.cn/models/LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "submitTime": "2026-09-21T03:59:26.348415+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4993324", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-customized-cambricon-mlu-370-x8"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/BAAI/AquilaMed-RL", "modelId": "BAAI/AquilaMed-RL", "submitTime": "2026-09-21T04:44:11.759434+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4994057", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-sunrise-pt-200-x1"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-9b-it-quantized.w4a16", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w4a16", "submitTime": "2026-09-21T04:55:59.065763+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4994202", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
@@ -580,7 +572,6 @@
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "submitTime": "2026-09-21T08:07:48.338371+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4996781", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm-customized", "modelAddress": "https://modelscope.cn/models/XHToken/Spark-X2.5-4B", "modelId": "XHToken/Spark-X2.5-4B", "submitTime": "2026-09-21T08:09:36.692226+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4996782", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-customized-cambricon-mlu-370-x8"}
|
||||
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/mlx-community/gemma-4-31B-it-qat-OptiQ-4bit", "modelId": "mlx-community/gemma-4-31B-it-qat-OptiQ-4bit", "submitTime": "2026-09-21T08:17:20.243727+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4996859", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "submitTime": "2026-09-21T08:17:20.220538+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4996858", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/blue2star/Qwen-Image-2.1-PE-T2I-ComfyUI", "modelId": "blue2star/Qwen-Image-2.1-PE-T2I-ComfyUI", "submitTime": "2026-09-21T08:17:20.209866+00:00", "targetGpu": "Biren_166m", "taskId": "4996857", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-biren-166m"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/aisingapore/Gemma-SEA-LION-v4-27B-IT-NVFP4", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-NVFP4", "submitTime": "2026-09-21T08:29:25.977872+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4997055", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-ascend-910-b3"}
|
||||
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/nm-testing/tinyllama-oneshot-w8w8-test-static-shape-change", "modelId": "nm-testing/tinyllama-oneshot-w8w8-test-static-shape-change", "submitTime": "2026-09-21T08:29:25.981023+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4997054", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}
|
||||
@@ -646,7 +637,6 @@
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/aisingapore/Llama-SEA-LION-v3.5-70B-R-NVFP4", "modelId": "aisingapore/Llama-SEA-LION-v3.5-70B-R-NVFP4", "submitTime": "2026-09-21T12:49:01.249214+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001061", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/mlx-community/LFM2.5-1.2B-Thinking-6bit", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-6bit", "submitTime": "2026-09-21T13:01:28.899242+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5001186", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-ascend-910-b3"}
|
||||
{"framework": "vllm-mlu", "modelAddress": "https://modelscope.cn/models/RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "submitTime": "2026-09-21T13:17:35.253765+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5001324", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-mlu-cambricon-mlu-370-x4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "submitTime": "2026-09-21T13:17:35.237740+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001319", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/inceptionai/Jais-2-8B-Chat", "modelId": "inceptionai/Jais-2-8B-Chat", "submitTime": "2026-09-21T14:18:50.561697+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "5002167", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-iluvatar-bi-150"}
|
||||
{"framework": "vllm-patch-tokenizer", "modelAddress": "https://modelscope.cn/models/neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "submitTime": "2026-09-21T14:20:50.936887+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5002188", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-patch-tokenizer-hygon-k100-ai"}
|
||||
{"framework": "vllm_tokenizer_patch", "modelAddress": "https://modelscope.cn/models/RedHatAI/gemma-2-2b-it-quantized.w8a8", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a8", "submitTime": "2026-09-21T14:48:11.802242+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5002502", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-tokenizer-patch-ascend-910-b3"}
|
||||
@@ -943,6 +933,7 @@
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ProCreations/BetterWright-K2-Horizon-7B-Uno", "modelId": "ProCreations/BetterWright-K2-Horizon-7B-Uno", "submitTime": "2026-09-22T19:09:59.409477+00:00", "targetGpu": "Ascend_910-b4", "taskId": "5023596", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-ascend-910-b4"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/neuralmagic/starcoder2-15b-quantized.w8a16", "modelId": "neuralmagic/starcoder2-15b-quantized.w8a16", "submitTime": "2026-09-21T12:49:01.235875+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001055", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/XHToken/Spark-X2.5-4B-FP8", "modelId": "XHToken/Spark-X2.5-4B-FP8", "submitTime": "2026-09-21T12:49:01.244975+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001063", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm", "modelAddress": "https://modelscope.cn/models/ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "submitTime": "2026-09-21T13:17:35.237740+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001319", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-kunlunxin-p-800"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/ddalcu/Muse-Glimmer-30B-MLX-Serve-8bit", "modelId": "ddalcu/Muse-Glimmer-30B-MLX-Serve-8bit", "submitTime": "2026-09-21T13:43:30.755806+00:00", "targetGpu": "Biren_166m", "taskId": "5001736", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-biren-166m"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "modelId": "logic65/Qwen3.8-Whittle-w50-18.3B-unrepaired", "submitTime": "2026-09-21T14:17:59.392335+00:00", "targetGpu": "Biren_166m", "taskId": "5002152", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-biren-166m"}
|
||||
{"framework": "vllm_fix_tokenizer", "modelAddress": "https://modelscope.cn/models/ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "submitTime": "2026-09-21T14:17:59.385934+00:00", "targetGpu": "Biren_166m", "taskId": "5002154", "taskType": "text-generation", "templateId": "modelhub-live-text-generation-vllm-fix-tokenizer-biren-166m"}
|
||||
|
||||
@@ -2,24 +2,24 @@
|
||||
"agentVersion": "2026.09.22.1",
|
||||
"checksums": {
|
||||
".modelhub_state/account_capacity.json": "930f9869793c3d1aa071c53f05b7cf82a4e372f002be88361d6d6189e27c12a1",
|
||||
".modelhub_state/architecture_compatibility_blacklist.json": "1a809badaeaa1bf58ac92b242910cdafc2febaa794b26419035d7758624cf513",
|
||||
".modelhub_state/architecture_compatibility_blacklist.json": "d5ef243efbf030d0f0747de28019e510a79a8eda76bde9b02e56634bb5e7ee85",
|
||||
".modelhub_state/architecture_history_backfill.json": "a92348206605da04f75c9e26e7c818b76f259e0b28983f18b6375ef94802f0ab",
|
||||
".modelhub_state/market_intelligence.json": "dba4c2c8ae706bf23d18f33d753477574893d938c36a9e7e64c87551b3388d57",
|
||||
".modelhub_state/official_capabilities.json": "5f7ef5bd13ca3b012c7dbb32d23890e937f06a0e9971b18300b4b9bd0287367a",
|
||||
".modelhub_state/outcome_checkpoint.json": "ebc136e014c341cea692031b63dbd40c167cbcbc048164521213960fe3e8c0b7",
|
||||
".modelhub_state/queue_cleanup_latest.json": "70b12a33050de2b7d18becf5530413055730ba9da8e6ce7b09a99877bc0fd590",
|
||||
".modelhub_state/recent_outcomes.jsonl": "79b6d55284c3f75cc08ad11b8ab20e4edff0f39ec4effad3abbee12ed1b7e175",
|
||||
".modelhub_state/recovery_active_tasks.jsonl": "b5d9541ecee3653a5ffa067ba4161cea423f781800cd308c6b054958d51ca296",
|
||||
".modelhub_state/recovery_intents.jsonl": "61a63b313523ad6a17dbb1bff3706277ba50a58996013d02253b596dd49a04f6",
|
||||
".modelhub_state/market_intelligence.json": "83ce0c30596082689f2bd11ee2026d0b68128449474dddbd9a212365c6d785ae",
|
||||
".modelhub_state/official_capabilities.json": "956ab0b63bcfd2d30e874234c438fff7d5ea84ac73a8d19aa9f9f52e39206ea3",
|
||||
".modelhub_state/outcome_checkpoint.json": "32ebfda4abfab6052679b0c6705844014cca32681f126bde621babf6ffed0741",
|
||||
".modelhub_state/queue_cleanup_latest.json": "25b072614686a8dea931431bf7c311a7f76d027c8b4f273ae90a556d9345992d",
|
||||
".modelhub_state/recent_outcomes.jsonl": "6684cca26023ab1863cd968bfbdd817493550ff274926a22a77747d93b7f5b95",
|
||||
".modelhub_state/recovery_active_tasks.jsonl": "dc057ef825158ccec5b0ba8b595bc81cebdfff3017e57d5502bad76c42ff8e73",
|
||||
".modelhub_state/recovery_intents.jsonl": "592248efa25645ca9a9b6476e9d8b81a4fe8ece9cab44e73dc02d974e1125bc4",
|
||||
".modelhub_state/routing_intelligence.json": "39ec7b5e77e11b96887ed828d68b094ab47d4d4e68b51e1ea92f4df7a81ab3e3",
|
||||
".modelhub_state/submission_exclusions.jsonl": "80315f370e9cc3cdadaf390e12573ef887794214779379c50b7b37b7631e8989",
|
||||
".modelhub_state/worker_crashes.jsonl": "81a279620f26a6ff30e7cd3dca4b86844ec41a0d30ffee7d8a4b03ab754d2d4c",
|
||||
"ledger/submissions.jsonl": "e41ca1b41c2d9fe7c553e3c10665298fcb8b0ac7fb7114270dbe689bfb7c6207",
|
||||
"outcomes/submissions.jsonl": "0f422d3d971d689e3511f5a4d53fb5c4dec02881de45841d5c4c8356bb93af49"
|
||||
"ledger/submissions.jsonl": "dae1e070159c8b2cae2773baa1f8f13efb2ee673e6d099ad54449477fb3fd09a",
|
||||
"outcomes/submissions.jsonl": "4f815c14602cb1af5c01ec33d4f8c47f96bf9d2e9ad12dcbf185be0d1ae4a94a"
|
||||
},
|
||||
"generation": 13252,
|
||||
"generation": 13253,
|
||||
"phase": "cycle",
|
||||
"schemaVersion": 1,
|
||||
"updatedAt": "2026-09-23T01:32:12.290526+00:00",
|
||||
"updatedAt": "2026-09-23T01:33:36.925943+00:00",
|
||||
"writerId": "0024d6b5efdf486184421da342f897ee"
|
||||
}
|
||||
|
||||
@@ -522,11 +522,8 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T01:52:39.246422+00:00", "modelId": "JANGQ-AI/Hy3-preview-JANGTQ", "modelProfile": {"architectures": ["HYV3ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 85146629560, "estimatedRequiredGiB": 95.18, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "hy_v3", "modelscopeFileSize": 85166002853, "modelscopeLicense": "other", "modelscopeParams": 21562957504, "modelscopeTags": ["license:other", "model_type:hy_v3", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:jang", "custom_tag:jangtq", "custom_tag:hy3", "custom_tag:hunyuan", "custom_tag:hy_v3", "custom_tag:moe", "custom_tag:apple-silicon", "custom_tag:2bit", "custom_tag:295b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 85166002853}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:43:28.088638+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984763", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T01:52:39.246416+00:00", "modelId": "neuralmagic/gemma-2-9b-it-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 11998647728, "estimatedRequiredGiB": 13.434, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 12020564058, "modelscopeLicense": "gemma", "modelscopeParams": 10159209984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 12020564058}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:44:06.835549+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4984803", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T01:52:39.246349+00:00", "modelId": "RedHatAI/starcoder2-15b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17785269848, "estimatedRequiredGiB": 19.88, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 17788663591, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 15957889024, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 17788663591}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:49:20.875289+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4984851", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867306+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683241105, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703487237, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703487237}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:55:49.585284+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984928", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867348+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20683239637, "estimatedRequiredGiB": 23.138, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 20703971805, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6086364400, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 20703971805}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:55:49.590451+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984931", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867168+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Bare-Speed-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16293473597, "estimatedRequiredGiB": 18.233, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 16314185171, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4665462000, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:chat", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b", "custom_tag:coding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16314185171}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:55:49.586551+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4984930", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867333+00:00", "modelId": "RedHatAI/Qwen2-0.5B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["Qwen2ForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 903168128, "estimatedRequiredGiB": 1.022, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen2", "modelscopeFileSize": 914658256, "modelscopeLicense": "apache-2.0", "modelscopeParams": 630167424, "modelscopeTags": ["license:apache-2.0", "model_type:qwen2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 914658256}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T17:59:59.555224+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4985000", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T02:15:17.867376+00:00", "modelId": "ornith-ai/Ornith-1.5-9B-NVFP4", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8818103892, "estimatedRequiredGiB": 9.883, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 8842796317, "modelscopeLicense": "mit", "modelscopeParams": 6728625904, "modelscopeTags": ["license:mit", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "modelopt", "repositoryOnDiskBytes": 8842796317}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T18:09:15.235288+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4985113", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T02:15:17.867362+00:00", "modelId": "neuralmagic/gemma-2-2b-quantized.w8a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6746735136, "estimatedRequiredGiB": 7.565, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 6768615496, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 6768615496}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T18:09:27.989767+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4985115", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T02:42:24.494006+00:00", "modelId": "neuralmagic/Llama-3.2-1B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 2024670536, "estimatedRequiredGiB": 2.273, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 2033824885, "modelscopeLicense": "llama3.2", "modelscopeParams": 1498482688, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm_tokenizer_patch", "lastTerminalAt": "2026-09-20T16:44:23.353846+00:00", "modelType": "llama", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "compressed-tensors", "successCount": 0, "successRate": 0.0, "targetGpu": "Ascend_910-b3", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 2033824885}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T18:39:41.567990+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4985663", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-21T02:42:24.494032+00:00", "modelId": "inceptionai/Jais-2-8B-Chat", "modelProfile": {"architectures": ["Jais2ForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16180861064, "estimatedRequiredGiB": 18.097, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "jais2", "modelscopeFileSize": 16193187697, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8090401280, "modelscopeTags": ["license:apache-2.0", "model_type:jais2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16193187697}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T18:39:53.164098+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4985692", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -540,7 +537,6 @@
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T03:02:20.757780+00:00", "modelId": "inclusionAI/Ling-3.0-tiny", "modelProfile": {"architectures": ["BailingMoeV3ForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15787992416, "estimatedRequiredGiB": 17.659, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "bailing_hybrid", "modelscopeFileSize": 15801179822, "modelscopeLicense": "mit", "modelscopeParams": 7893392800, "modelscopeTags": ["license:mit", "model_type:bailing_hybrid", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 15801179822}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:01:32.459840+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4985911", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T03:25:53.878044+00:00", "modelId": "neuralmagic/starcoder2-7b-quantized.w8a8", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7857274344, "estimatedRequiredGiB": 8.785, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 7860632064, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 7400416256, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7860632064}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:09:12.237648+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4986045", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T03:25:53.878006+00:00", "modelId": "RedHatAI/Phi-3-medium-128k-instruct-FP8", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 14289052464, "estimatedRequiredGiB": 15.972, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 14291553638, "modelscopeLicense": "mit", "modelscopeParams": 13960238080, "modelscopeTags": ["license:mit", "model_type:phi3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:fp8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 14291553638}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:12:09.492560+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4986098", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T03:25:53.878052+00:00", "modelId": "ewinregirgojr/Qwen3.8-14B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:23:30.183489+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4986237", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T03:45:50.542869+00:00", "modelId": "RedHatAI/Meta-Llama-3-8B-Instruct-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093325064, "modelscopeLicense": "llama3", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093325064}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:30:14.636274+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986332", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T03:45:50.542801+00:00", "modelId": "sbintuitions/sarashina2.2-3b-instruct-v0.1", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 6711252896, "estimatedRequiredGiB": 7.503, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 6713123803, "modelscopeLicense": "mit", "modelscopeParams": 3355609600, "modelscopeTags": ["license:mit", "model_type:llama", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 6713123803}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:31:34.206245+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986333", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T03:49:21.585684+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "2e9cebc01b20081f0839fa18fe9ca4f7d059509f235699027201fe328177b304", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 54864980000, "estimatedRequiredGiB": 61.361, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 54904930780, "modelscopeLicense": "gemma", "modelscopeParams": 27432406640, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 54904930780}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T19:46:10.235701+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4986586", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -549,12 +545,8 @@
|
||||
{"failReason": null, "framework": "transformers", "lastSyncTime": "2026-09-21T04:11:57.967204+00:00", "modelId": "BAAI/RoboBrain2.5-8B-NV", "modelProfile": {"architectures": ["Qwen3VLForConditionalGeneration"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 17534339552, "estimatedRequiredGiB": 19.609, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_vl", "modelscopeFileSize": 17545918210, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8767123696, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_vl", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 17545918210}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:09:23.973756+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986856", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "transformers", "lastSyncTime": "2026-09-21T04:11:57.967149+00:00", "modelId": "SyonLi/Qwen3-4B-Instruct-2507-Segmenter", "modelProfile": {"architectures": ["Qwen3ForCut"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8046243744, "estimatedRequiredGiB": 9.01, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3", "modelscopeFileSize": 8062181376, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 4023098880, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8062181376}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:09:24.163788+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986867", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "transformers", "lastSyncTime": "2026-09-21T04:11:57.967122+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-8bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "3086a71adcdca90005b4967eb30749c44cd82389aff8d599758e4b91675d40dc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1243645809, "estimatedRequiredGiB": 1.395, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 1248409935, "modelscopeLicense": "other", "modelscopeParams": 329251584, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 1248409935}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:09:24.460163+00:00", "targetGpu": "Iluvatar_bi-100", "taskId": "4986875", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T04:36:00.569189+00:00", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-6bit", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 22263760648, "estimatedRequiredGiB": 24.908, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 22286993822, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6019449856, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-serve", "custom_tag:qwen3_5", "custom_tag:apple-silicon", "custom_tag:mtp", "custom_tag:speculative-decoding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 22286993822}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:33:04.562192+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987155", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T04:36:00.569103+00:00", "modelId": "Mia-AiLab/Qwen3.8-27B-EXL3-3.5bpw", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 15338408461, "estimatedRequiredGiB": 17.168, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 15362064483, "modelscopeLicense": "apache-2.0", "modelscopeParams": 7669052656, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:exl3", "custom_tag:exllamav3", "custom_tag:quantization", "custom_tag:long-context", "custom_tag:speculative-decoding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "exl3", "repositoryOnDiskBytes": 15362064483}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:33:04.573215+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987156", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T04:51:55.164788+00:00", "modelId": "RedHatAI/Phi-3-mini-128k-instruct-quantized.w8a16", "modelProfile": {"architectures": ["Phi3ForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4020365960, "estimatedRequiredGiB": 4.496, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "phi3", "modelscopeFileSize": 4022810352, "modelscopeLicense": "mit", "modelscopeParams": 3821079552, "modelscopeTags": ["license:mit", "model_type:phi3", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4022810352}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:41:37.494440+00:00", "targetGpu": "hygon_k100-ai", "taskId": "4987371", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164858+00:00", "modelId": "TokenRhythm/NeoHorse-1-4B", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8411552560, "estimatedRequiredGiB": 9.428, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5_text", "modelscopeFileSize": 8435794426, "modelscopeLicense": "Apache License 2.0", "modelscopeParams": 4205751296, "modelscopeTags": ["license:Apache License 2.0", "model_type:qwen3_5_text", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:agentic", "custom_tag:tool-use", "custom_tag:coding", "custom_tag:reasoning", "custom_tag:instruction-following"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8435794426}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:51:02.382902+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987487", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164913+00:00", "modelId": "ewinregirgojr/Qwen3.8-9B-Instruct-Turbo", "modelProfile": {"architectures": ["Qwen3_5ForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 30679155168, "estimatedRequiredGiB": 34.312, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 30702164068, "modelscopeLicense": "apache-2.0", "modelscopeParams": 15180130288, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3", "custom_tag:14b", "custom_tag:deltanet", "custom_tag:linear-attention", "custom_tag:hybrid-attention", "custom_tag:distillation", "custom_tag:pruned", "custom_tag:reasoning", "custom_tag:tool-calling", "custom_tag:agent", "custom_tag:coding", "custom_tag:gguf", "deploy:swingdeploy"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 30702164068}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:51:02.378387+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987486", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T04:51:55.164866+00:00", "modelId": "VerStella/Qwen3.8-27B-heretic-ara-W8A8-DCU", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 31228126952, "estimatedRequiredGiB": 34.934, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 31258377392, "modelscopeLicense": "apache-2.0", "modelscopeParams": 27781427952, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:w8a8", "custom_tag:int8", "custom_tag:quantized"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 31258377392}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T20:51:02.374696+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987485", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:14:26.178375+00:00", "modelId": "JANGQ-AI/AppleScript-8B-JANG_4M", "modelProfile": {"architectures": ["ZayaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7915599888, "estimatedRequiredGiB": 8.885, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "zaya", "modelscopeFileSize": 7950463599, "modelscopeLicense": "other", "modelscopeParams": 2274126328, "modelscopeTags": ["license:other", "model_type:zaya", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:applescript", "custom_tag:macos", "custom_tag:computer-use", "custom_tag:agent", "custom_tag:tool-calling", "custom_tag:function-calling", "custom_tag:mlx", "custom_tag:moe", "custom_tag:osaurus"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 7950463599}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:13:06.834956+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4987784", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:25:05.959941+00:00", "modelId": "neuralmagic/Llama-3.2-3B-Instruct-FP8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4395007696, "estimatedRequiredGiB": 4.922, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 4404161088, "modelscopeLicense": "llama3.2", "modelscopeParams": 3606752256, "modelscopeTags": ["license:llama3.2", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:llama", "custom_tag:llama-3", "custom_tag:neuralmagic", "custom_tag:llmcompressor"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4404161088}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:22:47.667525+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4987928", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T05:42:03.443555+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v3-9B", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "1557c77dfe4d3dc1a68ae6b8349e7d094da7eec7752d29442a9d6d27e5f13121", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 18483466448, "estimatedRequiredGiB": 20.702, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 18523972484, "modelscopeLicense": "gemma", "modelscopeParams": 9241705984, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 18523972484}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-20T21:26:40.552644+00:00", "targetGpu": "Sunrise_pt-200-x1", "taskId": "4987963", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -605,7 +597,6 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T10:33:30.477829+00:00", "modelId": "RedHatAI/starcoder2-3b-quantized.w8a16", "modelProfile": {"architectures": ["Starcoder2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4093180600, "estimatedRequiredGiB": 4.578, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "starcoder2", "modelscopeFileSize": 4096479058, "modelscopeLicense": "bigcode-openrail-m", "modelscopeParams": 3181366272, "modelscopeTags": ["license:bigcode-openrail-m", "model_type:starcoder2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:code"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4096479058}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T02:10:00.507387+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991869", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T10:33:30.477823+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 3403624536, "estimatedRequiredGiB": 3.828, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 3425449433, "modelscopeLicense": "llama2", "modelscopeParams": 3204165888, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 3425449433}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T02:10:00.601936+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991870", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T10:33:30.477818+00:00", "modelId": "RedHatAI/gemma-2-9b-it-quantized.w4a16", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 7963207224, "estimatedRequiredGiB": 8.924, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 7985124576, "modelscopeLicense": "llama2", "modelscopeParams": 10159209984, "modelscopeTags": ["license:llama2", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 7985124576}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T02:10:00.705684+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4991871", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T10:33:30.477508+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality-FP16", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952487299, "estimatedRequiredGiB": 33.498, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29973200235, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:m1", "custom_tag:m2", "custom_tag:fp16", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29973200235}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T02:21:28.285988+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4992046", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T11:15:42.052931+00:00", "modelId": "RedHatAI/Meta-Llama-3.1-8B-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093261800, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093261800}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T03:11:17.186758+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4992751", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-21T11:35:03.480276+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "1566bbd6c69427ad32f7f86aeba4e39850a1a7af5c6f4ef3c9ca82f0e882f3c5", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093204103, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm", "custom_tag:quantized", "custom_tag:8-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093204103}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T03:31:36.734307+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4993022", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T12:09:51.363778+00:00", "modelId": "LiquidAI/LFM2.5-1.2B-Thinking-MLX-6bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 951093016, "estimatedRequiredGiB": 1.068, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 955963132, "modelscopeLicense": "other", "modelscopeParams": 256113408, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx", "custom_tag:reasoning"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "recentProfileFeedback": {"attributableFailureCount": 0, "consecutiveFailures": 0, "decisionFailureRate": 0.0, "decisionSuccessRate": 0.0, "decisionTotal": 0, "failureBreakdown": {"ambiguous_runtime": 1}, "failureCount": 1, "failureRate": 1.0, "framework": "vllm-customized", "lastTerminalAt": "2026-09-21T02:15:17.867294+00:00", "modelType": "lfm2", "pendingCount": 0, "pendingRate": 0.0, "platformFailureCount": 0, "quantizationMethod": "none", "successCount": 0, "successRate": 0.0, "targetGpu": "Cambricon_mlu-370-x8", "taskType": "text-generation", "total": 1, "unresolvedFailureCount": 1}, "repositoryOnDiskBytes": 955963132}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T03:59:26.348415+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4993324", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -628,7 +619,6 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T16:21:18.661378+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a8", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "ef4c08c913d3f64cae22550f0c4a189795901d4b11e26d8e640e07e857659fb9", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084013360, "estimatedRequiredGiB": 10.162, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093204103, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm", "custom_tag:quantized", "custom_tag:8-bit"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093204103}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:07:48.338371+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "4996781", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-customized", "lastSyncTime": "2026-09-21T16:21:18.661327+00:00", "modelId": "XHToken/Spark-X2.5-4B", "modelProfile": {"architectures": ["Spark2_5ForCausalLM"], "configFingerprint": "1b41437401598cde7836ccc25883635c2738b06a5af924b7497b6e747e655b4b", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 8224192408, "estimatedRequiredGiB": 9.209, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "spark2_5", "modelscopeFileSize": 8239718268, "modelscopeLicense": "apache-2.0", "modelscopeParams": 4112079360, "modelscopeTags": ["license:apache-2.0", "model_type:spark2_5", "library:safetensors", "library:", "task:text-generation", "custom_tag:llm", "custom_tag:sparkx2_5"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 8239718268}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:09:36.692226+00:00", "targetGpu": "Cambricon_mlu-370-x8", "taskId": "4996782", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T16:21:18.661292+00:00", "modelId": "mlx-community/gemma-4-31B-it-qat-OptiQ-4bit", "modelProfile": {"architectures": ["Gemma4ForConditionalGeneration"], "configFingerprint": "1c0a4f69628b2a0ad37a6eee139a4c98afed7d8c93529a9ef42a90756e4e1940", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 23524076260, "estimatedRequiredGiB": 26.327, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma4", "modelscopeFileSize": 23556606333, "modelscopeLicense": "apache-2.0", "modelscopeParams": 6649116524, "modelscopeTags": ["license:apache-2.0", "model_type:gemma4", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:quantized", "custom_tag:mixed-precision", "custom_tag:4bit", "custom_tag:8bit", "custom_tag:optiq", "custom_tag:qat", "custom_tag:apple-silicon", "custom_tag:text-generation", "custom_tag:image-text-to-text", "custom_tag:gemma-4"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 23556606333}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:17:20.243727+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4996859", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T16:21:18.661392+00:00", "modelId": "Youssofal/Qwen3.8-27B-MTPLX-Optimized-Quality", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29952488643, "estimatedRequiredGiB": 33.497, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29972714678, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8027131120, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:apple-silicon", "custom_tag:macos", "custom_tag:speculative-decoding", "custom_tag:multi-token-prediction", "custom_tag:qwen", "custom_tag:qwen3.8", "custom_tag:mtp", "custom_tag:mtplx", "custom_tag:local-ai", "custom_tag:coding", "custom_tag:qwen3-8", "custom_tag:qwen-3.8", "custom_tag:local-llm", "custom_tag:llm", "custom_tag:m5", "custom_tag:m5-max", "custom_tag:m4", "custom_tag:m3", "custom_tag:macbook-pro", "custom_tag:mac-studio", "custom_tag:opencode", "custom_tag:claude-code", "custom_tag:27b", "custom_tag:qwen3.8-27b", "custom_tag:qwen3-8-27b"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29972714678}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:17:20.220538+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "4996858", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T16:21:18.661372+00:00", "modelId": "blue2star/Qwen-Image-2.1-PE-T2I-ComfyUI", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 29306922460, "estimatedRequiredGiB": 32.775, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 29326950127, "modelscopeLicense": "other", "modelscopeParams": 18821595084, "modelscopeTags": ["license:other", "model_type:qwen3_5", "library:pytorch", "library:transformer", "library:lora", "library:safetensors", "task:text-generation", "custom_tag:qwen", "custom_tag:prompt-rewriting", "custom_tag:text-to-image"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 29326950127}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:17:20.209866+00:00", "targetGpu": "Biren_166m", "taskId": "4996857", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T16:50:30.559284+00:00", "modelId": "aisingapore/Gemma-SEA-LION-v4-27B-IT-NVFP4", "modelProfile": {"architectures": ["Gemma3ForConditionalGeneration"], "configFingerprint": "53aadead5bb69dd6c72976cf6353832080116244f86ffae86297646e5bb48fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 20886763512, "estimatedRequiredGiB": 23.388, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma3", "modelscopeFileSize": 20926892207, "modelscopeLicense": "gemma", "modelscopeParams": 28842037282, "modelscopeTags": ["license:gemma", "model_type:gemma3", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 20926892207}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:29:25.977872+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4997055", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T16:50:30.559229+00:00", "modelId": "nm-testing/tinyllama-oneshot-w8w8-test-static-shape-change", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "26bfee231695e882b96dee0191bc1dfec25bcad132af96a7ee5f0e26f3cb9fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 1231270112, "estimatedRequiredGiB": 1.378, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 1233120748, "modelscopeLicense": null, "modelscopeParams": 1100048384, "modelscopeTags": ["model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 1233120748}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T08:29:25.981023+00:00", "targetGpu": "Ascend_910-b3", "taskId": "4997054", "taskType": "text-generation", "verifyResult": null}
|
||||
@@ -701,7 +691,6 @@
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T20:50:04.063150+00:00", "modelId": "aisingapore/Llama-SEA-LION-v3.5-70B-R-NVFP4", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 42709318472, "estimatedRequiredGiB": 47.753, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 42728540075, "modelscopeLicense": "llama3.1", "modelscopeParams": 70553707056, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 42728540075}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T12:49:01.249214+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001061", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T21:02:46.265685+00:00", "modelId": "mlx-community/LFM2.5-1.2B-Thinking-6bit", "modelProfile": {"architectures": ["Lfm2ForCausalLM"], "configFingerprint": "53aadead5bb69dd6c72976cf6353832080116244f86ffae86297646e5bb48fdc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 951093016, "estimatedRequiredGiB": 1.068, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "lfm2", "modelscopeFileSize": 955857175, "modelscopeLicense": "other", "modelscopeParams": 256113408, "modelscopeTags": ["license:other", "model_type:lfm2", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:liquid", "custom_tag:lfm2.5", "custom_tag:edge", "custom_tag:mlx"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 955857175}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T13:01:28.899242+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5001186", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-mlu", "lastSyncTime": "2026-09-21T21:18:49.765030+00:00", "modelId": "RedHatAI/SmolLM-135M-Instruct-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "b3ef6bcf57e6bb4fc9ede989e67a765b0e40b19fbd407f1f8452833c12bd5c38", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 219876272, "estimatedRequiredGiB": 0.25, "gpuMemoryEvidence": {"memoryGiB": 24.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 24.0, "maximumRepositorySizeGiB": 20.0, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 223261617, "modelscopeLicense": "apache-2.0", "modelscopeParams": 162826560, "modelscopeTags": ["license:apache-2.0", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 223261617}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T13:17:35.253765+00:00", "targetGpu": "Cambricon_mlu-370-x4", "taskId": "5001324", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm", "lastSyncTime": "2026-09-21T21:18:49.765067+00:00", "modelId": "ddalcu/Qwen3.8-27B-MLX-Serve-iQ-MLX-3.8bpw", "modelProfile": {"architectures": ["Qwen3_5ForConditionalGeneration"], "configFingerprint": "2ebd1d3ee8ed952dd08e24120ad5d7342c9a018f55c478225a242b664721e8d0", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 12999919368, "estimatedRequiredGiB": 14.555, "gpuMemoryEvidence": {"memoryGiB": 100.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 100.0, "maximumRepositorySizeGiB": 83.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "qwen3_5", "modelscopeFileSize": 13023155398, "modelscopeLicense": "apache-2.0", "modelscopeParams": 3544073216, "modelscopeTags": ["license:apache-2.0", "model_type:qwen3_5", "library:mlx", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:mlx", "custom_tag:mlx-serve", "custom_tag:qwen3_5", "custom_tag:apple-silicon", "custom_tag:imatrix", "custom_tag:mtp", "custom_tag:speculative-decoding"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 13023155398}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T13:17:35.237740+00:00", "targetGpu": "Kunlunxin_p-800", "taskId": "5001319", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_fix_tokenizer", "lastSyncTime": "2026-09-21T22:22:17.565308+00:00", "modelId": "inceptionai/Jais-2-8B-Chat", "modelProfile": {"architectures": ["Jais2ForCausalLM"], "configFingerprint": "16f74ea0f409861ab0c8169a25d9484cb482e27d6a3ad7c529a70b7cdbfabee3", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 16180861064, "estimatedRequiredGiB": 18.097, "gpuMemoryEvidence": {"memoryGiB": 32.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 32.0, "maximumRepositorySizeGiB": 26.667, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "jais2", "modelscopeFileSize": 16193187697, "modelscopeLicense": "apache-2.0", "modelscopeParams": 8090401280, "modelscopeTags": ["license:apache-2.0", "model_type:jais2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation"], "modelscopeTasks": ["text-generation"], "quantizationMethod": null, "repositoryOnDiskBytes": 16193187697}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T14:18:50.561697+00:00", "targetGpu": "Iluvatar_bi-150", "taskId": "5002167", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm-patch-tokenizer", "lastSyncTime": "2026-09-21T22:23:42.064201+00:00", "modelId": "neuralmagic/Meta-Llama-3.1-8B-quantized.w8a16", "modelProfile": {"architectures": ["LlamaForCausalLM"], "configFingerprint": "0d5bc82787a89d4313672808d1a8629a1fc675c91aa6a08865f770aa24447fbc", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 9084040952, "estimatedRequiredGiB": 10.163, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "llama", "modelscopeFileSize": 9093251901, "modelscopeLicense": "llama3.1", "modelscopeParams": 8030261248, "modelscopeTags": ["license:llama3.1", "model_type:llama", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 9093251901}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T14:20:50.936887+00:00", "targetGpu": "hygon_k100-ai", "taskId": "5002188", "taskType": "text-generation", "verifyResult": null}
|
||||
{"failReason": null, "framework": "vllm_tokenizer_patch", "lastSyncTime": "2026-09-21T22:48:12.757474+00:00", "modelId": "RedHatAI/gemma-2-2b-it-quantized.w8a8", "modelProfile": {"architectures": ["Gemma2ForCausalLM"], "configFingerprint": "7b5500238d52e9e2a21471d722bc1636ca047e1a12ba49c491593b86bec602b8", "configOptimization": {"applied": false, "source": "official"}, "configSource": "modelhub_live", "estimatedLoadBytes": 4385522016, "estimatedRequiredGiB": 4.926, "gpuMemoryEvidence": {"memoryGiB": 64.0, "source": "local_modelhub_preflight_oom"}, "gpuMemoryGiB": 64.0, "maximumRepositorySizeGiB": 53.333, "memorySizingBasis": "recursive_repository_on_disk", "modelCard": {}, "modelType": "gemma2", "modelscopeFileSize": 4407346443, "modelscopeLicense": "gemma", "modelscopeParams": 3204165888, "modelscopeTags": ["license:gemma", "model_type:gemma2", "library:transformer", "library:safetensors", "library:pytorch", "task:text-generation", "custom_tag:int8", "custom_tag:vllm"], "modelscopeTasks": ["text-generation"], "quantizationMethod": "compressed-tensors", "repositoryOnDiskBytes": 4407346443}, "outcome": "pending", "status": "waiting", "submitTime": "2026-09-21T14:48:11.802242+00:00", "targetGpu": "Ascend_910-b3", "taskId": "5002502", "taskType": "text-generation", "verifyResult": null}
|
||||
|
||||
Reference in New Issue
Block a user